diff --git a/.config/nextest.toml b/.config/nextest.toml index cdf24654e..358ac5719 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -121,9 +121,12 @@ threads-required = 'num-test-threads' # suite while passing in 16.5s on its own, a 9x margin that rules out slowness. # `max-threads = 1` alone does not help, because the competition is the # unit-test fan-out rather than the other server tests. The cost is bounded — -# two tests, one of which the default filter already excludes. +# three tests, one of which the default filter already excludes. +# +# `pitr_restore` boots three servers in sequence and runs the restore three +# times, so it needs the same raised kill. [[profile.default.overrides]] -filter = 'binary(crash_wal_truncation) | binary(crash_ilp_timeseries_write)' +filter = 'binary(crash_wal_truncation) | binary(crash_ilp_timeseries_write) | binary(pitr_restore)' test-group = 'server-process-serial' threads-required = 'num-test-threads' slow-timeout = { period = "30s", terminate-after = 8 } @@ -147,7 +150,7 @@ slow-timeout = { period = "30s", terminate-after = 8 } # hide the cause rather than fix it. The tail this costs is bounded — roughly a # dozen crash/shutdown tests at about ten seconds each. [[profile.default.overrides]] -filter = 'binary(wal_direct_io) | binary(ilp_client_address) | binary(crash_recovery) | binary(crash_recovery_overlays) | binary(crash_recovery_analytics) | binary(crash_resp_kv_write) | binary(crash_metadata_applier_wedge) | binary(crash_dropped_collection_reclaim) | binary(crash_purge_not_resurrected) | binary(crash_mid_replay) | binary(crash_checkpoint_corruption) | binary(crash_checkpoint_truncate_window) | binary(crash_refused_write_not_resurrected) | binary(crash_replay_fail_stop) | binary(crash_core_stall) | binary(crash_replay_stamp) | binary(crash_replay_stamp_calvin) | binary(calvin_hold_liveness) | binary(apply_pipeline_group_independence) | binary(crash_kv_atomic_autocommit) | test(/^cases::startup_failure::/) | test(/^cases::shutdown_in_flight::/) | test(/^cases::shutdown_budget::/) | test(/^cases::shutdown_abort_offender::/) | test(/^cases::shutdown_idempotent::/)' +filter = 'binary(wal_direct_io) | binary(ilp_client_address) | binary(timeseries_write_events) | binary(crash_recovery) | binary(crash_recovery_overlays) | binary(crash_recovery_analytics) | binary(crash_resp_kv_write) | binary(crash_metadata_applier_wedge) | binary(crash_dropped_collection_reclaim) | binary(crash_purge_not_resurrected) | binary(crash_mid_replay) | binary(crash_checkpoint_corruption) | binary(crash_checkpoint_truncate_window) | binary(crash_refused_write_not_resurrected) | binary(crash_replay_fail_stop) | binary(crash_core_stall) | binary(crash_replay_stamp) | binary(crash_replay_stamp_calvin) | binary(calvin_hold_liveness) | binary(apply_pipeline_group_independence) | binary(crash_kv_atomic_autocommit) | test(/^cases::startup_failure::/) | test(/^cases::shutdown_in_flight::/) | test(/^cases::shutdown_budget::/) | test(/^cases::shutdown_abort_offender::/) | test(/^cases::shutdown_idempotent::/)' test-group = 'server-process' threads-required = 'num-test-threads' diff --git a/CHANGELOG.md b/CHANGELOG.md index bdefa6119..0a0310e93 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,6 +14,7 @@ NodeDB uses [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - Native-protocol `SELECT` returns nested objects and arrays as structured values, not JSON text. `nodedb_types::conversion::json_to_value_display` is replaced by `json_to_value_ref`. - JWT `metadata` claims keep their JSON type instead of being coerced to strings. - `document_get` for a missing id returns `Ok(None)` instead of a serialization error. +- **`[server] single_node_calvin` is removed.** A server without a `[cluster]` section always runs the single-node Calvin sequencer, so cross-core (cross-vShard) transactions always commit atomically. A config that still sets the key fails to load as an unknown field. Remove the line. ### Added diff --git a/Cargo.lock b/Cargo.lock index 63fdf1aab..b9c071cbe 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -4378,6 +4378,7 @@ dependencies = [ "faultbox", "futures", "getrandom 0.4.3", + "hex", "hmac 0.12.1", "nexar", "nodedb-raft", @@ -4427,6 +4428,7 @@ dependencies = [ "nodedb-test-support", "nodedb-types", "quinn", + "rdkafka", "redb", "rustls", "serde_json", @@ -4568,6 +4570,7 @@ dependencies = [ name = "nodedb-query" version = "0.5.0" dependencies = [ + "hex", "nodedb-fts", "nodedb-spatial", "nodedb-types", @@ -4620,6 +4623,7 @@ name = "nodedb-sql" version = "0.5.0" dependencies = [ "chrono", + "hex", "nodedb-query", "nodedb-spatial", "nodedb-types", @@ -4685,6 +4689,7 @@ dependencies = [ "bytemuck", "crc32c", "getrandom 0.4.3", + "hex", "nanoid", "nodedb-codec", "rand 0.10.2", diff --git a/docs/architecture.md b/docs/architecture.md index c132910af..2ffd45ea0 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -162,7 +162,7 @@ SET cross_shard_txn = 'best_effort_non_atomic'; (Bare `best_effort` is deliberately rejected; invalid values return SQLSTATE `22023`.) -**Single-node deployments run Calvin by default** (`[server] single_node_calvin = true`): a standalone node synthesizes a one-node sequencer group so transactions spanning multiple cores (vShards) commit atomically instead of being rejected. Set it `false` to force the legacy fast path. Uncontended single-shard point writes bypass the sequencer entirely and go directly through the relevant data-group Raft; contended or predicate/bulk writes route through the deterministic scheduler. +**Single-node deployments always run Calvin.** A node with no `[cluster]` section synthesizes a one-node cluster with its own sequencer group, so transactions spanning multiple cores (vShards) commit atomically. Uncontended single-shard point writes bypass the sequencer entirely and go directly through the relevant data-group Raft; contended or predicate/bulk writes route through the deterministic scheduler. **Overlay hygiene.** Per-transaction staging overlays are kept alive by every staged write/read; overlays orphaned by vanished clients are reaped after a 6-hour lease. The `nodedb_active_txn_overlays` Prometheus gauge tracks live overlays. Data-Plane resource rejection surfaces as SQLSTATE `53200` (backpressure — retry when pressure subsides). diff --git a/docs/databases.md b/docs/databases.md index 131cc8371..e5a17194c 100644 --- a/docs/databases.md +++ b/docs/databases.md @@ -389,8 +389,8 @@ Database operations are gated by role: | `CLONE DATABASE` | `Superuser` | | `MIRROR DATABASE` | `Superuser` | | `MOVE TENANT` | `Superuser` | -| `BACKUP DATABASE` | `DatabaseOwner` or higher | -| `RESTORE DATABASE` | `Superuser` | +| `BACKUP DATABASE` | `DatabaseOwner` or `Superuser` | +| `RESTORE DATABASE` | `DatabaseOwner` or `Superuser`; `Superuser` when the database does not exist | See [Roles & Permissions](security/rbac.md) for full role definitions. diff --git a/docs/getting-started.md b/docs/getting-started.md index 7f2ab69f9..2ef35d74d 100644 --- a/docs/getting-started.md +++ b/docs/getting-started.md @@ -295,12 +295,252 @@ listener is already plaintext. | -------------------------------- | ---------------------------------- | ------- | | `checkpoint.interval_secs` | `NODEDB_CHECKPOINT_INTERVAL_SECS` | `300` | | `checkpoint.wal_segment_target_mb` | `NODEDB_WAL_SEGMENT_TARGET_MB` | `64` | +| `checkpoint.wal_archive_interval_secs` | none | `10` | | `tuning.wal.direct_io` | `NODEDB_WAL_DIRECT_IO` | `true` | | `tuning.wal.write_buffer_size` | `NODEDB_WAL_WRITE_BUFFER_SIZE` | `2MiB` | -Both intervals must be positive. `write_buffer_size` accepts a memory size and -must be at least `64KiB`. Turn `direct_io` off only on a filesystem that -rejects `O_DIRECT`. +All intervals must be positive. A zero interval stops startup. +`write_buffer_size` accepts a memory size and must be at least `64KiB`. Turn +`direct_io` off only on a filesystem that rejects `O_DIRECT`. + +`wal_archive_interval_secs` applies only when `[cold_storage]` is configured. +Each interval, every sealed WAL segment the archive does not hold is uploaded +to cold storage. A sealed segment therefore reaches the archive within one +interval. The segment still being written is uploaded once it is sealed. +Checkpoints never delete a WAL segment the archive does not hold. + +**Point-in-time recovery:** + +| Config field | Environment variable | Default | +| ---------------------------------- | -------------------- | ------- | +| `pitr.enabled` | none | `false` | +| `pitr.base_snapshot_interval_secs` | none | `86400` | +| `pitr.base_snapshot_retention` | none | `7` | +| `pitr.restore_point_interval_secs` | none | `0` | + +```toml +[pitr] +enabled = true +base_snapshot_interval_secs = 86400 +base_snapshot_retention = 7 +restore_point_interval_secs = 3600 + +[cold_storage] +bucket = "my-nodedb-cold" + +[encryption] +key_path = "/etc/nodedb/keys/wal.key" +``` + +With `pitr.enabled = true`, startup stops unless `[cold_storage]` is +configured and opens, and unless `[encryption]` is configured. The WAL archive +is the only copy of a segment once a checkpoint deletes it locally. Base +snapshots are encrypted with the WAL key. With `pitr.enabled = false` and no +`[cold_storage]`, checkpoints delete WAL segments without archiving them. + +Archived segments are stored under `{prefix}wal/{node_id}/{incarnation}/`. +The incarnation is a random id stored in the data directory. A node whose +data directory is wiped gets a new incarnation, so it never mixes its +segments with those of its earlier life. + +Each `base_snapshot_interval_secs`, the node takes a base snapshot of every +Data Plane core and of its catalogs. Bases are stored in `[snapshot_storage]` +under `{node_id}/{incarnation}/`. The first base is taken once startup +completes when the node has none. A base must finish within one interval. + +Bases are incremental. Each image is cut into chunks of 256 KiB to 4 MiB at +content-defined boundaries. A chunk is stored once, under `chunks/{id}` in +the same directory. The id is an HMAC of the chunk content under a key +derived from the WAL key, so it reveals nothing about the content. A new base +uploads only the chunks the store lacks. Every base manifest still lists +every chunk it needs, so a restore reads one base and never its parent. + +After each base, the node keeps the newest `base_snapshot_retention` bases +and deletes older ones. It never deletes the base it just took. It then +deletes archived WAL segments whose records all lie below the oldest kept +base, so every kept base can replay forward. The segment holding that base's +start is always kept. It also deletes every chunk no kept base lists. A chunk +a base still being written relies on is never deleted. Both values must be +positive. A zero value stops startup. + +A failed run is retried at the next interval. These metrics report the task: + +- `nodedb_pitr_base_last_success_timestamp_seconds` +- `nodedb_pitr_base_snapshots` +- `nodedb_pitr_base_last_failure_timestamp_seconds` +- `nodedb_pitr_base_failures_total` +- `nodedb_pitr_wal_segments_collected_total` +- `nodedb_pitr_base_chunks_uploaded_total` +- `nodedb_pitr_base_chunk_bytes_uploaded_total` +- `nodedb_pitr_base_chunks_reused_total` +- `nodedb_pitr_base_chunks_collected_total` + +**Restoring one node:** + +```bash +nodedb restore --config /etc/nodedb/nodedb.toml --target-time 2026-09-01T12:00:00Z --dry-run +nodedb restore --config /etc/nodedb/nodedb.toml --target-lsn 48213 +``` + +The restore runs offline, with the server stopped. It writes the newest base +at or below the target, plus the archived WAL cut at the target, into the +data directory of the config. The data directory must be empty. A time +target is RFC 3339 or epoch seconds, milliseconds or microseconds. +`--dry-run` prints the plan and writes nothing. `--incarnation` picks the +node life when several hold bases. Start the server afterwards. It needs no +flag. + +**Cluster restore points:** + +A cluster rewinds only to a restore point: one instant every Raft group of +the cluster agrees on. + +```sql +CREATE RESTORE POINT; +SHOW RESTORE POINTS; +``` + +- `CREATE RESTORE POINT` takes a point now and returns its row. +- `SHOW RESTORE POINTS` lists every point, oldest first. +- Both are superuser only, and both refuse on a node outside a cluster. +- A row holds `id`, `hlc` (the point's watermark, HLC nanoseconds) and + `created_at_ms`. + +With `restore_point_interval_secs` above `0`, the cluster also takes a point +each interval. A point cuts every data group and the Calvin sequencer at its +watermark. Each node records every group's place at the point in its WAL. +Every write a group's log places after its cut records a commit time above +the watermark, on every replica. + +Each node also copies the metadata group's committed log to +`{prefix}raft/{node_id}/{incarnation}/` in cold storage. The metadata log +never compacts past the entries that copy holds. + +**Restoring a cluster:** + +1. Pick the point with `SHOW RESTORE POINTS`. +2. Stop every node of the cluster. +3. On every node, empty the data directory except `tls/`. +4. On every node, run the restore: + + ```bash + nodedb restore --config /etc/nodedb/nodedb.toml --cluster --restore-point 5812 + ``` + +5. Start every node. + +The restore needs the point in the node's archived WAL. After a node records +a point, it seals the WAL segment that holds the records, so they reach the +archive within one `wal_archive_interval_secs`. A point taken less than one +interval before the stop can be missing from the archive. The restore then +refuses and names the node and the point. + +Each node's restore writes: + +- The newest base holding nothing the restore drops, in data or metadata. +- The archived WAL through its last kept record. Each record is kept or + dropped by one rule: + - A write keeps its commit HLC. It is kept when that HLC is below the + watermark, wherever the WAL placed it. + - A record that a metadata entry's apply appended carries the entry's + stamp, and follows the entry. + - Any other record (a checkpoint, a tombstone) carries no clock. It is + kept when it lies before this node's cut of its vShard's group in the + WAL, or before the metadata group's cut when no group homes the vShard. +- Each data group and the sequencer, started at its place at the point. The + sequencer resumes at the first epoch after the point. +- The metadata group, started at the base's catalogs, with the archived + metadata log entries after the base, through the point. The first boot + applies them. A range this node's archive lacks is read from another + node's archive. + +Every metadata entry a node proposes carries the node's HLC. The restore +keeps an entry only when that stamp is below the watermark, so the catalog +matches the data exactly. A DDL issued after the watermark is dropped even +when it applied before the point's entry, and so is every write that +depends on it. A dropped entry keeps its index as an empty entry. + +The restore also sets two counters at the point: + +- The surrogate high-water mark rises to the highest surrogate any node + ever reserved, after the point too. No surrogate is issued twice. +- Each tenant write mark is the newest kept write. A mark the base holds at + or above the watermark stops just below it. + +The generation claim needs a cold store with conditional create (put if +absent). A store without one refuses the restore and names the store. + +Every group of every node starts at one new term, and the cluster epoch +starts at the same value. The value comes from a generation the restore +claims in cold storage under `{prefix}raft/restore/`. Every node restoring +the same point before the cluster starts takes the same generation. Each +node's first boot seals the generation, and the next restore takes a newer +one. A node the restore missed holds lower terms and a lower epoch. Raft +refuses its log, and the epoch fence stands it down. A restored data +directory holds the file `restore_generation` until its first boot. That +boot needs `[cold_storage]` to seal the generation. + +A restore that fails empties the data directory except `tls/`. Run it +again. A group a node hosts that recorded no place at the point starts with +no log and catches up from its leader. The report lists such groups. + +**Scheduled backups:** + +| Config field | Environment variable | Default | +| -------------------------- | -------------------- | ------- | +| `backup.schedule.database` | none | none | +| `backup.schedule.target` | none | none | +| `backup.schedule.cron` | none | none | +| `backup.schedule.keep` | none | none | + +```toml +[[backup.schedule]] +database = "sales" +target = "s3://my-backups/nightly/sales" +cron = "0 3 * * *" +keep = 7 + +[backup_encryption] +key_path = "/etc/nodedb/keys/backup.key" +``` + +Each `[[backup.schedule]]` entry runs `BACKUP DATABASE ` on its +cron schedule. The cron is 5-field and uses `scheduler.cron_timezone`. A run +writes one envelope named `-.ndbb` under `target`. +`` is the scheduled minute, not the time the run started. The run +then deletes the oldest envelopes of that database under `target` beyond +`keep`. Other objects under `target` are never deleted. + +- `target` is `s3:///` or `file:///`. It resolves + against `[backup_storage]`, like a `BACKUP DATABASE ... TO` URI. +- A `file://` target must name a directory inside `[backup_storage] + local_root`. +- Every field is required. `keep` must be positive. +- Two entries with the same `database` and `target` stop startup. +- A schedule needs `[backup_encryption]`. Without it, startup stops. + +In a cluster, only the leader of vShard 0 runs scheduled backups, and only +while its leader lease is valid. A leader cut off from its peers stops once +its lease lapses, before another node can take over. The lease is checked +again before the envelope write and before the record below. After each +completed run, the leader records the scheduled minute through the metadata +group, so every node holds the same record. A node reads that record only +after it has applied the metadata group through a read index its leader +confirmed. When that read cannot be confirmed, the tick is skipped. A new +leader of vShard 0 runs any due minute with no completed record. A minute that runs twice across a leader +change writes the same envelope again. Several missed minutes fire one run. +A new entry, or an entry whose `database`, `target`, or `cron` changes, +starts fresh from the current minute. A failed run retries after 60 seconds. +A run never overlaps an earlier run of the same entry. Each run is recorded +in the job history under `backup::`. These metrics report +the runs: + +- `nodedb_backup_schedule_runs_total` +- `nodedb_backup_schedule_failures_total` +- `nodedb_backup_schedule_last_success_timestamp_seconds` +- `nodedb_backup_schedule_last_failure_timestamp_seconds` +- `nodedb_backup_schedule_envelopes_deleted_total` +- `nodedb_backup_schedule_ticks_skipped_total` **Timeseries memtable settings:** @@ -319,17 +559,25 @@ between flushes. **Cluster settings** (each needs a `[cluster]` section in the config file): -| Config field | Environment variable | Default | -| ------------------------------------- | ------------------------------------ | ------- | -| `cluster.node_id` | `NODEDB_NODE_ID` | none | -| `cluster.seed_nodes` | `NODEDB_SEED_NODES` | none | -| `cluster.join_retry_max_attempts` | `NODEDB_JOIN_RETRY_MAX_ATTEMPTS` | `8` | -| `cluster.join_retry_max_backoff_secs` | `NODEDB_JOIN_RETRY_MAX_BACKOFF_SECS` | `32` | +| Config field | Environment variable | Default | +| ------------------------------------- | ------------------------------------ | --------------------------- | +| `cluster.node_id` | `NODEDB_NODE_ID` | none | +| `cluster.seed_nodes` | `NODEDB_SEED_NODES` | none | +| `cluster.swim_listen` | `NODEDB_SWIM_LISTEN` | `cluster.listen` port + 1 | +| `cluster.join_retry_max_attempts` | `NODEDB_JOIN_RETRY_MAX_ATTEMPTS` | `8` | +| `cluster.join_retry_max_backoff_secs` | `NODEDB_JOIN_RETRY_MAX_BACKOFF_SECS` | `32` | `NODEDB_SEED_NODES` takes a comma-separated `host:port` list. Both join-retry values must be positive. Setting any of these without a `[cluster]` section stops startup. +`cluster.swim_listen` is the UDP `host:port` of the SWIM failure detector. +By default it uses the `cluster.listen` IP, one port above the `cluster.listen` +port. Each node advertises its bound address to its peers, so nodes can set it +independently. Startup fails if the address cannot be bound. Open this UDP port +between every pair of nodes. Without it, nodes cannot detect a failed peer, and +a crashed node's descriptor leases block DDL until they expire. + **Maintenance loop settings:** | Config field | Environment variable | Default | diff --git a/docs/query-language.md b/docs/query-language.md index e89590b76..bb264bab9 100644 --- a/docs/query-language.md +++ b/docs/query-language.md @@ -631,7 +631,7 @@ SHOW CHANGE STREAMS; CREATE CONSUMER GROUP processors ON order_changes; -- Commit offset for a specific partition -COMMIT OFFSET PARTITION 0 AT 42 ON order_changes CONSUMER GROUP processors; +COMMIT OFFSET PARTITION 0 AT 0:42 ON order_changes CONSUMER GROUP processors; -- Batch commit all partitions at their latest consumed position COMMIT OFFSETS ON order_changes CONSUMER GROUP processors; @@ -1071,7 +1071,7 @@ Isolation level: **Snapshot Isolation (SI)**. Reads see a consistent snapshot fr ### Cross-Shard Transactions -An interactive `BEGIN ... COMMIT` block whose statements span multiple vShards or nodes commits **atomically** by default — the whole block flushes through the Calvin sequencer's durable vote/verdict barrier at COMMIT. Reads taken during the transaction (point reads, predicate scans, index probes, both sides of distributed JOINs) are OCC-validated at COMMIT; a stale read aborts with `40001` (retry the transaction). `SET cross_shard_txn = 'best_effort_non_atomic'` opts bulk loads out of cross-shard atomicity. Single-node deployments run the same path by default (`single_node_calvin = true`), so transactions spanning cores commit atomically too. See [Architecture — Cross-Shard Transactions](architecture.md#cross-shard-transactions). +An interactive `BEGIN ... COMMIT` block whose statements span multiple vShards or nodes commits **atomically** by default — the whole block flushes through the Calvin sequencer's durable vote/verdict barrier at COMMIT. Reads taken during the transaction (point reads, predicate scans, index probes, both sides of distributed JOINs) are OCC-validated at COMMIT; a stale read aborts with `40001` (retry the transaction). `SET cross_shard_txn = 'best_effort_non_atomic'` opts bulk loads out of cross-shard atomicity. Single-node deployments always run the same path, so transactions spanning cores commit atomically too. See [Architecture — Cross-Shard Transactions](architecture.md#cross-shard-transactions). ### Read-Your-Own-Writes diff --git a/docs/real-time.md b/docs/real-time.md index 409f8cc28..c0336bdd0 100644 --- a/docs/real-time.md +++ b/docs/real-time.md @@ -110,13 +110,20 @@ SHOW CHANGE STREAMS; Consumer groups track read positions independently, enabling multiple consumers to process the same stream at their own pace. +Every event carries an `offset` token `::`. In a cluster, `index` is the Raft log index of the write, so every replica gives an event the same offset. `epoch` rises when a partition moves to another data group, so offsets never go backwards. A committed offset is replicated to every node. A consumer can move to another node, or continue after a leader change, and resume exactly after its last commit. + +A node that joined a partition from a snapshot does not hold the events before the snapshot. A consumer whose offset lies below them gets a `reset_required` error naming the offset the node holds events from. It never skips events silently. When another replica still holds those events, the consume goes there instead. + ```sql -- Create a consumer group CREATE CONSUMER GROUP analytics ON order_changes; CREATE CONSUMER GROUP billing ON order_changes; -- Commit offset for a specific partition -COMMIT OFFSET PARTITION 0 AT 42 ON order_changes CONSUMER GROUP analytics; +COMMIT OFFSET PARTITION 0 AT 0:42:2 ON order_changes CONSUMER GROUP analytics; + +-- : acknowledges every event of that write +COMMIT OFFSET PARTITION 0 AT 0:42 ON order_changes CONSUMER GROUP analytics; -- Or batch commit all partitions at their latest consumed position COMMIT OFFSETS ON order_changes CONSUMER GROUP analytics; @@ -219,7 +226,16 @@ WITH ( ); ``` -Each POST includes `X-Idempotency-Key`, `X-Event-Sequence`, `X-Partition`, and `X-LSN` headers. 4xx client errors (except 429) are not retried. +Each POST includes `X-Idempotency-Key`, `X-Fencing-Token`, `X-CDC-Offset`, `X-Event-Sequence`, `X-Partition`, and `X-LSN` headers. 4xx client errors (except 429) are not retried. + +In a cluster, one node delivers a stream at a time: the node that holds the leader lease of the stream's owning Raft group. It reads every partition, including partitions whose rows live on other nodes. It checks its lease right before each POST and right before each offset commit. When that node fails, the next lease holder resumes from the committed offsets. Delivery is at-least-once: a POST can succeed after the old owner lost its lease but before it committed the offset, and the new owner then sends that event again. Two headers let an endpoint apply each event once: + +| Header | Value | What the endpoint does with it | +|---|---|---| +| `X-Idempotency-Key` | `:::`: the event's partition and position | Store each applied key. Drop a request whose key is already stored. Every owner and every retry sends the same key for the same event. | +| `X-Fencing-Token` | The Raft term of the delivering node's leader lease on the owning group | Store the highest token accepted. Reject a request with a lower token: it comes from an owner a later owner replaced. Each new owner delivers with a higher token. | + +A single node without Raft always sends token `0`. ### Kafka Bridge @@ -237,6 +253,8 @@ WITH ( Supports transactional exactly-once semantics via `enable.idempotence` and `transactional.id`. +In a cluster, one node publishes a stream at a time, under the same lease and checks as webhook delivery. Each record's key is the event's `:::`, the same for every owner and retry. Each record carries a `fencing-token` header with the publishing node's lease term, which rises with each new owner. + ### SSE Streaming HTTP Server-Sent Events for CDC consumers that can't use WebSocket: @@ -260,11 +278,11 @@ The response includes gap-detection fields alongside the events: { "events": [...], "evicted_since_last_poll": 0, - "oldest_available_lsn": 19240 + "oldest_available_offset": "0:19240:2" } ``` -`evicted_since_last_poll` is non-zero when the consumer fell behind and the stream buffer wrapped — events in that gap are lost for this consumer. `oldest_available_lsn` lets consumers detect gaps without waiting for the next event to arrive by comparing it against their last-seen LSN. +`evicted_since_last_poll` is non-zero when the consumer fell behind and the stream buffer wrapped — events in that gap are lost for this consumer. `oldest_available_offset` lets consumers detect gaps without waiting for the next event to arrive by comparing it against their last-seen offset. The `nodedb_cdc_events_dropped_total{tenant,stream}` Prometheus counter tracks drops per named stream. Alert on this counter increasing for a stream whose consumer is active. diff --git a/docs/security/rbac.md b/docs/security/rbac.md index 0b5af99fd..090c491e4 100644 --- a/docs/security/rbac.md +++ b/docs/security/rbac.md @@ -135,8 +135,8 @@ Who can execute cluster and database-scoped DDL: | `MIRROR DATABASE` | Superuser | | `ALTER DATABASE ... PROMOTE` | Superuser (irreversible; requires vault access) | | `MOVE TENANT` | Superuser | -| `BACKUP DATABASE` | Superuser, ClusterAdmin, or DatabaseOwner | -| `RESTORE DATABASE` | Superuser | +| `BACKUP DATABASE` | Superuser or DatabaseOwner | +| `RESTORE DATABASE` | Superuser or DatabaseOwner; Superuser for a database that does not exist | | `KILL SESSION` | Superuser, ClusterAdmin, or session owner | | `CREATE/ALTER/DROP OIDC PROVIDER` | Superuser or ClusterAdmin | diff --git a/docs/security/tenants.md b/docs/security/tenants.md index f4fd4087d..38e41cf59 100644 --- a/docs/security/tenants.md +++ b/docs/security/tenants.md @@ -85,6 +85,43 @@ COPY tenant_restore(acme) FROM STDIN; Backups cover all 7 engines: documents, indexes, vectors, graph edges, KV tables, timeseries, and CRDT state. Payloads are encrypted with AES-256-GCM under the tenant WAL key. +Each backup records a row count and a digest per collection and engine. A restore checks them twice: + +- Before its first write, against the backup's own rows. A mismatch refuses the restore, and nothing changes. +- After its last write, against a capture of the destination. A mismatch fails the restore with an error that names every mismatched collection. The restore does not roll back: the restored data stays in place for inspection. + +`DRY RUN` runs the first check only. + +## Database Backup/Restore + +`BACKUP DATABASE` writes every tenant's rows in one database to an object-store URI. `RESTORE DATABASE` reads it back through the tenant restore, with the same two checks. + +```sql +BACKUP DATABASE shop TO 's3://backups/shop/nightly.ndbb'; +RESTORE DATABASE shop FROM 's3://backups/shop/nightly.ndbb' DRY RUN; +RESTORE DATABASE shop FROM 'file:///srv/nodedb/backups/shop.ndbb' FORCE; +``` + +- `s3:///` uses the `[backup_storage]` endpoint, region and keys. Empty keys use IAM credentials. +- `file:///` must lie inside `[backup_storage] local_root`, with every symlink on the path followed. A symlink that leaves the root refuses the URI. Without `local_root`, every `file://` URI is refused. +- The server reads, writes, lists and deletes a `file://` object below a directory handle of the root and follows no symlink on the way. On Linux it opens through `openat2` with `RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS`. A component swapped for a symlink after the URI resolved refuses the read or write with SQLSTATE `22023`. Scheduled-backup retention never lists a symlink under the target and never deletes through one. +- The backup takes one cut for every tenant of the database and records it in the manifest. Each Raft data group captures the database's tenants when it applies the cut's barrier entry. The capture holds every entry at or below the barrier and none above it, so no row written after the cut is in the backup. A backup whose group changed leader before the capture was collected fails with a retryable error that names the group. On a server with no Raft groups the cut is the instant the backup dispatches its snapshots to the Data Plane cores. Every write, whatever path sends it, reaches a core through one dispatcher, so the backup holds every write dispatched before the cut and none after it. Each core runs the backup's snapshot as a barrier: it sees every request dispatched to that core before it, in any priority tier, and none after it. A write's restore-staleness mark, a RESTORE's included, lands on the same side of the cut as the write. +- Each tenant's backup quota admits and meters the backup and the restore, as for `COPY`. +- A restore also counts against the write quota of every collection it restores, under the name DML charges. A hard write cap on one collection refuses the restore before any write. A grant on `*` is charged once per restored row. A grant on the `tenant:` marker is charged the tenant's restored rows. A `DRY RUN` charges no write quota. +- `DEFINE SCOPE '' AS BACKUP ON 'tenant:'` defines a backup scope for one tenant. +- A malformed URI, an unknown scheme, or a path outside `local_root` fails with SQLSTATE `22023` before any store is touched. +- A backup restores only under its own database name. +- Credentials never come from the SQL text. + +```toml +[backup_storage] +local_root = "/srv/nodedb/backups" +endpoint = "" +access_key = "" +secret_key = "" +region = "us-east-1" +``` + ## Tenant Purge (GDPR Erasure) ```sql diff --git a/docs/security/threat-model.md b/docs/security/threat-model.md index a6d9f57fd..068beee6e 100644 --- a/docs/security/threat-model.md +++ b/docs/security/threat-model.md @@ -46,7 +46,7 @@ All client payloads, headers, credentials, cursors, and offsets are untrusted in | **pgwire** | Startup parameters, passwords/TLS client certs, SQL, prepared values, COPY filenames/paths and row data | `control::server::pgwire::{listener,connection,connection_identity}` authenticates the connection; `session_auth::{authenticate,build_auth_context_with_session}` binds identity/session/database; SQL routes through `nodedb_sql::plan_sql`, `authorize_task_set`, and authorized dispatch. COPY and file-oriented operations must be treated as server-side path authority and checked before opening/allocating. Parser, protocol, numeric-narrowing, request, and stream limits reject malformed or excessive input. | | **HTTP JSON, NDJSON, query streams, and resource APIs** | `Authorization`, `X-NodeDB-Database`, `X-On-Deny`, query parameters, SQL, JSON/NDJSON body, uploads, and stream chunks | `http::auth::{resolve_identity,resolve_auth}` validates bearer credentials; `http::routes::query::resolve_database_id` resolves the selected database; handlers build an authenticated context, authorize the target operation, then use the gateway/authorized dispatcher. This row also covers the dedicated `document`, `auth`, `session`, `key`, `wasm`, `subscribe`, `status`, and `cluster`/debug route modules: each must make an explicit role/capability and body-limit decision rather than inheriting authority merely from routing. WASM upload validates administrator authority and bounded module content before persistence. `http::auth::apply_on_deny_header` accepts only the trimmed, case-insensitive `SILENT` or `ERROR` presentation tokens; SQL/session denial parsing is separate. Body/stream limits, JSON extraction, bounded response collection, deadlines, and startup gating are the availability boundary. | | **HTTP health and drain** | Probe requests, drain authorization, and drain trigger | `http::server::build_router` exposes `/healthz`, `/health/live`, `/health/ready`, and `/health/drain`; `startup_gate_middleware` intentionally leaves health paths reachable while reporting non-ready state. `http::routes::health::drain` requires `ResolvedIdentity` with superuser authority. Drain is a lifecycle control, not a data path: `control::shutdown::ShutdownBus`, listener drain guards, and readiness gates stop admission and report state. Deploy the drain endpoint only behind the operator/network controls appropriate for its availability impact. | -| **CDC poll, SSE, and named streams** | Identity, collection/stream names, cursors, offsets, consumer-group commits, SSE reconnects | The transient `/v1/cdc/{collection}` and `/v1/cdc/{collection}/poll` poll/SSE surface uses an opaque, epoch-aware `control::change_stream::ChangeCursor` that preserves publication order and resets on malformed, stale, future, or wrong-epoch input. Durable named streams and consumer groups use `event::cdc::CdcOffset`, a composite `(lsn, sequence)` position whose per-partition commits reject regression and preserve same-LSN siblings. `event::cdc::{StreamRegistry,CdcRouter}` keys durable streams/buffers by `(database_id, tenant_id, stream_name)`. Both token types are positions, never authorization: handlers authorize the collection/source first and apply a token only within that separately authorized database/tenant scope. Buffer retention, polling limits, SSE cancellation, and Event Plane backpressure bound delivery. | +| **CDC poll, SSE, and named streams** | Identity, collection/stream names, cursors, offsets, consumer-group commits, SSE reconnects | The transient `/v1/cdc/{collection}` and `/v1/cdc/{collection}/poll` poll/SSE surface uses an opaque `control::change_stream::ChangeCursor`: one replicated position per feed (data group, Calvin vShard, or single-node process), valid on every node that holds the feed. It is bounded in size, rejects malformed input, and resets when the serving node lacks events above it. Durable named streams and consumer groups use `event::cdc::CdcOffset`, a composite `(epoch, index, sequence)` position whose per-partition commits reject regression and preserve same-write siblings. `event::cdc::{StreamRegistry,CdcRouter}` keys durable streams/buffers by `(database_id, tenant_id, stream_name)`. Both token types are positions, never authorization: handlers authorize the collection/source first and apply a token only within that separately authorized database/tenant scope. Buffer retention, polling limits, SSE cancellation, and Event Plane backpressure bound delivery. | | **WS-RPC and LIVE** | Upgrade headers, bearer credentials, JSON-RPC frames, subscription/filter expressions, live cursors | `http::routes::ws_rpc::ws_handler` is a streaming route; it must resolve the connection identity and route each operation through the same plan/authorization path as HTTP query. `LIVE`/subscription state is scoped to the authenticated identity, database, tenant, and authorized collection; bounded websocket frames, subscriptions, queues, and cancellation prevent an upgrade from becoming an unlimited execution channel. | | **Native protocol** | Handshake/authenticate messages, MessagePack frames, SQL/operation payloads, streaming chunks | `control::server::native::{handshake,codec,session,dispatch}` and `session_auth::native::authenticate` authenticate before creating a session. MessagePack is decoded with framing and size checks, planned/authorized into capabilities, and dispatched through the SPSC bridge. Native clients do not supply an identity or physical task. | | **RESP** | TLS/plain TCP setup, `AUTH` credentials, RESP arrays/bulk strings, collection selection, KV/hash/sorted-set/pub-sub commands | `control::server::resp::{listener,codec,session,handler}` parses bounded RESP frames and keeps data operations fail-closed until `AUTH` establishes an `AuthenticatedIdentity`. Command handlers derive the tenant from that identity, use the default database and selected collection, authorize the corresponding read/write task, and dispatch only the resulting capability through `resp::gateway_dispatch`. Connection limits, parser limits, startup gating, and the critical listener drain bound this compatibility surface. Plain RESP must be restricted to a trusted network; use TLS when credentials cross an untrusted network. | diff --git a/nodedb-array/src/codec/tile_decode.rs b/nodedb-array/src/codec/tile_decode.rs index 2f729d4db..8ba84ecb4 100644 --- a/nodedb-array/src/codec/tile_decode.rs +++ b/nodedb-array/src/codec/tile_decode.rs @@ -17,7 +17,7 @@ use crate::codec::limits::{ use crate::codec::tag::{CodecTag, peek_tag}; use crate::error::{ArrayError, ArrayResult}; use crate::tile::mbr::TileMBR; -use crate::tile::sparse_tile::SparseTile; +use crate::tile::sparse_tile::{RowKind, SparseTile}; const SUPPORTED_PAYLOAD_VERSION: u8 = 1; @@ -82,9 +82,12 @@ pub fn decode_sparse_tile(payload: &[u8]) -> ArrayResult { } fn decode_raw(body: &[u8]) -> ArrayResult { - zerompk::from_msgpack(body).map_err(|e| ArrayError::SegmentCorruption { - detail: format!("raw tile decode: {e}"), - }) + let tile: SparseTile = + zerompk::from_msgpack(body).map_err(|e| ArrayError::SegmentCorruption { + detail: format!("raw tile decode: {e}"), + })?; + tile.check_stored_identities()?; + Ok(tile) } fn decode_structural(body: &[u8]) -> ArrayResult { @@ -125,9 +128,9 @@ fn decode_structural(body: &[u8]) -> ArrayResult { dim_dicts.push(dict); } - // Surrogates. + // Surrogates: live rows only, in row order. let surr_bytes = read_framed(body, &mut pos)?; - let surrogates = decode_surrogates(surr_bytes)?; + let live_surrogates = decode_surrogates(surr_bytes)?; // Row kinds. let rk_bytes = read_framed(body, &mut pos)?; @@ -178,16 +181,17 @@ fn decode_structural(body: &[u8]) -> ArrayResult { let mbr = TileMBR::new(axis_count, attr_count); // Validate sizes match cell_count. - if surrogates.len() != cell_count { + if row_kinds.len() != cell_count { return Err(ArrayError::SegmentCorruption { detail: format!( - "structural tile: surrogate count {surr} != cell_count {cell_count}", - surr = surrogates.len() + "structural tile: row kind count {kinds} != cell_count {cell_count}", + kinds = row_kinds.len() ), }); } + let surrogates = spread_live_surrogates(&row_kinds, live_surrogates)?; - Ok(SparseTile { + let tile = SparseTile { dim_dicts, attr_cols, surrogates, @@ -195,7 +199,39 @@ fn decode_structural(body: &[u8]) -> ArrayResult { valid_until_ms, row_kinds, mbr, - }) + }; + tile.check_stored_identities()?; + Ok(tile) +} + +/// Place the live-row surrogate column back onto every row: a live row takes +/// the next surrogate, a tombstone or erasure row holds none. The column must +/// hold exactly one surrogate per live row. +fn spread_live_surrogates( + row_kinds: &[u8], + live_surrogates: Vec, +) -> ArrayResult>> { + let live_count = live_surrogates.len(); + let mut live = live_surrogates.into_iter(); + let mut surrogates = Vec::with_capacity(row_kinds.len()); + for (row, &kind) in row_kinds.iter().enumerate() { + if RowKind::from_u8(kind)? == RowKind::Live { + let surrogate = live.next().ok_or_else(|| ArrayError::SegmentCorruption { + detail: format!( + "structural tile: live row {row} has no surrogate ({live_count} stored)" + ), + })?; + surrogates.push(Some(surrogate)); + } else { + surrogates.push(None); + } + } + if live.next().is_some() { + return Err(ArrayError::SegmentCorruption { + detail: format!("structural tile: {live_count} surrogates stored for fewer live rows"), + }); + } + Ok(surrogates) } #[cfg(test)] @@ -230,7 +266,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(i as i64)], attrs: &[CellValue::Int64(i as i64 * 2)], - surrogate: Surrogate::ZERO, + surrogate: Some(Surrogate::new(i as u32 + 1)), valid_from_ms: i as i64, valid_until_ms: OPEN_UPPER, kind: RowKind::Live, @@ -293,7 +329,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(i)], attrs: &[CellValue::Int64(i)], - surrogate: Surrogate::ZERO, + surrogate: Some(Surrogate::new(i as u32 + 1)), valid_from_ms: 0, valid_until_ms: OPEN_UPPER, kind: RowKind::Live, @@ -303,7 +339,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(99)], attrs: &[], - surrogate: Surrogate::ZERO, + surrogate: None, valid_from_ms: 0, valid_until_ms: OPEN_UPPER, kind: RowKind::Tombstone, @@ -312,6 +348,32 @@ mod tests { let tile = b.build(); let out = roundtrip(&tile); assert_eq!(out.row_kinds, tile.row_kinds); + assert_eq!(out.surrogates, tile.surrogates); + assert_eq!(out.surrogates[20], None); + } + + /// A structural payload whose live-row surrogate column is one short is + /// refused: a live row never decodes without its identity. + #[test] + fn a_live_row_without_a_stored_surrogate_is_refused() { + let kinds = [ + RowKind::Live.as_u8(), + RowKind::Tombstone.as_u8(), + RowKind::Live.as_u8(), + ]; + assert!(spread_live_surrogates(&kinds, vec![Surrogate::new(1)]).is_err()); + assert!( + spread_live_surrogates( + &kinds, + vec![Surrogate::new(1), Surrogate::new(2), Surrogate::new(3)] + ) + .is_err(), + "a surrogate left over after the last live row is refused" + ); + assert_eq!( + spread_live_surrogates(&kinds, vec![Surrogate::new(1), Surrogate::new(2)]).unwrap(), + vec![Some(Surrogate::new(1)), None, Some(Surrogate::new(2))] + ); } #[test] @@ -342,7 +404,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(1)], attrs: &[CellValue::Int64(10)], - surrogate: Surrogate::ZERO, + surrogate: Some(Surrogate::new(1)), valid_from_ms: 100, valid_until_ms: 500, kind: RowKind::Live, @@ -351,7 +413,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(2)], attrs: &[CellValue::Int64(20)], - surrogate: Surrogate::ZERO, + surrogate: Some(Surrogate::new(2)), valid_from_ms: 200, valid_until_ms: OPEN_UPPER, kind: RowKind::Live, @@ -362,7 +424,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(i)], attrs: &[CellValue::Int64(i)], - surrogate: Surrogate::ZERO, + surrogate: Some(Surrogate::new(i as u32)), valid_from_ms: i * 10, valid_until_ms: OPEN_UPPER, kind: RowKind::Live, diff --git a/nodedb-array/src/codec/tile_encode.rs b/nodedb-array/src/codec/tile_encode.rs index 1e86c62d8..95abb069d 100644 --- a/nodedb-array/src/codec/tile_encode.rs +++ b/nodedb-array/src/codec/tile_encode.rs @@ -10,7 +10,7 @@ // [u32 LE cell_count] // [u32 LE axis_count] // per axis: [u32 LE encoded_len][coord_rle payload] -// [u32 LE surrogates_len][fastlanes payload] +// [u32 LE surrogates_len][fastlanes payload — live rows only] // [u32 LE row_kinds_len][raw u8s] // [u32 LE system_from_ms_len][gorilla payload — absent for Raw tag] // [u32 LE valid_from_ms_len][gorilla payload] @@ -58,6 +58,9 @@ fn write_framed(chunk: &[u8], out: &mut Vec) { /// Encode a `SparseTile` into `out`. The segment writer wraps this payload /// in BlockFraming (length + CRC). pub fn encode_sparse_tile(tile: &SparseTile, out: &mut Vec) -> ArrayResult<()> { + // A stored live cell always holds a bound surrogate. A derived result tile + // holds none and is refused here, before any byte is written. + tile.check_stored_identities()?; let tag = choose_tag(tile); out.push(tag.as_byte()); out.push(PAYLOAD_VERSION); @@ -89,8 +92,10 @@ fn encode_structural(tile: &SparseTile, out: &mut Vec) -> ArrayResult<()> { write_framed(&axis_buf, out); } - // Surrogates. - let surr_bytes = encode_surrogates(&tile.surrogates)?; + // Surrogates: live rows only, in row order. A tombstone or erasure row + // holds none, and the decoder restores the gaps from the row kinds. + let live_surrogates: Vec<_> = tile.surrogates.iter().copied().flatten().collect(); + let surr_bytes = encode_surrogates(&live_surrogates)?; write_framed(&surr_bytes, out); // Row kinds. @@ -157,7 +162,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(i as i64), CoordValue::Int64(i as i64 * 2)], attrs: &[CellValue::Int64(i as i64)], - surrogate: Surrogate::ZERO, + surrogate: Some(Surrogate::new(i as u32 + 1)), valid_from_ms: i as i64 * 10, valid_until_ms: OPEN_UPPER, kind: RowKind::Live, @@ -217,7 +222,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(i), CoordValue::Int64(i)], attrs: &[], - surrogate: Surrogate::ZERO, + surrogate: None, valid_from_ms: 0, valid_until_ms: OPEN_UPPER, kind: RowKind::Tombstone, @@ -248,4 +253,51 @@ mod tests { let decoded = decode_sparse_tile(&buf).unwrap(); assert_eq!(decoded.surrogates, tile.surrogates); } + + /// Tombstone rows interleaved with live rows keep no identity through the + /// structural codec, and every live row keeps its own. + #[test] + fn structural_roundtrip_keeps_sentinel_rows_without_identity() { + let s = schema(); + let mut b = SparseTileBuilder::new(&s); + for i in 0..20i64 { + let coord = [CoordValue::Int64(i), CoordValue::Int64(i)]; + if i % 3 == 0 { + b.push_row(SparseRow::sentinel(&coord, RowKind::Tombstone)) + .unwrap(); + } else { + b.push_row(SparseRow::live( + &coord, + &[CellValue::Int64(i)], + Surrogate::new(100 + i as u32), + 0, + OPEN_UPPER, + )) + .unwrap(); + } + } + let tile = b.build(); + let mut buf = Vec::new(); + encode_sparse_tile(&tile, &mut buf).unwrap(); + assert_eq!(buf[0], CodecTag::Structural.as_byte()); + let decoded = decode_sparse_tile(&buf).unwrap(); + assert_eq!(decoded.surrogates, tile.surrogates); + assert_eq!(decoded.surrogates[0], None); + assert_eq!(decoded.surrogates[1], Some(Surrogate::new(101))); + } + + /// A derived result tile holds no identity, so it is never written. + #[test] + fn a_derived_tile_is_refused_by_the_encoder() { + let s = schema(); + let mut b = SparseTileBuilder::new(&s); + b.push( + &[CoordValue::Int64(1), CoordValue::Int64(1)], + &[CellValue::Int64(1)], + ) + .unwrap(); + let mut buf = Vec::new(); + assert!(encode_sparse_tile(&b.build(), &mut buf).is_err()); + assert!(buf.is_empty(), "nothing is written for a refused tile"); + } } diff --git a/nodedb-array/src/lib.rs b/nodedb-array/src/lib.rs index 7086e6252..0e49d0148 100644 --- a/nodedb-array/src/lib.rs +++ b/nodedb-array/src/lib.rs @@ -36,8 +36,7 @@ pub use sync::{ SnapshotHeader, SnapshotSink, TileSnapshot, }; pub use tile::{ - AttrStats, DENSE_PROMOTION_THRESHOLD, DenseTile, SparseTile, TileMBR, should_promote_to_dense, - sparse_to_dense, tile_id_for_cell, tile_indices_for_cell, + AttrStats, DenseTile, SparseTile, TileMBR, tile_id_for_cell, tile_indices_for_cell, }; pub use types::{ArrayId, CellValue, Coord, Domain, TileId}; pub use wal::ArrayWalRecord; diff --git a/nodedb-array/src/query/ceiling.rs b/nodedb-array/src/query/ceiling.rs index af8da3281..ca6f99393 100644 --- a/nodedb-array/src/query/ceiling.rs +++ b/nodedb-array/src/query/ceiling.rs @@ -99,7 +99,7 @@ mod tests { valid_from_ms: valid_from, valid_until_ms: valid_until, attrs: vec![CellValue::Int64(val)], - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), } .encode() .unwrap() diff --git a/nodedb-array/src/query/elementwise.rs b/nodedb-array/src/query/elementwise.rs index e1bf4890d..207849755 100644 --- a/nodedb-array/src/query/elementwise.rs +++ b/nodedb-array/src/query/elementwise.rs @@ -14,9 +14,9 @@ use std::collections::BTreeMap; -use crate::error::ArrayResult; +use crate::error::{ArrayError, ArrayResult}; use crate::schema::ArraySchema; -use crate::tile::sparse_tile::{SparseTile, SparseTileBuilder}; +use crate::tile::sparse_tile::{RowKind, SparseTile, SparseTileBuilder}; use crate::types::cell_value::value::CellValue; use crate::types::coord::value::CoordValue; @@ -37,8 +37,8 @@ pub fn elementwise( op: BinaryOp, ) -> ArrayResult { let n_attrs = schema.attrs.len(); - let by_coord_a = index_rows(a); - let by_coord_b = index_rows(b); + let by_coord_a = index_rows(a)?; + let by_coord_b = index_rows(b)?; let mut keys: BTreeMap, ()> = BTreeMap::new(); for k in by_coord_a.keys().chain(by_coord_b.keys()) { keys.insert(k.clone(), ()); @@ -136,19 +136,38 @@ fn decode_key(k: &[CoordKey]) -> Vec { .collect() } -fn index_rows(tile: &SparseTile) -> BTreeMap, Vec> { +/// The live cells of `tile` by coordinate. Tombstone and erasure rows hold +/// no attribute entries, so attributes are read at the live-row index. +fn index_rows(tile: &SparseTile) -> ArrayResult, Vec>> { + let corrupt = |row: usize| ArrayError::SegmentCorruption { + detail: format!("elementwise: sparse tile row {row} is out of range"), + }; let mut out = BTreeMap::new(); - let n = tile.nnz() as usize; - for row in 0..n { - let coord: Vec = tile + let mut live_row = 0usize; + for row in 0..tile.row_count() { + if tile.row_kind(row)? != RowKind::Live { + continue; + } + let coord = tile .dim_dicts .iter() - .map(|d| d.values[d.indices[row] as usize].clone()) - .collect(); - let attrs: Vec = tile.attr_cols.iter().map(|col| col[row].clone()).collect(); + .map(|d| { + d.indices + .get(row) + .and_then(|&i| d.values.get(i as usize)) + .cloned() + .ok_or_else(|| corrupt(row)) + }) + .collect::>>()?; + let attrs = tile + .attr_cols + .iter() + .map(|col| col.get(live_row).cloned().ok_or_else(|| corrupt(row))) + .collect::>>()?; + live_row += 1; out.insert(encode_key(&coord), attrs); } - out + Ok(out) } #[cfg(test)] @@ -217,6 +236,42 @@ mod tests { assert_eq!(out.attr_cols[0][0], CellValue::Null); } + /// A tombstone row before a live row neither joins the result nor + /// shifts the live row's attributes. + #[test] + fn a_tombstone_row_is_skipped() { + use crate::tile::sparse_tile::SparseRow; + use nodedb_types::{OPEN_UPPER, Surrogate}; + + let s = schema(); + let mut builder = SparseTileBuilder::new(&s); + builder + .push_row(SparseRow { + coord: &[CoordValue::Int64(0)], + attrs: &[], + surrogate: None, + valid_from_ms: 0, + valid_until_ms: OPEN_UPPER, + kind: RowKind::Tombstone, + }) + .unwrap(); + builder + .push_row(SparseRow { + coord: &[CoordValue::Int64(1)], + attrs: &[CellValue::Int64(4)], + surrogate: Some(Surrogate::new(1)), + valid_from_ms: 0, + valid_until_ms: OPEN_UPPER, + kind: RowKind::Live, + }) + .unwrap(); + let a = builder.build(); + let b = tile(&[(1, 6)]); + let out = elementwise(&s, &a, &b, BinaryOp::Add).unwrap(); + assert_eq!(out.nnz(), 1); + assert_eq!(out.attr_cols[0][0], CellValue::Int64(10)); + } + #[test] fn sub_and_mul() { let s = schema(); diff --git a/nodedb-array/src/query/rechunk.rs b/nodedb-array/src/query/rechunk.rs index cf3974d4c..0d70ac09d 100644 --- a/nodedb-array/src/query/rechunk.rs +++ b/nodedb-array/src/query/rechunk.rs @@ -49,11 +49,8 @@ pub fn rechunk_sparse( .iter() .map(|col| col[attr_row].clone()) .collect(); - let surrogate = tile - .surrogates - .get(row) - .copied() - .unwrap_or(nodedb_types::Surrogate::ZERO); + // A stored cell keeps its identity; a derived row keeps none. + let surrogate = tile.row_surrogate(row)?; let valid_from_ms = tile.valid_from_ms.get(row).copied().unwrap_or(0); let valid_until_ms = tile .valid_until_ms diff --git a/nodedb-array/src/query/retention.rs b/nodedb-array/src/query/retention.rs index 4c2a90ed7..f351496c1 100644 --- a/nodedb-array/src/query/retention.rs +++ b/nodedb-array/src/query/retention.rs @@ -18,7 +18,6 @@ use crate::tile::cell_payload::CellPayload; use crate::tile::sparse_tile::{RowKind, SparseRow, SparseTile, SparseTileBuilder}; use crate::types::TileId; use crate::types::coord::value::CoordValue; -use nodedb_types::{OPEN_UPPER, Surrogate}; // ── Result type ───────────────────────────────────────────────────────────── @@ -119,7 +118,7 @@ pub fn decode_sparse_rows(tile: &SparseTile) -> ArrayResult> { }) .collect::>>()?; - let surrogate = tile.surrogates.get(row).copied().unwrap_or(Surrogate::ZERO); + let surrogate = tile.live_surrogate(row)?; let valid_from_ms = tile.valid_from_ms.get(row).copied().ok_or_else(|| { ArrayError::SegmentCorruption { detail: format!("decode_sparse_rows: valid_from_ms row {row} out of range"), @@ -272,28 +271,20 @@ pub fn merge_for_retention( // Drop entirely — GDPR erasure removes the cell from the ceiling. } RowKind::Tombstone => { - builder.push_row(SparseRow { - coord: &coord, - attrs: &[], - surrogate: Surrogate::ZERO, - valid_from_ms: 0, - valid_until_ms: OPEN_UPPER, - kind: RowKind::Tombstone, - })?; + builder.push_row(SparseRow::sentinel(&coord, RowKind::Tombstone))?; cells_carried_forward += 1; } RowKind::Live => { let p = payload.ok_or_else(|| ArrayError::SegmentCorruption { detail: "Live row in ceiling has no CellPayload".into(), })?; - builder.push_row(SparseRow { - coord: &coord, - attrs: &p.attrs, - surrogate: p.surrogate, - valid_from_ms: p.valid_from_ms, - valid_until_ms: p.valid_until_ms, - kind: RowKind::Live, - })?; + builder.push_row(SparseRow::live( + &coord, + &p.attrs, + p.surrogate, + p.valid_from_ms, + p.valid_until_ms, + ))?; cells_carried_forward += 1; } } @@ -338,6 +329,7 @@ mod tests { use crate::types::cell_value::value::CellValue; use crate::types::coord::value::CoordValue; use crate::types::domain::{Domain, DomainBound}; + use nodedb_types::{OPEN_UPPER, Surrogate}; fn schema() -> ArraySchema { ArraySchemaBuilder::new("t") @@ -368,7 +360,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(x)], attrs: &[CellValue::Int64(v)], - surrogate: Surrogate::ZERO, + surrogate: Some(Surrogate::new(x as u32 + 1)), valid_from_ms: 0, valid_until_ms: OPEN_UPPER, kind: RowKind::Live, @@ -383,7 +375,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(x)], attrs: &[], - surrogate: Surrogate::ZERO, + surrogate: None, valid_from_ms: 0, valid_until_ms: OPEN_UPPER, kind: RowKind::Tombstone, @@ -398,7 +390,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(x)], attrs: &[], - surrogate: Surrogate::ZERO, + surrogate: None, valid_from_ms: 0, valid_until_ms: OPEN_UPPER, kind: RowKind::GdprErased, diff --git a/nodedb-array/src/query/slice.rs b/nodedb-array/src/query/slice.rs index 4cad1139e..7ee98c72b 100644 --- a/nodedb-array/src/query/slice.rs +++ b/nodedb-array/src/query/slice.rs @@ -119,11 +119,8 @@ pub fn slice_sparse( .iter() .map(|col| col[attr_row].clone()) .collect(); - let surrogate = tile - .surrogates - .get(row) - .copied() - .unwrap_or(nodedb_types::Surrogate::ZERO); + // A stored cell keeps its identity; a derived row keeps none. + let surrogate = tile.row_surrogate(row)?; let valid_from_ms = tile.valid_from_ms.get(row).copied().unwrap_or(0); let valid_until_ms = tile .valid_until_ms diff --git a/nodedb-array/src/segment/reader.rs b/nodedb-array/src/segment/reader.rs index f8c2c27d3..7d1ba3c6d 100644 --- a/nodedb-array/src/segment/reader.rs +++ b/nodedb-array/src/segment/reader.rs @@ -17,7 +17,6 @@ use crate::tile::cell_payload::{CELL_GDPR_ERASURE_SENTINEL, CELL_TOMBSTONE_SENTI use crate::tile::dense_tile::DenseTile; use crate::tile::sparse_tile::{RowKind, SparseTile}; use crate::types::coord::value::CoordValue; -use nodedb_types::Surrogate; /// Extract the encoded `CellPayload` bytes for a specific `coord` from a /// `SparseTile`. @@ -63,19 +62,27 @@ pub fn extract_cell_bytes(tile: &SparseTile, coord: &[CoordValue]) -> ArrayResul RowKind::GdprErased => return Ok(Some(CELL_GDPR_ERASURE_SENTINEL.to_vec())), RowKind::Live => {} } - // Live row — build attrs from all attr columns. + // Attribute columns hold live rows only: index by the live-row index. + let mut live_row = 0usize; + for earlier in 0..row { + if tile.row_kind(earlier)? == RowKind::Live { + live_row += 1; + } + } let attrs: Vec<_> = tile .attr_cols .iter() .map(|col| { - col.get(row) + col.get(live_row) .cloned() .ok_or_else(|| ArrayError::SegmentCorruption { - detail: format!("extract_cell_bytes: attr col row {row} out of range"), + detail: format!( + "extract_cell_bytes: attr col live row {live_row} out of range" + ), }) }) .collect::>>()?; - let surrogate = tile.surrogates.get(row).copied().unwrap_or(Surrogate::ZERO); + let surrogate = tile.live_surrogate(row)?; let valid_from_ms = tile.valid_from_ms .get(row) @@ -406,10 +413,13 @@ mod tests { fn make_sparse(s: &crate::schema::ArraySchema, base: i64) -> SparseTile { let mut b = SparseTileBuilder::new(s); - b.push( + b.push_row(crate::tile::sparse_tile::SparseRow::live( &[CoordValue::Int64(base), CoordValue::Int64(base + 1)], &[CellValue::Int64(base * 10)], - ) + nodedb_types::Surrogate::new(base as u32 + 1), + 0, + nodedb_types::OPEN_UPPER, + )) .unwrap(); b.build() } @@ -537,7 +547,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(1), CoordValue::Int64(2)], attrs: &[CellValue::Int64(99)], - surrogate: Surrogate::ZERO, + surrogate: Some(Surrogate::new(3)), valid_from_ms: 100, valid_until_ms: 200, kind: crate::tile::sparse_tile::RowKind::Live, @@ -603,7 +613,7 @@ mod tests { b.push_row(crate::tile::sparse_tile::SparseRow { coord: &[CoordValue::Int64(7), CoordValue::Int64(8)], attrs: &[], - surrogate: Surrogate::ZERO, + surrogate: None, valid_from_ms: 0, valid_until_ms: nodedb_types::OPEN_UPPER, kind: RowKind::Tombstone, @@ -619,6 +629,41 @@ mod tests { assert_eq!(bytes, CELL_TOMBSTONE_SENTINEL); } + /// A live row after a sentinel row reads its own attributes: the + /// attribute columns skip sentinel rows. + #[test] + fn extract_cell_bytes_reads_a_live_row_after_a_sentinel() { + use crate::tile::sparse_tile::{RowKind, SparseRow, SparseTileBuilder}; + use nodedb_types::Surrogate; + + let s = schema(); + let mut b = SparseTileBuilder::new(&s); + b.push_row(SparseRow { + coord: &[CoordValue::Int64(1), CoordValue::Int64(1)], + attrs: &[], + surrogate: None, + valid_from_ms: 0, + valid_until_ms: nodedb_types::OPEN_UPPER, + kind: RowKind::Tombstone, + }) + .unwrap(); + b.push_row(SparseRow { + coord: &[CoordValue::Int64(2), CoordValue::Int64(2)], + attrs: &[CellValue::Int64(20)], + surrogate: Some(Surrogate::new(9)), + valid_from_ms: 5, + valid_until_ms: nodedb_types::OPEN_UPPER, + kind: RowKind::Live, + }) + .unwrap(); + let tile = b.build(); + let coord = vec![CoordValue::Int64(2), CoordValue::Int64(2)]; + let bytes = extract_cell_bytes(&tile, &coord).unwrap().unwrap(); + let payload = CellPayload::decode(&bytes).unwrap(); + assert_eq!(payload.attrs, vec![CellValue::Int64(20)]); + assert_eq!(payload.surrogate, Surrogate::new(9)); + } + #[test] fn extract_cell_bytes_returns_erasure_sentinel_for_erased_row() { use crate::tile::cell_payload::{CELL_GDPR_ERASURE_SENTINEL, is_cell_gdpr_erasure}; @@ -629,7 +674,7 @@ mod tests { b.push_row(crate::tile::sparse_tile::SparseRow { coord: &[CoordValue::Int64(4), CoordValue::Int64(5)], attrs: &[], - surrogate: Surrogate::ZERO, + surrogate: None, valid_from_ms: 0, valid_until_ms: nodedb_types::OPEN_UPPER, kind: RowKind::GdprErased, diff --git a/nodedb-array/src/segment/writer.rs b/nodedb-array/src/segment/writer.rs index d2031aeae..848c87450 100644 --- a/nodedb-array/src/segment/writer.rs +++ b/nodedb-array/src/segment/writer.rs @@ -120,7 +120,7 @@ mod tests { use crate::schema::ArraySchemaBuilder; use crate::schema::attr_spec::{AttrSpec, AttrType}; use crate::schema::dim_spec::{DimSpec, DimType}; - use crate::tile::sparse_tile::SparseTileBuilder; + use crate::tile::sparse_tile::{SparseRow, SparseTileBuilder}; use crate::types::cell_value::value::CellValue; use crate::types::coord::value::CoordValue; use crate::types::domain::{Domain, DomainBound}; @@ -150,15 +150,21 @@ mod tests { fn sparse_tile(s: &crate::schema::ArraySchema) -> SparseTile { let mut b = SparseTileBuilder::new(s); - b.push( + b.push_row(SparseRow::live( &[CoordValue::Int64(1), CoordValue::Int64(2)], &[CellValue::Int64(10)], - ) + nodedb_types::Surrogate::new(1), + 0, + nodedb_types::OPEN_UPPER, + )) .unwrap(); - b.push( + b.push_row(SparseRow::live( &[CoordValue::Int64(3), CoordValue::Int64(0)], &[CellValue::Int64(20)], - ) + nodedb_types::Surrogate::new(2), + 0, + nodedb_types::OPEN_UPPER, + )) .unwrap(); b.build() } diff --git a/nodedb-array/src/tile/cell_payload.rs b/nodedb-array/src/tile/cell_payload.rs index 17a61f561..b7ffe9b96 100644 --- a/nodedb-array/src/tile/cell_payload.rs +++ b/nodedb-array/src/tile/cell_payload.rs @@ -40,16 +40,32 @@ pub struct CellPayload { } impl CellPayload { + /// Encode a live cell. A live cell always holds a bound surrogate, so a + /// payload that carries `Surrogate::ZERO` is refused. pub fn encode(&self) -> ArrayResult> { + self.check_bound()?; zerompk::to_msgpack_vec(self).map_err(|e| ArrayError::SegmentCorruption { detail: format!("encode CellPayload: {e}"), }) } + /// Decode a live cell, refusing one that carries `Surrogate::ZERO`. pub fn decode(bytes: &[u8]) -> ArrayResult { - zerompk::from_msgpack(bytes).map_err(|e| ArrayError::SegmentCorruption { - detail: format!("decode CellPayload: {e}"), - }) + let payload: Self = + zerompk::from_msgpack(bytes).map_err(|e| ArrayError::SegmentCorruption { + detail: format!("decode CellPayload: {e}"), + })?; + payload.check_bound()?; + Ok(payload) + } + + fn check_bound(&self) -> ArrayResult<()> { + if self.surrogate == Surrogate::ZERO { + return Err(ArrayError::SegmentCorruption { + detail: "live cell payload carries Surrogate::ZERO, which names no row".into(), + }); + } + Ok(()) } } @@ -78,10 +94,19 @@ mod tests { valid_from_ms: 1_000, valid_until_ms: OPEN_UPPER, attrs: vec![CellValue::Int64(42)], - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(7), } } + #[test] + fn a_payload_under_zero_is_refused() { + let mut p = sample_payload(); + p.surrogate = Surrogate::ZERO; + assert!(p.encode().is_err()); + let raw = zerompk::to_msgpack_vec(&p).unwrap(); + assert!(CellPayload::decode(&raw).is_err()); + } + #[test] fn payload_msgpack_roundtrip() { let p = sample_payload(); @@ -90,7 +115,7 @@ mod tests { assert_eq!(decoded.valid_from_ms, 1_000); assert_eq!(decoded.valid_until_ms, OPEN_UPPER); assert_eq!(decoded.attrs, vec![CellValue::Int64(42)]); - assert_eq!(decoded.surrogate, Surrogate::ZERO); + assert_eq!(decoded.surrogate, Surrogate::new(7)); } #[test] diff --git a/nodedb-array/src/tile/dense_tile.rs b/nodedb-array/src/tile/dense_tile.rs index 3932eb58b..ab0daf7c2 100644 --- a/nodedb-array/src/tile/dense_tile.rs +++ b/nodedb-array/src/tile/dense_tile.rs @@ -2,10 +2,11 @@ //! Dense tile payload — flat row-major attribute arrays. //! -//! Used when a tile's fill ratio crosses -//! [`super::DENSE_PROMOTION_THRESHOLD`]. The dense layout drops -//! coordinate columns entirely: cell `i`'s coordinates are recovered -//! from `i` and the tile's per-dim extents. +//! The dense layout drops coordinate columns entirely: cell `i`'s +//! coordinates are recovered from `i` and the tile's per-dim extents. It +//! holds no row kinds, surrogates or valid times, so the bitemporal array +//! store never writes it, and the engine refuses it as corruption wherever +//! it reads a segment tile. use serde::{Deserialize, Serialize}; diff --git a/nodedb-array/src/tile/mod.rs b/nodedb-array/src/tile/mod.rs index 151cfa7ac..9ef2272fc 100644 --- a/nodedb-array/src/tile/mod.rs +++ b/nodedb-array/src/tile/mod.rs @@ -4,7 +4,6 @@ pub mod cell_payload; pub mod dense_tile; pub mod layout; pub mod mbr; -pub mod promotion; pub mod sparse_tile; pub use cell_payload::{ @@ -14,5 +13,4 @@ pub use cell_payload::{ pub use dense_tile::DenseTile; pub use layout::{tile_id_for_cell, tile_indices_for_cell}; pub use mbr::{AttrStats, TileMBR}; -pub use promotion::{DENSE_PROMOTION_THRESHOLD, should_promote_to_dense, sparse_to_dense}; pub use sparse_tile::SparseTile; diff --git a/nodedb-array/src/tile/promotion.rs b/nodedb-array/src/tile/promotion.rs deleted file mode 100644 index c47617b67..000000000 --- a/nodedb-array/src/tile/promotion.rs +++ /dev/null @@ -1,193 +0,0 @@ -// SPDX-License-Identifier: Apache-2.0 - -//! Sparse → dense promotion at fill ratio > [`DENSE_PROMOTION_THRESHOLD`]. -//! -//! The threshold is fixed in code rather than in the schema because -//! it's a storage-level decision, not a workload one — at fill ratios -//! above 70%, the per-cell overhead of coordinate columns exceeds the -//! cost of storing nulls densely. - -use super::dense_tile::{DenseTile, cells_per_tile}; -use super::mbr::MbrBuilder; -use super::sparse_tile::SparseTile; -use crate::error::{ArrayError, ArrayResult}; -use crate::schema::ArraySchema; -use crate::types::cell_value::value::CellValue; - -/// Fill ratio above which a sparse tile is rewritten into a dense -/// tile (auto-promotion threshold). -pub const DENSE_PROMOTION_THRESHOLD: f64 = 0.7; - -/// True if `nnz / cells_per_tile > threshold`. -pub fn should_promote_to_dense(tile: &SparseTile, schema: &ArraySchema) -> bool { - let total = cells_per_tile(&schema.tile_extents); - if total == 0 { - return false; - } - (tile.nnz() as f64) / (total as f64) > DENSE_PROMOTION_THRESHOLD -} - -/// Convert a sparse tile to a dense tile by materialising every cell. -/// -/// Cells absent in the sparse payload become [`CellValue::Null`]. The -/// caller is responsible for the integer-only-dim precondition (dense -/// indexing relies on integer cell offsets) — this is enforced -/// because non-`Int64`/`TimestampMs` dims do not have well-defined -/// `tile_extent` semantics in dense layout. -pub fn sparse_to_dense(tile: &SparseTile, schema: &ArraySchema) -> ArrayResult { - use crate::schema::dim_spec::DimType; - use crate::types::coord::value::CoordValue; - for d in &schema.dims { - if !matches!(d.dtype, DimType::Int64 | DimType::TimestampMs) { - return Err(ArrayError::InvalidSchema { - array: schema.name.clone(), - detail: format!( - "dense promotion requires integer dims; '{}' is {:?}", - d.name, d.dtype - ), - }); - } - } - let mut dense = DenseTile::empty(schema); - let mut mbr = MbrBuilder::new(schema.arity(), schema.attrs.len()); - let n_rows = tile.nnz() as usize; - for row in 0..n_rows { - // Reconstruct row's coord from the dim dictionaries. - let coord: Vec = schema - .dims - .iter() - .enumerate() - .map(|(i, _)| { - let dict = &tile.dim_dicts[i]; - let idx = dict.indices[row] as usize; - dict.values[idx].clone() - }) - .collect(); - let attrs: Vec = (0..schema.attrs.len()) - .map(|i| tile.attr_cols[i][row].clone()) - .collect(); - let flat = flat_index_for_coord(schema, &coord)?; - for (i, a) in attrs.iter().enumerate() { - dense.attr_cols[i][flat] = a.clone(); - } - mbr.fold(&coord, &attrs); - } - dense.mbr = mbr.build(); - Ok(dense) -} - -/// Row-major flat index `((c0 - lo0) * extent1 * extent2 ...) + ...`. -fn flat_index_for_coord( - schema: &ArraySchema, - coord: &[crate::types::coord::value::CoordValue], -) -> ArrayResult { - use crate::schema::dim_spec::DimType; - use crate::types::coord::value::CoordValue; - use crate::types::domain::DomainBound; - let mut flat: usize = 0; - for (i, dim) in schema.dims.iter().enumerate() { - let extent = schema.tile_extents[i] as usize; - let lo = match (&dim.dtype, &dim.domain.lo) { - (DimType::Int64, DomainBound::Int64(v)) - | (DimType::TimestampMs, DomainBound::TimestampMs(v)) => *v, - _ => 0, - }; - let off = match coord.get(i) { - Some(CoordValue::Int64(v)) | Some(CoordValue::TimestampMs(v)) => { - ((*v - lo) as usize) % extent - } - _ => { - return Err(ArrayError::CoordOutOfDomain { - array: schema.name.clone(), - dim: dim.name.clone(), - detail: "non-integer coord in dense promotion".to_string(), - }); - } - }; - flat = flat * extent + off; - } - Ok(flat) -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::schema::ArraySchemaBuilder; - use crate::schema::attr_spec::{AttrSpec, AttrType}; - use crate::schema::dim_spec::{DimSpec, DimType}; - use crate::tile::sparse_tile::SparseTileBuilder; - use crate::types::coord::value::CoordValue; - use crate::types::domain::{Domain, DomainBound}; - - fn schema_2x2() -> ArraySchema { - ArraySchemaBuilder::new("g") - .dim(DimSpec::new( - "x", - DimType::Int64, - Domain::new(DomainBound::Int64(0), DomainBound::Int64(1)), - )) - .dim(DimSpec::new( - "y", - DimType::Int64, - Domain::new(DomainBound::Int64(0), DomainBound::Int64(1)), - )) - .attr(AttrSpec::new("v", AttrType::Int64, true)) - .tile_extents(vec![2, 2]) - .build() - .unwrap() - } - - fn build_tile(s: &ArraySchema, n: usize) -> SparseTile { - let mut b = SparseTileBuilder::new(s); - let cells = [(0i64, 0i64, 10i64), (0, 1, 20), (1, 0, 30), (1, 1, 40)]; - for (x, y, v) in cells.iter().take(n) { - b.push( - &[CoordValue::Int64(*x), CoordValue::Int64(*y)], - &[CellValue::Int64(*v)], - ) - .unwrap(); - } - b.build() - } - - #[test] - fn promotes_above_threshold() { - let s = schema_2x2(); - let t = build_tile(&s, 3); // 3/4 = 0.75 > 0.7 - assert!(should_promote_to_dense(&t, &s)); - } - - #[test] - fn does_not_promote_at_or_below_threshold() { - let s = schema_2x2(); - let t = build_tile(&s, 2); // 2/4 = 0.5 - assert!(!should_promote_to_dense(&t, &s)); - } - - #[test] - fn dense_conversion_places_cells_at_flat_index() { - let s = schema_2x2(); - let t = build_tile(&s, 4); - let d = sparse_to_dense(&t, &s).unwrap(); - // Row-major (x, y) for 2x2 with extents [2, 2]: (0,0)=0, - // (0,1)=1, (1,0)=2, (1,1)=3. - assert_eq!(d.attr_cols[0][0], CellValue::Int64(10)); - assert_eq!(d.attr_cols[0][1], CellValue::Int64(20)); - assert_eq!(d.attr_cols[0][2], CellValue::Int64(30)); - assert_eq!(d.attr_cols[0][3], CellValue::Int64(40)); - assert_eq!(d.mbr.nnz, 4); - } - - #[test] - fn dense_conversion_leaves_absent_cells_null() { - let s = schema_2x2(); - let t = build_tile(&s, 2); - let d = sparse_to_dense(&t, &s).unwrap(); - // Two cells populated, two should remain Null. - let nulls = d.attr_cols[0] - .iter() - .filter(|c| matches!(c, CellValue::Null)) - .count(); - assert_eq!(nulls, 2); - } -} diff --git a/nodedb-array/src/tile/sparse_tile.rs b/nodedb-array/src/tile/sparse_tile.rs index 091bd920c..c92ca81f7 100644 --- a/nodedb-array/src/tile/sparse_tile.rs +++ b/nodedb-array/src/tile/sparse_tile.rs @@ -116,11 +116,15 @@ pub struct SparseTile { pub dim_dicts: Vec, /// One column per schema attr, parallel to `schema.attrs`. pub attr_cols: Vec>, - /// Per-cell global surrogate (one entry per row, parallel to the - /// dim-dict index streams and `attr_cols`). Cross-engine bitmap - /// joins read this column directly without translating coords back - /// to user-visible primary keys. - pub surrogates: Vec, + /// Per-row global surrogate, parallel to the dim-dict index streams. + /// Cross-engine bitmap joins read this column directly without + /// translating coords back to user-visible primary keys. + /// + /// A stored live cell holds `Some` bound surrogate. A tombstone or + /// erasure row holds `None`: it names a coordinate, not a row. A + /// derived query result row, such as an elementwise output, holds + /// `None` and is never stored. `Surrogate::ZERO` never appears. + pub surrogates: Vec>, /// Per-cell valid-time lower bound in milliseconds (inclusive). /// Parallel to `surrogates`. pub valid_from_ms: Vec, @@ -130,9 +134,6 @@ pub struct SparseTile { pub valid_until_ms: Vec, /// Per-row [`RowKind`] encoded as `u8`. Parallel to `surrogates`. /// `0` = Live, `1` = Tombstone, `2` = GdprErased. - /// - /// Older segments written before this column existed will deserialise - /// with an empty `Vec`; readers treat a missing entry as `Live`. pub row_kinds: Vec, pub mbr: TileMBR, } @@ -185,21 +186,78 @@ impl SparseTile { Some(&b) => RowKind::from_u8(b), } } + + /// The surrogate `row` holds: `Some` for a stored live cell, `None` for a + /// tombstone, an erasure, or a derived result row. A row index past the + /// surrogate column is a malformed tile. + pub fn row_surrogate(&self, row: usize) -> ArrayResult> { + self.surrogates + .get(row) + .copied() + .ok_or_else(|| ArrayError::SegmentCorruption { + detail: format!( + "tile row {row} is past the surrogate column ({} rows)", + self.surrogates.len() + ), + }) + } + + /// The surrogate of live `row`. A live row read from storage always holds + /// one, so a row without it is a malformed tile. + pub fn live_surrogate(&self, row: usize) -> ArrayResult { + self.row_surrogate(row)? + .ok_or_else(|| ArrayError::SegmentCorruption { + detail: format!("live tile row {row} carries no surrogate"), + }) + } + + /// Check the identity invariant of a stored tile: every live row holds a + /// bound surrogate and every tombstone or erasure row holds none. The + /// segment codec runs it on every tile it encodes or decodes. + pub fn check_stored_identities(&self) -> ArrayResult<()> { + if self.row_kinds.len() != self.surrogates.len() { + return Err(ArrayError::SegmentCorruption { + detail: format!( + "stored tile carries {} row kinds but {} surrogates", + self.row_kinds.len(), + self.surrogates.len() + ), + }); + } + for (row, surrogate) in self.surrogates.iter().enumerate() { + let kind = self.row_kind(row)?; + match (kind, surrogate) { + (RowKind::Live, Some(s)) if *s != Surrogate::ZERO => {} + (RowKind::Tombstone | RowKind::GdprErased, None) => {} + (kind, surrogate) => { + return Err(ArrayError::SegmentCorruption { + detail: format!( + "stored tile row {row} of kind {kind:?} holds surrogate \ + {surrogate:?}; a live row holds a bound surrogate and a \ + tombstone or erasure row holds none" + ), + }); + } + } + } + Ok(()) + } } /// All row-level data passed to [`SparseTileBuilder::push_row`]. pub struct SparseRow<'a> { pub coord: &'a [CoordValue], pub attrs: &'a [CellValue], - pub surrogate: Surrogate, + /// `Some` bound surrogate for a stored live cell. `None` for a tombstone, + /// an erasure, or a derived result row that is never stored. + pub surrogate: Option, pub valid_from_ms: i64, pub valid_until_ms: i64, pub kind: RowKind, } impl<'a> SparseRow<'a> { - /// Construct a live row. Convenience wrapper so callers that only write - /// live data don't have to spell out `kind: RowKind::Live` explicitly. + /// Construct a live row that holds its bound surrogate. pub fn live( coord: &'a [CoordValue], attrs: &'a [CellValue], @@ -210,12 +268,25 @@ impl<'a> SparseRow<'a> { Self { coord, attrs, - surrogate, + surrogate: Some(surrogate), valid_from_ms, valid_until_ms, kind: RowKind::Live, } } + + /// Construct a tombstone or erasure row. It carries no attributes and no + /// identity. + pub fn sentinel(coord: &'a [CoordValue], kind: RowKind) -> Self { + Self { + coord, + attrs: &[], + surrogate: None, + valid_from_ms: 0, + valid_until_ms: OPEN_UPPER, + kind, + } + } } /// Streaming builder. Folds `(coord, attrs)` pairs into the tile body @@ -224,7 +295,7 @@ pub struct SparseTileBuilder<'a> { schema: &'a ArraySchema, dim_dicts: Vec, attr_cols: Vec>, - surrogates: Vec, + surrogates: Vec>, valid_from_ms: Vec, valid_until_ms: Vec, row_kinds: Vec, @@ -259,6 +330,26 @@ impl<'a> SparseTileBuilder<'a> { valid_until_ms, kind, } = row; + match (kind, surrogate) { + (_, Some(s)) if s == Surrogate::ZERO => { + return Err(ArrayError::InvalidOp { + detail: format!( + "row pushed into array '{}' carries Surrogate::ZERO, which names no row", + self.schema.name + ), + }); + } + (RowKind::Tombstone | RowKind::GdprErased, Some(s)) => { + return Err(ArrayError::InvalidOp { + detail: format!( + "{kind:?} row pushed into array '{}' carries surrogate {s}; a \ + tombstone or erasure row carries no identity", + self.schema.name + ), + }); + } + _ => {} + } if coord.len() != self.schema.arity() { return Err(ArrayError::CoordArityMismatch { array: self.schema.name.clone(), @@ -299,14 +390,14 @@ impl<'a> SparseTileBuilder<'a> { Ok(()) } - /// Push a live row whose surrogate is unknown to the caller (recovery, - /// pure-shape ops). The slot is filled with [`Surrogate::ZERO`]; - /// callers that have a real surrogate should use [`Self::push_row`]. + /// Push a derived live row, such as an elementwise result. It holds no + /// surrogate, so the tile it lands in is a query result and never stored. + /// A stored cell goes through [`Self::push_row`] with its bound surrogate. pub fn push(&mut self, coord: &[CoordValue], attrs: &[CellValue]) -> ArrayResult<()> { self.push_row(SparseRow { coord, attrs, - surrogate: Surrogate::ZERO, + surrogate: None, valid_from_ms: 0, valid_until_ms: OPEN_UPPER, kind: RowKind::Live, @@ -432,7 +523,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(1), CoordValue::Int64(100)], attrs: &[CellValue::String("A".into()), CellValue::Float64(1.0)], - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: Some(Surrogate::new(1)), valid_from_ms: 50, valid_until_ms: 150, kind: RowKind::Live, @@ -441,7 +532,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(2), CoordValue::Int64(200)], attrs: &[CellValue::String("B".into()), CellValue::Float64(2.0)], - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: Some(Surrogate::new(2)), valid_from_ms: 300, valid_until_ms: 900, kind: RowKind::Live, @@ -465,7 +556,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(1), CoordValue::Int64(10)], attrs: &[CellValue::String("X".into()), CellValue::Float64(1.0)], - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: Some(Surrogate::new(1)), valid_from_ms: 0, valid_until_ms: OPEN_UPPER, kind: RowKind::Live, @@ -475,7 +566,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(2), CoordValue::Int64(20)], attrs: &[], - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: None, valid_from_ms: 0, valid_until_ms: OPEN_UPPER, kind: RowKind::Tombstone, @@ -485,7 +576,7 @@ mod tests { b.push_row(SparseRow { coord: &[CoordValue::Int64(3), CoordValue::Int64(30)], attrs: &[], - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: None, valid_from_ms: 0, valid_until_ms: OPEN_UPPER, kind: RowKind::GdprErased, @@ -509,5 +600,54 @@ mod tests { RowKind::from_u8(decoded.row_kinds[2]).unwrap(), RowKind::GdprErased ); + assert_eq!( + decoded.surrogates, + vec![Some(Surrogate::new(1)), None, None] + ); + decoded.check_stored_identities().unwrap(); + } + + /// `Surrogate::ZERO` names no row, so the builder refuses it on any row. + #[test] + fn builder_refuses_a_zero_surrogate() { + let s = schema(); + let mut b = SparseTileBuilder::new(&s); + let r = b.push_row(SparseRow::live( + &[CoordValue::Int64(1), CoordValue::Int64(10)], + &[CellValue::String("X".into()), CellValue::Float64(1.0)], + Surrogate::ZERO, + 0, + OPEN_UPPER, + )); + assert!(r.is_err(), "a live row under ZERO is refused"); + assert_eq!(b.build().row_count(), 0, "the refused row is not stored"); + } + + /// A tombstone or erasure row names a coordinate, never a row identity. + #[test] + fn builder_refuses_a_sentinel_row_with_an_identity() { + let s = schema(); + let mut b = SparseTileBuilder::new(&s); + let coord = [CoordValue::Int64(1), CoordValue::Int64(10)]; + let mut row = SparseRow::sentinel(&coord, RowKind::Tombstone); + row.surrogate = Some(Surrogate::new(4)); + assert!(b.push_row(row).is_err()); + } + + /// A derived row holds no surrogate, so its tile fails the stored-tile + /// check: it can be read but never written to a segment. + #[test] + fn a_derived_row_fails_the_stored_identity_check() { + let s = schema(); + let mut b = SparseTileBuilder::new(&s); + b.push( + &[CoordValue::Int64(1), CoordValue::Int64(10)], + &[CellValue::String("X".into()), CellValue::Float64(1.0)], + ) + .unwrap(); + let tile = b.build(); + assert_eq!(tile.row_surrogate(0).unwrap(), None); + assert!(tile.live_surrogate(0).is_err()); + assert!(tile.check_stored_identities().is_err()); } } diff --git a/nodedb-cluster-tests/Cargo.toml b/nodedb-cluster-tests/Cargo.toml index fd6403f2c..f36ab2553 100644 --- a/nodedb-cluster-tests/Cargo.toml +++ b/nodedb-cluster-tests/Cargo.toml @@ -39,6 +39,7 @@ hex = { workspace = true } loro = { workspace = true } nexar = { workspace = true } quinn = { workspace = true } +rdkafka = { workspace = true } redb = { workspace = true } rustls = { workspace = true } serde_json = { workspace = true } diff --git a/nodedb-cluster-tests/tests/cluster_common/calvin_test_node.rs b/nodedb-cluster-tests/tests/cluster_common/calvin_test_node.rs index b421c72a3..96666acef 100644 --- a/nodedb-cluster-tests/tests/cluster_common/calvin_test_node.rs +++ b/nodedb-cluster-tests/tests/cluster_common/calvin_test_node.rs @@ -258,6 +258,7 @@ async fn spawn_one_calvin_node( install_snapshot_chunk_bytes: 4 * 1024 * 1024, orphan_partial_max_age_secs: 300, log_compaction_threshold: None, + wire_build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), }; let lifecycle = ClusterLifecycleTracker::new(); @@ -416,7 +417,7 @@ pub async fn wait_for_sequencer_leader( pub fn try_recv_txn(rx: &mut mpsc::Receiver) -> Option { while let Ok(input) = rx.try_recv() { if let SchedulerInput::Txn(txn) = input { - return Some(txn); + return Some(*txn); } } None diff --git a/nodedb-cluster-tests/tests/cluster_common/test_node.rs b/nodedb-cluster-tests/tests/cluster_common/test_node.rs index f3fd6a340..091f3c7f5 100644 --- a/nodedb-cluster-tests/tests/cluster_common/test_node.rs +++ b/nodedb-cluster-tests/tests/cluster_common/test_node.rs @@ -122,6 +122,31 @@ impl TestNode { Self::spawn_with_transport(node_id, transport, seed_nodes).await } + /// Like [`TestNode::spawn`], but with the `JoinRequest` build identity + /// overridden to `wire_build_id` instead of the real + /// `nodedb_types::wire_version::WIRE_BUILD_ID`. Used to exercise + /// `handle_join_request`'s build-id rejection path end to end: the + /// joiner's `start_cluster` call is expected to return `Err` once the + /// seed rejects the mismatched build. + pub async fn spawn_with_wire_build_id( + node_id: u64, + seed_nodes: Vec, + wire_build_id: String, + ) -> Result> { + let transport = Arc::new(test_transport(node_id)?); + let data_dir = tempfile::tempdir()?; + let data_dir_path = data_dir.path().to_path_buf(); + Self::spawn_inner( + node_id, + transport, + seed_nodes, + data_dir_path, + Some(data_dir), + wire_build_id, + ) + .await + } + /// Use a pre-bound transport so the caller knows the listen /// address before start_cluster runs. Fresh temp data dir. pub async fn spawn_with_transport( @@ -137,6 +162,7 @@ impl TestNode { seed_nodes, data_dir_path, Some(data_dir), + nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), ) .await } @@ -155,7 +181,15 @@ impl TestNode { seed_nodes: Vec, ) -> Result> { let transport = Arc::new(test_transport(node_id)?); - Self::spawn_inner(node_id, transport, seed_nodes, data_dir.to_path_buf(), None).await + Self::spawn_inner( + node_id, + transport, + seed_nodes, + data_dir.to_path_buf(), + None, + nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), + ) + .await } pub async fn spawn_inner( @@ -164,6 +198,7 @@ impl TestNode { seed_nodes: Vec, data_dir_path: PathBuf, owned_data_dir: Option, + wire_build_id: String, ) -> Result> { let catalog = Arc::new(ClusterCatalog::open(&data_dir_path.join("cluster.redb"))?); let listen_addr = transport.local_addr(); @@ -207,6 +242,7 @@ impl TestNode { install_snapshot_chunk_bytes: 4 * 1024 * 1024, orphan_partial_max_age_secs: 300, log_compaction_threshold: None, + wire_build_id, }; let lifecycle = ClusterLifecycleTracker::new(); diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/cluster_join_build_id_mismatch.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/cluster_join_build_id_mismatch.rs new file mode 100644 index 000000000..9d3a9fbe3 --- /dev/null +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/cluster_join_build_id_mismatch.rs @@ -0,0 +1,32 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Integration test: a joiner advertising a different `WIRE_BUILD_ID` is +//! refused by `handle_join_request` and never joins the cluster. +//! +//! Exercises the config-level `wire_build_id` override end to end: the +//! joiner's `JoinRequest` carries the overridden build, the seed's +//! `handle_join_request` compares it against its own real +//! `nodedb_types::wire_version::WIRE_BUILD_ID`, and rejects. + +use std::time::Duration; + +use super::cluster_common::TestNode; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn joiner_with_mismatched_build_id_is_refused() { + let node1 = TestNode::spawn(1, vec![]).await.expect("node 1 bootstrap"); + tokio::time::sleep(Duration::from_millis(200)).await; + + let seeds = vec![node1.listen_addr()]; + let result = TestNode::spawn_with_wire_build_id(2, seeds, "some-other-build".to_owned()).await; + + assert!( + result.is_err(), + "joiner with a mismatched build_id must fail to join" + ); + + // The seed's topology must not have admitted the rejected joiner. + assert_eq!(node1.topology_size(), 1); + + node1.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/cluster_suite/cases/mod.rs b/nodedb-cluster-tests/tests/cluster_suite/cases/mod.rs index 80fe38369..eb8e7a6e1 100644 --- a/nodedb-cluster-tests/tests/cluster_suite/cases/mod.rs +++ b/nodedb-cluster-tests/tests/cluster_suite/cases/mod.rs @@ -11,6 +11,7 @@ mod calvin_e2e_ollp; mod calvin_e2e_pgwire; mod calvin_sequencer_failover; mod cluster_join; +mod cluster_join_build_id_mismatch; mod cluster_join_idempotent; mod cluster_join_leader_crash; mod cluster_join_race; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/array_install_snapshot_cluster.rs b/nodedb-cluster-tests/tests/common_suite/cases/array_install_snapshot_cluster.rs new file mode 100644 index 000000000..7c9b85045 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/array_install_snapshot_cluster.rs @@ -0,0 +1,150 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A Raft group snapshot carries array cells. +//! +//! Array cells are written while the data groups compact their logs past the +//! writes. A fresh learner then joins. It cannot catch up by `AppendEntries`, +//! so it installs group snapshots, and afterwards its own core holds exactly +//! the cell versions the original nodes hold: every put, overwrite and +//! tombstone, with the same system times and surrogates. +//! +//! The comparison reads each node's local core, never a pgwire query: the +//! gateway forwards array reads to the shard owners, so a query on the +//! learner proves nothing about its own state. + +use std::time::{Duration, Instant}; + +use nodedb::engine::array::export::ArrayCellVersion; +use nodedb::types::ArrayCellsBlob; +use nodedb_types::TenantId; + +use crate::common::cluster_harness::{TestCluster, wait_for}; + +/// Low enough that the writes below compact the data-group logs. +const COMPACTION_THRESHOLD: u64 = 4; +const CELLS: i64 = 48; +const TENANT: u64 = 1; + +/// Every cell version of `blobs`, canonical and sorted. +fn versions(blobs: &[ArrayCellsBlob]) -> Vec<(String, u32, Vec)> { + let mut out: Vec<(String, u32, Vec)> = blobs + .iter() + .flat_map(|blob| { + let decoded: Vec = + zerompk::from_msgpack(&blob.cells).expect("decode array cells"); + decoded.into_iter().map(move |version| { + let bytes = zerompk::to_msgpack_vec(&version).expect("encode version"); + (blob.array.clone(), blob.vshard, bytes) + }) + }) + .collect(); + out.sort(); + out +} + +/// cluster/array_install_snapshot +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_learner_installs_array_cells_from_group_snapshots() { + // `replication_factor = 4`, the post-join node count, places every node, + // the learner included, on every data group, so every node's core holds + // every cell. + let mut cluster = + TestCluster::spawn_three_with_compaction_threshold_and_rf(COMPACTION_THRESHOLD, 4) + .await + .expect("3-node cluster with low compaction threshold and rf=4"); + + cluster + .exec_ddl_on_any_leader( + "CREATE ARRAY snap_grid DIMS (x INT64 [0..63], y INT64 [0..63]) \ + ATTRS (v INT64) TILE_EXTENTS (8, 8)", + ) + .await + .expect("CREATE ARRAY"); + + // One Raft entry per statement, spread over the grid's vShards. + for i in 0..CELLS { + cluster.nodes[0] + .exec(&format!( + "INSERT INTO ARRAY snap_grid COORDS ({}, {}) VALUES ({i})", + i % 64, + (i * 7) % 64 + )) + .await + .unwrap_or_else(|e| panic!("insert cell {i}: {e}")); + } + // An overwrite and a tombstone ride the snapshot too. + tokio::time::sleep(Duration::from_millis(10)).await; + cluster.nodes[0] + .exec("INSERT INTO ARRAY snap_grid COORDS (1, 7) VALUES (1000)") + .await + .expect("overwrite"); + cluster.nodes[0] + .exec("DELETE FROM ARRAY snap_grid WHERE COORDS IN ((2, 14))") + .await + .expect("delete"); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + + let expected = versions(&cluster.nodes[0].array_cells(TenantId::new(TENANT)).await); + assert!( + expected.len() > CELLS as usize, + "the original node holds every put, the overwrite and the tombstone: {}", + expected.len() + ); + for node in &cluster.nodes[1..] { + assert_eq!( + versions(&node.array_cells(TenantId::new(TENANT)).await), + expected, + "node {} holds the same cells before the learner joins", + node.node_id + ); + } + let compacted = cluster + .nodes + .iter() + .map(|n| n.max_data_group_snapshot_index()) + .max() + .unwrap_or(0); + assert!( + compacted > 0, + "a data group must compact before the learner joins, so it installs a snapshot" + ); + + let learner_id = cluster + .add_learner_node() + .await + .expect("add learner node") + .node_id; + let learner = cluster + .nodes + .iter() + .find(|n| n.node_id == learner_id) + .expect("learner present"); + + wait_for( + "the learner installs a data-group snapshot", + Duration::from_secs(30), + Duration::from_millis(200), + || learner.max_data_group_snapshot_index() > 0, + ) + .await; + + let deadline = Instant::now() + Duration::from_secs(30); + loop { + let found = versions(&learner.array_cells(TenantId::new(TENANT)).await); + if found == expected { + break; + } + if Instant::now() >= deadline { + panic!( + "learner {learner_id} holds {} cell versions, the original nodes {}", + found.len(), + expected.len() + ); + } + tokio::time::sleep(Duration::from_millis(200)).await; + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/array_raft_snapshot_install.rs b/nodedb-cluster-tests/tests/common_suite/cases/array_raft_snapshot_install.rs index c2df49f8a..88abc5ce8 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/array_raft_snapshot_install.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/array_raft_snapshot_install.rs @@ -6,9 +6,10 @@ //! schema are accepted and replicated correctly. //! //! This exercises the full path through `ReplicatedWrite::ArraySchema` in -//! `run_apply_loop` → `apply_array_schema` → `OriginSchemaRegistry::import_snapshot`. -//! After that, `ReplicatedWrite::ArrayOp` entries are accepted because every -//! node now has the schema. +//! `run_apply_loop` → `apply_array_schema` → `OriginSchemaRegistry::import_snapshot`, +//! and the `PutArray` catalog entry the receiving node proposes. After that, +//! `ReplicatedWrite::ArrayOp` entries are accepted because every node holds +//! the schema and the catalog row. use crate::common; @@ -23,9 +24,7 @@ use nodedb_array::types::cell_value::value::CellValue; use nodedb_array::types::coord::value::CoordValue; use nodedb_types::sync::wire::array::{ArrayDeltaMsg, ArraySchemaSyncMsg}; -use common::array_sync::{ - build_schema_snapshot, hlc, import_schema_snapshot, register_catalog_entry, -}; +use common::array_sync::{build_schema_snapshot, hlc, import_schema_snapshot}; use common::cluster_harness::{TestCluster, wait_for}; fn shareds(cluster: &TestCluster) -> Vec<&Arc> { @@ -83,8 +82,9 @@ fn op_log_count(shared: &Arc) -> u64 { /// cluster/array_raft_snapshot_install /// /// Proposes a schema snapshot on node 1 only (node 2 and 3 start without -/// the schema). After the schema proposal is committed, all nodes must have -/// the schema. Subsequently, data ops are replicated correctly to all nodes. +/// the schema or a catalog row). After the proposal commits, every node must +/// hold the schema and the replicated catalog row. Subsequently, data ops are +/// replicated correctly to all nodes. #[tokio::test(flavor = "multi_thread")] async fn cluster_array_raft_snapshot_install() { let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); @@ -94,12 +94,6 @@ async fn cluster_array_raft_snapshot_install() { // Register the schema on node 1 only (nodes 2+3 start unaware). let (snap_bytes, schema_hlc) = build_schema_snapshot("snparray"); import_schema_snapshot(shareds[0], "snparray", &snap_bytes, schema_hlc); - register_catalog_entry(shareds[0], "snparray"); - - // Also register catalog entries on all nodes so the Data Plane can - // accept ops — the catalog is local and must be pre-populated. - register_catalog_entry(shareds[1], "snparray"); - register_catalog_entry(shareds[2], "snparray"); let inbound = make_inbound(shareds[0]); @@ -132,6 +126,30 @@ async fn cluster_array_raft_snapshot_install() { ) .await; + // The catalog row replicated to every node, durable and in the mirror. + wait_for( + "all 3 nodes hold the snparray catalog row", + Duration::from_secs(10), + Duration::from_millis(50), + || { + shareds.iter().all(|s| { + let durable = s + .credentials + .catalog() + .load_all_arrays() + .map(|rows| rows.iter().any(|a| a.name == "snparray")) + .unwrap_or(false); + let mirror = s + .array_catalog + .read() + .map(|c| c.all_entries().iter().any(|a| a.name == "snparray")) + .unwrap_or(false); + durable && mirror + }) + }, + ) + .await; + // Now propose data ops — all nodes must accept them. for i in 0u64..3 { let msg = put_delta("snparray", i as i64, i as f64, schema_hlc, 5000 + i); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/assign_surrogate_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/assign_surrogate_cross_node.rs index bff04e519..83b9a6f00 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/assign_surrogate_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/assign_surrogate_cross_node.rs @@ -54,36 +54,33 @@ async fn assign_remote_surrogate_is_authoritative_and_idempotent() { ) .await; - // Find a (coordinator, pk) pair where the pk's home vShard (`VShardId::from_key`) - // leader is a DIFFERENT node than the coordinator, so the routed assign actually - // crosses to a remote leader (not a local short circuit). This must work under - // ANY leadership distribution — including this harness's single-data-leader - // topology where ONE node leads every data vShard: from that node's own snapshot - // every key is local, but from a FOLLOWER's snapshot the leader resolves to a - // remote node. So we try each node as the candidate coordinator and pick the - // first whose snapshot yields a remote-led key. The owner is resolved exactly - // like the production helper: `leader_for_vshard` on the coordinator's snapshot. - let mut picked: Option<(u64, String, String, VShardId, u64)> = None; - 'outer: for cand in &cluster.nodes { + // A key's home is its collection's home vShard (`VShardId::from_collection`). + // Find a coordinator whose snapshot names a DIFFERENT node as that vShard's + // leader, so the routed assign crosses to a remote leader (not a local short + // circuit). The leader node's own snapshot sees the home as local, so each + // node is tried as the coordinator and the first follower is picked. The + // owner is resolved exactly like the production helper: `leader_for_vshard` + // on the coordinator's snapshot. + let collection = "people".to_string(); + let home = VShardId::from_collection(nodedb_types::CollectionKey::from_bare(DB, &collection)); + let pk = "person:0".to_string(); + let mut picked: Option<(u64, u64)> = None; + for cand in &cluster.nodes { let cand_id = cand.shared.node_id; let Some(routing) = cand.shared.cluster_routing.as_ref() else { continue; }; let guard = routing.read().unwrap_or_else(|p| p.into_inner()); - for i in 0..10_000u32 { - let pk = format!("person:{i}"); - let vshard = VShardId::from_key(pk.as_bytes()); - if let Ok(leader) = guard.leader_for_vshard(vshard.as_u32()) - && leader != 0 - && leader != cand_id - { - picked = Some((cand_id, "people".to_string(), pk, vshard, leader)); - break 'outer; - } + if let Ok(leader) = guard.leader_for_vshard(home.as_u32()) + && leader != 0 + && leader != cand_id + { + picked = Some((cand_id, leader)); + break; } } - let (coordinator_id, collection, pk, vshard, owner) = - picked.expect("some node's snapshot resolves a pk to a remote-led home vShard"); + let (coordinator_id, owner) = + picked.expect("some node's snapshot names a remote leader for the collection's home"); let coordinator = cluster .nodes @@ -100,7 +97,6 @@ async fn assign_remote_surrogate_is_authoritative_and_idempotent() { // authoritative surrogate. let s1 = assign_surrogate_routed( &coordinator.shared, - vshard, nodedb_types::CollectionKey::from_bare(DB, &collection), TENANT, pk.as_bytes(), @@ -119,7 +115,6 @@ async fn assign_remote_surrogate_is_authoritative_and_idempotent() { // source, so a repeat resolves the already-bound value. let s2 = assign_surrogate_routed( &coordinator.shared, - vshard, nodedb_types::CollectionKey::from_bare(DB, &collection), TENANT, pk.as_bytes(), diff --git a/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/fixture.rs b/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/fixture.rs new file mode 100644 index 000000000..ba40afc02 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/fixture.rs @@ -0,0 +1,341 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The cluster, schedule and per-node tickers the backup schedule failover +//! tests drive. + +use crate::common; +use common::cluster_harness::shared_steps::holds_vshard0_lease; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use std::path::PathBuf; +use std::time::{Duration, Instant}; + +use nodedb::config::server::BackupScheduleSettings; +use nodedb::control::backup::schedule::marks::settled_through; +use nodedb::event::scheduler::backup_job::BackupJobs; +use nodedb::event::scheduler::dispatcher::{JobDispatcher, JobDispatcherConfig}; +use nodedb_types::DatabaseId; + +const COLLECTION: &str = "bsf_orders"; +pub(super) const DATABASE: &str = "default"; +const TARGET_DIR: &str = "nightly"; +pub(super) const CONVERGE: Duration = Duration::from_secs(30); +pub(super) const STEP: Duration = Duration::from_millis(100); + +/// The leader of vShard 0's Raft group as `node`'s Raft status reports it, +/// `0` while none is known. The scheduler decides who fires from the same +/// status. +fn vshard0_leader(node: &TestClusterNode) -> u64 { + let group = node + .shared + .cluster_routing + .as_ref() + .expect("cluster routing") + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(0) + .expect("vShard 0 maps to a group"); + node.all_group_leaders() + .into_iter() + .find_map(|(id, leader)| (id == group).then_some(leader)) + .unwrap_or(0) +} + +/// One node's scheduler: its backup jobs and dispatcher. +struct Ticker { + jobs: BackupJobs, + dispatcher: JobDispatcher, +} + +impl Ticker { + fn new(schedule: &BackupScheduleSettings) -> Self { + Self { + jobs: BackupJobs::new(std::slice::from_ref(schedule)), + dispatcher: JobDispatcher::new(JobDispatcherConfig { + max_concurrent_jobs: 4, + max_result_bytes: u64::MAX, + }), + } + } + + /// Fire one tick at `now_secs` on `node` and wait for its work. + async fn tick(&self, node: &TestClusterNode, now_secs: u64) { + self.jobs.fire( + &node.shared, + &self.dispatcher, + &node.shared.job_history, + now_secs, + ); + let deadline = Instant::now() + Duration::from_secs(120); + while self.jobs.in_flight() > 0 { + assert!( + Instant::now() < deadline, + "node {}: a scheduled backup did not finish", + node.node_id + ); + tokio::time::sleep(Duration::from_millis(50)).await; + } + } +} + +/// A running cluster with one row to back up, its schedule, and the due +/// minute `M`. +pub(super) struct Fixture { + _root: tempfile::TempDir, + root: PathBuf, + pub(super) cluster: TestCluster, + pub(super) schedule: BackupScheduleSettings, + tickers: Vec, + pub(super) due: u64, +} + +impl Fixture { + pub(super) async fn new() -> Self { + let dir = tempfile::tempdir().expect("backup root"); + let root = dir.path().canonicalize().expect("canonical backup root"); + let cluster = TestCluster::spawn_three_with_backup_root(root.clone()) + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {COLLECTION}")) + .await + .expect("create collection"); + let deadline = Instant::now() + CONVERGE; + loop { + match cluster.nodes[0] + .client + .simple_query(&format!("INSERT INTO {COLLECTION} {{ id: 'o1', n: 1 }}")) + .await + { + Ok(_) => break, + Err(e) if Instant::now() < deadline => { + tracing::debug!(error = %e, "insert not accepted yet; retrying"); + tokio::time::sleep(Duration::from_millis(200)).await; + } + Err(e) => panic!("insert: {e}"), + } + } + cluster.wait_for_full_apply_convergence(CONVERGE).await; + + let schedule = BackupScheduleSettings { + database: DATABASE.into(), + target: format!("file://{}/{TARGET_DIR}", root.display()), + cron: "*/5 * * * *".into(), + keep: 3, + }; + let now_min = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .expect("clock after the epoch") + .as_secs() + / 60; + let tickers = cluster + .nodes + .iter() + .map(|_| Ticker::new(&schedule)) + .collect(); + wait_for("one node holds the vShard 0 lease", CONVERGE, STEP, || { + let leader = vshard0_leader(&cluster.nodes[0]); + leader != 0 + && cluster + .nodes + .iter() + .all(|node| vshard0_leader(node) == leader) + && cluster + .nodes + .iter() + .filter(|node| holds_vshard0_lease(node)) + .count() + == 1 + }) + .await; + Self { + _root: dir, + root, + cluster, + schedule, + tickers, + // A multiple of 5, so neither neighbouring minute matches. + due: (now_min / 5 + 1) * 5, + } + } + + /// The node holding the vShard 0 lease, else the leader Raft reports. + pub(super) fn leader(&self) -> u64 { + self.cluster + .nodes + .iter() + .find(|node| holds_vshard0_lease(node)) + .map_or_else( + || vshard0_leader(&self.cluster.nodes[0]), + |node| node.node_id, + ) + } + + /// Tick every node at `now_secs`. + pub(super) async fn tick_all(&self, now_secs: u64) { + for (node, ticker) in self.cluster.nodes.iter().zip(&self.tickers) { + ticker.tick(node, now_secs).await; + } + } + + /// Tick every node but `node_id` at `now_secs`. + pub(super) async fn tick_except(&self, node_id: u64, now_secs: u64) { + for (node, ticker) in self.cluster.nodes.iter().zip(&self.tickers) { + if node.node_id != node_id { + ticker.tick(node, now_secs).await; + } + } + } + + /// Cut node `node_id` off from every other node, both ways, or heal it. + pub(super) fn partition(&self, node_id: u64, severed: bool) { + let transport = |node: &TestClusterNode| { + std::sync::Arc::clone( + node.shared + .cluster_transport + .as_ref() + .expect("cluster transport"), + ) + }; + let cut = self + .cluster + .nodes + .iter() + .find(|node| node.node_id == node_id) + .map(transport) + .expect("the node is a member"); + for node in self + .cluster + .nodes + .iter() + .filter(|node| node.node_id != node_id) + { + let peer = transport(node); + if severed { + peer.sever(node_id); + cut.sever(node.node_id); + } else { + peer.heal(node_id); + cut.heal(node.node_id); + } + } + } + + /// Wait until one node other than `excluded` holds the vShard 0 lease. + pub(super) async fn wait_new_coordinator(&self, excluded: u64) { + let nodes = &self.cluster.nodes; + wait_for( + "a new node holds the vShard 0 lease", + CONVERGE, + STEP, + || { + nodes + .iter() + .filter(|node| node.node_id != excluded && holds_vshard0_lease(node)) + .count() + == 1 + }, + ) + .await; + } + + /// Tick node `node_id` alone at `now_secs`. + pub(super) async fn tick_node(&self, node_id: u64, now_secs: u64) { + for (node, ticker) in self.cluster.nodes.iter().zip(&self.tickers) { + if node.node_id == node_id { + ticker.tick(node, now_secs).await; + } + } + } + + /// The mark in each node's local catalog, in node order. + pub(super) fn local_marks(&self) -> Vec> { + self.cluster + .nodes + .iter() + .map(|node| settled_through(&node.shared, &self.schedule).expect("read mark")) + .collect() + } + + pub(super) async fn wait_marks(&self, desc: &str, mark: u64) { + wait_for(desc, CONVERGE, STEP, || { + self.local_marks().iter().all(|local| *local == Some(mark)) + }) + .await; + } + + /// Kill the vShard 0 leader, then wait for a new one and a live leader + /// on every group: the backup reads every data group. + pub(super) async fn kill_leader(&mut self) -> u64 { + let leader = self.leader(); + let idx = self + .cluster + .nodes + .iter() + .position(|node| node.node_id == leader) + .expect("the vShard 0 leader is a member"); + let dead = self.cluster.nodes.remove(idx); + self.tickers.remove(idx); + dead.shutdown().await; + let nodes = &self.cluster.nodes; + wait_for( + "the survivors elect a new vShard 0 leader", + CONVERGE, + STEP, + || { + let next = vshard0_leader(&nodes[0]); + next != 0 && next != leader && nodes.iter().all(|node| vshard0_leader(node) == next) + }, + ) + .await; + wait_for("every group has a live leader", CONVERGE, STEP, || { + nodes.iter().all(|node| { + node.all_group_leaders() + .into_iter() + .all(|(_, group_leader)| group_leader != 0 && group_leader != leader) + }) + }) + .await; + self.wait_new_coordinator(leader).await; + leader + } + + /// Successful and failed runs `node` recorded for the schedule. + pub(super) fn runs(&self, node: &TestClusterNode) -> (usize, usize) { + let runs = node.shared.job_history.last_runs( + DatabaseId::DEFAULT.as_u64(), + 0, + &self.schedule.job_name(), + 100, + ); + let ok = runs.iter().filter(|run| run.success).count(); + (ok, runs.len() - ok) + } + + /// Every envelope under the target, sorted. + pub(super) fn envelopes(&self) -> Vec { + let dir = self.root.join(TARGET_DIR); + if !dir.exists() { + return Vec::new(); + } + let mut names: Vec = std::fs::read_dir(dir) + .expect("list target") + .map(|entry| { + entry + .expect("entry") + .file_name() + .to_string_lossy() + .into_owned() + }) + .collect(); + names.sort(); + names + } + + pub(super) async fn shutdown(self) { + self.cluster.shutdown().await; + for ticker in self.tickers { + ticker.dispatcher.shutdown_and_drain().await; + } + } +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/isolated_leader.rs b/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/isolated_leader.rs new file mode 100644 index 000000000..1df512bd1 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/isolated_leader.rs @@ -0,0 +1,121 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A vShard 0 leader cut off from both peers fires nothing once its lease +//! lapses, and the new leader runs the due minute once. + +use crate::common; +use common::cluster_harness::shared_steps::holds_vshard0_lease; +use common::cluster_harness::wait_for; + +use nodedb::control::backup::schedule::envelope_name; +use nodedb::control::backup::schedule::marks::settled_through; + +use super::fixture::{CONVERGE, DATABASE, Fixture, STEP}; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn an_isolated_leader_stops_firing_and_the_new_leader_runs_the_minute_once() { + let fx = Fixture::new().await; + let due = fx.due; + let leader = fx.leader(); + wait_for( + "the leader holds the vShard 0 lease", + CONVERGE, + STEP, + || { + fx.cluster + .nodes + .iter() + .any(|node| node.node_id == leader && holds_vshard0_lease(node)) + }, + ) + .await; + + fx.tick_all((due - 1) * 60 + 5).await; + fx.wait_marks("every node holds the armed mark", due - 1) + .await; + + // `due` comes due while the leader is cut off from both peers. + fx.partition(leader, true); + let isolated = fx + .cluster + .nodes + .iter() + .find(|node| node.node_id == leader) + .expect("isolated node"); + wait_for("the isolated leader's lease lapses", CONVERGE, STEP, || { + !holds_vshard0_lease(isolated) + }) + .await; + fx.wait_new_coordinator(leader).await; + + // The isolated node ticks through `due` and fires nothing. + fx.tick_node(leader, due * 60 + 5).await; + fx.tick_node(leader, due * 60 + 35).await; + assert_eq!( + fx.runs(isolated), + (0, 0), + "a leader without its lease fires nothing" + ); + assert!(fx.envelopes().is_empty(), "{:?}", fx.envelopes()); + + let peers = &fx.cluster.nodes; + wait_for( + "every group the peers host has a live leader", + CONVERGE, + STEP, + || { + peers + .iter() + .filter(|node| node.node_id != leader) + .all(|node| { + node.all_group_leaders() + .into_iter() + .all(|(_, group_leader)| group_leader != 0 && group_leader != leader) + }) + }, + ) + .await; + + // The peers tick after `due`, with the partition still in place. Each + // peer's routing hints follow its own Raft, so the backup takes every + // group from a reachable leader. A failed attempt waits out its retry + // delay on the scheduler clock. No minute from `due + 1` through + // `due + 4` matches, so every attempt runs `due` itself. + let mut now_secs = (due + 1) * 60 + 5; + for _ in 0..3 { + fx.tick_except(leader, now_secs).await; + if fx.local_marks().contains(&Some(due)) { + break; + } + now_secs += 61; + } + let peers_marked = fx + .cluster + .nodes + .iter() + .filter(|node| node.node_id != leader) + .any(|node| settled_through(&node.shared, &fx.schedule).ok().flatten() == Some(due)); + assert!( + peers_marked, + "the new leader runs the due minute while the partition lasts" + ); + assert_eq!(fx.envelopes(), [envelope_name(DATABASE, due * 60_000)]); + assert_eq!( + fx.runs(isolated), + (0, 0), + "the isolated node still fires nothing" + ); + fx.partition(leader, false); + + // Healed, the former leader catches up and finds nothing due. + fx.wait_marks("every node holds the mark of the due minute", due) + .await; + fx.tick_all((due + 4) * 60 + 30).await; + + let successes: usize = fx.cluster.nodes.iter().map(|node| fx.runs(node).0).sum(); + assert_eq!(successes, 1, "the due minute runs exactly once"); + assert_eq!(fx.runs(isolated).0, 0); + assert_eq!(fx.envelopes(), [envelope_name(DATABASE, due * 60_000)]); + + fx.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/lagging_catalog.rs b/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/lagging_catalog.rs new file mode 100644 index 000000000..6f5a71565 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/lagging_catalog.rs @@ -0,0 +1,87 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A new vShard 0 leader whose catalog lags the schedule mark skips the tick +//! instead of rerunning the due minute. + +use nodedb::control::backup::schedule::envelope_name; +use nodedb::control::cluster::metadata_applier::backup_mark_fail_point; +use nodedb_types::fail_point::FailGuard; + +use super::fixture::{DATABASE, Fixture}; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_new_leader_whose_catalog_lags_never_reruns_the_due_minute() { + let mut fx = Fixture::new().await; + let due = fx.due; + let leader = fx.leader(); + + fx.tick_all((due - 1) * 60 + 5).await; + fx.wait_marks("every node holds the armed mark", due - 1) + .await; + + // Hold back every other node's apply of the next mark. + let guards: Vec = fx + .cluster + .nodes + .iter() + .filter(|node| node.node_id != leader) + .map(|node| { + FailGuard::fail( + &backup_mark_fail_point(node.node_id), + "held back by the test", + ) + }) + .collect(); + + // The leader runs `due`. Only its own catalog holds the new mark. + fx.tick_node(leader, due * 60 + 5).await; + let leader_node = fx + .cluster + .nodes + .iter() + .find(|node| node.node_id == leader) + .expect("leader node"); + assert_eq!( + fx.runs(leader_node), + (1, 0), + "the leader runs the due minute" + ); + assert_eq!(fx.envelopes(), [envelope_name(DATABASE, due * 60_000)]); + + fx.kill_leader().await; + assert!( + fx.local_marks().iter().all(|mark| *mark == Some(due - 1)), + "the survivors' catalogs lag the mark: {:?}", + fx.local_marks() + ); + + // The new leader's catalog shows `due` as due, but its read of the mark + // cannot be confirmed while its catalog lags, so the tick is skipped. + fx.tick_all((due + 1) * 60 + 5).await; + fx.tick_all((due + 1) * 60 + 35).await; + for node in &fx.cluster.nodes { + assert_eq!( + fx.runs(node), + (0, 0), + "node {} runs nothing on a lagging catalog", + node.node_id + ); + } + + // Released, the survivors apply the mark, and nothing is due. + drop(guards); + fx.wait_marks("every survivor applies the mark of the due minute", due) + .await; + fx.tick_all((due + 2) * 60 + 5).await; + for node in &fx.cluster.nodes { + assert_eq!( + fx.runs(node), + (0, 0), + "node {} never reruns the due minute", + node.node_id + ); + } + assert_eq!(fx.envelopes(), [envelope_name(DATABASE, due * 60_000)]); + + fx.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/mod.rs b/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/mod.rs new file mode 100644 index 000000000..9bf4fccfb --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/mod.rs @@ -0,0 +1,33 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A scheduled backup survives a leader change of vShard 0. +//! +//! Only the leader of vShard 0 fires scheduled backups. It picks the due +//! minute from the schedule mark the metadata group replicates, read after +//! it applied the group through a confirmed read index. The tests drive each +//! node's scheduler tick with a chosen clock. +//! +//! - The leader arms the schedule one minute before a due minute `M`, then +//! dies before its tick fires `M`. The new leader runs `M` once, the other +//! survivor runs nothing, and later ticks run nothing more. +//! - The leader runs `M`, the survivors' catalogs are held back from +//! applying that mark, and the leader dies. The new leader's catalog still +//! shows `M` as due, but its read cannot confirm the mark, so it skips the +//! tick. Once the survivors apply the mark, nothing is due. `M` is never +//! run twice. This one needs `--features failpoints`. +//! - The leader is cut off from both peers when `M` comes due. Its leader +//! lease lapses, so it fires nothing, while the peers elect a new leader +//! that runs `M` once, before the partition heals. +//! +//! A local "last fired" marker fails the first. A read of the local catalog +//! without the read-index barrier fails the second. A coordinator check on +//! the Raft role alone fails the third: the cut-off leader keeps its role. +//! +//! All three run on the multi-thread runtime: the cluster harness serves DDL +//! and metadata proposals on its nodes' async tasks with `block_in_place`. + +mod fixture; +mod isolated_leader; +#[cfg(feature = "failpoints")] +mod lagging_catalog; +mod new_leader; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/new_leader.rs b/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/new_leader.rs new file mode 100644 index 000000000..4e0b415c4 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/backup_schedule_failover/new_leader.rs @@ -0,0 +1,64 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A new vShard 0 leader runs the due minute the dead leader never fired. + +use nodedb::control::backup::schedule::envelope_name; + +use super::fixture::{DATABASE, Fixture}; + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_new_vshard0_leader_runs_the_due_minute_exactly_once() { + let mut fx = Fixture::new().await; + let due = fx.due; + + // One minute before `due`, every node ticks. Only the leader arms. + fx.tick_all((due - 1) * 60 + 5).await; + fx.wait_marks("every node holds the armed mark", due - 1) + .await; + + // `due` comes due, and the leader dies before its tick fires it. + fx.kill_leader().await; + + // The survivors tick after `due`, until the mark reaches it. A tick that + // fails waits out its retry delay on the scheduler clock, so each + // attempt moves the clock past it. No minute from `due + 1` through + // `due + 4` matches, so every attempt runs `due` itself. + let mut now_secs = (due + 1) * 60 + 5; + for _ in 0..3 { + fx.tick_all(now_secs).await; + if fx.local_marks().contains(&Some(due)) { + break; + } + now_secs += 61; + } + fx.wait_marks("every survivor holds the mark of the due minute", due) + .await; + + // Later ticks find nothing due. + fx.tick_all((due + 4) * 60 + 30).await; + + // Exactly one run of `due`, on the new leader, and one envelope for it. + let new_leader = fx.leader(); + let mut successes = 0; + for node in &fx.cluster.nodes { + let (ok, failed) = fx.runs(node); + if node.node_id == new_leader { + assert_eq!( + ok, 1, + "the new leader runs the due minute once ({failed} failed)" + ); + } else { + assert_eq!( + (ok, failed), + (0, 0), + "node {} is no leader and runs nothing", + node.node_id + ); + } + successes += ok; + } + assert_eq!(successes, 1); + assert_eq!(fx.envelopes(), [envelope_name(DATABASE, due * 60_000)]); + + fx.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_cdc_net_kinds.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_cdc_net_kinds.rs new file mode 100644 index 000000000..54803aa5b --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_cdc_net_kinds.rs @@ -0,0 +1,352 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A Calvin commit publishes each row's net kind on the Control-Plane change +//! stream of every node. +//! +//! - A transaction that writes two collections on distinct vShards commits +//! through Calvin. Several writes to one row publish one event: +//! insert-then-update publishes `Insert`, an update publishes `Update`, +//! update-then-delete publishes `Delete`, and insert-then-delete publishes +//! nothing. +//! - An autocommit UPDATE that moves an implicit edge to another vShard +//! commits through Calvin, and publishes `Update`. +//! - A columnar collection and an array publish one `*` event per kind of +//! net change a Calvin commit makes to them. +//! +//! Each node's feed is read by replay, so the check does not depend on when a +//! subscription opened. Events of different partitions have no fixed order, +//! so each check compares the set of `(row, operation)` a collection holds. + +use std::collections::BTreeSet; +use std::time::{Duration, Instant}; + +use nodedb::control::change_stream::ReplayStart; +use nodedb_types::{DatabaseId, TenantId}; + +use super::calvin_multishard_fixture::{Fixture, keyed_ddl, schemaless_ddl}; +use super::vshard_names::{distinct_vshard_collections, key_on_other_vshard}; +use crate::common::cluster_harness::TestClusterNode; + +const ARRIVAL: Duration = Duration::from_secs(20); + +/// Every `(row, operation)` `node`'s feed holds for `collection`. +fn feed(node: &TestClusterNode, collection: &str) -> BTreeSet<(String, &'static str)> { + node.shared + .change_stream + .query_changes_in_database( + TenantId::new(1), + DatabaseId::DEFAULT, + Some(collection), + ReplayStart::Timestamp(0), + 1024, + ) + .unwrap_or_else(|e| panic!("node {}: replay refused: {e:?}", node.node_id)) + .events + .iter() + .map(|change| { + ( + change.document_id.as_str().to_owned(), + change.operation.as_str(), + ) + }) + .collect() +} + +/// Wait until every node's feed of `collection` holds `expected`, then +/// check that it holds nothing more. +async fn expect_feeds(fx: &Fixture, collection: &str, expected: &BTreeSet<(String, &'static str)>) { + let deadline = Instant::now() + ARRIVAL; + while fx + .cluster + .nodes + .iter() + .any(|node| !feed(node, collection).is_superset(expected)) + { + assert!( + Instant::now() < deadline, + "every node's feed of {collection} holds {expected:?}" + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } + // A duplicate or a stray event is published in the same apply round. + tokio::time::sleep(Duration::from_secs(1)).await; + for node in &fx.cluster.nodes { + assert_eq!( + &feed(node, collection), + expected, + "node {}: {collection} publishes each row's net kind once", + node.node_id + ); + } +} + +fn rows(entries: &[(&str, &'static str)]) -> BTreeSet<(String, &'static str)> { + entries + .iter() + .map(|(row, op)| ((*row).to_owned(), *op)) + .collect() +} + +async fn run(fx: &Fixture, sql: &str) { + fx.coordinator() + .client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e:?}")); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_calvin_transaction_publishes_net_kinds() { + let (left, right) = distinct_vshard_collections("calvin_net_left", "calvin_net_right"); + let fx = Fixture::spawn(&[keyed_ddl(&left), keyed_ddl(&right)]).await; + fx.wait_group_mounted(&left).await; + fx.wait_group_mounted(&right).await; + + for id in ["u1", "d1"] { + run( + &fx, + &format!("INSERT INTO {left} (id, v) VALUES ('{id}', 'seed')"), + ) + .await; + } + + // A transaction that spans vShards commits through Calvin. + run(&fx, "SET cross_shard_txn = 'strict'").await; + run(&fx, "BEGIN").await; + for sql in [ + format!("INSERT INTO {right} (id, v) VALUES ('n1', 'new')"), + format!("UPDATE {right} SET v = 'changed' WHERE id = 'n1'"), + format!("UPDATE {left} SET v = 'changed' WHERE id = 'u1'"), + format!("UPDATE {left} SET v = 'changed' WHERE id = 'd1'"), + format!("DELETE FROM {left} WHERE id = 'd1'"), + format!("INSERT INTO {left} (id, v) VALUES ('x1', 'new')"), + format!("DELETE FROM {left} WHERE id = 'x1'"), + ] { + run(&fx, &sql).await; + } + run(&fx, "COMMIT").await; + fx.converge().await; + + expect_feeds( + &fx, + &left, + &rows(&[ + ("u1", "INSERT"), + ("d1", "INSERT"), + ("u1", "UPDATE"), + ("d1", "DELETE"), + ]), + ) + .await; + expect_feeds(&fx, &right, &rows(&[("n1", "INSERT")])).await; + + fx.cluster.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_calvin_routed_autocommit_update_publishes_update() { + let coll = "calvin_net_edges"; + let fx = Fixture::spawn(&[schemaless_ddl(coll)]).await; + fx.wait_group_mounted(coll).await; + + let src = key_on_other_vshard(coll, "src_e1"); + let old_hub = key_on_other_vshard(coll, "hub_old"); + let new_hub = key_on_other_vshard(coll, "hub_new"); + run( + &fx, + &format!( + "INSERT INTO {coll} \ + {{ id: 'e1', _from: '{src}', _to: '{old_hub}', _type: 'l', mark: 'move' }}" + ), + ) + .await; + run( + &fx, + &format!("UPDATE {coll} SET _to = '{new_hub}' WHERE id = 'e1'"), + ) + .await; + fx.converge().await; + + expect_feeds(&fx, coll, &rows(&[("e1", "INSERT"), ("e1", "UPDATE")])).await; + + fx.cluster.shutdown().await; +} + +/// How many events of each `(row, operation)` `node`'s feed holds for +/// `collection`. A whole-collection event names every row as `*`, so the +/// count tells two inserts apart. +fn feed_counts( + node: &TestClusterNode, + collection: &str, +) -> std::collections::BTreeMap<(String, &'static str), usize> { + let mut counts = std::collections::BTreeMap::new(); + for change in node + .shared + .change_stream + .query_changes_in_database( + TenantId::new(1), + DatabaseId::DEFAULT, + Some(collection), + ReplayStart::Timestamp(0), + 1024, + ) + .unwrap_or_else(|e| panic!("node {}: replay refused: {e:?}", node.node_id)) + .events + { + *counts + .entry(( + change.document_id.as_str().to_owned(), + change.operation.as_str(), + )) + .or_insert(0) += 1; + } + counts +} + +/// Wait until every node's feed of `collection` holds exactly `expected` +/// events of each kind. +async fn expect_feed_counts( + fx: &Fixture, + collection: &str, + expected: &std::collections::BTreeMap<(String, &'static str), usize>, +) { + let deadline = Instant::now() + ARRIVAL; + while fx + .cluster + .nodes + .iter() + .any(|node| &feed_counts(node, collection) != expected) + { + assert!( + Instant::now() < deadline, + "every node's feed of {collection} holds {expected:?}; node 0 holds {:?}", + feed_counts(&fx.cluster.nodes[0], collection) + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } + // A duplicate or a stray event is published in the same apply round. + tokio::time::sleep(Duration::from_secs(1)).await; + for node in &fx.cluster.nodes { + assert_eq!( + &feed_counts(node, collection), + expected, + "node {}: {collection} publishes each net change once", + node.node_id + ); + } +} + +/// Run `sql` on the coordinator until it is accepted: a statement on a +/// freshly created object waits for the object's group to serve it. +async fn run_until_accepted(fx: &Fixture, sql: &str) { + let deadline = Instant::now() + ARRIVAL; + loop { + match fx.coordinator().client.simple_query(sql).await { + Ok(_) => return, + Err(error) if Instant::now() < deadline => { + tracing::debug!(%error, sql, "statement not accepted yet; retrying"); + tokio::time::sleep(Duration::from_millis(200)).await; + } + Err(error) => panic!("{sql}: {error:?}"), + } + } +} + +fn counts( + entries: &[((&str, &'static str), usize)], +) -> std::collections::BTreeMap<(String, &'static str), usize> { + entries + .iter() + .map(|((row, op), count)| (((*row).to_owned(), *op), *count)) + .collect() +} + +/// A columnar collection publishes one `*` event per kind of net change a +/// Calvin commit makes to it: the seed row's insert, then the +/// transaction's update of it and its insert of a new row. The row it +/// inserted and deleted adds no kind. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_calvin_transaction_publishes_columnar_net_kinds() { + let (columnar, side) = distinct_vshard_collections("calvin_net_cols", "calvin_net_cols_side"); + let fx = Fixture::spawn(&[ + format!( + "CREATE COLLECTION {columnar} (id TEXT PRIMARY KEY, v TEXT) WITH (engine='columnar')" + ), + keyed_ddl(&side), + ]) + .await; + fx.wait_group_mounted(&columnar).await; + fx.wait_group_mounted(&side).await; + run( + &fx, + &format!("INSERT INTO {columnar} (id, v) VALUES ('u1', 'seed')"), + ) + .await; + + run(&fx, "SET cross_shard_txn = 'strict'").await; + run(&fx, "BEGIN").await; + for sql in [ + format!("UPDATE {columnar} SET v = 'changed' WHERE id = 'u1'"), + format!("INSERT INTO {columnar} (id, v) VALUES ('n1', 'new')"), + format!("INSERT INTO {columnar} (id, v) VALUES ('x1', 'new')"), + format!("DELETE FROM {columnar} WHERE id = 'x1'"), + format!("INSERT INTO {side} (id, v) VALUES ('s1', 'side')"), + ] { + run(&fx, &sql).await; + } + run(&fx, "COMMIT").await; + fx.converge().await; + + expect_feed_counts( + &fx, + &columnar, + &counts(&[(("*", "INSERT"), 2), (("*", "UPDATE"), 1)]), + ) + .await; + + fx.cluster.shutdown().await; +} + +/// An array publishes one `*` event per kind of net change a Calvin commit +/// makes to its cells: the seed cell's insert, then the transaction's update +/// of that cell and its insert of a new one. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_calvin_transaction_publishes_array_net_kinds() { + let grid = "calvin_net_grid"; + let side = "calvin_net_grid_side"; + let fx = Fixture::spawn(&[keyed_ddl(side)]).await; + fx.wait_group_mounted(side).await; + fx.cluster + .exec_ddl_on_any_leader(&format!( + "CREATE ARRAY {grid} DIMS (x INT64 [0..63], y INT64 [0..63]) \ + ATTRS (v INT64) TILE_EXTENTS (8, 8)" + )) + .await + .expect("CREATE ARRAY"); + run_until_accepted( + &fx, + &format!("INSERT INTO ARRAY {grid} COORDS (1, 1) VALUES (5)"), + ) + .await; + + run(&fx, "SET cross_shard_txn = 'strict'").await; + run(&fx, "BEGIN").await; + for sql in [ + format!("INSERT INTO ARRAY {grid} COORDS (1, 1) VALUES (6)"), + format!("INSERT INTO ARRAY {grid} COORDS (40, 40) VALUES (7)"), + format!("INSERT INTO {side} (id, v) VALUES ('s1', 'side')"), + ] { + run(&fx, &sql).await; + } + run(&fx, "COMMIT").await; + fx.converge().await; + + expect_feed_counts( + &fx, + grid, + &counts(&[(("*", "INSERT"), 2), (("*", "UPDATE"), 1)]), + ) + .await; + + fx.cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_joiner_after_sequencer_compaction.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_joiner_after_sequencer_compaction.rs new file mode 100644 index 000000000..94ede1cd7 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_joiner_after_sequencer_compaction.rs @@ -0,0 +1,301 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A node that joins a data group after the sequencer compacted its log holds +//! every Calvin-written row and edge of the group. +//! +//! A Calvin transaction runs outside its data groups' logs: the sequencer +//! orders it, and each participant's scheduler installs its slice. Calvin +//! writes barely grow a data group's log, so the group need not compact, and +//! a new replica can catch up by log replay. Its scheduler catches up from +//! the first index the sequencer log still holds, so a transaction the +//! sequencer compacted away never reaches it. +//! +//! On 3 nodes with RF 2 the test picks a data group `G_e` that a fourth +//! node's placement names. It writes edges and cross-shard rows homed on +//! `G_e`, then compacts the sequencer log past them with filler edges homed +//! on other groups. The fourth node joins, and its replica of `G_e` must hold +//! every edge and row: the group's snapshot carries them with the Calvin cut +//! they were captured at. + +use std::time::Duration; + +use nodedb::control::security::catalog::calvin_base::CalvinBase; +use nodedb::types::{DatabaseId, TenantDataSnapshot, TenantId, VShardId}; +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; + +use crate::common::cluster_harness::shared_steps::{ + db_detail, group_members, group_of_key, group_status, key_collection, +}; +use crate::common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +const COMPACTION_THRESHOLD: u64 = 4; +const RF: usize = 2; +const GROUPS: u64 = 4; +const NODES_WITH_JOINER: [u64; 4] = [1, 2, 3, 4]; +const PAIRS: usize = 4; +const ROWS: usize = 4; +const MAX_FILLER: usize = 400; +const TENANT: u64 = 1; +/// SQLSTATE `serialization_failure`: the transaction aborted and a client +/// runs it again. +const SERIALIZATION_FAILURE: &str = "40001"; +/// How long a write retries a `40001` before the test fails. +const RETRY_DEADLINE: Duration = Duration::from_secs(30); +/// The pause between two attempts of a write. +const RETRY_BACKOFF: Duration = Duration::from_millis(50); + +/// `count` node keys whose group satisfies `on`, named `{prefix}{i}`. +fn keys_where( + node: &TestClusterNode, + prefix: &str, + count: usize, + on: impl Fn(u64) -> bool, +) -> Vec { + (0..1_000_000) + .map(|i| format!("{prefix}{i}")) + .filter(|k| on(group_of_key(node, k))) + .take(count) + .collect() +} + +/// Run `sql` on `node` as one client request. A `40001` failure aborted the +/// transaction and wrote nothing: a real client runs it again, and so does +/// this until [`RETRY_DEADLINE`]. Any other error, or a `40001` past the +/// deadline, panics with `what`, the SQLSTATE and the detail. +async fn run_retrying(node: &TestClusterNode, what: &str, sql: &str) { + let deadline = tokio::time::Instant::now() + RETRY_DEADLINE; + loop { + let error = match node.client.simple_query(sql).await { + Ok(_) => return, + Err(error) => error, + }; + let retryable = error + .as_db_error() + .is_some_and(|db| db.code().code() == SERIALIZATION_FAILURE); + if !retryable || tokio::time::Instant::now() >= deadline { + panic!("{what}: {}", db_detail(&error)); + } + // A failed block can leave the session in an aborted transaction. + // Outside one, ROLLBACK only warns, so its result does not matter. + let _ = node.client.simple_query("ROLLBACK").await; + tokio::time::sleep(RETRY_BACKOFF).await; + } +} + +async fn insert_edge(node: &TestClusterNode, collection: &str, src: &str, dst: &str) { + run_retrying( + node, + &format!("insert {src} -> {dst}"), + &format!("GRAPH INSERT EDGE IN '{collection}' FROM '{src}' TO '{dst}' TYPE 'l'"), + ) + .await; +} + +/// Every edge key `node` stores locally, from its own tenant snapshot. +async fn local_edge_keys(node: &TestClusterNode) -> Vec { + let bytes = node.create_tenant_snapshot(TenantId::new(TENANT)).await; + let snapshot: TenantDataSnapshot = zerompk::from_msgpack(&bytes).unwrap_or_default(); + snapshot.edges.into_iter().map(|(key, _)| key).collect() +} + +/// Whether `edges` holds an edge from `src` to `dst`. +fn holds_edge(edges: &[String], src: &str, dst: &str) -> bool { + let from = format!("\u{0}{src}\u{0}"); + let to = format!("\u{0}{dst}\u{0}"); + edges + .iter() + .any(|key| key.contains(&from) && key.contains(&to)) +} + +/// The number of documents of `collection` `node` stores locally. +async fn local_rows(node: &TestClusterNode, collection: &str) -> usize { + let mut count = 0; + for core in 0..node.num_cores() { + count += node + .document_keys_on_core(core, TenantId::new(TENANT)) + .await + .iter() + .filter(|key| key_collection(key) == Some(collection)) + .count(); + } + count +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_joiner_after_sequencer_compaction_holds_every_calvin_write() { + let mut cluster = TestCluster::spawn_three_with_groups_compaction_threshold_and_rf( + GROUPS, + COMPACTION_THRESHOLD, + RF, + ) + .await + .expect("3-node cluster with 4 groups, low compaction threshold and rf=2"); + + let view = &cluster.nodes[0]; + let data_groups: Vec = (1..=GROUPS).collect(); + let placement = nodedb_cluster::rebalancer::placement::compute_placement( + &NODES_WITH_JOINER, + &data_groups, + RF as u32, + ); + let joiner_id = NODES_WITH_JOINER[3]; + let joins = |g: u64| placement.get(&g).is_some_and(|p| p.contains(&joiner_id)); + let endpoint_group = data_groups + .iter() + .copied() + .find(|g| joins(*g)) + .expect("placement names the joiner in a group"); + let named = |prefix: &str, on: &dyn Fn(u64) -> bool| { + (0..10_000) + .map(|i| format!("{prefix}{i}")) + .find(|name| view.group_id_for_collection(name).is_some_and(on)) + .unwrap_or_else(|| panic!("a {prefix} collection name on the wanted group")) + }; + // The edge collection homes off G_e: its filler edges' binds then grow + // no G_e log. + let edges = named("cj_edges_", &|g| g != endpoint_group); + let rows_here = named("cj_rows_e_", &|g| g == endpoint_group); + let rows_there = named("cj_rows_o_", &|g| g != endpoint_group); + + for name in [&edges, &rows_here, &rows_there] { + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {name}")) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION {name}: {e}")); + } + wait_for( + "all nodes see the collections", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 3) + }, + ) + .await; + + // Edges whose endpoints home on G_e, and cross-shard rows with one + // slice on G_e. Each is a Calvin transaction. + let writer = &cluster.nodes[0]; + let endpoints = keys_where(view, "cj_ep_", PAIRS * 2, |g| g == endpoint_group); + for pair in endpoints.chunks(2) { + insert_edge(writer, &edges, &pair[0], &pair[1]).await; + } + writer + .client + .simple_query("SET cross_shard_txn = 'strict'") + .await + .expect("SET cross_shard_txn = strict"); + for i in 0..ROWS { + run_retrying( + writer, + &format!("cross-shard COMMIT {i}"), + &format!( + "BEGIN; \ + INSERT INTO {rows_here} {{ id: 'r{i}', v: 'here' }}; \ + INSERT INTO {rows_there} {{ id: 'r{i}', v: 'there' }}; \ + COMMIT" + ), + ) + .await; + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + let sequencer_after_writes = cluster + .nodes + .iter() + .filter_map(|n| group_status(n, SEQUENCER_GROUP_ID)) + .map(|s| s.last_applied) + .max() + .expect("the sequencer group is hosted"); + + // Compact the sequencer log past the writes with Calvin edges homed on + // other groups, so G_e's own log does not grow. + let sequencer_compacted = |cluster: &TestCluster| { + cluster + .nodes + .iter() + .filter_map(|n| group_status(n, SEQUENCER_GROUP_ID)) + .all(|s| s.snapshot_index >= sequencer_after_writes) + }; + let filler = keys_where(view, "cj_fill_", MAX_FILLER * 2, |g| g != endpoint_group); + let mut written = 0; + for pair in filler.chunks(2) { + if sequencer_compacted(&cluster) { + break; + } + insert_edge(writer, &edges, &pair[0], &pair[1]).await; + written += 1; + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + assert!( + sequencer_compacted(&cluster), + "every sequencer replica compacted past {sequencer_after_writes} after {written} \ + filler edges" + ); + + let joined = cluster + .add_learner_node() + .await + .expect("add the fourth node") + .node_id; + assert_eq!( + joined, joiner_id, + "the fourth node joins as the placed node" + ); + let joiner = cluster + .nodes + .iter() + .find(|n| n.node_id == joiner_id) + .expect("joiner present"); + + // The joiner's scheduler of every G_e vShard started from a Calvin base + // that reaches its sequencer log. + let endpoint_vshards: Vec = endpoints + .iter() + .map(|k| VShardId::from_key(k.as_bytes()).as_u32()) + .chain(std::iter::once( + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &rows_here) + .vshard() + .as_u32(), + )) + .collect(); + wait_for( + "the joiner keeps the Calvin state of every G_e vShard", + Duration::from_secs(60), + Duration::from_millis(100), + || { + group_members(&cluster.nodes[0], endpoint_group).contains(&joiner_id) + && endpoint_vshards + .iter() + .all(|v| CalvinBase::is_kept(joiner.shared.calvin.bases.base(*v))) + }, + ) + .await; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + + let held = local_edge_keys(joiner).await; + for pair in endpoints.chunks(2) { + assert!( + holds_edge(&held, &pair[0], &pair[1]), + "joiner {joiner_id} holds the edge {} -> {} written before the sequencer \ + compacted", + pair[0], + pair[1] + ); + } + assert_eq!( + local_rows(joiner, &rows_here).await, + ROWS, + "joiner {joiner_id} holds every cross-shard row written on G_e" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_remote_participant_report.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_remote_participant_report.rs new file mode 100644 index 000000000..929a7bafe --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_remote_participant_report.rs @@ -0,0 +1,355 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A Calvin participant on another node reports its apply to the +//! coordinator through its completion ack. +//! +//! With a replication factor of 1, each data group lives on one node. The +//! client's session runs on the sequencer leader, which coordinates every +//! Calvin transaction. Each collection below lives on another node, so the +//! coordinator holds no replica of the participant, and no local apply +//! result reaches it. Only the participant's completion ack does. +//! +//! - A strict cross-shard transaction writes a timeseries row and a +//! document. The transaction parks after its ingest resolved. A write on +//! the owner then gives the ingest's new column another type, so the +//! ingest's install rejects its row at its log position. The COMMIT +//! reports that rejected row. +//! +//! A fresh column comes only from a raw ILP line (see `ts_native_ingest`). +//! So the collection declares only its `ts` time key, and the +//! transaction's line and the owner's line both go through the native +//! `TimeseriesIngest` opcode. +//! - A `DELETE ... RETURNING` on an edge-bearing collection commits through +//! the dependent Calvin path. Its reply carries the deleted row. +//! +//! File name contains "calvin" so nextest applies the cluster test-group +//! serialization. + +#![cfg(feature = "failpoints")] + +use std::time::Duration; + +use crate::common; +use common::cluster_harness::shared_steps::{fail_stopped, sequencer_admitted}; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use nodedb_test_support::native_harness::send_sql; +use nodedb_types::fail_point::{FailAction, FailGuard}; +use nodedb_types::{DatabaseId, TenantId}; + +use super::ts_native_ingest::{ + assert_native_ok, ingest_native, native_session, rejection_warnings, +}; +use super::vshard_names::distinct_vshard_collections; + +const TENANT: u64 = 1; +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(50); +/// Candidate collection names tried before giving up. +const MAX_TRIES: u32 = 512; +/// Implicit-edge documents seeded before the `RETURNING` delete. +const SOURCES: usize = 4; + +fn pg_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +async fn exec(node: &TestClusterNode, sql: &str) -> Vec { + node.client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {}", pg_detail(&e))) +} + +/// Whether `collection` is edge-bearing in `node`'s local catalog. +fn edge_bearing(node: &TestClusterNode, collection: &str) -> bool { + node.shared + .credentials + .catalog() + .load_collections_for_tenant(DatabaseId::DEFAULT, TENANT) + .map(|collections| { + collections + .iter() + .any(|c| c.name == collection && c.has_implicit_edges) + }) + .unwrap_or(false) +} + +/// The node that alone replicates the data group of `collection`, once +/// placement converged to one replica. +async fn owner_of<'a>(cluster: &'a TestCluster, collection: &str) -> &'a TestClusterNode { + let group_id = cluster.nodes[0] + .group_id_for_collection(collection) + .unwrap_or_else(|| panic!("the data group of {collection}")); + wait_for( + &format!("exactly one node replicates group {group_id}"), + CONVERGE, + STEP, + || { + cluster + .nodes + .iter() + .filter(|node| node.replicates_data_group(group_id)) + .count() + == 1 + }, + ) + .await; + cluster + .nodes + .iter() + .find(|node| node.replicates_data_group(group_id)) + .unwrap_or_else(|| panic!("the replica of group {group_id}")) +} + +/// A collection name `{prefix}_{i}` whose data group lives on a node other +/// than `coordinator`. +async fn collection_off(cluster: &TestCluster, coordinator: u64, prefix: &str) -> String { + for i in 0..MAX_TRIES { + let name = format!("{prefix}_{i}"); + if owner_of(cluster, &name).await.node_id != coordinator { + return name; + } + } + panic!("no collection under {prefix} lives off node {coordinator} in {MAX_TRIES} tries"); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn remote_participants_report_counts_and_rows_through_their_acks() { + let cluster = TestCluster::spawn_three_with_replication_factor(1) + .await + .expect("spawn 3-node cluster"); + wait_for( + "every node sees one sequencer leader", + CONVERGE, + STEP, + || { + let leader = cluster.nodes[0].sequencer_leader(); + leader != 0 && cluster.nodes.iter().all(|n| n.sequencer_leader() == leader) + }, + ) + .await; + let leader_id = cluster.nodes[0].sequencer_leader(); + let coordinator = cluster + .nodes + .iter() + .find(|node| node.node_id == leader_id) + .expect("the sequencer leader is a cluster node"); + + remote_ingest_reports_its_rejected_row(&cluster, coordinator).await; + remote_delete_returns_its_row(&cluster, coordinator).await; + + for node in &cluster.nodes { + assert!( + !fail_stopped(node), + "node {} fail-stopped a core", + node.node_id + ); + } + cluster.shutdown().await; +} + +/// The timeseries participant's install rejects a row its resolve accepted. +/// The COMMIT on the coordinator reports it. +async fn remote_ingest_reports_its_rejected_row( + cluster: &TestCluster, + coordinator: &TestClusterNode, +) { + let series = collection_off(cluster, coordinator.node_id, "remote_report_ts").await; + let (_, documents) = distinct_vshard_collections(&series, "remote_report_doc"); + let owner = owner_of(cluster, &series).await; + + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {series} (ts BIGINT TIME_KEY) WITH (engine='timeseries')" + )) + .await + .expect("create the timeseries collection"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {documents}")) + .await + .expect("create the document collection"); + wait_for("every node sees both collections", CONVERGE, STEP, || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 2) + }) + .await; + let mut session = native_session(coordinator).await; + assert_native_ok( + &ingest_native( + &mut session, + 1, + &series, + &format!("{series} value=0.5 1000000000"), + ) + .await, + "the warm-up ingest", + ); + + // The transaction parks after its ingest resolved `extra` as a float, + // before the sequencer admits it. + let gate_dir = tempfile::tempdir().expect("gate tempdir"); + let release = gate_dir.path().join("release-commit"); + let parked = gate_dir.path().join("release-commit.parked"); + let gate = FailGuard::install( + &format!("calvin::after_stamp::{series}"), + FailAction::WaitForFile(release.clone()), + ); + + assert_native_ok( + &send_sql(&mut session, 2, "SET cross_shard_txn = 'strict'").await, + "SET cross_shard_txn", + ); + let admitted_before = sequencer_admitted(coordinator); + assert_native_ok(&send_sql(&mut session, 3, "BEGIN").await, "BEGIN"); + // The transaction's line gives `extra`, which no schema holds yet, a + // float. + assert_native_ok( + &ingest_native( + &mut session, + 4, + &series, + &format!("{series} value=1.0,extra=1.5 2000000000"), + ) + .await, + "the staged ingest", + ); + assert_native_ok( + &send_sql( + &mut session, + 5, + &format!("INSERT INTO {documents} {{ id: 'd-1', n: 1 }}"), + ) + .await, + "the staged document insert", + ); + let commit = tokio::spawn(async move { send_sql(&mut session, 6, "COMMIT").await }); + wait_for( + "the transaction parks after its stamp", + CONVERGE, + STEP, + || parked.exists(), + ) + .await; + + // The owner's replica types `extra` as a string. + let mut on_owner = native_session(owner).await; + assert_native_ok( + &ingest_native( + &mut on_owner, + 1, + &series, + &format!("{series} value=2.0,extra=\"text\" 3000000000"), + ) + .await, + "the owner's ingest", + ); + std::fs::write(&release, b"release").expect("release the transaction"); + let committed = commit.await.expect("transaction task"); + drop(gate); + + assert_native_ok(&committed, "COMMIT"); + assert!( + sequencer_admitted(coordinator) > admitted_before, + "the cross-shard COMMIT is sequenced through Calvin" + ); + let notices = rejection_warnings(&committed, &series); + assert!( + notices.iter().any(|notice| notice.contains("1 line(s)")), + "the COMMIT reports the row the remote install rejected, got {notices:?}" + ); + assert!( + !coordinator.replicates_data_group( + coordinator + .group_id_for_collection(&series) + .expect("the series data group") + ), + "the coordinator holds no replica of the timeseries participant" + ); + + // The owner stores the warm-up row and the string row. The transaction's + // row never lands. + wait_for_stored_rows(owner, &series, 2).await; +} + +/// Wait until `node`'s own replica of `collection` stores `expected` rows, +/// then check it stores exactly that many. +async fn wait_for_stored_rows(node: &TestClusterNode, collection: &str, expected: usize) { + let deadline = tokio::time::Instant::now() + CONVERGE; + let mut rows = Vec::new(); + while tokio::time::Instant::now() < deadline { + rows = node + .timeseries_rows_local(TenantId::new(TENANT), collection) + .await; + if rows.len() == expected { + break; + } + tokio::time::sleep(STEP).await; + } + assert_eq!( + rows.len(), + expected, + "node {} stores {rows:?} of {collection}", + node.node_id + ); +} + +/// The document participant's `RETURNING` rows reach the coordinator only +/// through its completion ack. +async fn remote_delete_returns_its_row(cluster: &TestCluster, coordinator: &TestClusterNode) { + let documents = collection_off(cluster, coordinator.node_id, "remote_report_edges").await; + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {documents} WITH (engine='document_schemaless')" + )) + .await + .expect("create the edge-bearing collection"); + wait_for("every node sees the collection", CONVERGE, STEP, || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 3) + }) + .await; + for i in 0..SOURCES { + exec( + coordinator, + &format!( + "INSERT INTO {documents} {{ id: 'edge_{i}', _from: 'src_{i}', _to: 'hub', _type: 'l' }}" + ), + ) + .await; + } + wait_for("the collection is edge-bearing", CONVERGE, STEP, || { + edge_bearing(coordinator, &documents) + }) + .await; + + let admitted_before = sequencer_admitted(coordinator); + let messages = exec( + coordinator, + &format!("DELETE FROM {documents} WHERE id = 'edge_2' RETURNING *"), + ) + .await; + assert!( + sequencer_admitted(coordinator) > admitted_before, + "the RETURNING delete commits through Calvin" + ); + let ids: Vec = messages + .iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get("id").map(str::to_owned), + _ => None, + }) + .collect(); + assert_eq!( + ids, + vec!["edge_2".to_owned()], + "the delete answers the row its remote participant deleted" + ); +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/calvin_superseded_collection_cluster.rs b/nodedb-cluster-tests/tests/common_suite/cases/calvin_superseded_collection_cluster.rs new file mode 100644 index 000000000..f2e21a287 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/calvin_superseded_collection_cluster.rs @@ -0,0 +1,125 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A Calvin transaction staged against a collection that a purge and a +//! same-name create replaced never applies, in the new collection or in any +//! other one. +//! +//! 1. A 3-node cluster holds two collections on distinct vShards. A +//! coordinator that is not the sequencer leader runs +//! `BEGIN; INSERT a; INSERT b; COMMIT` in strict cross-shard mode. +//! 2. The fail gate `calvin::after_stamp::` parks the transaction on the +//! sequencer leader after its collection incarnations are stamped and +//! before the sequencer admits it. +//! 3. `a` is purged and created again while the transaction is parked. +//! 4. Released, the transaction reaches every replica with `a`'s old +//! incarnation. Every replica refuses it, and the global verdict aborts it: +//! the client gets the retryable serialization failure, and neither +//! collection holds a row on any node. +//! 5. The same transaction, retried, commits. +//! +//! Requires `--features failpoints`. File name contains "cluster" so nextest +//! applies the cluster test group. + +#![cfg(feature = "failpoints")] + +use std::time::Duration; + +use nodedb_types::fail_point::{FailAction, FailGuard}; +use tokio_postgres::error::SqlState; + +use super::calvin_multishard_fixture::{Fixture, keyed_ddl}; +use super::vshard_names::distinct_vshard_collections; +use crate::common::cluster_harness::{TestClusterNode, wait_for}; + +/// A fresh session on `node` in strict cross-shard mode, apart from the +/// harness client the DDL helpers use. +async fn strict_session(node: &TestClusterNode) -> tokio_postgres::Client { + let (client, connection) = tokio_postgres::connect( + &format!( + "host=127.0.0.1 port={} user=nodedb dbname=default", + node.pg_addr.port() + ), + tokio_postgres::NoTls, + ) + .await + .expect("connect a session to the coordinator"); + tokio::spawn(async move { + let _ = connection.await; + }); + client + .simple_query("SET cross_shard_txn = 'strict'") + .await + .expect("SET cross_shard_txn = strict"); + client +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_txn_staged_before_a_recreate_applies_nowhere() { + let (col_a, col_b) = distinct_vshard_collections("calvin_superseded_0", "calvin_superseded"); + let fx = Fixture::spawn(&[keyed_ddl(&col_a), keyed_ddl(&col_b)]).await; + fx.wait_group_mounted(&col_a).await; + fx.wait_group_mounted(&col_b).await; + + let txn = format!( + "BEGIN; \ + INSERT INTO {col_a} (id, v) VALUES ('k1', 'a'); \ + INSERT INTO {col_b} (id, v) VALUES ('k2', 'b'); \ + COMMIT" + ); + + let gate_dir = tempfile::tempdir().expect("gate directory"); + let release = gate_dir.path().join("release"); + let parked = gate_dir.path().join("release.parked"); + let session = strict_session(fx.coordinator()).await; + let refused = { + let _gate = FailGuard::install( + &format!("calvin::after_stamp::{col_a}"), + FailAction::WaitForFile(release.clone()), + ); + let commit = session.simple_query(&txn); + let recreate = async { + wait_for( + "the transaction parked after its incarnation stamp", + Duration::from_secs(15), + Duration::from_millis(20), + || parked.exists(), + ) + .await; + for ddl in [format!("DROP COLLECTION {col_a} PURGE"), keyed_ddl(&col_a)] { + fx.cluster + .exec_ddl_on_any_leader(&ddl) + .await + .unwrap_or_else(|e| panic!("{ddl}: {e}")); + } + std::fs::write(&release, b"release").expect("release the parked transaction"); + }; + let (refused, ()) = tokio::join!(commit, recreate); + refused + }; + + let error = refused.expect_err("a transaction staged before the recreate must not commit"); + assert_eq!( + error.code(), + Some(&SqlState::T_R_SERIALIZATION_FAILURE), + "the client retries a superseded transaction: {error:?}" + ); + + fx.converge().await; + for coll in [&col_a, &col_b] { + fx.wait_rows_on_every_node(&format!("SELECT id FROM {coll}"), 0) + .await; + } + + let retry = strict_session(fx.coordinator()).await; + retry + .simple_query(&txn) + .await + .expect("the retried transaction commits against the recreated collection"); + fx.converge().await; + for coll in [&col_a, &col_b] { + fx.wait_rows_on_every_node(&format!("SELECT id FROM {coll}"), 1) + .await; + } + + fx.cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/catalog_put_if_absent.rs b/nodedb-cluster-tests/tests/common_suite/cases/catalog_put_if_absent.rs index b46582761..eac6d91ca 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/catalog_put_if_absent.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/catalog_put_if_absent.rs @@ -10,7 +10,7 @@ //! locally-authored definition. //! //! No SQL DDL emits this variant yet, so the tests propose the entry -//! directly through `metadata_proposer::propose_catalog_entry` (which +//! directly through `metadata_proposer::propose_catalog_entry_async` (which //! forwards to the metadata-group leader) and assert idempotency + //! no-clobber by reading the replicated record on every node. @@ -21,7 +21,7 @@ use std::time::Duration; use common::cluster_harness::{TestCluster, wait_for}; use nodedb::control::catalog_entry::CatalogEntry; -use nodedb::control::metadata_proposer::propose_catalog_entry; +use nodedb::control::metadata_proposer::propose_catalog_entry_async; use nodedb::control::security::catalog::StoredCollection; use nodedb_types::DatabaseId; @@ -47,11 +47,11 @@ fn coll_fields( /// Propose a `PutCollectionIfAbsent` for `coll`, trying each node /// until one accepts (the proposer forwards to the metadata leader, /// so any node works — the loop mirrors `exec_ddl_on_any_leader`). -fn propose_if_absent(cluster: &TestCluster, coll: StoredCollection) -> Result<(), String> { +async fn propose_if_absent(cluster: &TestCluster, coll: StoredCollection) -> Result<(), String> { let entry = CatalogEntry::PutCollectionIfAbsent(Box::new(coll)); let mut last_err = String::new(); for node in &cluster.nodes { - match propose_catalog_entry(&node.shared, &entry) { + match propose_catalog_entry_async(&node.shared, &entry).await { Ok(_) => return Ok(()), Err(e) => last_err = e.to_string(), } @@ -71,7 +71,9 @@ async fn put_if_absent_creates_then_no_clobbers_then_idempotent() { a.declared_primary_key = Some("a_key".to_string()); // 1. Create via PutCollectionIfAbsent (collection is absent). - propose_if_absent(&cluster, a.clone()).expect("propose A"); + propose_if_absent(&cluster, a.clone()) + .await + .expect("propose A"); // Assert A materialized on all three nodes with A's fields. wait_for( @@ -92,7 +94,7 @@ async fn put_if_absent_creates_then_no_clobbers_then_idempotent() { let mut b = StoredCollection::new(TENANT, COLL, "tester"); b.bitemporal = false; b.declared_primary_key = Some("b_key".to_string()); - propose_if_absent(&cluster, b).expect("propose B"); + propose_if_absent(&cluster, b).await.expect("propose B"); // Wait for B's proposal to have applied cluster-wide, then assert // every node STILL shows A's fields — B was skipped, no clobber. @@ -119,7 +121,7 @@ async fn put_if_absent_creates_then_no_clobbers_then_idempotent() { // 3. Re-propose A verbatim — idempotent no-op. Still exactly one // collection, unchanged. - propose_if_absent(&cluster, a).expect("re-propose A"); + propose_if_absent(&cluster, a).await.expect("re-propose A"); wait_for( "idempotent: still exactly one collection with A's fields", Duration::from_secs(10), diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cdc_consume_across_leader_change.rs b/nodedb-cluster-tests/tests/common_suite/cases/cdc_consume_across_leader_change.rs new file mode 100644 index 000000000..dbe0ed734 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cdc_consume_across_leader_change.rs @@ -0,0 +1,319 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A change-stream consumer keeps its place across a data-group leader +//! change. +//! +//! Every replica routes a committed write's change events into its own +//! buffer at the entry's Raft log position, and `COMMIT OFFSET` is a +//! replicated catalog entry. So: +//! +//! - every replica serves the same events at the same positions; +//! - a consumer that reads part of the stream from the leader, commits, and +//! resumes on another node after the leader dies receives every event +//! exactly once, in order. +//! +//! A position taken from a node-local WAL LSN, or an offset kept only on the +//! committing node, fails the resume: the survivor compares the cursor +//! against its own positions and skips or re-reads events. + +use crate::common; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use std::collections::{BTreeMap, BTreeSet}; +use std::time::{Duration, Instant}; + +use nodedb::event::cdc::CdcOffset; +use nodedb::event::cdc::consume::{ConsumeError, ConsumeParams, consume_local}; +use nodedb_types::DatabaseId; + +const COLLECTION: &str = "cdc_failover"; +const STREAM: &str = "cdc_failover_feed"; +const GROUP: &str = "cdc_failover_readers"; +const TENANT: u64 = 1; + +/// Rows written before the leader dies. +const BEFORE: usize = 6; +/// Rows written after the leader dies. +const AFTER: usize = 3; +/// Events the consumer reads, from one partition, and commits before the +/// leader dies. +const FIRST_READ: usize = 3; + +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(100); + +/// One consumed event: its partition, position, and row id. +type Seen = (u32, CdcOffset, String); + +/// The events `node` serves after the group's committed offsets, from one +/// partition or from all of them. +fn read_from(node: &TestClusterNode, partition: Option, limit: usize) -> Vec { + let params = ConsumeParams { + database_id: DatabaseId::DEFAULT, + tenant_id: TENANT, + stream_name: STREAM, + group_name: GROUP, + partition, + limit, + }; + match consume_local(&node.shared, ¶ms) { + Ok(result) => result + .events + .iter() + .map(|event| (event.partition, event.position(), event.row_id.clone())) + .collect(), + Err(ConsumeError::BufferEmpty(_)) => Vec::new(), + Err(error) => panic!("node {}: consume failed: {error}", node.node_id), + } +} + +/// Every partition's events `node` serves after the committed offsets. +fn read(node: &TestClusterNode) -> Vec { + read_from(node, None, 1_000) +} + +/// `events` by partition, each partition in the order it was served. +/// Replicas interleave partitions in their own arrival order, so only the +/// per-partition sequences are comparable across nodes. +fn by_partition(events: &[Seen]) -> BTreeMap> { + let mut out: BTreeMap> = BTreeMap::new(); + for (partition, position, row_id) in events { + out.entry(*partition) + .or_default() + .push((*position, row_id.clone())); + } + out +} + +/// Insert `row` through `node`, retrying while the cluster elects a leader. +async fn insert(node: &TestClusterNode, row: usize) { + let sql = format!("INSERT INTO {COLLECTION} {{ id: 'row-{row}', n: {row} }}"); + let deadline = Instant::now() + CONVERGE; + loop { + match node.client.simple_query(&sql).await { + Ok(_) => return, + Err(error) if Instant::now() < deadline => { + tracing::debug!(row, %error, "insert not accepted yet; retrying"); + tokio::time::sleep(Duration::from_millis(200)).await; + } + Err(error) => panic!("insert row-{row}: {error}"), + } + } +} + +/// The leader of the data group that owns `partition`, as `node` sees it. +fn partition_leader(node: &TestClusterNode, partition: u32) -> (u64, u64) { + let group = node + .shared + .cluster_routing + .as_ref() + .expect("cluster routing") + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(partition) + .expect("partition maps to a data group"); + let leader = node + .all_group_leaders() + .into_iter() + .find_map(|(id, leader)| (id == group).then_some(leader)) + .unwrap_or(0); + (group, leader) +} + +/// The highest position per partition in `events`. +fn tails(events: &[Seen]) -> BTreeMap { + let mut tails = BTreeMap::new(); + for (partition, position, _) in events { + let tail = tails.entry(*partition).or_insert(CdcOffset::ZERO); + if *position > *tail { + *tail = *position; + } + } + tails +} + +fn row_number(row_id: &str) -> usize { + row_id + .strip_prefix("row-") + .and_then(|n| n.parse().ok()) + .unwrap_or_else(|| panic!("unexpected row id {row_id}")) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn consumer_resumes_on_another_node_after_the_leader_dies() { + let mut cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {COLLECTION}")) + .await + .expect("create collection"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE CHANGE STREAM {STREAM} ON {COLLECTION}")) + .await + .expect("create change stream"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE CONSUMER GROUP {GROUP} ON {STREAM}")) + .await + .expect("create consumer group"); + wait_for( + "every node registers the stream and the group", + CONVERGE, + STEP, + || { + cluster.nodes.iter().all(|node| { + node.has_change_stream(DatabaseId::DEFAULT, TENANT, STREAM) + && node + .shared + .group_registry + .get(DatabaseId::DEFAULT, TENANT, STREAM, GROUP) + .is_some() + }) + }, + ) + .await; + + for row in 0..BEFORE { + insert(&cluster.nodes[0], row).await; + } + cluster.wait_for_full_apply_convergence(CONVERGE).await; + wait_for("every replica buffers every event", CONVERGE, STEP, || { + cluster.nodes.iter().all(|node| read(node).len() == BEFORE) + }) + .await; + + // A replica-served consume returns the same events, at the same + // positions, as every other replica, the leader included. + let reference = by_partition(&read(&cluster.nodes[0])); + for node in &cluster.nodes { + assert_eq!( + by_partition(&read(node)), + reference, + "node {} serves a different change sequence", + node.node_id + ); + } + + // Consume part of one partition from the leader of its data group. + let (&partition, expected) = reference.iter().next().expect("one partition"); + let (group, leader) = partition_leader(&cluster.nodes[0], partition); + let leader_idx = cluster + .nodes + .iter() + .position(|node| node.node_id == leader) + .unwrap_or_else(|| panic!("no live node leads data group {group}")); + let first = read_from(&cluster.nodes[leader_idx], Some(partition), FIRST_READ); + let expected_first: Vec<(CdcOffset, String)> = + expected.iter().take(FIRST_READ).cloned().collect(); + assert_eq!( + by_partition(&first).remove(&partition).unwrap_or_default(), + expected_first + ); + + for (partition, offset) in tails(&first) { + cluster.nodes[leader_idx] + .client + .simple_query(&format!( + "COMMIT OFFSET PARTITION {partition} AT {offset} ON {STREAM} CONSUMER GROUP {GROUP}" + )) + .await + .unwrap_or_else(|e| panic!("commit offset {offset} on partition {partition}: {e}")); + } + let committed = tails(&first); + wait_for( + "every node holds the committed offsets", + CONVERGE, + STEP, + || { + cluster.nodes.iter().all(|node| { + committed.iter().all(|(partition, offset)| { + node.shared.offset_store.get_offset( + DatabaseId::DEFAULT, + TENANT, + STREAM, + GROUP, + *partition, + ) == *offset + }) + }) + }, + ) + .await; + + // Kill the leader the consumer read from. + let dead = cluster.nodes.remove(leader_idx); + let dead_id = dead.node_id; + dead.shutdown().await; + wait_for( + "the survivors elect a new leader for the data group", + CONVERGE, + STEP, + || { + cluster.nodes.iter().all(|node| { + node.all_group_leaders() + .into_iter() + .any(|(id, leader)| id == group && leader != 0 && leader != dead_id) + }) + }, + ) + .await; + + for row in BEFORE..BEFORE + AFTER { + insert(&cluster.nodes[0], row).await; + } + let remaining = BEFORE - first.len() + AFTER; + wait_for( + "every survivor buffers the new events", + CONVERGE, + STEP, + || { + cluster + .nodes + .iter() + .all(|node| read(node).len() == remaining) + }, + ) + .await; + + // Resume on a survivor. Both survivors serve the same continuation. + let resumed = read(&cluster.nodes[0]); + for node in &cluster.nodes { + assert_eq!( + by_partition(&read(node)), + by_partition(&resumed), + "survivor {} resumes with a different sequence", + node.node_id + ); + } + + // Every row arrives exactly once across the two reads. + let delivered: Vec<&Seen> = first.iter().chain(resumed.iter()).collect(); + let rows: Vec = delivered.iter().map(|(_, _, id)| row_number(id)).collect(); + let distinct: BTreeSet = rows.iter().copied().collect(); + assert_eq!( + rows.len(), + BEFORE + AFTER, + "duplicate or missing events: {rows:?}" + ); + assert_eq!( + distinct, + (0..BEFORE + AFTER).collect::>(), + "missing rows: {rows:?}" + ); + + // In order: within each partition, positions rise with the insert order. + let mut last: BTreeMap = BTreeMap::new(); + for (partition, position, id) in delivered { + let row = row_number(id); + if let Some((previous, previous_row)) = last.get(partition) { + assert!( + position > previous && row > *previous_row, + "partition {partition}: row-{row} at {position} follows row-{previous_row} at {previous}" + ); + } + last.insert(*partition, (*position, row)); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/clone_materialize_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/clone_materialize_cross_node.rs new file mode 100644 index 000000000..0fb899e14 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/clone_materialize_cross_node.rs @@ -0,0 +1,211 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! `ALTER DATABASE ... MATERIALIZE` across a 3-node cluster. +//! +//! The materialization runs on one node. Afterwards every node's catalog +//! holds the clone's collections as `Materialized` with no clone origin, and +//! the clone answers from its own storage on every node once its source is +//! dropped. + +use crate::common; + +use std::time::Duration; + +use common::cluster_harness::shared_steps::{database_id, use_database}; +use common::cluster_harness::{TestCluster, TestClusterNode}; +use nodedb_types::{CloneStatus, DatabaseId}; + +const SOURCE: &str = "cm_src"; +const CLONE: &str = "cm_clone"; +const ROWS: usize = 5; + +/// Every collection of `db` on `node` that is not a finished materialization. +fn unmaterialized(node: &TestClusterNode, db: DatabaseId) -> Vec { + node.shared + .credentials + .catalog() + .load_all_collections(db) + .expect("load collections") + .into_iter() + .filter(|c| c.cloned_from.is_some() || c.clone_status != CloneStatus::Materialized) + .map(|c| c.name) + .collect() +} + +async fn row_count(node: &TestClusterNode, sql: &str) -> usize { + node.client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql} on node {}: {e}", node.node_id)) + .iter() + .filter(|m| matches!(m, tokio_postgres::SimpleQueryMessage::Row(_))) + .count() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn materialize_on_one_node_matches_every_catalog_and_replica() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + + cluster + .exec_ddl_on_any_leader(&format!("CREATE DATABASE {SOURCE}")) + .await + .unwrap_or_else(|e| panic!("CREATE DATABASE: {e}")); + use_database(&cluster, SOURCE).await; + cluster + .exec_ddl_on_any_leader( + "CREATE COLLECTION cm_records (key STRING PRIMARY KEY, data STRING) \ + WITH (engine='kv')", + ) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION: {e}")); + for i in 0..ROWS { + cluster.nodes[0] + .exec(&format!( + "INSERT INTO cm_records (key, data) VALUES ('k{i}', 'v{i}')" + )) + .await + .unwrap_or_else(|e| panic!("INSERT k{i}: {e}")); + } + use_database(&cluster, "default").await; + + cluster + .exec_ddl_on_any_leader(&format!("CLONE DATABASE {CLONE} FROM {SOURCE}")) + .await + .unwrap_or_else(|e| panic!("CLONE DATABASE: {e}")); + cluster + .exec_ddl_on_any_leader(&format!("ALTER DATABASE {CLONE} MATERIALIZE")) + .await + .unwrap_or_else(|e| panic!("ALTER DATABASE MATERIALIZE: {e}")); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + + for node in &cluster.nodes { + let clone_id = database_id(node, CLONE); + let pending = unmaterialized(node, clone_id); + assert!( + pending.is_empty(), + "node {} still holds unmaterialized clone collections: {pending:?}", + node.node_id + ); + } + + // With the source gone, only the clone's own replicated storage answers. + cluster + .exec_ddl_on_any_leader(&format!("DROP DATABASE {SOURCE} CASCADE")) + .await + .unwrap_or_else(|e| panic!("DROP DATABASE source: {e}")); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + use_database(&cluster, CLONE).await; + for node in &cluster.nodes { + assert_eq!( + row_count(node, "SELECT key FROM cm_records").await, + ROWS, + "node {} must read every materialized row from the clone", + node.node_id + ); + } + + cluster.shutdown().await; +} + +/// Every `(key, value)` row `sql` returns on `node`, sorted. +async fn rows(node: &TestClusterNode, sql: &str) -> Vec<(String, String)> { + let mut out: Vec<(String, String)> = node + .client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql} on node {}: {e}", node.node_id)) + .into_iter() + .filter_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(row) => Some(( + row.get(0).unwrap_or_default().to_owned(), + row.get(1).unwrap_or_default().to_owned(), + )), + _ => None, + }) + .collect(); + out.sort(); + out +} + +/// An UPDATE on a shadowed clone copies the source row up through one node. +/// Every node then holds the copy-up mapping, and every node reads the +/// updated row once, never next to the stale source copy. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn copy_up_through_one_node_is_readable_on_every_node() { + const SRC: &str = "cu_src"; + const CLN: &str = "cu_clone"; + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + + cluster + .exec_ddl_on_any_leader(&format!("CREATE DATABASE {SRC}")) + .await + .unwrap_or_else(|e| panic!("CREATE DATABASE: {e}")); + use_database(&cluster, SRC).await; + cluster + .exec_ddl_on_any_leader( + "CREATE COLLECTION cu_docs (id TEXT PRIMARY KEY, content TEXT) \ + WITH (engine='document_strict')", + ) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION: {e}")); + for i in 0..3 { + cluster.nodes[0] + .exec(&format!( + "INSERT INTO cu_docs (id, content) VALUES ('d{i}', 'old{i}')" + )) + .await + .unwrap_or_else(|e| panic!("INSERT d{i}: {e}")); + } + use_database(&cluster, "default").await; + cluster + .exec_ddl_on_any_leader(&format!("CLONE DATABASE {CLN} FROM {SRC}")) + .await + .unwrap_or_else(|e| panic!("CLONE DATABASE: {e}")); + use_database(&cluster, CLN).await; + + cluster.nodes[1] + .exec("UPDATE cu_docs SET content = 'new1' WHERE id = 'd1'") + .await + .unwrap_or_else(|e| panic!("UPDATE on the clone: {e}")); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + + let expected = vec![ + ("d0".to_string(), "old0".to_string()), + ("d1".to_string(), "new1".to_string()), + ("d2".to_string(), "old2".to_string()), + ]; + for node in &cluster.nodes { + let clone_id = database_id(node, CLN); + let key = + nodedb::control::planner::sql_plan_convert::convert::db_qualified(clone_id, "cu_docs"); + let mappings = node + .shared + .credentials + .catalog() + .list_clone_copyups(&key) + .expect("list copy-ups"); + assert_eq!( + mappings.len(), + 1, + "node {} must hold the replicated copy-up mapping", + node.node_id + ); + assert_eq!( + rows(node, "SELECT id, content FROM cu_docs").await, + expected, + "node {} must read the copied-up row once, updated", + node.node_id + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_array.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_array.rs index 5cdb27e39..af2c15c38 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_array.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_array.rs @@ -6,18 +6,16 @@ //! therefore runs in the `cluster` test group (max-threads = 1, //! threads-required = num-test-threads). They run strictly serially and alone. //! -//! ## Architecture Note: Local Array Catalog +//! ## Replicated array catalog //! -//! `CREATE ARRAY` writes to the local in-memory `ArrayCatalog` on the -//! executing node only — it is NOT replicated through Raft (unlike -//! `CREATE COLLECTION`). As a result, all array queries (ARRAY_SLICE, -//! ARRAY_AGG, etc.) must be issued on the same node that executed the -//! `CREATE ARRAY` DDL. +//! `CREATE ARRAY` proposes a `PutArray` catalog entry through the metadata +//! group, so every node holds the array. The fixture creates the array on +//! one node, writes its cells through another, and every test queries from +//! every node. //! -//! The "distributed" aspect of these tests is that cell data is stored -//! across multiple vShards on different nodes (Hilbert-partitioned). The -//! coordinator on the DDL node fans out to peer shards via the array RPC -//! path and merges the results. +//! Cells are stored across multiple vShards on different nodes +//! (Hilbert-partitioned). The coordinator on the querying node fans out to +//! peer shards via the array RPC path and merges the results. //! //! Tests: //! 1. `cluster_array_slice_spans_multiple_shards` — ARRAY_SLICE fan-out @@ -27,9 +25,6 @@ //! correct per-group sums. //! 4. `cluster_array_vector_prefilter_distributed` — fused vector+slice //! query wires end-to-end without error. -//! 5. `cluster_array_routing_retry_on_owner_change` — stale routing table -//! on the coordinator node (poisoned via `force_stale_route_for_test`) -//! recovers and the array query succeeds on retry. use crate::common; @@ -75,10 +70,8 @@ async fn query_named_rows( /// Spin up a 3-node cluster with a pre-populated genome array. /// -/// Returns `(cluster, leader_idx)` where `leader_idx` is the index into -/// `cluster.nodes` of the node that executed the `CREATE ARRAY` DDL. All -/// array queries must be issued on this node because the array catalog is -/// local (not replicated through Raft). +/// The array is created on one node and its cells are written and flushed +/// through another, so the fixture itself depends on the replicated catalog. /// /// Schema: /// DIMS (chr INT64 [0..9], pos INT64 [0..99]) @@ -91,12 +84,12 @@ async fn query_named_rows( /// chr=1: pos=10/20/30, qual=10.0/20.0/30.0 → sum=60.0 /// chr=2: pos=10/20/30, qual=100.0/200.0/300.0 → sum=600.0 /// total qual sum: 666.0 -async fn spawn_cluster_with_genome() -> (TestCluster, usize) { +async fn spawn_cluster_with_genome() -> TestCluster { let cluster = TestCluster::spawn_three() .await .expect("3-node cluster spawn"); - let leader_idx = cluster + let ddl_idx = cluster .exec_ddl_on_any_leader( "CREATE ARRAY genome \ DIMS (chr INT64 [0..9], pos INT64 [0..99]) \ @@ -107,8 +100,9 @@ async fn spawn_cluster_with_genome() -> (TestCluster, usize) { .await .expect("CREATE ARRAY genome"); - // Insert 9 cells from the DDL node (the one with the local array catalog). - cluster.nodes[leader_idx] + // Write through a node that did not run the DDL. + let writer_idx = (ddl_idx + 1) % cluster.nodes.len(); + cluster.nodes[writer_idx] .exec( "INSERT INTO ARRAY genome \ COORDS (0, 10) VALUES (1.0), \ @@ -125,21 +119,29 @@ async fn spawn_cluster_with_genome() -> (TestCluster, usize) { .expect("INSERT INTO ARRAY genome"); // Flush so reads exercise the segment-scan path, not just the memtable. - cluster.nodes[leader_idx] + cluster.nodes[writer_idx] .exec("SELECT ARRAY_FLUSH('genome')") .await .expect("ARRAY_FLUSH"); + cluster + .wait_for_full_apply_convergence(std::time::Duration::from_secs(10)) + .await; - (cluster, leader_idx) + cluster } // ── Test 1: slice fan-out to peer shards ───────────────────────────────────── #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn cluster_array_slice_spans_multiple_shards() { - let (cluster, node_idx) = spawn_cluster_with_genome().await; - let client = &cluster.nodes[node_idx].client; + let cluster = spawn_cluster_with_genome().await; + for (node_idx, node) in cluster.nodes.iter().enumerate() { + assert_chr1_slice(&node.client, node_idx).await; + } + cluster.shutdown().await; +} +async fn assert_chr1_slice(client: &tokio_postgres::Client, node_idx: usize) { // Slice chr=1, pos 0..99 — should return exactly the 3 cells for chr=1. // ARRAY_SLICE projects one pgwire field per declared column: a // `coords` JSON-array column followed by an `attrs` JSON-array column @@ -153,7 +155,7 @@ async fn cluster_array_slice_spans_multiple_shards() { assert_eq!( rows.len(), 3, - "expected 3 cells for chr=1, pos 0..99; got {rows:?}" + "node {node_idx}: expected 3 cells for chr=1, pos 0..99; got {rows:?}" ); let mut quals: Vec = rows @@ -175,22 +177,29 @@ async fn cluster_array_slice_spans_multiple_shards() { assert_eq!( quals, vec![10.0, 20.0, 30.0], - "chr=1 qual values must be [10.0, 20.0, 30.0]; got {quals:?}" + "node {node_idx}: chr=1 qual values must be [10.0, 20.0, 30.0]; got {quals:?}" ); - - cluster.shutdown().await; } // ── Test 2: agg sum across all shards ──────────────────────────────────────── #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn cluster_array_agg_sum_across_shards() { - let (cluster, node_idx) = spawn_cluster_with_genome().await; - let client = &cluster.nodes[node_idx].client; + let cluster = spawn_cluster_with_genome().await; + for (node_idx, node) in cluster.nodes.iter().enumerate() { + assert_total_sum(&node.client, node_idx).await; + } + cluster.shutdown().await; +} +async fn assert_total_sum(client: &tokio_postgres::Client, node_idx: usize) { let rows = query_named_rows(client, "SELECT * FROM ARRAY_AGG('genome', 'qual', 'sum')").await; - assert_eq!(rows.len(), 1, "scalar agg must return exactly one row"); + assert_eq!( + rows.len(), + 1, + "node {node_idx}: scalar agg must return one row" + ); // ARRAY_AGG projects a `result` column carrying the aggregate value. let result_text = rows[0] @@ -202,19 +211,22 @@ async fn cluster_array_agg_sum_across_shards() { assert!( (result - 666.0).abs() < 1e-4, - "expected sum=666.0, got {result}" + "node {node_idx}: expected sum=666.0, got {result}" ); - - cluster.shutdown().await; } // ── Test 3: agg grouped by chr dimension ───────────────────────────────────── #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn cluster_array_agg_grouped_by_chr() { - let (cluster, node_idx) = spawn_cluster_with_genome().await; - let client = &cluster.nodes[node_idx].client; + let cluster = spawn_cluster_with_genome().await; + for (node_idx, node) in cluster.nodes.iter().enumerate() { + assert_grouped_sums(&node.client, node_idx).await; + } + cluster.shutdown().await; +} +async fn assert_grouped_sums(client: &tokio_postgres::Client, node_idx: usize) { let rows = query_named_rows( client, "SELECT * FROM ARRAY_AGG('genome', 'qual', 'sum', 'chr')", @@ -224,7 +236,7 @@ async fn cluster_array_agg_grouped_by_chr() { assert_eq!( rows.len(), 3, - "group-by-chr must return 3 rows (one per chromosome); got {rows:?}" + "node {node_idx}: group-by-chr must return 3 rows; got {rows:?}" ); // Group-by-key projects two columns: `group` (the dimension value) @@ -267,15 +279,13 @@ async fn cluster_array_agg_grouped_by_chr() { "chr=2 sum must be 600.0, got {}", groups[2].1 ); - - cluster.shutdown().await; } // ── Test 4: vector prefilter fused with distributed slice ───────────────────── #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn cluster_array_vector_prefilter_distributed() { - let (cluster, node_idx) = spawn_cluster_with_genome().await; + let cluster = spawn_cluster_with_genome().await; // Create a document collection and vector index to back the fused query. // DDL must be issued and accepted on any leader. @@ -290,32 +300,29 @@ async fn cluster_array_vector_prefilter_distributed() { .await .expect("CREATE VECTOR INDEX"); - // The fused query: ORDER BY vector_distance + JOIN ARRAY_SLICE. - // Issued on the DDL node because the array catalog is local there. - // The vector index is empty — the assertion is that the query wires - // through every distributed layer without error: + // The fused query: ORDER BY vector_distance + JOIN ARRAY_SLICE, issued + // on every node. The vector index is empty — the assertion is that the + // query wires through every distributed layer without error: // planner fusion → convert → ArrayOp::SurrogateBitmapScan (distributed // fan-out to peer shards) + VectorOp::Search with inline_prefilter_plan. - let result = cluster.nodes[node_idx] - .client - .simple_query( - "SELECT id FROM genes \ - JOIN ARRAY_SLICE('genome', '{chr: [1, 1], pos: [0, 99]}') AS s \ - ON id = s.qual \ - ORDER BY vector_distance(embedding, [1.0, 0.0, 0.0]) \ - LIMIT 10", - ) - .await; - - match result { - Ok(_) => {} - Err(e) => { + for (node_idx, node) in cluster.nodes.iter().enumerate() { + let result = node + .client + .simple_query( + "SELECT id FROM genes \ + JOIN ARRAY_SLICE('genome', '{chr: [1, 1], pos: [0, 99]}') AS s \ + ON id = s.qual \ + ORDER BY vector_distance(embedding, [1.0, 0.0, 0.0]) \ + LIMIT 10", + ) + .await; + if let Err(e) = result { let msg = format!("{e}"); // Tolerate empty-index / no-rows errors. Codec panics and planner // errors indicate the fused distributed path was never attempted. assert!( !msg.contains("codec") && !msg.contains("panic") && !msg.contains("plan error"), - "unexpected error from fused distributed array+vector query: {msg}" + "node {node_idx}: unexpected error from fused array+vector query: {msg}" ); } } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_body_read.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_body_read.rs new file mode 100644 index 000000000..1e5de5ea0 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_body_read.rs @@ -0,0 +1,338 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A stored-procedure body in a cluster writes an array and reads it back in +//! the same body. +//! +//! The body runs as one system transaction. Its `INSERT INTO ARRAY` stages a +//! per-shard write. Its `ARRAY_AGG` and `ARRAY_SLICE` are cluster array +//! reads, which the transaction dispatches through this node's array +//! coordinator, carrying the transaction id so each shard folds in the +//! transaction's staged cells. A cluster array read has no Data-Plane +//! handler: dispatched to a core, it panics there and fails the statement. +//! +//! The test checks: +//! - `CALL` succeeds, so each body read ran through the coordinator; +//! - no node fail-stopped a core; +//! - both writes commit with the body, and every node reads them. +//! +//! A body statement's rows are discarded and a procedural expression cannot +//! query, so no body can report what its read returned. The body read asks +//! for the same staged-cell fold a client transaction's cluster array read +//! uses, which the second test checks: inside `BEGIN`, `ARRAY_AGG` and +//! `ARRAY_SLICE` see the transaction's staged cells on every shard they +//! span, another connection sees none of them, and after `COMMIT` both see +//! all of them. + +use crate::common; + +use common::cluster_harness::shared_steps::fail_stopped; +use common::cluster_harness::{TestCluster, TestClusterNode}; + +const CREATE_ARRAY: &str = "CREATE ARRAY bodygrid \ + DIMS (x INT64 [0..63], y INT64 [0..63]) \ + ATTRS (v FLOAT64) \ + TILE_EXTENTS (8, 8) \ + CELL_ORDER HILBERT"; + +/// Writes a cell, reads the array twice with that cell staged, then writes a +/// cell on a distant tile. +const CREATE_PROCEDURE: &str = "CREATE PROCEDURE fill_and_read_bodygrid() AS \ + BEGIN \ + INSERT INTO ARRAY bodygrid COORDS (1, 1) VALUES (1.5); \ + SELECT * FROM ARRAY_AGG('bodygrid', 'v', 'sum'); \ + SELECT * FROM ARRAY_SLICE('bodygrid', '{\"x\":[0,63],\"y\":[0,63]}', '*', 100); \ + INSERT INTO ARRAY bodygrid COORDS (60, 60) VALUES (2.5); \ + END"; + +const EXPECTED_SUM: f64 = 4.0; + +/// Every row `sql` returns on `client`. +async fn rows_of( + client: &tokio_postgres::Client, + sql: &str, +) -> Vec { + client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e:?}")) + .into_iter() + .filter_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => Some(r), + _ => None, + }) + .collect() +} + +/// The sum `ARRAY_AGG` reports over `array`'s `v` attribute on `client`. +/// `None` when the aggregate has no row or an empty result, as it has over +/// no cells. +async fn agg_sum(client: &tokio_postgres::Client, array: &str) -> Option { + let rows = rows_of( + client, + &format!("SELECT * FROM ARRAY_AGG('{array}', 'v', 'sum')"), + ) + .await; + assert!( + rows.len() <= 1, + "a scalar ARRAY_AGG returns at most one row" + ); + let text = rows.first()?.get("result")?; + if text.is_empty() { + return None; + } + Some( + text.parse() + .unwrap_or_else(|e| panic!("ARRAY_AGG result {text} is not a float: {e}")), + ) +} + +/// How many cells an `ARRAY_SLICE` over the whole of `array` returns on +/// `client`. +async fn slice_count(client: &tokio_postgres::Client, array: &str) -> usize { + rows_of( + client, + &format!("SELECT * FROM ARRAY_SLICE('{array}', '{{\"x\":[0,63],\"y\":[0,63]}}', '*', 100)"), + ) + .await + .len() +} + +/// The `result` column of a one-row `ARRAY_AGG` sum over pgwire. +async fn pgwire_sum(node: &TestClusterNode) -> f64 { + agg_sum(&node.client, "bodygrid") + .await + .unwrap_or_else(|| panic!("node {}: ARRAY_AGG over bodygrid is empty", node.node_id)) +} + +/// A second pgwire connection to `node`, as the harness superuser. +async fn second_connection(node: &TestClusterNode) -> tokio_postgres::Client { + let (client, connection) = tokio_postgres::connect( + &format!( + "host=127.0.0.1 port={} user=nodedb dbname=default", + node.pg_addr.port() + ), + tokio_postgres::NoTls, + ) + .await + .unwrap_or_else(|e| panic!("second connection to node {}: {e:?}", node.node_id)); + tokio::spawn(async move { + let _ = connection.await; + }); + client +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_procedure_body_reads_the_array_it_writes() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(CREATE_ARRAY) + .await + .expect("CREATE ARRAY bodygrid"); + let caller_idx = cluster + .exec_ddl_on_any_leader(CREATE_PROCEDURE) + .await + .expect("CREATE PROCEDURE fill_and_read_bodygrid"); + let caller = &cluster.nodes[caller_idx]; + + caller + .client + .simple_query("CALL fill_and_read_bodygrid()") + .await + .unwrap_or_else(|e| { + panic!( + "CALL on node {}: a body's cluster array read must run through \ + the array coordinator: {e:?}", + caller.node_id + ) + }); + + for node in &cluster.nodes { + assert!( + !fail_stopped(node), + "node {} fail-stopped a core", + node.node_id + ); + } + + cluster + .wait_for_full_apply_convergence(std::time::Duration::from_secs(10)) + .await; + for node in &cluster.nodes { + let sum = pgwire_sum(node).await; + assert!( + (sum - EXPECTED_SUM).abs() < 1e-9, + "node {} reads both cells the body committed: sum {sum}", + node.node_id + ); + } + + cluster.shutdown().await; +} + +/// A cross-shard Calvin transaction that writes array cells on distant tiles +/// and a document row commits, and every node reads the committed cells. +/// +/// Each staged cell write homes to the vShard its tile hashes to, and the +/// transaction routes and locks it there. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_calvin_transaction_commits_array_cells_on_their_tiles() { + const ARRAY: &str = "calvingrid"; + const SIDE: &str = "calvingrid_side"; + const COMMITTED_SUM: f64 = 9.0; + const COMMITTED_CELLS: usize = 3; + + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + let ddl_idx = cluster + .exec_ddl_on_any_leader(&CREATE_ARRAY.replace("bodygrid", ARRAY)) + .await + .expect("CREATE ARRAY calvingrid"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {SIDE}")) + .await + .expect("CREATE COLLECTION calvingrid_side"); + let node = &cluster.nodes[(ddl_idx + 1) % cluster.nodes.len()]; + let client: &tokio_postgres::Client = &node.client; + + // The side collection's first write retries until this node serves it. + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(30); + loop { + match client + .simple_query(&format!("INSERT INTO {SIDE} (id, v) VALUES ('seed', 'x')")) + .await + { + Ok(_) => break, + Err(e) if std::time::Instant::now() < deadline => { + tracing::debug!(error = %e, "side collection not served yet; retrying"); + tokio::time::sleep(std::time::Duration::from_millis(200)).await; + } + Err(e) => panic!("seed the side collection: {e:?}"), + } + } + + client + .simple_query("SET cross_shard_txn = 'strict'") + .await + .expect("SET cross_shard_txn"); + client.simple_query("BEGIN").await.expect("BEGIN"); + // Cells on distant tiles, so they stage on several shards. + client + .simple_query(&format!( + "INSERT INTO ARRAY {ARRAY} \ + COORDS (1, 1) VALUES (2.0), \ + COORDS (40, 3) VALUES (3.0), \ + COORDS (60, 60) VALUES (4.0)" + )) + .await + .unwrap_or_else(|e| panic!("staged INSERT INTO ARRAY: {e:?}")); + client + .simple_query(&format!("INSERT INTO {SIDE} (id, v) VALUES ('s1', 'side')")) + .await + .unwrap_or_else(|e| panic!("staged side insert: {e:?}")); + client + .simple_query("COMMIT") + .await + .unwrap_or_else(|e| panic!("COMMIT of the array transaction: {e:?}")); + + cluster + .wait_for_full_apply_convergence(std::time::Duration::from_secs(10)) + .await; + for reader in &cluster.nodes { + assert!( + !fail_stopped(reader), + "node {} fail-stopped a core", + reader.node_id + ); + assert_eq!( + agg_sum(&reader.client, ARRAY).await, + Some(COMMITTED_SUM), + "node {}: ARRAY_AGG sees every committed cell", + reader.node_id + ); + assert_eq!( + slice_count(&reader.client, ARRAY).await, + COMMITTED_CELLS, + "node {}: ARRAY_SLICE returns every committed cell", + reader.node_id + ); + } + + cluster.shutdown().await; +} + +/// A client transaction's cluster array reads see its own staged cells, on +/// every shard the cells span. Another connection sees none of them until +/// COMMIT, and then both see all of them. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_client_transaction_reads_its_own_staged_array_cells() { + const ARRAY: &str = "txngrid"; + const STAGED_SUM: f64 = 7.0; + const STAGED_CELLS: usize = 3; + + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + let ddl_idx = cluster + .exec_ddl_on_any_leader(&CREATE_ARRAY.replace("bodygrid", ARRAY)) + .await + .expect("CREATE ARRAY txngrid"); + let node = &cluster.nodes[(ddl_idx + 1) % cluster.nodes.len()]; + let txn: &tokio_postgres::Client = &node.client; + let other = second_connection(node).await; + + txn.simple_query("BEGIN").await.expect("BEGIN"); + // Cells on distant tiles, so they stage on several shards. + txn.simple_query(&format!( + "INSERT INTO ARRAY {ARRAY} \ + COORDS (1, 1) VALUES (1.5), \ + COORDS (40, 3) VALUES (2.5), \ + COORDS (60, 60) VALUES (3.0)" + )) + .await + .unwrap_or_else(|e| panic!("staged INSERT INTO ARRAY: {e:?}")); + + assert_eq!( + agg_sum(txn, ARRAY).await, + Some(STAGED_SUM), + "ARRAY_AGG inside the transaction sums its staged cells" + ); + assert_eq!( + slice_count(txn, ARRAY).await, + STAGED_CELLS, + "ARRAY_SLICE inside the transaction returns its staged cells" + ); + + let outside_sum = agg_sum(&other, ARRAY).await; + assert!( + outside_sum.is_none_or(|sum| sum == 0.0), + "another connection's ARRAY_AGG sees no staged cell: {outside_sum:?}" + ); + assert_eq!( + slice_count(&other, ARRAY).await, + 0, + "another connection's ARRAY_SLICE sees no staged cell" + ); + + txn.simple_query("COMMIT").await.expect("COMMIT"); + cluster + .wait_for_full_apply_convergence(std::time::Duration::from_secs(10)) + .await; + + for (label, client) in [("the committing", txn), ("the other", &other)] { + assert_eq!( + agg_sum(client, ARRAY).await, + Some(STAGED_SUM), + "{label} connection's ARRAY_AGG sees the committed cells" + ); + assert_eq!( + slice_count(client, ARRAY).await, + STAGED_CELLS, + "{label} connection's ARRAY_SLICE sees the committed cells" + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_cell_raft_replication.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_cell_raft_replication.rs index 134c037da..7cf464eb8 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_cell_raft_replication.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_cell_raft_replication.rs @@ -10,11 +10,9 @@ //! distributed apply loop (opening the array + dispatching to its local Data //! Plane) and binds each cell's carried surrogate to its coord tuple. //! -//! The array catalog is per-node (local `CREATE ARRAY`, not Raft-replicated), -//! and a follower's `ensure_array_open` on apply needs that catalog — so the -//! `CREATE ARRAY` DDL runs on ALL three nodes (each registers an identical -//! catalog), mirroring how `array_raft_replication.rs` registers the schema on -//! every node before driving the sync path. +//! `CREATE ARRAY` replicates through the metadata group, so the DDL runs once. +//! Every node applies the catalog entry, which a follower's +//! `ensure_array_open` needs on apply. //! //! ## What is proven //! @@ -113,14 +111,12 @@ async fn cluster_array_cell_write_replicates_to_all_replicas() { .await .expect("spawn 3-node cluster"); - // The array catalog is local-only, so register it on EVERY node — each - // follower needs it to `ensure_array_open` when applying the replicated - // cell write. - for (idx, node) in cluster.nodes.iter().enumerate() { - node.exec(CREATE_ARRAY_DDL) - .await - .unwrap_or_else(|e| panic!("CREATE ARRAY on node {idx}: {e}")); - } + // CREATE ARRAY replicates through the metadata group. Every node applies + // the catalog entry and opens the array before a cell write reaches it. + cluster + .exec_ddl_on_any_leader(CREATE_ARRAY_DDL) + .await + .unwrap_or_else(|e| panic!("CREATE ARRAY: {e}")); // Insert two cells via node 0. In a cluster this routes through the array // coordinator → per-shard owner → Raft propose to the owning data group. diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_ws_rpc.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_ws_rpc.rs new file mode 100644 index 000000000..3deabf598 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_array_ws_rpc.rs @@ -0,0 +1,130 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Array statements sent over WebSocket RPC work in a cluster. +//! +//! One node runs `CREATE ARRAY`. Another node, which ran no DDL, takes an +//! `INSERT INTO ARRAY` and an `ARRAY_AGG` through the WebSocket RPC SQL entry +//! point. That node's array coordinator fans the insert out to the shards +//! that own the cells, and gathers the aggregate back from them. A third +//! node reads the same sum over pgwire. +//! +//! Before the transport intercepted cluster array ops, the insert went to +//! the gateway, which cannot encode a `ClusterArray` plan for the wire. + +use crate::common; + +use common::cluster_harness::TestCluster; +use common::cluster_harness::TestClusterNode; +use common::cluster_harness::node::lifecycle::HARNESS_SUPERUSER; +use nodedb::control::planner::context::QueryContext; +use nodedb::control::security::identity::AuthMethod; +use nodedb::control::server::http::routes::ws_rpc::execute_sql::execute_sql; +use nodedb::types::{DatabaseId, TraceId}; + +const CREATE_ARRAY: &str = "CREATE ARRAY wsgrid \ + DIMS (x INT64 [0..63], y INT64 [0..63]) \ + ATTRS (v FLOAT64) \ + TILE_EXTENTS (8, 8) \ + CELL_ORDER HILBERT"; + +/// Cells spread across the grid, so they land on several shards. +const INSERT: &str = "INSERT INTO ARRAY wsgrid \ + COORDS (1, 1) VALUES (1.5), \ + COORDS (40, 3) VALUES (2.5), \ + COORDS (60, 60) VALUES (3.0)"; + +const EXPECTED_SUM: f64 = 7.0; + +/// Run `sql` through `node`'s WebSocket RPC SQL entry point as the harness +/// superuser. +async fn ws_rpc_sql(node: &TestClusterNode, sql: &str) -> serde_json::Value { + let identity = node + .shared + .credentials + .to_identity(HARNESS_SUPERUSER, AuthMethod::Trust) + .expect("the harness superuser exists on every node"); + let query_ctx = QueryContext::for_state_with_lease(&node.shared); + execute_sql( + &node.shared, + &query_ctx, + &identity, + DatabaseId::DEFAULT, + sql, + TraceId::generate(), + "127.0.0.1:40000", + ) + .await + .unwrap_or_else(|e| panic!("ws_rpc {sql} on node {}: {e:?}", node.node_id)) +} + +/// Whether any number anywhere in `value` equals `expected`. +fn holds_number(value: &serde_json::Value, expected: f64) -> bool { + match value { + serde_json::Value::Number(n) => n.as_f64().is_some_and(|f| (f - expected).abs() < 1e-9), + serde_json::Value::Array(items) => items.iter().any(|v| holds_number(v, expected)), + serde_json::Value::Object(map) => map.values().any(|v| holds_number(v, expected)), + serde_json::Value::Null | serde_json::Value::Bool(_) | serde_json::Value::String(_) => { + false + } + } +} + +/// The `result` column of a one-row `ARRAY_AGG` read over pgwire. +async fn pgwire_sum(node: &TestClusterNode) -> f64 { + let msgs = node + .client + .simple_query("SELECT * FROM ARRAY_AGG('wsgrid', 'v', 'sum')") + .await + .unwrap_or_else(|e| panic!("ARRAY_AGG on node {}: {e:?}", node.node_id)); + let rows: Vec = msgs + .into_iter() + .filter_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => Some(r), + _ => None, + }) + .collect(); + assert_eq!(rows.len(), 1, "a scalar ARRAY_AGG returns one row"); + let text = rows[0] + .get("result") + .unwrap_or_else(|| panic!("ARRAY_AGG row carries no result column")); + text.parse() + .unwrap_or_else(|e| panic!("ARRAY_AGG result {text} is not a float: {e}")) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn ws_rpc_array_insert_and_agg_on_a_node_that_ran_no_ddl() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + let ddl_idx = cluster + .exec_ddl_on_any_leader(CREATE_ARRAY) + .await + .expect("CREATE ARRAY wsgrid"); + let writer = &cluster.nodes[(ddl_idx + 1) % cluster.nodes.len()]; + let reader = &cluster.nodes[(ddl_idx + 2) % cluster.nodes.len()]; + + let inserted = ws_rpc_sql(writer, INSERT).await; + assert!( + holds_number(&inserted, 3.0), + "the insert reports its 3 cells: {inserted}" + ); + + cluster + .wait_for_full_apply_convergence(std::time::Duration::from_secs(10)) + .await; + + let summed = ws_rpc_sql(writer, "SELECT * FROM ARRAY_AGG('wsgrid', 'v', 'sum')").await; + assert!( + holds_number(&summed, EXPECTED_SUM), + "ws_rpc ARRAY_AGG sums every shard's cells to {EXPECTED_SUM}: {summed}" + ); + + let over_pgwire = pgwire_sum(reader).await; + assert!( + (over_pgwire - EXPECTED_SUM).abs() < 1e-9, + "pgwire on node {} reads the cells ws_rpc wrote: sum {over_pgwire}", + reader.node_id + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_remote_cut.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_remote_cut.rs index 532dd1a56..3ce94ab65 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_remote_cut.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_remote_cut.rs @@ -124,9 +124,11 @@ fn envelope_holds_kv_row(envelope: &[u8], collection: &str, key: &[u8]) -> bool }) .flat_map(|snapshot| snapshot.kv_tables) .filter(|(name, _)| *name == table_key) - .filter_map(|(_, rows)| zerompk::from_msgpack::, Vec, u64)>>(&rows).ok()) + .filter_map(|(_, rows)| { + zerompk::from_msgpack::, Vec, u64, u32)>>(&rows).ok() + }) .flatten() - .any(|(row_key, _, _)| row_key.windows(key.len()).any(|window| window == key)) + .any(|(row_key, _, _, _)| row_key.windows(key.len()).any(|window| window == key)) } async fn connect(pg_addr: std::net::SocketAddr) -> tokio_postgres::Client { diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore_databases.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore_databases.rs index eed4c9148..4579c6b38 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore_databases.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore_databases.rs @@ -17,8 +17,9 @@ use std::time::Duration; use bytes::Bytes; use futures::{SinkExt, StreamExt}; use nodedb_types::backup_envelope::{ - DEFAULT_MAX_TOTAL_BYTES, DatabaseBlob, DatabaseDataSection, SECTION_ORIGIN_CATALOG_ROWS, - SECTION_ORIGIN_DATABASES, parse_encrypted as parse_envelope, + CollectionVerification, DEFAULT_MAX_TOTAL_BYTES, DatabaseBlob, DatabaseDataSection, + SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_DATABASES, SECTION_ORIGIN_VERIFICATION, + VerifiedPart, parse_encrypted as parse_envelope, }; use crate::common; @@ -199,6 +200,42 @@ async fn three_node_backup_gathers_one_section_per_node_and_database() { ); } + // The verification section counts each collection's rows once, summed + // over the source nodes, never once per replica. + let verification: Vec = env + .sections + .iter() + .filter(|s| s.origin_node_id == SECTION_ORIGIN_VERIFICATION) + .flat_map(|s| { + zerompk::from_msgpack::>(&s.body) + .expect("decode verification") + }) + .collect(); + for (database, rows) in DATABASES { + let database_id = listed + .iter() + .find(|b| b.name == database) + .map(|b| b.database_id) + .unwrap_or_else(|| panic!("database {database} is listed")); + for (collection, part) in [ + ("cl_docs", VerifiedPart::Documents), + ("cl_kv", VerifiedPart::KeyValue), + ("cl_cols", VerifiedPart::Columnar), + ] { + let count = verification + .iter() + .find(|r| { + r.database_id == database_id && r.collection == collection && r.part == part + }) + .map(|r| r.tally.count); + assert_eq!( + count, + Some(rows as u64), + "{database}.{collection} ({part}) must record {rows} rows" + ); + } + } + cluster.shutdown().await; } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore_graph_edges.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore_graph_edges.rs new file mode 100644 index 000000000..93b4e8c69 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_backup_restore_graph_edges.rs @@ -0,0 +1,228 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Cluster BACKUP / RESTORE of graph edges when nodes outnumber the +//! replication factor. +//! +//! An edge lives on the key vShard of each endpoint, `from_key(src)` and +//! `from_key(dst)`, never on its collection's vShard. With 3 nodes and RF 2 the +//! leader of the collection's group does not hold every edge. The backup must +//! still capture each edge once, and the restore must write it to both endpoint +//! homes, so a traversal reaches it from either endpoint on any node. +//! +//! No one node holds every endpoint's PK→surrogate bind either: each endpoint +//! is bound on the group of `from_key(endpoint)`. The backup captures each +//! bind from the source node of its home, so every restored endpoint keeps its +//! source surrogate. +//! +//! The restore target is a fresh cluster with the same RF: it has no live data +//! and a zero tenant write HLC, so no restore guard fires. + +use std::collections::{BTreeMap, HashSet}; +use std::time::Duration; + +use nodedb::types::{DatabaseId, TenantId}; +use nodedb_types::CollectionKey; + +use crate::common::cluster_harness::shared_steps::{db_detail, drain_backup, push_restore}; +use crate::common::cluster_harness::{TestCluster, wait_for, wait_for_async}; + +const TENANT: u64 = 1; +const RF: usize = 2; +const COLLECTION: &str = "graph_br"; +const FAN: usize = 12; + +/// The node ids a 1-hop traversal from `start` reaches, including `start`, or +/// `None` while the query fails (the collection not yet visible on the node). +async fn reached( + client: &tokio_postgres::Client, + start: &str, + direction: &str, +) -> Option> { + let sql = format!( + "GRAPH TRAVERSE IN '{COLLECTION}' FROM '{start}' DEPTH 1 LABEL 'l' DIRECTION {direction}" + ); + let msgs = client.simple_query(&sql).await.ok()?; + let raw = msgs.iter().find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => r.get("result").map(str::to_string), + _ => None, + })?; + let value: serde_json::Value = serde_json::from_str(&raw).ok()?; + let ids = value + .get("nodes")? + .as_array()? + .iter() + .filter_map(|n| n.get("id").and_then(|id| id.as_str()).map(str::to_string)) + .collect(); + Some(ids) +} + +/// The surrogate every node of `cluster` that binds `endpoint` binds it to. +/// `None` while no node binds it. Two nodes disagreeing fails the test. +fn cluster_bind(cluster: &TestCluster, endpoint: &str) -> Option { + let mut seen: Option = None; + for node in &cluster.nodes { + let bound = node + .shared + .surrogate_assigner + .lookup_bound( + CollectionKey::from_bare(DatabaseId::DEFAULT, COLLECTION), + TenantId::new(TENANT), + endpoint.as_bytes(), + ) + .expect("surrogate lookup") + .map(|s| s.as_u32()); + if let Some(s) = bound { + assert!( + seen.is_none_or(|prev| prev == s), + "node {} binds {endpoint} to {s}, another node to {seen:?}", + node.node_id + ); + seen = Some(s); + } + } + seen +} + +fn set(ids: impl IntoIterator) -> HashSet { + ids.into_iter().collect() +} + +fn sources() -> Vec { + (0..FAN).map(|i| format!("src_{i}")).collect() +} + +fn sinks() -> Vec { + (0..FAN).map(|i| format!("dst_{i}")).collect() +} + +/// Every traversal the test checks: `(start, direction, expected reach)`. +fn expectations() -> Vec<(String, &'static str, HashSet)> { + let hub = || "hub".to_string(); + let mut out = vec![ + (hub(), "in", set(std::iter::once(hub()).chain(sources()))), + (hub(), "out", set(std::iter::once(hub()).chain(sinks()))), + ]; + for src in sources() { + out.push((src.clone(), "out", set([src, hub()]))); + } + for dst in sinks() { + out.push((dst.clone(), "in", set([dst, hub()]))); + } + out +} + +/// Wait until node `idx` answers every traversal in `expectations()` exactly. +async fn wait_all_traversals(cluster: &TestCluster, idx: usize, phase: &str) { + let checks = expectations(); + let checks = &checks; + wait_for_async( + &format!("{phase}: node {idx} reaches every edge from both endpoints"), + Duration::from_secs(30), + Duration::from_millis(100), + || async move { + for (start, direction, want) in checks { + let got = reached(&cluster.nodes[idx].client, start, direction).await; + if got.as_ref() != Some(want) { + return false; + } + } + true + }, + ) + .await; +} + +/// Edges into and out of one hub survive BACKUP on a 3-node RF-2 cluster and +/// RESTORE into a fresh 3-node RF-2 cluster: every edge is traversable from +/// both of its endpoints on every node. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn restored_edges_are_traversable_from_both_endpoints_when_nodes_outnumber_rf() { + let source = TestCluster::spawn_three_with_replication_factor(RF) + .await + .expect("source cluster"); + source + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {COLLECTION}")) + .await + .expect("CREATE COLLECTION"); + wait_for( + "every source node sees the collection", + Duration::from_secs(10), + Duration::from_millis(50), + || { + source + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 1) + }, + ) + .await; + + for src in sources() { + source.nodes[0] + .client + .simple_query(&format!( + "GRAPH INSERT EDGE IN '{COLLECTION}' FROM '{src}' TO 'hub' TYPE 'l'" + )) + .await + .unwrap_or_else(|e| panic!("insert {src} -> hub: {}", db_detail(&e))); + } + for dst in sinks() { + source.nodes[0] + .client + .simple_query(&format!( + "GRAPH INSERT EDGE IN '{COLLECTION}' FROM 'hub' TO '{dst}' TYPE 'l'" + )) + .await + .unwrap_or_else(|e| panic!("insert hub -> {dst}: {}", db_detail(&e))); + } + for idx in 0..source.nodes.len() { + wait_all_traversals(&source, idx, "source").await; + } + let endpoints: Vec = std::iter::once("hub".to_string()) + .chain(sources()) + .chain(sinks()) + .collect(); + let source_binds: BTreeMap = endpoints + .iter() + .map(|ep| { + let s = cluster_bind(&source, ep) + .unwrap_or_else(|| panic!("some source node binds endpoint {ep}")); + (ep.clone(), s) + }) + .collect(); + + let bytes = drain_backup(&source.nodes[0].client, TENANT).await; + assert!(!bytes.is_empty(), "backup must produce bytes"); + source.shutdown().await; + + let target = TestCluster::spawn_three_with_replication_factor(RF) + .await + .expect("target cluster"); + push_restore(&target.nodes[0].client, TENANT, bytes).await; + target + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + + for idx in 0..target.nodes.len() { + wait_all_traversals(&target, idx, "target").await; + } + for idx in 0..target.nodes.len() { + for (start, direction, want) in expectations() { + let got = reached(&target.nodes[idx].client, &start, direction).await; + assert_eq!( + got.as_ref(), + Some(&want), + "node {idx}: {direction}-traversal from {start} after restore" + ); + } + } + for (endpoint, surrogate) in &source_binds { + assert_eq!( + cluster_bind(&target, endpoint), + Some(*surrogate), + "restored endpoint {endpoint} keeps its source surrogate" + ); + } + + target.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_cdc_publish_once.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_cdc_publish_once.rs index 11398b953..a436ec75d 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_cdc_publish_once.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_cdc_publish_once.rs @@ -1,47 +1,43 @@ // SPDX-License-Identifier: BUSL-1.1 -//! A replicated SQL write must raise EXACTLY ONE Control-Plane change event -//! per subscriber. +//! The Control-Plane change stream serves every replicated write once, at +//! the same position, on every node. //! -//! ## The bug this guards against +//! Every replica of a data group publishes the group's writes as its apply +//! loop settles them, in log order, at their log positions. So: //! -//! A Raft-replicated write is submitted to the local Data Plane independently -//! by EVERY replica's apply loop. If the change event were published from that -//! apply site, a replication factor of N would publish N events AND fan out N -//! cluster-wide NOTIFY broadcasts, one from each replica. `deliver_remote_notify` -//! forwards every NOTIFY straight to local subscribers and there is no dedup on -//! either side, so each subscriber would silently see the same write repeated — -//! no error, no warning, just duplicated CDC. The event is therefore published -//! once, by the node that handled the write (`dispatch_replicated_write`), after -//! commit + apply; replicas publish nothing. +//! - one write raises exactly one event per node: the node that proposed the +//! write publishes nothing of its own; +//! - every node gives the event the same partition and position; +//! - a cursor taken on one node resumes on another at the next event. //! //! ## Test shape //! -//! Bring up 3 nodes, subscribe on ALL of them, do ONE INSERT through one node, -//! and assert every node's subscription yields exactly one event: one -//! `recv_filtered` succeeds and a second one TIMES OUT. +//! Bring up 3 nodes, subscribe on ALL of them, do ONE INSERT through one +//! node, and assert every node's subscription yields exactly one event at +//! one shared position: one `recv_sequenced` succeeds and a second one +//! TIMES OUT. Then page the stream on one node and resume the cursor on +//! another, and on every node after the whole cluster restarts. //! -//! `ChangeStream::events_published()` is deliberately NOT used as the counter — -//! `deliver_remote_notify` sends straight to subscribers without touching it, -//! so a NOTIFY storm would not move it. Counting what a subscriber actually -//! receives is the only measure that sees the failure. +//! `ChangeStream::events_published()` is deliberately NOT used as the +//! counter. Counting what a subscriber receives is what a client sees. use crate::common; -use common::cluster_harness::TestCluster; +use common::cluster_harness::{TestCluster, TestClusterNode}; use std::time::Duration; -use nodedb::control::change_stream::{ChangeOperation, Subscription}; +use nodedb::control::change_stream::{ChangeOperation, ReplaySnapshot, ReplayStart, Subscription}; +use nodedb_types::{DatabaseId, TenantId}; const COLLECTION: &str = "cdc_once"; -/// How long to wait for the write's event to reach a node. Generous: on the -/// two non-origin nodes it travels as a QUIC NOTIFY. +/// How long to wait for the write's event to reach a node. const ARRIVAL_TIMEOUT: Duration = Duration::from_secs(10); -/// How long to wait to prove NO second event follows. A duplicate from a -/// replica's apply loop is published in the same apply round as the original, -/// so it arrives well within this window. +/// How long to wait to prove NO second event follows. A duplicate is +/// published in the same apply round as the original, so it arrives well +/// within this window. const NO_DUPLICATE_WINDOW: Duration = Duration::from_secs(3); #[tokio::test(flavor = "multi_thread", worker_threads = 4)] @@ -83,8 +79,9 @@ async fn replicated_write_publishes_exactly_one_change_event_per_node() { .wait_for_full_apply_convergence(Duration::from_secs(15)) .await; + let mut positions = Vec::new(); for (idx, sub) in subs.iter_mut().enumerate() { - let event = match tokio::time::timeout(ARRIVAL_TIMEOUT, sub.recv_filtered()).await { + let event = match tokio::time::timeout(ARRIVAL_TIMEOUT, sub.recv_sequenced()).await { Ok(Ok(e)) => e, Ok(Err(e)) => panic!("node {idx}: change stream closed: {e}"), Err(_) => panic!("node {idx}: the replicated INSERT published no change event"), @@ -95,6 +92,7 @@ async fn replicated_write_publishes_exactly_one_change_event_per_node() { ChangeOperation::Insert, "node {idx}: wrong operation kind" ); + positions.push((event.partition(), event.position())); // The publish-once guard: a second event for the same write means the // apply loop published per replica. @@ -102,14 +100,140 @@ async fn replicated_write_publishes_exactly_one_change_event_per_node() { Err(_) => {} Ok(Ok(dup)) => panic!( "node {idx}: one INSERT produced a SECOND change event \ - ({:?} on {} doc {}) — the write is being published once per \ - replica instead of once by the node that handled it", + ({:?} on {} doc {})", dup.operation, dup.collection, dup.document_id ), Ok(Err(e)) => panic!("node {idx}: change stream closed: {e}"), } } + assert!( + positions.windows(2).all(|pair| pair[0] == pair[1]), + "every node must give the write one shared position: {positions:?}" + ); + drop(subs); cluster.shutdown().await; } + +/// Rows the cursor tests write. +const ROWS: usize = 4; + +/// Replay `node`'s change stream of [`COLLECTION`]. +fn replay(node: &TestClusterNode, start: ReplayStart, limit: usize) -> ReplaySnapshot { + node.shared + .change_stream + .query_changes_in_database( + TenantId::new(1), + DatabaseId::DEFAULT, + Some(COLLECTION), + start, + limit, + ) + .unwrap_or_else(|e| panic!("node {}: replay refused: {e:?}", node.node_id)) +} + +/// A 3-node cluster whose [`COLLECTION`] took [`ROWS`] writes through node 0, +/// once every node holds every write's event. +async fn cluster_with_rows() -> TestCluster { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLLECTION} \ + (id TEXT PRIMARY KEY, payload TEXT) WITH (engine='document_strict')" + )) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION {COLLECTION}: {e}")); + for row in 0..ROWS { + cluster.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {COLLECTION} (id, payload) VALUES ('row-{row}', 'payload-{row}')" + )) + .await + .unwrap_or_else(|e| panic!("insert row-{row}: {e}")); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + let deadline = std::time::Instant::now() + ARRIVAL_TIMEOUT; + while cluster + .nodes + .iter() + .any(|node| replay(node, ReplayStart::Timestamp(0), ROWS).events.len() < ROWS) + { + assert!( + std::time::Instant::now() < deadline, + "every node must hold every write's event" + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } + cluster +} + +/// The sorted document ids of two replays. +fn documents(first: &ReplaySnapshot, rest: &ReplaySnapshot) -> Vec { + let mut documents: Vec = first + .events + .iter() + .chain(rest.events.iter()) + .map(|change| change.document_id.as_str().to_owned()) + .collect(); + documents.sort(); + documents +} + +fn every_row() -> Vec { + (0..ROWS).map(|row| format!("row-{row}")).collect() +} + +/// A cursor from one node's replay resumes on another node at the next +/// event, and no event is served twice or skipped. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_change_cursor_from_one_node_resumes_on_another() { + let cluster = cluster_with_rows().await; + let first = replay(&cluster.nodes[0], ReplayStart::Timestamp(0), 2); + assert!(first.has_more); + let rest = replay( + &cluster.nodes[1], + ReplayStart::Cursor(first.cursor.clone()), + ROWS, + ); + assert_eq!( + documents(&first, &rest), + every_row(), + "the cursor from node 0 must resume on node 1 with exactly the remaining events" + ); + cluster.shutdown().await; +} + +/// A cursor taken before every node restarts resumes after the restart at +/// the next event, with no reset: each node rebuilds its feeds from its +/// journal. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_change_cursor_resumes_after_the_cluster_restarts() { + let cluster = cluster_with_rows().await; + let first = replay(&cluster.nodes[0], ReplayStart::Timestamp(0), 2); + assert!(first.has_more); + + let cluster = cluster.restart_all().await.expect("restart the cluster"); + for node in &cluster.nodes { + // A refused cursor panics in `replay`: the restart must not reset it. + let deadline = std::time::Instant::now() + ARRIVAL_TIMEOUT; + let mut rest = replay(node, ReplayStart::Cursor(first.cursor.clone()), ROWS); + while rest.events.len() < ROWS - first.events.len() && std::time::Instant::now() < deadline + { + tokio::time::sleep(Duration::from_millis(100)).await; + rest = replay(node, ReplayStart::Cursor(first.cursor.clone()), ROWS); + } + assert_eq!( + documents(&first, &rest), + every_row(), + "node {} must resume the pre-restart cursor with exactly the remaining events", + node.node_id + ); + } + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_cdc_transaction_events.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_cdc_transaction_events.rs new file mode 100644 index 000000000..fe99d474f --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_cdc_transaction_events.rs @@ -0,0 +1,185 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A committed transaction publishes its rows on the Control-Plane change +//! stream of every node, once each, at the commit's position. A rolled-back +//! transaction publishes nothing. +//! +//! ## Test shape +//! +//! Bring up 3 nodes and subscribe on all of them. Through node 0, commit one +//! transaction (INSERT, INSERT, UPDATE), roll back another, then write one +//! autocommit sentinel row. Every node must yield the two committed rows at +//! one shared position, then the sentinel: nothing of the rolled-back +//! transaction and no duplicate arrives in between. A replay on the last +//! node, from the start of its feed, must serve the committed rows too. + +use crate::common; +use common::cluster_harness::TestCluster; + +use std::collections::BTreeSet; +use std::time::Duration; + +use nodedb::control::change_stream::{ReplayStart, SequencedChangeEvent, Subscription}; +use nodedb_types::{DatabaseId, TenantId}; + +const COLLECTION: &str = "cdc_txn"; + +/// How long to wait for an event to reach a node. +const ARRIVAL_TIMEOUT: Duration = Duration::from_secs(10); + +/// The next event `sub` receives on node `idx`, or a test error. +async fn next_event(sub: &mut Subscription, idx: usize, what: &str) -> SequencedChangeEvent { + match tokio::time::timeout(ARRIVAL_TIMEOUT, sub.recv_sequenced()).await { + Ok(Ok(event)) => event, + Ok(Err(e)) => panic!("node {idx}: change stream closed while awaiting {what}: {e}"), + Err(_) => panic!("node {idx}: no change event for {what}"), + } +} + +/// Run `sql` on node 0's session, failing the test on error. +async fn exec(cluster: &TestCluster, sql: &str) { + cluster.nodes[0] + .client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn committed_transaction_publishes_once_on_every_node() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLLECTION} \ + (id TEXT PRIMARY KEY, payload TEXT) WITH (engine='document_strict')" + )) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION {COLLECTION}: {e}")); + + // Subscribe on every node BEFORE the writes: the bus replays nothing to + // a receiver that did not yet exist. + let mut subs: Vec = cluster + .nodes + .iter() + .map(|node| { + node.shared + .change_stream + .subscribe(Some(COLLECTION.to_string()), None) + }) + .collect(); + + exec(&cluster, "BEGIN").await; + exec( + &cluster, + &format!("INSERT INTO {COLLECTION} (id, payload) VALUES ('t-1', 'new')"), + ) + .await; + exec( + &cluster, + &format!("INSERT INTO {COLLECTION} (id, payload) VALUES ('t-2', 'new')"), + ) + .await; + exec( + &cluster, + &format!("UPDATE {COLLECTION} SET payload = 'updated' WHERE id = 't-1'"), + ) + .await; + exec(&cluster, "COMMIT").await; + + exec(&cluster, "BEGIN").await; + exec( + &cluster, + &format!("INSERT INTO {COLLECTION} (id, payload) VALUES ('r-1', 'rolled back')"), + ) + .await; + exec(&cluster, "ROLLBACK").await; + + exec( + &cluster, + &format!("INSERT INTO {COLLECTION} (id, payload) VALUES ('s-1', 'sentinel')"), + ) + .await; + + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let committed = BTreeSet::from(["t-1".to_owned(), "t-2".to_owned()]); + let mut commit_positions = Vec::new(); + for (idx, sub) in subs.iter_mut().enumerate() { + let first = next_event(sub, idx, "the first committed row").await; + let second = next_event(sub, idx, "the second committed row").await; + let rows: BTreeSet = [&first, &second] + .iter() + .map(|event| event.document_id.as_str().to_owned()) + .collect(); + assert_eq!( + rows, committed, + "node {idx}: the commit publishes each row it wrote once" + ); + assert_eq!( + ( + first.partition(), + first.position().epoch, + first.position().index + ), + ( + second.partition(), + second.position().epoch, + second.position().index + ), + "node {idx}: every row of the commit publishes at the commit's position" + ); + commit_positions.push((first.partition(), first.position())); + + let sentinel = next_event(sub, idx, "the sentinel").await; + assert_eq!( + sentinel.document_id.as_str(), + "s-1", + "node {idx}: an event of the rolled-back transaction, or a duplicate, \ + arrived before the sentinel" + ); + } + assert!( + commit_positions.windows(2).all(|pair| pair[0] == pair[1]), + "every node gives the commit one shared position: {commit_positions:?}" + ); + + // A cursor on another node sees the committed rows. + let replayed = cluster.nodes[2] + .shared + .change_stream + .query_changes_in_database( + TenantId::new(1), + DatabaseId::DEFAULT, + Some(COLLECTION), + ReplayStart::Timestamp(0), + 16, + ) + .unwrap_or_else(|e| panic!("node 2: replay refused: {e:?}")); + let replayed_rows: Vec = replayed + .events + .iter() + .map(|change| change.document_id.as_str().to_owned()) + .collect(); + let mut committed_replayed: Vec = replayed_rows + .iter() + .filter(|row| row.starts_with("t-")) + .cloned() + .collect(); + committed_replayed.sort(); + assert_eq!( + committed_replayed, + vec!["t-1".to_owned(), "t-2".to_owned()], + "node 2's feed serves each committed row once: {replayed_rows:?}" + ); + assert!( + !replayed_rows.iter().any(|row| row == "r-1"), + "node 2's feed serves a rolled-back row: {replayed_rows:?}" + ); + + drop(subs); + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_execute_request.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_execute_request.rs index 73c924b54..42c04b8d1 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_execute_request.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_execute_request.rs @@ -39,7 +39,7 @@ fn make_kv_put_request( key: b"test-key".to_vec(), value: value_bytes, ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"test-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -58,6 +58,8 @@ fn make_kv_put_request( version: descriptor_version, }], txn_id: None, + vshard_id: None, + read_groups: Vec::new(), } } @@ -168,6 +170,8 @@ async fn execute_request_read_carries_watermark_lsn() { trace_id: [0u8; 16], descriptor_versions: vec![], txn_id: None, + vshard_id: None, + read_groups: Vec::new(), }; let resp = send_execute_request(transport, node1.listen_addr, req).await; @@ -284,7 +288,7 @@ async fn execute_request_cross_node_dispatch() { key: b"k1".to_vec(), value: value_bytes, ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"k1".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -300,6 +304,8 @@ async fn execute_request_cross_node_dispatch() { version: 0, // Accept any version (pre-B.1 sentinel bypass) }], txn_id: None, + vshard_id: None, + read_groups: Vec::new(), }; let resp = send_execute_request(sender_transport, target_addr, req).await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_pitr_restore_point.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_pitr_restore_point.rs new file mode 100644 index 000000000..7ab977aca --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_pitr_restore_point.rs @@ -0,0 +1,313 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A cluster rewinds to a restore point with `nodedb restore --cluster`. +//! +//! Three nodes share one cold store and one snapshot store. Rows land in +//! collections spread over both data groups, every node takes a base, more +//! rows land, and a restore point is taken. After the point, more rows land +//! and a collection is created. Every node stops, restores its data +//! directory to the point, and starts. Each node's own replica then holds +//! exactly the rows written before the point, the later collection is gone, +//! and the cluster takes new writes. + +use std::time::Duration; + +use crate::common; +use common::cluster_harness::shared_steps::db_detail; +use common::cluster_harness::{ + PitrStorage, TestCluster, TestClusterNode, read_once_a_leader_exists, +}; + +const COLLECTIONS: &[&str] = &["pitr_a", "pitr_b", "pitr_c", "pitr_d"]; + +/// The first column of every row `sql` returns on `node`, sorted. +async fn column(node: &TestClusterNode, sql: &str) -> Vec { + let messages = read_once_a_leader_exists( + sql, + Duration::from_secs(30), + Duration::from_millis(100), + || node.client.simple_query(sql), + ) + .await; + let mut values: Vec = messages + .iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .collect(); + values.sort(); + values +} + +async fn insert_rows(cluster: &TestCluster, prefix: &str) { + for collection in COLLECTIONS { + for i in 0..3 { + let sql = format!("INSERT INTO {collection} (id, v) VALUES ('{prefix}-{i}', {i})"); + cluster.nodes[0] + .exec(&sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + } +} + +fn ids(prefixes: &[&str]) -> Vec { + let mut ids: Vec = prefixes + .iter() + .flat_map(|prefix| (0..3).map(move |i| format!("{prefix}-{i}"))) + .collect(); + ids.sort(); + ids +} + +/// Take a restore point through `node` and return its id. +async fn create_restore_point(node: &TestClusterNode) -> u64 { + let messages = node + .client + .simple_query("CREATE RESTORE POINT") + .await + .unwrap_or_else(|e| panic!("CREATE RESTORE POINT: {}", db_detail(&e))); + messages + .iter() + .find_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .and_then(|id| id.parse().ok()) + .unwrap_or_else(|| panic!("CREATE RESTORE POINT returned no id: {messages:?}")) +} + +/// A PITR cluster with every collection created, rows `before-base` +/// written, and a base taken on every node. +async fn cluster_with_bases(pitr: &PitrStorage) -> TestCluster { + let cluster = TestCluster::spawn_three_with_pitr(pitr.clone()) + .await + .expect("cluster"); + for collection in COLLECTIONS { + create_collection(&cluster, collection).await; + } + insert_rows(&cluster, "before-base").await; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + for node in &cluster.nodes { + nodedb::control::pitr::take_base_now(&node.shared) + .await + .unwrap_or_else(|e| panic!("node {}: base: {e}", node.node_id)); + } + cluster +} + +async fn create_collection(cluster: &TestCluster, collection: &str) { + let sql = format!( + "CREATE COLLECTION {collection} (id STRING PRIMARY KEY, v INT) \ + WITH (engine='document_strict')" + ); + cluster + .exec_ddl_on_any_leader(&sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); +} + +/// Wait until every node archived `point`, stop every node, restore each to +/// the point, and start them. +async fn restore_every_node(cluster: TestCluster, pitr: &PitrStorage, point: u64) -> TestCluster { + for node in &cluster.nodes { + pitr.await_point_archived(node, point) + .await + .unwrap_or_else(|e| panic!("node {}: {e}", node.node_id)); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + let stopped = cluster.stop_all().await.expect("stop every node"); + for node in stopped.nodes() { + let report = pitr + .restore_node(&node, point) + .await + .unwrap_or_else(|e| panic!("node {}: restore: {e}", node.node_id)); + assert!( + report.contains("restore complete"), + "node {}: {report}", + node.node_id + ); + } + let cluster = stopped.start_all().await.expect("start every node"); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + for node in &cluster.nodes { + node.client + .simple_query("SET default_read_consistency = 'eventual'") + .await + .unwrap_or_else(|e| { + panic!( + "node {}: set eventual reads: {}", + node.node_id, + db_detail(&e) + ) + }); + } + cluster +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_cluster_restores_every_group_to_a_restore_point() { + let root = tempfile::tempdir().expect("storage root"); + let pitr = PitrStorage::create(root.path()).expect("PITR storage"); + let cluster = cluster_with_bases(&pitr).await; + + insert_rows(&cluster, "before-point").await; + let point = create_restore_point(&cluster.nodes[0]).await; + for node in &cluster.nodes { + pitr.await_point_archived(node, point) + .await + .unwrap_or_else(|e| panic!("node {}: {e}", node.node_id)); + } + insert_rows(&cluster, "after-point").await; + create_collection(&cluster, "pitr_late").await; + + let cluster = restore_every_node(cluster, &pitr, point).await; + let expected = ids(&["before-base", "before-point"]); + for node in &cluster.nodes { + let id = node.node_id; + for collection in COLLECTIONS { + assert_eq!( + column(node, &format!("SELECT id FROM {collection}")).await, + expected, + "node {id}: {collection} holds exactly the rows written before the point" + ); + } + assert!( + node.exec("SELECT id FROM pitr_late").await.is_err(), + "node {id}: the collection created after the point is gone" + ); + } + + cluster.nodes[1] + .exec("INSERT INTO pitr_a (id, v) VALUES ('after-restore', 9)") + .await + .unwrap_or_else(|e| panic!("a write after the restore: {e}")); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + let mut with_new = expected.clone(); + with_new.push("after-restore".into()); + with_new.sort(); + for node in &cluster.nodes { + assert_eq!( + column(node, "SELECT id FROM pitr_a").await, + with_new, + "node {}: the restored cluster replicates a new write", + node.node_id + ); + } + + cluster.shutdown().await; +} + +/// A DDL issued after the point's watermark and applied before the point's +/// metadata entry is absent after the restore, and so are its rows. The +/// point parks at the gate `restore_point::after_watermark` after it takes +/// the watermark. The test issues the DDL while the point is parked, then +/// releases it, so the DDL lands first in the metadata log. +#[cfg(feature = "failpoints")] +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_ddl_after_the_watermark_is_absent_though_it_precedes_the_point() { + use nodedb_types::fail_point::{FailAction, FailGuard}; + + let root = tempfile::tempdir().expect("storage root"); + let pitr = PitrStorage::create(root.path()).expect("PITR storage"); + let cluster = cluster_with_bases(&pitr).await; + + let gate_dir = tempfile::tempdir().expect("gate dir"); + let release = gate_dir.path().join("release-point"); + let parked = gate_dir.path().join("release-point.parked"); + let hold = FailGuard::install( + "restore_point::after_watermark", + FailAction::WaitForFile(release.clone()), + ); + let points_before = column(&cluster.nodes[0], "SHOW RESTORE POINTS").await; + let conn_str = format!( + "host=127.0.0.1 port={} user=nodedb dbname=default", + cluster.nodes[0].pg_addr.port() + ); + let creator = tokio::spawn(async move { + let (client, connection) = tokio_postgres::connect(&conn_str, tokio_postgres::NoTls) + .await + .expect("connect for the restore point"); + tokio::spawn(connection); + let messages = client + .simple_query("CREATE RESTORE POINT") + .await + .unwrap_or_else(|e| panic!("CREATE RESTORE POINT: {}", db_detail(&e))); + messages + .iter() + .find_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .and_then(|id| id.parse::().ok()) + .expect("CREATE RESTORE POINT returns its id") + }); + // The point took its watermark and is parked. This DDL and its rows + // carry later HLCs and apply first. + tokio::time::timeout(Duration::from_secs(30), async { + while !parked.exists() { + tokio::time::sleep(Duration::from_millis(20)).await; + } + }) + .await + .expect("the restore point reaches its gate"); + create_collection(&cluster, "pitr_mid").await; + for i in 0..3 { + let sql = format!("INSERT INTO pitr_mid (id, v) VALUES ('mid-{i}', {i})"); + cluster.nodes[0] + .exec(&sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + } + assert!( + !creator.is_finished(), + "the point stays parked while the DDL and its rows apply" + ); + assert_eq!( + column(&cluster.nodes[0], "SHOW RESTORE POINTS").await, + points_before, + "the parked point has not proposed its entry" + ); + std::fs::write(&release, b"").expect("release the restore point"); + let point = creator.await.expect("restore point task"); + drop(hold); + + let cluster = restore_every_node(cluster, &pitr, point).await; + let expected = ids(&["before-base"]); + for node in &cluster.nodes { + let id = node.node_id; + assert!( + node.exec("SELECT id FROM pitr_mid").await.is_err(), + "node {id}: the collection created after the watermark is gone" + ); + assert_eq!( + column(node, "SELECT id FROM pitr_a").await, + expected, + "node {id}: the collections created before the watermark hold their rows" + ); + } + // A same-name collection starts empty: no row of the dropped DDL came + // back in any engine. + create_collection(&cluster, "pitr_mid").await; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + for node in &cluster.nodes { + assert!( + column(node, "SELECT id FROM pitr_mid").await.is_empty(), + "node {}: no row written after the watermark survives", + node.node_id + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_post_apply_follower_dispatch.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_post_apply_follower_dispatch.rs index 9a933ff26..6ee31903a 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cluster_post_apply_follower_dispatch.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_post_apply_follower_dispatch.rs @@ -124,7 +124,7 @@ async fn async_dispatch_fires_on_follower_for_put_and_purge() { && *lsn > 0), "follower (node_id={}) must observe the WAL tombstone for the purged collection. \ If this fails, something reintroduced leader gating on the async post-apply \ - lane — `spawn_post_apply_async_side_effects` must run on every node. \ + lane — `run_post_apply_async_side_effects` must run on every node. \ Follower tombstones observed: {latest:?}", follower.node_id, ); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_retry.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_retry.rs new file mode 100644 index 000000000..93b1612cc --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_retry.rs @@ -0,0 +1,159 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A RESTORE that failed part-way is retried without `FORCE` and converges. +//! +//! The fail point `restore::reissue::before_edges` fails the first attempt +//! after the document rows committed and before any edge. Those rows raised +//! write marks newer than the envelope's watermark. Each mark carries the id +//! of the restore that wrote it, so the retry of the same envelope passes the +//! staleness guard and writes every row and edge. A client write after that +//! still refuses a further restore of the envelope. +//! +//! Requires `--features failpoints`. + +#![cfg(feature = "failpoints")] + +use std::collections::HashSet; +use std::time::Duration; + +use nodedb_types::fail_point::FailGuard; + +use crate::common::cluster_harness::shared_steps::{db_detail, drain_backup, try_push_restore}; +use crate::common::cluster_harness::{TestCluster, wait_for, wait_for_async}; + +const TENANT: u64 = 1; +const DOCS: &str = "rr_docs"; +const GRAPH: &str = "rr_graph"; +const ROWS: usize = 6; +const FAULT: &str = "restore::reissue::before_edges"; + +async fn count_docs(client: &tokio_postgres::Client) -> Option { + let rows = client + .simple_query(&format!("SELECT COUNT(*) FROM {DOCS}")) + .await + .ok()?; + rows.iter().find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => r.get(0).and_then(|s| s.parse().ok()), + _ => None, + }) +} + +/// The sources a reverse 1-hop traversal from `hub` reaches, or `None` while +/// the query fails. +async fn hub_sources(client: &tokio_postgres::Client) -> Option> { + let sql = format!("GRAPH TRAVERSE IN '{GRAPH}' FROM 'hub' DEPTH 1 LABEL 'l' DIRECTION in"); + let msgs = client.simple_query(&sql).await.ok()?; + let raw = msgs.iter().find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => r.get("result").map(str::to_string), + _ => None, + })?; + let value: serde_json::Value = serde_json::from_str(&raw).ok()?; + Some( + value + .get("nodes")? + .as_array()? + .iter() + .filter_map(|n| n.get("id").and_then(|id| id.as_str()).map(str::to_string)) + .filter(|id| id != "hub") + .collect(), + ) +} + +fn sources() -> HashSet { + (0..ROWS).map(|i| format!("src_{i}")).collect() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_restore_failed_part_way_retries_without_force() { + let source = TestCluster::spawn_three().await.expect("source cluster"); + source + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {DOCS} (id TEXT PRIMARY KEY, v TEXT) WITH (engine='document_strict')" + )) + .await + .expect("CREATE docs"); + source + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {GRAPH}")) + .await + .expect("CREATE graph"); + wait_for( + "every source node sees both collections", + Duration::from_secs(10), + Duration::from_millis(50), + || { + source + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 2) + }, + ) + .await; + for i in 0..ROWS { + for sql in [ + format!("INSERT INTO {DOCS} (id, v) VALUES ('k{i}', 'v{i}')"), + format!("GRAPH INSERT EDGE IN '{GRAPH}' FROM 'src_{i}' TO 'hub' TYPE 'l'"), + ] { + source.nodes[0] + .client + .simple_query(&sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {}", db_detail(&e))); + } + } + source + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + let bytes = drain_backup(&source.nodes[0].client, TENANT).await; + source.shutdown().await; + + let target = TestCluster::spawn_three().await.expect("target cluster"); + + // First attempt: the rows commit, then the re-issue fails before edges. + { + let _fault = FailGuard::fail(FAULT, "injected before the edge re-issue"); + let refused = try_push_restore(&target.nodes[0].client, TENANT, bytes.clone()).await; + assert!(refused.is_err(), "the first RESTORE must fail at {FAULT}"); + } + target + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + + // Retry without FORCE: the guard knows the first attempt's writes. + try_push_restore(&target.nodes[0].client, TENANT, bytes.clone()) + .await + .unwrap_or_else(|e| panic!("the retried RESTORE must pass the guard: {e}")); + target + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + + for idx in 0..target.nodes.len() { + let client = &target.nodes[idx].client; + wait_for_async( + &format!("node {idx} holds every restored row and edge"), + Duration::from_secs(30), + Duration::from_millis(100), + || async move { + count_docs(client).await == Some(ROWS) + && hub_sources(client).await == Some(sources()) + }, + ) + .await; + } + + // A client write after the restore still refuses a further restore of + // the same envelope. + target.nodes[0] + .client + .simple_query(&format!("INSERT INTO {DOCS} (id, v) VALUES ('late', 'x')")) + .await + .unwrap_or_else(|e| panic!("late insert: {}", db_detail(&e))); + let refused = try_push_restore(&target.nodes[0].client, TENANT, bytes) + .await + .expect_err("a client write newer than the envelope refuses the restore"); + assert!( + refused.contains("restore refused"), + "the guard names the refusal: {refused}" + ); + + target.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_surrogate_conflict.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_surrogate_conflict.rs new file mode 100644 index 000000000..44813445f --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_surrogate_conflict.rs @@ -0,0 +1,115 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A restore whose carried surrogate another key already holds fails with a +//! typed conflict error, and binds nothing. +//! +//! Binds are first-wins per key, so without the check the restored row +//! installs over the row the other key names. The target binds a restored row's +//! surrogate to a different key on every node before the restore runs. + +use std::time::Duration; + +use nodedb::types::{DatabaseId, TenantId}; +use nodedb_types::{CollectionKey, Surrogate}; + +use crate::common::cluster_harness::shared_steps::{db_detail, drain_backup, try_push_restore}; +use crate::common::cluster_harness::{TestCluster, wait_for}; + +const TENANT: u64 = 1; +const DOCS: &str = "sc_docs"; +const ROWS: usize = 4; + +fn key() -> CollectionKey<'static> { + CollectionKey::from_bare(DatabaseId::DEFAULT, DOCS) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_restored_surrogate_bound_to_another_key_fails_the_restore() { + let source = TestCluster::spawn_three().await.expect("source cluster"); + source + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {DOCS} (id TEXT PRIMARY KEY, v TEXT) WITH (engine='document_strict')" + )) + .await + .expect("CREATE COLLECTION"); + wait_for( + "every source node sees the collection", + Duration::from_secs(10), + Duration::from_millis(50), + || { + source + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 1) + }, + ) + .await; + for i in 0..ROWS { + source.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {DOCS} (id, v) VALUES ('k{i}', 'v{i}')" + )) + .await + .unwrap_or_else(|e| panic!("insert k{i}: {}", db_detail(&e))); + } + source + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + let restored = source + .nodes + .iter() + .find_map(|n| { + n.shared + .surrogate_assigner + .lookup_bound(key(), TenantId::new(TENANT), b"k0") + .ok() + .flatten() + }) + .expect("a source node binds k0") + .as_u32(); + let bytes = drain_backup(&source.nodes[0].client, TENANT).await; + source.shutdown().await; + + let target = TestCluster::spawn_three().await.expect("target cluster"); + for node in &target.nodes { + let held = node + .shared + .surrogate_assigner + .bind( + key(), + TenantId::new(TENANT), + b"intruder", + Surrogate::new(restored), + ) + .expect("pre-existing bind"); + assert_eq!(held.as_u32(), restored); + } + + let refused = try_push_restore(&target.nodes[0].client, TENANT, bytes) + .await + .expect_err("a restored surrogate another key holds fails the restore"); + for needle in [DOCS, "surrogate_identity", "'k0'", "'intruder'"] { + assert!( + refused.contains(needle), + "the conflict names {needle}: {refused}" + ); + } + assert!( + refused.contains(&restored.to_string()), + "the conflict names surrogate {restored}: {refused}" + ); + for node in &target.nodes { + assert_eq!( + node.shared + .surrogate_assigner + .lookup_bound(key(), TenantId::new(TENANT), b"k0") + .expect("lookup"), + None, + "node {} binds nothing of the refused restore", + node.node_id + ); + } + + target.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_surrogate_floor.rs b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_surrogate_floor.rs new file mode 100644 index 000000000..1eb2fb792 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/cluster_restore_surrogate_floor.rs @@ -0,0 +1,135 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A restore raises the target's surrogate floor above every restored +//! surrogate, so a row inserted after it never reuses one. +//! +//! The source and the fresh target both issue surrogates from 1, so without +//! the raise the target's first new rows take the restored rows' surrogates. +//! New rows are inserted through every node, so each node's allocator is +//! exercised after the restore. + +use std::collections::BTreeMap; +use std::time::Duration; + +use nodedb::types::{DatabaseId, TenantId}; +use nodedb_types::CollectionKey; + +use crate::common::cluster_harness::shared_steps::{db_detail, drain_backup, push_restore}; +use crate::common::cluster_harness::{TestCluster, wait_for, wait_for_async}; + +const TENANT: u64 = 1; +const DOCS: &str = "sf_docs"; +const ROWS: usize = 20; + +/// The surrogate the nodes of `cluster` bind `pk` to, `None` while no node +/// binds it. Two nodes disagreeing fails the test. +fn bound(cluster: &TestCluster, pk: &str) -> Option { + let mut seen: Option = None; + for node in &cluster.nodes { + let s = node + .shared + .surrogate_assigner + .lookup_bound( + CollectionKey::from_bare(DatabaseId::DEFAULT, DOCS), + TenantId::new(TENANT), + pk.as_bytes(), + ) + .expect("surrogate lookup") + .map(|s| s.as_u32()); + if let Some(s) = s { + assert!( + seen.is_none_or(|prev| prev == s), + "nodes bind {pk} to different surrogates: {seen:?} and {s}" + ); + seen = Some(s); + } + } + seen +} + +async fn insert(cluster: &TestCluster, node_idx: usize, pk: &str) { + cluster.nodes[node_idx] + .client + .simple_query(&format!("INSERT INTO {DOCS} (id, v) VALUES ('{pk}', 'x')")) + .await + .unwrap_or_else(|e| panic!("insert {pk} on node {node_idx}: {}", db_detail(&e))); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn rows_inserted_after_a_restore_reuse_no_restored_surrogate() { + let source = TestCluster::spawn_three().await.expect("source cluster"); + source + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {DOCS} (id TEXT PRIMARY KEY, v TEXT) WITH (engine='document_strict')" + )) + .await + .expect("CREATE COLLECTION"); + wait_for( + "every source node sees the collection", + Duration::from_secs(10), + Duration::from_millis(50), + || { + source + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 1) + }, + ) + .await; + for i in 0..ROWS { + insert(&source, i % source.nodes.len(), &format!("old{i}")).await; + } + source + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + let bytes = drain_backup(&source.nodes[0].client, TENANT).await; + source.shutdown().await; + + let target = TestCluster::spawn_three().await.expect("target cluster"); + push_restore(&target.nodes[0].client, TENANT, bytes).await; + target + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + let target_ref = ⌖ + wait_for_async( + "every restored row is bound on the target", + Duration::from_secs(30), + Duration::from_millis(100), + || async move { (0..ROWS).all(|i| bound(target_ref, &format!("old{i}")).is_some()) }, + ) + .await; + let restored: BTreeMap = (0..ROWS) + .map(|i| { + let pk = format!("old{i}"); + (bound(&target, &pk).expect("restored bind"), pk) + }) + .collect(); + assert_eq!( + restored.len(), + ROWS, + "restored rows keep distinct surrogates" + ); + let floor = restored.keys().copied().max().unwrap_or(0); + + for i in 0..ROWS { + insert(&target, i % target.nodes.len(), &format!("new{i}")).await; + } + target + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + for i in 0..ROWS { + let pk = format!("new{i}"); + let s = bound(&target, &pk).unwrap_or_else(|| panic!("{pk} is bound")); + assert!( + !restored.contains_key(&s), + "{pk} reuses surrogate {s} of restored row {}", + restored[&s] + ); + assert!( + s > floor, + "{pk} got surrogate {s}, at or below the restored floor {floor}" + ); + } + + target.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/committed_publish_failover.rs b/nodedb-cluster-tests/tests/common_suite/cases/committed_publish_failover.rs new file mode 100644 index 000000000..f4737d4c1 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/committed_publish_failover.rs @@ -0,0 +1,273 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A committed transaction's message reaches its topic exactly once across a +//! leader kill. +//! +//! A SYNC trigger body publishes one message per inserted row. The message +//! rides the insert's redo record, so every replica of the source +//! collection's group holds it once the insert commits. Only the node that +//! holds that group's leader lease delivers, from a replicated cursor. +//! +//! - `a_publish_committed_before_a_leader_kill_is_delivered_once` parks the +//! leader's delivery before it sends anything, then kills the leader. The +//! new lease holder delivers every message from the cursor. +//! - `a_publish_sent_before_a_leader_kill_is_not_appended_twice` holds the +//! leader's delivery until every message is committed, lets it send them +//! all in one pass, parks it before it commits the cursor, then kills it. The new lease holder +//! sends the messages again from the old cursor, and the topic appends none of them twice. + +#![cfg(feature = "failpoints")] + +use crate::common; +use common::cluster_harness::shared_steps::{kill_node, leader_index_of}; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use std::collections::BTreeMap; +use std::path::Path; +use std::time::Duration; + +use nodedb::event::cdc::CdcOffset; +use nodedb_types::DatabaseId; +use nodedb_types::fail_point::{FailAction, FailGuard}; + +const TOPIC: &str = "committed_failover_feed"; +const SRC: &str = "committed_failover_src"; +const TENANT: u64 = 1; +const ROWS: [&str; 3] = ["r1", "r2", "r3"]; + +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(100); + +/// The payloads `node`'s topic log holds, with how many times each appears. +fn topic_payloads(node: &TestClusterNode) -> BTreeMap { + let messages = node + .shared + .credentials + .catalog() + .load_ep_topic_messages(DatabaseId::DEFAULT, TENANT, TOPIC) + .unwrap_or_else(|e| panic!("node {}: read topic: {e}", node.node_id)); + let mut payloads = BTreeMap::new(); + for message in messages { + let row = ROWS + .iter() + .find(|row| message.payload.contains(*row)) + .unwrap_or_else(|| panic!("unexpected topic message {:?}", message.payload)); + *payloads.entry((*row).to_owned()).or_insert(0) += 1; + } + payloads +} + +/// Every row's message, once. +fn every_message_once() -> BTreeMap { + ROWS.iter().map(|row| ((*row).to_owned(), 1)).collect() +} + +/// How many committed messages `node` holds for delivery. +fn held(node: &TestClusterNode) -> usize { + let Some(ledgers) = node.shared.sink_ledgers.get() else { + return 0; + }; + let partitions = ledgers.publishes.partitions().unwrap_or_default(); + partitions + .into_iter() + .map(|partition| { + ledgers + .publishes + .held_after(partition, CdcOffset::ZERO, 1_000) + .map(|held| held.len()) + .unwrap_or(0) + }) + .sum() +} + +/// A three-node cluster with the topic, the source collection, and its SYNC +/// publishing trigger on every node. +async fn cluster_with_publishing_trigger() -> TestCluster { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + wait_for("every data group has a leader", CONVERGE, STEP, || { + cluster.nodes[0] + .all_group_leaders() + .iter() + .all(|(_, leader)| *leader != 0) + }) + .await; + for ddl in [ + format!("CREATE TOPIC {TOPIC}"), + format!( + "CREATE COLLECTION {SRC} (id TEXT PRIMARY KEY, v BIGINT) \ + WITH (engine='document_strict')" + ), + format!( + "CREATE SYNC TRIGGER {SRC}_pub AFTER INSERT ON {SRC} FOR EACH ROW \ + BEGIN PUBLISH TO {TOPIC} NEW.id; END" + ), + ] { + cluster + .exec_ddl_on_any_leader(&ddl) + .await + .unwrap_or_else(|e| panic!("{ddl}: {e}")); + } + wait_for("every node registers the topic", CONVERGE, STEP, || { + cluster.nodes.iter().all(|node| { + node.shared + .ep_topic_registry + .get(DatabaseId::DEFAULT, TENANT, TOPIC) + .is_some() + }) + }) + .await; + cluster +} + +/// Insert every row through `node`: each commits one message. +async fn insert_rows(node: &TestClusterNode) { + for (n, row) in ROWS.iter().enumerate() { + node.exec(&format!("INSERT INTO {SRC} (id, v) VALUES ('{row}', {n})")) + .await + .unwrap_or_else(|e| panic!("insert {row}: {e}")); + } +} + +/// Wait until `marker` exists: the gate's task reached it. +async fn wait_parked(marker: &Path, what: &str) { + wait_for(what, CONVERGE, Duration::from_millis(20), || { + marker.exists() + }) + .await; +} + +/// The survivors' topic logs converge on every message once, and the +/// delivery cursor passes every message on every survivor. +async fn survivors_hold_every_message_once(cluster: &TestCluster) { + wait_for( + "every survivor's topic holds every message", + CONVERGE, + STEP, + || { + cluster + .nodes + .iter() + .all(|node| topic_payloads(node).len() == ROWS.len()) + }, + ) + .await; + wait_for( + "the delivery cursor passes every message on every survivor", + CONVERGE, + STEP, + || cluster.nodes.iter().all(|node| held(node) == 0), + ) + .await; + // A delivery pass runs every 100ms. Several passes later the topic still + // holds each message once. + tokio::time::sleep(Duration::from_millis(1_000)).await; + for node in &cluster.nodes { + assert_eq!( + topic_payloads(node), + every_message_once(), + "node {} holds a message twice or misses one", + node.node_id + ); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_publish_committed_before_a_leader_kill_is_delivered_once() { + let mut cluster = cluster_with_publishing_trigger().await; + let leader = leader_index_of(&cluster, SRC); + let leader_id = cluster.nodes[leader].node_id; + let writer = (leader + 1) % cluster.nodes.len(); + + let gate_dir = tempfile::tempdir().expect("gate directory"); + // Never created: the leader stays parked until it dies. + let release = gate_dir.path().join("release"); + let parked = gate_dir.path().join("release.parked"); + let _gate = FailGuard::install( + &format!("publish::before_delivery::node{leader_id}"), + FailAction::WaitForFile(release), + ); + wait_parked(&parked, "the leader's delivery parks").await; + + insert_rows(&cluster.nodes[writer]).await; + wait_for( + "every replica holds every committed message", + CONVERGE, + STEP, + || cluster.nodes.iter().all(|node| held(node) == ROWS.len()), + ) + .await; + for node in &cluster.nodes { + assert!( + topic_payloads(node).is_empty(), + "node {}: nothing is delivered while the lease holder is parked", + node.node_id + ); + } + + kill_node(&mut cluster, leader).await; + survivors_hold_every_message_once(&cluster).await; + cluster.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_publish_sent_before_a_leader_kill_is_not_appended_twice() { + let mut cluster = cluster_with_publishing_trigger().await; + let leader = leader_index_of(&cluster, SRC); + let leader_id = cluster.nodes[leader].node_id; + let writer = (leader + 1) % cluster.nodes.len(); + + // Hold the leader's delivery until every message is committed, so one + // pass sends them all. + let hold_dir = tempfile::tempdir().expect("gate directory"); + let hold_release = hold_dir.path().join("release"); + let hold_parked = hold_dir.path().join("release.parked"); + let _hold = FailGuard::install( + &format!("publish::before_delivery::node{leader_id}"), + FailAction::WaitForFile(hold_release.clone()), + ); + wait_parked(&hold_parked, "the leader's delivery parks").await; + insert_rows(&cluster.nodes[writer]).await; + wait_for( + "every replica holds every committed message", + CONVERGE, + STEP, + || cluster.nodes.iter().all(|node| held(node) == ROWS.len()), + ) + .await; + + // Then let it send, and park it before its cursor commit. The release + // file of the second gate is never created: the leader dies parked. + let commit_dir = tempfile::tempdir().expect("gate directory"); + let commit_parked = commit_dir.path().join("release.parked"); + let _commit = FailGuard::install( + &format!("publish::before_cursor_commit::node{leader_id}"), + FailAction::WaitForFile(commit_dir.path().join("release")), + ); + std::fs::write(&hold_release, b"release").expect("release the delivery"); + wait_parked(&commit_parked, "the leader parks before its cursor commit").await; + wait_for( + "every node's topic holds every message the leader sent", + CONVERGE, + STEP, + || { + cluster + .nodes + .iter() + .all(|node| topic_payloads(node) == every_message_once()) + }, + ) + .await; + for node in &cluster.nodes { + assert!( + held(node) > 0, + "node {}: the cursor never passed the sent messages", + node.node_id + ); + } + + kill_node(&mut cluster, leader).await; + survivors_hold_every_message_once(&cluster).await; + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/committed_publish_metadata_lag.rs b/nodedb-cluster-tests/tests/common_suite/cases/committed_publish_metadata_lag.rs new file mode 100644 index 000000000..cefee65c5 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/committed_publish_metadata_lag.rs @@ -0,0 +1,221 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A committed transaction's message waits for a lagging catalog and is +//! never dropped for it. +//! +//! Node B leads the source collection's data group, so it holds the lease +//! that delivers the collection's committed messages. B's metadata apply is +//! held. The topic is created, and a SYNC trigger body publishes to it from +//! an insert on node A, while B's catalog does not know the topic yet. B's +//! Calvin scheduler holds the transaction until B's catalog reaches the +//! metadata floor A planned it at, so the insert waits for B. B delivers +//! nothing and drops nothing while held. Once released, B's catalog reaches +//! the topic, the insert commits, and B delivers the message exactly once. + +#![cfg(feature = "failpoints")] + +use crate::common; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use std::sync::atomic::Ordering; +use std::time::{Duration, Instant}; + +use nodedb::control::cluster::metadata_applier::metadata_apply_hold_point; +use nodedb::event::cdc::CdcOffset; +use nodedb_types::DatabaseId; +use nodedb_types::fail_point::{FailAction, FailGuard}; + +const TOPIC: &str = "metadata_lag_feed"; +const SRC: &str = "metadata_lag_src"; +const TENANT: u64 = 1; + +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(100); + +/// How many messages `node`'s topic log holds. +fn topic_messages(node: &TestClusterNode) -> usize { + node.shared + .credentials + .catalog() + .load_ep_topic_messages(DatabaseId::DEFAULT, TENANT, TOPIC) + .map(|messages| messages.len()) + .unwrap_or(0) +} + +/// How many committed messages `node` holds for delivery. +fn held(node: &TestClusterNode) -> usize { + let Some(ledgers) = node.shared.sink_ledgers.get() else { + return 0; + }; + ledgers + .publishes + .partitions() + .unwrap_or_default() + .into_iter() + .map(|partition| { + ledgers + .publishes + .held_after(partition, CdcOffset::ZERO, 1_000) + .map(|held| held.len()) + .unwrap_or(0) + }) + .sum() +} + +/// Committed messages `node` counted as dropped for a missing topic. +fn dropped(node: &TestClusterNode) -> u64 { + node.shared.system_metrics.as_ref().map_or(0, |metrics| { + metrics.committed_publishes_dropped.load(Ordering::Relaxed) + }) +} + +fn knows_topic(node: &TestClusterNode) -> bool { + node.shared + .ep_topic_registry + .get(DatabaseId::DEFAULT, TENANT, TOPIC) + .is_some() +} + +/// Run `sql` on `node`, retrying while the cluster settles. +async fn exec_on(node: &TestClusterNode, sql: &str) { + let deadline = Instant::now() + CONVERGE; + loop { + match node.exec(sql).await { + Ok(()) => return, + Err(error) if Instant::now() < deadline => { + tracing::debug!(%error, sql, "statement not accepted yet; retrying"); + tokio::time::sleep(Duration::from_millis(200)).await; + } + Err(error) => panic!("{sql}: {error}"), + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_lagging_lease_holder_waits_for_the_topic_and_delivers_once() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + wait_for("every data group has a leader", CONVERGE, STEP, || { + cluster.nodes[0] + .all_group_leaders() + .iter() + .all(|(_, leader)| *leader != 0) + }) + .await; + for ddl in [ + format!( + "CREATE COLLECTION {SRC} (id TEXT PRIMARY KEY, v BIGINT) \ + WITH (engine='document_strict')" + ), + format!( + "CREATE SYNC TRIGGER {SRC}_pub AFTER INSERT ON {SRC} FOR EACH ROW \ + BEGIN PUBLISH TO {TOPIC} NEW.id; END" + ), + ] { + cluster + .exec_ddl_on_any_leader(&ddl) + .await + .unwrap_or_else(|e| panic!("{ddl}: {e}")); + } + + // B leads the source group and delivers its committed messages. A is + // another node, where the statements under test run. + let probe = &cluster.nodes[0]; + let group = probe + .group_id_for_collection(SRC) + .unwrap_or_else(|| panic!("no group mapping for {SRC}")); + let b_id = probe + .all_group_leaders() + .into_iter() + .find_map(|(id, leader)| (id == group).then_some(leader)) + .unwrap_or(0); + let b = cluster + .nodes + .iter() + .position(|node| node.node_id == b_id) + .unwrap_or_else(|| panic!("no live node leads {SRC}'s group {group}")); + let a = (b + 1) % cluster.nodes.len(); + + // Hold B's metadata apply, then create the topic through A. + let gate_dir = tempfile::tempdir().expect("gate directory"); + let release = gate_dir.path().join("release"); + let parked = gate_dir.path().join("release.parked"); + let _hold = FailGuard::install( + &metadata_apply_hold_point(b_id), + FailAction::WaitForFile(release.clone()), + ); + exec_on(&cluster.nodes[a], &format!("CREATE TOPIC {TOPIC}")).await; + wait_for( + "every node but B knows the topic, and B's metadata apply is held", + CONVERGE, + STEP, + || { + parked.exists() + && cluster + .nodes + .iter() + .enumerate() + .all(|(i, node)| knows_topic(node) == (i != b)) + }, + ) + .await; + + // A's trigger body finds the topic. The transaction carries A's metadata + // floor, which covers the topic, so B's Calvin scheduler holds it until + // B's catalog reaches that floor: the insert completes only once B is + // released. + let insert = format!("INSERT INTO {SRC} (id, v) VALUES ('r1', 1)"); + let (inserted, ()) = tokio::join!(cluster.nodes[a].exec(&insert), async { + // Several delivery passes later, B still delivered and dropped nothing. + tokio::time::sleep(Duration::from_secs(2)).await; + assert!(!knows_topic(&cluster.nodes[b]), "B's catalog still lags"); + for node in &cluster.nodes { + assert_eq!( + topic_messages(node), + 0, + "node {}: nothing is delivered while the lease holder lags", + node.node_id + ); + } + assert_eq!( + dropped(&cluster.nodes[b]), + 0, + "B drops nothing while it lags" + ); + + // Release B. Its catalog reaches the topic, B runs the held + // transaction, and B delivers the message. + std::fs::write(&release, b"release").expect("release B's metadata apply"); + }); + inserted.unwrap_or_else(|e| panic!("{insert}: {e}")); + wait_for( + "every node's topic holds the message and every ledger drains", + CONVERGE, + STEP, + || { + cluster + .nodes + .iter() + .all(|node| topic_messages(node) == 1 && held(node) == 0) + }, + ) + .await; + tokio::time::sleep(Duration::from_secs(1)).await; + for node in &cluster.nodes { + assert_eq!( + topic_messages(node), + 1, + "node {} holds the message a number of times other than once", + node.node_id + ); + assert_eq!( + dropped(node), + 0, + "node {} dropped the message", + node.node_id + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/cross_node_pk_write.rs b/nodedb-cluster-tests/tests/common_suite/cases/cross_node_pk_write.rs index 46c56f55c..55aa0fabe 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/cross_node_pk_write.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/cross_node_pk_write.rs @@ -7,8 +7,7 @@ //! //! This 3-node test cluster runs replication factor 3: EVERY node is a voter of //! every data group and locally binds surrogates for every committed write. -//! There is NO "non-member" coordinator that has to ship `Surrogate::ZERO` for -//! these keys — every node resolves pk → surrogate from its own local catalog. +//! Every node resolves pk → surrogate from its own local catalog. //! A point WRITE (UPDATE / DELETE by PK) issued anywhere routes via Raft //! propose → apply and lands on all three replicas. //! @@ -21,11 +20,12 @@ //! before any read-back. //! //! (b) NO GHOST / PHANTOM POLLUTION: a PK that is only ever DELETEd or only -//! ever READ (never INSERTed) must NEVER acquire a surrogate binding. On -//! apply, the `decode.rs` `bind_or_lookup` path re-resolves a carried -//! ZERO surrogate READ-ONLY and NEVER binds ZERO, so a missing pk stays -//! unbound. A subsequent INSERT of that pk therefore allocates a FRESH -//! surrogate and resolves correctly — no `pk → ZERO` phantom corrupts it. +//! ever READ (never INSERTed) must NEVER acquire a surrogate binding. A +//! delete of an unbound pk carries no surrogate (`None`), and the apply +//! path resolves it READ-ONLY, so a missing pk stays unbound. A write +//! that carries `Surrogate::ZERO` is refused outright. A subsequent +//! INSERT of that pk therefore allocates a FRESH surrogate and resolves +//! correctly. //! //! ## Test shape //! @@ -37,9 +37,9 @@ //! gone on every node, and that all three members agree. //! 4. GHOST/PHANTOM (the anti-pollution regression): //! - DELETE a key that was NEVER inserted, from every node (each delete -//! resolves to an unbound key → ZERO; apply must NOT bind it), then -//! INSERT that key and assert it reads back as its real value on every -//! node. A phantom `ghost → ZERO` binding would corrupt this read. +//! carries no surrogate; apply must NOT bind one), then INSERT that +//! key and assert it reads back as its real value on every node. A +//! phantom binding for `ghost` corrupts this read. //! - Assert a never-written, never-deleted pk returns no row on every //! node — proof that merely reading/deleting an absent key created no //! spurious binding. @@ -296,9 +296,9 @@ async fn cross_node_pk_write_converges_and_does_not_pollute() { // --- ANTI-POLLUTION (ghost) regression — load-bearing ---------------- // DELETE a key that was NEVER inserted, from EVERY node. Each delete - // resolves to an unbound key (ZERO carry); apply must re-resolve READ-ONLY - // and NEVER bind `ghost → ZERO`. The delete is a correct no-op either way, - // but a phantom ZERO binding here would corrupt the INSERT below. + // resolves to an unbound key and carries no surrogate; apply resolves it + // READ-ONLY and binds nothing. The delete is a correct no-op either way, + // but a phantom binding here corrupts the INSERT below. for (idx, node) in cluster.nodes.iter().enumerate() { exec_dml( &node.client, @@ -313,11 +313,10 @@ async fn cross_node_pk_write_converges_and_does_not_pollute() { .wait_for_full_apply_convergence(Duration::from_secs(15)) .await; - // INSERT the ghost key. With pollution, a phantom `ghost → ZERO` binding - // wins (first-wins) and the row lands under surrogate ZERO → the read - // resolves wrong/empty. With the correct apply path no binding was ever - // written, so the INSERT allocates a fresh surrogate and resolves on all - // replicas. All members must agree. + // INSERT the ghost key. With pollution, a phantom binding for `ghost` + // wins (first-wins) and the read resolves wrong or empty. With the + // correct apply path no binding was ever written, so the INSERT allocates + // a fresh surrogate and resolves on all replicas. All members must agree. cluster.nodes[0] .client .simple_query("INSERT INTO xn_pk_w (id, payload) VALUES ('ghost', 'ghost-val')") @@ -333,7 +332,7 @@ async fn cross_node_pk_write_converges_and_does_not_pollute() { "xn_pk_w", "ghost", Some("ghost-val"), - "ghost no-op DELETE + INSERT (a phantom `ghost → ZERO` binding would corrupt this)", + "ghost no-op DELETE + INSERT (a phantom `ghost` binding would corrupt this)", ) .await; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/database_lifecycle_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/database_lifecycle_cross_node.rs new file mode 100644 index 000000000..9243044bd --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/database_lifecycle_cross_node.rs @@ -0,0 +1,229 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! Database lifecycle across a 3-node cluster. +//! +//! - `DROP DATABASE ... CASCADE` on one node leaves no row of the database in +//! any node's catalog, and the database stays dropped after every node +//! restarts and replays the metadata log. +//! - Database ids come from the replicated metadata log: every node agrees +//! on each id, and no id repeats across a full restart. + +use crate::common; + +use std::time::Duration; + +use common::cluster_harness::shared_steps::use_database; +use common::cluster_harness::{TestCluster, TestClusterNode}; +use nodedb_types::DatabaseId; + +const TENANT: u64 = 1; +const DROPPED: &str = "dl_dropped"; + +/// The id every node's catalog holds for `name`. Panics when a node lacks +/// the database or two nodes disagree on its id. +fn agreed_database_id(cluster: &TestCluster, name: &str) -> DatabaseId { + let ids: Vec = cluster + .nodes + .iter() + .map(|node| { + node.shared + .credentials + .catalog() + .get_database_id_by_name(name) + .expect("look up database id") + .unwrap_or_else(|| panic!("node {} lacks database '{name}'", node.node_id)) + }) + .collect(); + assert!( + ids.windows(2).all(|pair| pair[0] == pair[1]), + "nodes disagree on the id of database '{name}': {ids:?}" + ); + ids[0] +} + +/// Every catalog row this node still holds for database `id`, named by kind. +fn rows_for_database(node: &TestClusterNode, name: &str, id: DatabaseId) -> Vec { + let catalog = node.shared.credentials.catalog(); + let db = id.as_u64(); + let mut rows = Vec::new(); + if catalog + .get_database_id_by_name(name) + .expect("look up database name") + .is_some() + { + rows.push(format!("databases_by_name:{name}")); + } + if catalog.get_database(id).expect("read database").is_some() { + rows.push(format!("databases:{db}")); + } + for coll in catalog.load_all_collections(id).expect("load collections") { + rows.push(format!("collection:{}", coll.name)); + } + for seq in catalog + .load_sequences_in_database(db) + .expect("load sequences") + { + rows.push(format!("sequence:{}", seq.name)); + } + for owner in catalog + .load_all_owners() + .expect("load owners") + .into_iter() + .filter(|owner| owner.database_id == db) + { + rows.push(format!("owner:{}:{}", owner.object_type, owner.object_name)); + } + if node.shared.sequence_registry.exists(db, TENANT, "dl_seq") { + rows.push("sequence_registry:dl_seq".to_string()); + } + rows +} + +/// Poll every node until it holds no row of database `id`. +async fn assert_dropped_everywhere(cluster: &TestCluster, id: DatabaseId, stage: &str) { + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + for node in &cluster.nodes { + // The collection purge finishes in the apply's post-apply lane, so + // poll briefly for it before reporting what is left. + let deadline = std::time::Instant::now() + Duration::from_secs(10); + let mut left = rows_for_database(node, DROPPED, id); + while !left.is_empty() && std::time::Instant::now() < deadline { + tokio::time::sleep(Duration::from_millis(50)).await; + left = rows_for_database(node, DROPPED, id); + } + assert!( + left.is_empty(), + "{stage}: node {} still holds rows of dropped database {}: {left:?}", + node.node_id, + id.as_u64() + ); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn drop_database_cascade_removes_rows_on_every_node_and_survives_restart() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + + cluster + .exec_ddl_on_any_leader(&format!("CREATE DATABASE {DROPPED}")) + .await + .unwrap_or_else(|e| panic!("CREATE DATABASE: {e}")); + let dropped_id = agreed_database_id(&cluster, DROPPED); + + use_database(&cluster, DROPPED).await; + for ddl in [ + "CREATE COLLECTION dl_docs (id TEXT PRIMARY KEY, content TEXT) \ + WITH (engine='document_strict')", + "CREATE COLLECTION dl_kv (key STRING PRIMARY KEY, value STRING) WITH (engine='kv')", + "CREATE SEQUENCE dl_seq START 1", + ] { + cluster + .exec_ddl_on_any_leader(ddl) + .await + .unwrap_or_else(|e| panic!("{ddl}: {e}")); + } + cluster.nodes[0] + .exec("INSERT INTO dl_docs (id, content) VALUES ('k1', 'v1')") + .await + .unwrap_or_else(|e| panic!("INSERT: {e}")); + for node in &cluster.nodes { + assert!( + !rows_for_database(node, DROPPED, dropped_id).is_empty(), + "node {} must hold the database's rows before the drop", + node.node_id + ); + } + use_database(&cluster, "default").await; + + cluster + .exec_ddl_on_any_leader(&format!("DROP DATABASE {DROPPED} CASCADE")) + .await + .unwrap_or_else(|e| panic!("DROP DATABASE CASCADE: {e}")); + assert_dropped_everywhere(&cluster, dropped_id, "after the drop").await; + + let cluster = cluster + .restart_all() + .await + .unwrap_or_else(|e| panic!("restart every node: {e}")); + assert_dropped_everywhere(&cluster, dropped_id, "after every node restarted").await; + + // An id issued after the restart never reuses the dropped database's id. + cluster + .exec_ddl_on_any_leader("CREATE DATABASE dl_after_restart") + .await + .unwrap_or_else(|e| panic!("CREATE DATABASE after restart: {e}")); + let after = agreed_database_id(&cluster, "dl_after_restart"); + assert!( + after.as_u64() > dropped_id.as_u64(), + "database id {} issued after the restart does not exceed the dropped id {}", + after.as_u64(), + dropped_id.as_u64() + ); + + cluster.shutdown().await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn database_ids_agree_across_nodes_and_never_repeat_after_restart() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + + let mut issued = Vec::new(); + for name in ["dl_ids_a", "dl_ids_b"] { + cluster + .exec_ddl_on_any_leader(&format!("CREATE DATABASE {name}")) + .await + .unwrap_or_else(|e| panic!("CREATE DATABASE {name}: {e}")); + issued.push(agreed_database_id(&cluster, name)); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + // Every node advanced its hwm, so any node that leads next continues + // past every id already issued. + for node in &cluster.nodes { + let hwm = node.shared.database_registry.current_hwm(); + assert!( + hwm >= issued[1].as_u64(), + "node {} hwm {hwm} lags the replicated reservation of {}", + node.node_id, + issued[1].as_u64() + ); + } + + let cluster = cluster + .restart_all() + .await + .unwrap_or_else(|e| panic!("restart every node: {e}")); + for name in ["dl_ids_c", "dl_ids_d"] { + cluster + .exec_ddl_on_any_leader(&format!("CREATE DATABASE {name}")) + .await + .unwrap_or_else(|e| panic!("CREATE DATABASE {name}: {e}")); + issued.push(agreed_database_id(&cluster, name)); + } + for name in ["dl_ids_a", "dl_ids_b"] { + agreed_database_id(&cluster, name); + } + + let mut unique: Vec = issued.iter().map(|id| id.as_u64()).collect(); + unique.sort_unstable(); + unique.dedup(); + assert_eq!( + unique.len(), + issued.len(), + "a database id repeated across the restart: {issued:?}" + ); + assert!( + issued + .windows(2) + .all(|pair| pair[0].as_u64() < pair[1].as_u64()), + "database ids are not increasing across the restart: {issued:?}" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/ddl_prepare_lease_reclaim.rs b/nodedb-cluster-tests/tests/common_suite/cases/ddl_prepare_lease_reclaim.rs new file mode 100644 index 000000000..6282f3e7d --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/ddl_prepare_lease_reclaim.rs @@ -0,0 +1,134 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A node that dies holding the DDL preparation lease stops blocking DDL once +//! the metadata leader knows it is dead, long before the lease's 60s +//! stuck-owner fallback. Its late entries apply nothing. + +use crate::common; + +use std::time::{Duration, Instant}; + +use common::cluster_harness::shared_steps::propose_and_apply; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; +use nodedb_cluster::MetadataEntry; + +const POLL: Duration = Duration::from_millis(50); +/// SWIM declares the owner Dead within a probe round plus a 3-node +/// suspicion timeout. The leader then waits the dead-holder grace, and polls. +const RECLAIM_BUDGET: Duration = Duration::from_secs(45); +/// The stuck-owner fallback a dead owner must not have to wait out. +const STUCK_OWNER_LEASE: Duration = Duration::from_secs(60); +const DEAD_TOKEN: u64 = 0x00dd_1ea5; + +fn owner_token(node: &TestClusterNode) -> Option { + node.shared + .metadata_ddl_owner + .lock() + .unwrap_or_else(|p| p.into_inner()) + .map(|owner| owner.token) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_dead_ddl_lease_owner_is_reclaimed_before_its_lease_runs_out() { + let mut cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + + // The owner is a node that does not lead the metadata group, so the + // survivors keep their metadata leader and quorum. + let metadata_leader = cluster.nodes[0].metadata_group_leader(); + let owner_idx = cluster + .nodes + .iter() + .position(|n| n.node_id != metadata_leader) + .expect("a node that does not lead the metadata group"); + let owner_id = cluster.nodes[owner_idx].node_id; + + // The owner is mid-prepare: it holds the lease and has reserved a + // pending DDL under it. + propose_and_apply( + &cluster.nodes[owner_idx], + &MetadataEntry::DdlPrepareAcquire { + token: DEAD_TOKEN, + node_id: owner_id, + }, + ) + .await; + propose_and_apply( + &cluster.nodes[owner_idx], + &MetadataEntry::DdlPendingPropose { + token: DEAD_TOKEN, + objects: Vec::new(), + proposed_at: cluster.nodes[owner_idx].shared.hlc_clock.now(), + }, + ) + .await; + wait_for( + "every node sees the owner's lease and pending record", + Duration::from_secs(10), + POLL, + || { + cluster.nodes.iter().all(|n| { + owner_token(n) == Some(DEAD_TOKEN) && n.shared.pending_ddl.contains(DEAD_TOKEN) + }) + }, + ) + .await; + + // A harness shutdown never releases the lease, so the owner dies with it. + let owner = cluster.nodes.remove(owner_idx); + owner.shutdown().await; + let killed_at = Instant::now(); + + wait_for( + "the metadata leader reclaims the dead owner's lease", + RECLAIM_BUDGET, + POLL, + || { + cluster.nodes.iter().all(|n| { + owner_token(n) != Some(DEAD_TOKEN) && !n.shared.pending_ddl.contains(DEAD_TOKEN) + }) + }, + ) + .await; + assert!( + killed_at.elapsed() < STUCK_OWNER_LEASE, + "the reclaim took {:?}: the dead owner waited out the stuck-owner fallback", + killed_at.elapsed() + ); + + // A survivor's DDL proceeds. + cluster + .exec_ddl_on_any_leader("CREATE COLLECTION reclaim_after_dead_owner") + .await + .expect("a survivor's DDL proceeds once the lease is reclaimed"); + let survivor = &cluster.nodes[0]; + + // The dead owner's late entries, replayed through the log, apply nothing. + propose_and_apply( + survivor, + &MetadataEntry::DdlPendingPropose { + token: DEAD_TOKEN, + objects: Vec::new(), + proposed_at: survivor.shared.hlc_clock.now(), + }, + ) + .await; + propose_and_apply( + survivor, + &MetadataEntry::DdlPendingFinalize { token: DEAD_TOKEN }, + ) + .await; + assert!( + !survivor.shared.pending_ddl.contains(DEAD_TOKEN), + "a reclaimed token reserves nothing" + ); + assert_ne!( + survivor + .shared + .metadata_ddl_applied_token + .load(std::sync::atomic::Ordering::Acquire), + DEAD_TOKEN, + "a reclaimed token's finalize applies nothing" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_dead_holder.rs b/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_dead_holder.rs new file mode 100644 index 000000000..b78478624 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_dead_holder.rs @@ -0,0 +1,177 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A crashed lease holder that SWIM declares Dead stops blocking DDL once the +//! dead-holder grace passes, long before its lease expires. + +use crate::common; + +use std::sync::Arc; +use std::time::Duration; + +use common::cluster_harness::{TestCluster, wait_for}; +use nodedb_cluster::DescriptorKind; + +const TENANT: u64 = 1; +const WAIT_BUDGET: Duration = Duration::from_secs(3); +const POLL: Duration = Duration::from_millis(20); + +/// A crashed holder stays in topology, so only its SWIM Dead verdict can end +/// its hold early. The survivors' real SWIM detectors must declare it Dead. +/// The ALTER on the metadata leader must then commit once the dead grace +/// passes, while the holder's lease still outlives the drain timeout. The +/// leader's lease GC must then release the dead holder's lease. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn ddl_proceeds_after_dead_holder_clamp() { + use nodedb_cluster::DEAD_HOLDER_LEASE_GRACE; + + /// Longer than the drain timeout, so only the dead-holder clamp can clear it. + const HOLDER_LEASE: Duration = Duration::from_secs(120); + /// Default SWIM needs a probe round plus a 3-node suspicion timeout. + const SWIM_DEAD_BUDGET: Duration = Duration::from_secs(30); + /// One lease-GC sweep after the ALTER commits. + const GC_BUDGET: Duration = Duration::from_secs(10); + /// The drain clears no earlier than the dead grace after the Dead + /// verdict, and the ALTER starts right at that verdict. The extra time + /// covers the drain propose, the Raft-silence check and the commit. + const ALTER_BUDGET: Duration = DEAD_HOLDER_LEASE_GRACE.saturating_add(Duration::from_secs(15)); + + let mut cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + cluster + .exec_ddl_on_any_leader("CREATE COLLECTION dead_holder") + .await + .expect("create"); + wait_for( + "collection stamped v1 on every node", + WAIT_BUDGET, + POLL, + || { + cluster + .nodes + .iter() + .all(|n| n.collection_descriptor(TENANT, "dead_holder").map(|s| s.0) == Some(1)) + }, + ) + .await; + + // Kill a node that does not lead the metadata group, so the survivors + // keep their metadata leader and quorum. + let metadata_leader = cluster.nodes[0].metadata_group_leader(); + let holder_idx = cluster + .nodes + .iter() + .position(|n| n.node_id != metadata_leader) + .expect("a node that does not lead the metadata group"); + let holder_id = cluster.nodes[holder_idx].node_id; + cluster.nodes[holder_idx] + .acquire_lease( + DescriptorKind::Collection, + TENANT, + "dead_holder", + 1, + HOLDER_LEASE, + ) + .await + .expect("holder acquires v1"); + wait_for( + "every node sees the holder's lease", + WAIT_BUDGET, + POLL, + || { + cluster.nodes.iter().all(|n| { + n.has_lease( + DescriptorKind::Collection, + TENANT, + "dead_holder", + holder_id, + 1, + ) + }) + }, + ) + .await; + + // A harness shutdown never releases leases, so the holder dies holding one. + let holder = cluster.nodes.remove(holder_idx); + holder.shutdown().await; + + let drainer_idx = cluster + .nodes + .iter() + .position(|n| n.node_id == metadata_leader) + .expect("metadata leader survives"); + wait_for( + "the metadata leader's SWIM declares the holder Dead", + SWIM_DEAD_BUDGET, + POLL, + || { + cluster.nodes[drainer_idx] + .shared + .lease_runtime + .holder_liveness + .dead_since(holder_id) + .is_some() + }, + ) + .await; + + let drainer = &cluster.nodes[drainer_idx]; + let existing = drainer + .shared + .credentials + .catalog() + .get_collection(nodedb_types::DatabaseId::DEFAULT, TENANT, "dead_holder") + .expect("read existing") + .expect("exists"); + let alter_shared = Arc::clone(&drainer.shared); + let alter_result = tokio::spawn(async move { + let entry = nodedb::control::catalog_entry::CatalogEntry::PutCollection(Box::new(existing)); + match tokio::time::timeout( + ALTER_BUDGET, + nodedb::control::metadata_proposer::propose_catalog_entry_async(&alter_shared, &entry), + ) + .await + { + Ok(result) => result.map_err(|e| e.to_string()), + Err(_) => Err(format!( + "propose_catalog_entry_async timed out after {ALTER_BUDGET:?}" + )), + } + }) + .await + .expect("join"); + + assert!( + alter_result.is_ok(), + "ALTER must commit once the dead-holder clamp passes: {:?}", + alter_result.err() + ); + let dead_for = drainer + .shared + .lease_runtime + .holder_liveness + .dead_since(holder_id) + .expect("holder still recorded Dead") + .elapsed(); + assert!( + dead_for >= DEAD_HOLDER_LEASE_GRACE, + "the drain cleared {dead_for:?} after the Dead verdict, before the \ + {DEAD_HOLDER_LEASE_GRACE:?} grace" + ); + + wait_for( + "collection stamped v2 and the dead holder's lease released", + GC_BUDGET, + POLL, + || { + cluster.nodes.iter().all(|n| { + n.collection_descriptor(TENANT, "dead_holder").map(|s| s.0) == Some(2) + && n.leases_for_descriptor(DescriptorKind::Collection, TENANT, "dead_holder") + .iter() + .all(|l| l.node_id != holder_id) + }) + }, + ) + .await; + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_drain.rs b/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_drain.rs index 3ec63bf89..6ec2df0e0 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_drain.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_drain.rs @@ -49,17 +49,22 @@ async fn drain_blocks_new_acquires_at_drained_version() { // cancel it at the end of the test. let shared = Arc::clone(&leader.shared); let drain_id = id.clone(); - tokio::task::spawn_blocking(move || { + tokio::spawn(async move { let now_hlc = shared.hlc_clock.now(); let expires_at = nodedb_types::Hlc::new(now_hlc.wall_ns.saturating_add(60_000_000_000), 0); let entry = nodedb_cluster::MetadataEntry::DescriptorDrainStart { descriptor_id: drain_id, up_to_version: 1, expires_at, + proposer_node_id: shared.node_id, + owner: nodedb_cluster::DrainOwner::Ddl, }; let raw = nodedb_cluster::encode_entry(&entry).expect("encode"); let handle = shared.metadata_raft.get().expect("metadata raft handle"); - handle.propose(raw).expect("propose drain start"); + handle + .propose_async(raw) + .await + .expect("propose drain start"); }) .await .expect("join"); @@ -118,7 +123,7 @@ async fn drain_clears_after_end_entry() { // every node's tracker should be empty for this descriptor. let shared = Arc::clone(&leader.shared); let drain_id = id.clone(); - tokio::task::spawn_blocking(move || { + tokio::spawn(async move { let now_hlc = shared.hlc_clock.now(); let expires_at = nodedb_types::Hlc::new(now_hlc.wall_ns.saturating_add(60_000_000_000), 0); let handle = shared.metadata_raft.get().expect("handle"); @@ -126,15 +131,20 @@ async fn drain_clears_after_end_entry() { descriptor_id: drain_id.clone(), up_to_version: 5, expires_at, + proposer_node_id: shared.node_id, + owner: nodedb_cluster::DrainOwner::Ddl, }; handle - .propose(nodedb_cluster::encode_entry(&start).unwrap()) + .propose_async(nodedb_cluster::encode_entry(&start).unwrap()) + .await .expect("start"); let end = nodedb_cluster::MetadataEntry::DescriptorDrainEnd { descriptor_id: drain_id, + owner: nodedb_cluster::DrainOwner::Ddl, }; handle - .propose(nodedb_cluster::encode_entry(&end).unwrap()) + .propose_async(nodedb_cluster::encode_entry(&end).unwrap()) + .await .expect("end"); }) .await @@ -186,21 +196,16 @@ async fn ddl_waits_for_existing_lease_to_release() { ) .await; - // Acquire a lease on v1 — this is what drain must wait for. - leader - .acquire_lease( - DescriptorKind::Collection, - TENANT, - "drainable", - 1, - Duration::from_secs(60), - ) + // Hold a v1 lease the way a running statement does. This is what the + // drain must wait for: a drain start releases an idle lease at once. + let hold = leader + .hold_lease(DescriptorKind::Collection, TENANT, "drainable", 1) .await - .expect("acquire v1 lease"); + .expect("hold v1 lease"); - // Kick off an ALTER directly via `propose_catalog_entry` + // Kick off an ALTER directly via `propose_catalog_entry_async` // rather than pgwire so the test can run it in a - // spawn_blocking task while the main task polls for drain + // spawned task while the main task polls for drain // state. We build a fresh `PutCollection` with the same // content as the existing v1 record — the applier will see // a Put* for an existing descriptor, run drain for prior=1, @@ -214,13 +219,17 @@ async fn ddl_waits_for_existing_lease_to_release() { .expect("exists"); let alter_shared = Arc::clone(&leader.shared); - let alter_handle = tokio::task::spawn_blocking(move || { + let alter_handle = tokio::spawn(async move { let entry = nodedb::control::catalog_entry::CatalogEntry::PutCollection(Box::new(existing)); - nodedb::control::metadata_proposer::propose_catalog_entry_with_timeout( - &alter_shared, - &entry, + match tokio::time::timeout( Duration::from_secs(10), + nodedb::control::metadata_proposer::propose_catalog_entry_async(&alter_shared, &entry), ) + .await + { + Ok(result) => result.map_err(|e| e.to_string()), + Err(_) => Err("propose_catalog_entry_async timed out after 10s".to_string()), + } }); // Give the ALTER a chance to start draining. Poll until the @@ -245,12 +254,10 @@ async fn ddl_waits_for_existing_lease_to_release() { "DDL must not have committed while drain is waiting" ); - // Release the lease. Drain should complete; the Put* should - // commit; version should bump to 2 on every node. - leader - .release_leases(vec![coll_id("drainable")]) - .await - .expect("release"); + // End the statement. Its last hold ending under the drain releases the + // lease; the drain completes, the Put* commits, and the version bumps to + // 2 on every node. + drop(hold); let alter_result = alter_handle.await.expect("join"); assert!( @@ -289,29 +296,27 @@ async fn drain_timeout_clears_state() { // the way so drain cannot complete. The drain_for_ddl call // times out, emits DrainEnd, and returns an error. // - // Setup: acquire a long-lived lease at v1 so drain has - // something to wait for that it cannot clear on its own. - leader - .acquire_lease( - DescriptorKind::Collection, - TENANT, - "stuck", - 1, - Duration::from_secs(60), - ) + // Setup: hold a v1 lease the way a running statement does, so the drain + // has something to wait for that it cannot clear on its own. + let hold = leader + .hold_lease(DescriptorKind::Collection, TENANT, "stuck", 1) .await - .expect("acquire"); + .expect("hold"); let shared = Arc::clone(&leader.shared); let drain_id = id.clone(); - let result = tokio::task::spawn_blocking(move || { - // 0 own holds: the lease under test belongs to another holder, so - // nothing here may be excluded from the drain. - nodedb::control::lease::drain_for_ddl(&shared, drain_id, 1, Duration::from_millis(200), 0) - }) - .await - .expect("join"); + // 0 own holds: the lease under test belongs to another holder, so + // the drain excludes nothing here. + let result = nodedb::control::lease::drain_for_ddl_async( + &shared, + drain_id, + 1, + Duration::from_millis(200), + 0, + ) + .await; + drop(hold); assert!(result.is_err(), "drain should have timed out"); let err_msg = format!("{}", result.unwrap_err()); assert!( diff --git a/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_forwarding_and_renewal.rs b/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_forwarding_and_renewal.rs index 317ace7d6..81a92a72d 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_forwarding_and_renewal.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_forwarding_and_renewal.rs @@ -116,6 +116,16 @@ async fn lease_renews_before_expiry() { let leader = &cluster.nodes[0]; + // The renewal loop renews a held lease and lets an idle one lapse, so a + // statement's hold stands in for the query running across the window. + let held = nodedb_cluster::DescriptorId::new( + 0, + TENANT, + DescriptorKind::Collection, + "renewable".to_string(), + ); + leader.shared.lease_refcount.increment(&held, 1); + // Acquire on the leader. Lease has ~3s expiry from now. let initial = leader .acquire_lease( @@ -148,6 +158,7 @@ async fn lease_renews_before_expiry() { initial_expiry, renewed.expires_at ); + leader.shared.lease_refcount.decrement(&held, 1); cluster.shutdown().await; } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_planner_integration.rs b/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_planner_integration.rs index 0ab45f660..44d980e19 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_planner_integration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/descriptor_lease_planner_integration.rs @@ -95,17 +95,22 @@ async fn drain_forces_plan_retry_surfacing_typed_error() { let shared = Arc::clone(&leader.shared); let drain_id = coll_id("retry_me"); let drain_id_clone = drain_id.clone(); - tokio::task::spawn_blocking(move || { + tokio::spawn(async move { let now_hlc = shared.hlc_clock.now(); let expires_at = nodedb_types::Hlc::new(now_hlc.wall_ns.saturating_add(60_000_000_000), 0); let entry = nodedb_cluster::MetadataEntry::DescriptorDrainStart { descriptor_id: drain_id_clone, up_to_version: 1, expires_at, + proposer_node_id: shared.node_id, + owner: nodedb_cluster::DrainOwner::Ddl, }; let raw = nodedb_cluster::encode_entry(&entry).expect("encode"); let handle = shared.metadata_raft.get().expect("handle"); - handle.propose(raw).expect("propose drain start"); + handle + .propose_async(raw) + .await + .expect("propose drain start"); }) .await .expect("join"); @@ -132,13 +137,14 @@ async fn drain_forces_plan_retry_surfacing_typed_error() { // cluster (not that there are any) aren't affected. let cleanup_shared = Arc::clone(&leader.shared); let cleanup_id = drain_id.clone(); - tokio::task::spawn_blocking(move || { + tokio::spawn(async move { let entry = nodedb_cluster::MetadataEntry::DescriptorDrainEnd { descriptor_id: cleanup_id, + owner: nodedb_cluster::DrainOwner::Ddl, }; let raw = nodedb_cluster::encode_entry(&entry).expect("encode"); let handle = cleanup_shared.metadata_raft.get().expect("handle"); - handle.propose(raw).expect("propose drain end"); + handle.propose_async(raw).await.expect("propose drain end"); }) .await .expect("join"); @@ -168,17 +174,22 @@ async fn drain_cleared_mid_retry_succeeds() { // Install drain. let shared = Arc::clone(&leader.shared); let dstart = drain_id.clone(); - tokio::task::spawn_blocking(move || { + tokio::spawn(async move { let now_hlc = shared.hlc_clock.now(); let expires_at = nodedb_types::Hlc::new(now_hlc.wall_ns.saturating_add(60_000_000_000), 0); let entry = nodedb_cluster::MetadataEntry::DescriptorDrainStart { descriptor_id: dstart, up_to_version: 1, expires_at, + proposer_node_id: shared.node_id, + owner: nodedb_cluster::DrainOwner::Ddl, }; let raw = nodedb_cluster::encode_entry(&entry).expect("encode"); let handle = shared.metadata_raft.get().expect("handle"); - handle.propose(raw).expect("propose drain start"); + handle + .propose_async(raw) + .await + .expect("propose drain start"); }) .await .expect("join"); @@ -194,13 +205,14 @@ async fn drain_cleared_mid_retry_succeeds() { let dend = drain_id.clone(); tokio::spawn(async move { tokio::time::sleep(Duration::from_millis(100)).await; - tokio::task::spawn_blocking(move || { + tokio::spawn(async move { let entry = nodedb_cluster::MetadataEntry::DescriptorDrainEnd { descriptor_id: dend, + owner: nodedb_cluster::DrainOwner::Ddl, }; let raw = nodedb_cluster::encode_entry(&entry).expect("encode"); let handle = cleanup_shared.metadata_raft.get().expect("handle"); - handle.propose(raw).expect("propose drain end"); + handle.propose_async(raw).await.expect("propose drain end"); }) .await .expect("join"); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/descriptor_replay_incarnation.rs b/nodedb-cluster-tests/tests/common_suite/cases/descriptor_replay_incarnation.rs new file mode 100644 index 000000000..c68c655e0 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/descriptor_replay_incarnation.rs @@ -0,0 +1,267 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Group-0 metadata replay after a graceful restart of a single-node +//! metadata group keeps the latest descriptor of every name. +//! +//! - A historical descriptor entry never lowers the persisted latest +//! version. +//! - A replayed `PurgeCollection` or `DeleteMaterializedView` fences to the +//! incarnation it dropped and leaves a recreated one intact. + +use crate::common; + +use std::time::Duration; + +use common::cluster_harness::shared_steps::wait_for_single_node_ready; +use common::cluster_harness::{TestClusterNode, read_once_a_leader_exists, wait_for}; + +const TENANT: u64 = 1; + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn historical_descriptor_entries_replay_without_regressing_the_latest_version() { + let data_dir = tempfile::tempdir().expect("tempdir"); + let data_path = data_dir.path().to_path_buf(); + let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path.clone()) + .await + .expect("spawn single-node metadata group"); + wait_for_single_node_ready(&node).await; + + node.client + .simple_query( + "CREATE COLLECTION replay_graph (id TEXT PRIMARY KEY, name TEXT) \ + WITH (engine='document_strict')", + ) + .await + .expect("create graph-bearing collection"); + wait_for( + "collection descriptor reaches version 1", + Duration::from_secs(10), + Duration::from_millis(50), + || { + node.collection_descriptor(TENANT, "replay_graph") + .map(|v| v.0) + == Some(1) + }, + ) + .await; + + node.client + .simple_query( + "GRAPH INSERT EDGE IN replay_graph FROM 'a' TO 'b' \ + TYPE 'knows' PROPERTIES '{}'", + ) + .await + .expect("insert edge and mark collection edge-bearing"); + wait_for( + "edge-bearing descriptor reaches version 2", + Duration::from_secs(10), + Duration::from_millis(50), + || { + node.collection_descriptor(TENANT, "replay_graph") + .map(|v| v.0) + == Some(2) + }, + ) + .await; + + node.graceful_shutdown_wal_only().await; + let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path) + .await + .expect("restart against the persisted catalog and full metadata log"); + wait_for_single_node_ready(&node).await; + + wait_for( + "latest collection descriptor remains visible after replay", + Duration::from_secs(10), + Duration::from_millis(50), + || { + node.collection_descriptor(TENANT, "replay_graph") + .map(|v| v.0) + == Some(2) + }, + ) + .await; + + node.client + .simple_query( + "CREATE COLLECTION ddl_after_descriptor_replay \ + (id TEXT PRIMARY KEY) WITH (engine='document_strict')", + ) + .await + .expect( + "historical metadata replay must advance its watermark so later DDL remains usable", + ); + assert_eq!( + node.collection_descriptor(TENANT, "replay_graph") + .map(|version| version.0), + Some(2), + "replaying historical version 1 must not overwrite the persisted latest version 2" + ); + + node.shutdown().await; +} + +/// `SELECT COUNT(*) FROM ` against the single-node harness's driving +/// pgwire client. A freshly restarted node refuses reads until each range +/// has a serving leader, so the count retries on that refusal alone. +async fn row_count(client: &tokio_postgres::Client, name: &str) -> usize { + let query = format!("SELECT COUNT(*) FROM {name}"); + let query = query.as_str(); + read_once_a_leader_exists( + &format!("count rows of {name}"), + Duration::from_secs(10), + Duration::from_millis(50), + || client.simple_query(query), + ) + .await + .into_iter() + .find_map(|msg| match msg { + tokio_postgres::SimpleQueryMessage::Row(row) => row + .get(0) + .map(|s| s.parse::().expect("COUNT(*) parse")), + _ => None, + }) + .expect("COUNT(*) returned no rows") +} + +/// CREATE, `DROP COLLECTION ... PURGE`, and CREATE again on the same name, +/// all before a graceful restart of the single-node metadata +/// group. The replayed `PurgeCollection` must fence to the first incarnation: +/// the second incarnation's descriptor and rows must survive replay, and the +/// collection must still take DDL afterward. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn replayed_purge_does_not_remove_a_recreated_collection() { + let data_dir = tempfile::tempdir().expect("tempdir"); + let data_path = data_dir.path().to_path_buf(); + let node = TestClusterNode::spawn_single_node_calvin_on_path(10, data_path.clone()) + .await + .expect("spawn single-node metadata group"); + wait_for_single_node_ready(&node).await; + + node.client + .simple_query("CREATE COLLECTION recreated (id BIGINT PRIMARY KEY, amount BIGINT)") + .await + .expect("create first incarnation"); + node.client + .simple_query("INSERT INTO recreated (id, amount) VALUES (1, 10)") + .await + .expect("insert into first incarnation"); + + node.client + .simple_query("DROP COLLECTION recreated PURGE") + .await + .expect("purge first incarnation"); + + node.client + .simple_query("CREATE COLLECTION recreated (id BIGINT PRIMARY KEY, amount BIGINT)") + .await + .expect("create second incarnation"); + for id in 1..=3i64 { + node.client + .simple_query(&format!( + "INSERT INTO recreated (id, amount) VALUES ({id}, {})", + id * 10 + )) + .await + .expect("insert into second incarnation"); + } + assert_eq!( + row_count(&node.client, "recreated").await, + 3, + "second incarnation must hold its 3 rows before restart" + ); + + node.graceful_shutdown_wal_only().await; + let node = TestClusterNode::spawn_single_node_calvin_on_path(10, data_path) + .await + .expect("restart against the persisted metadata log"); + wait_for_single_node_ready(&node).await; + + assert!( + node.collection_descriptor(TENANT, "recreated").is_some(), + "the second incarnation's descriptor must survive replay of the group-0 log" + ); + assert_eq!( + row_count(&node.client, "recreated").await, + 3, + "replaying the first incarnation's PurgeCollection must not reclaim the second \ + incarnation's rows" + ); + + node.client + .simple_query("INSERT INTO recreated (id, amount) VALUES (4, 40)") + .await + .expect("the recreated collection must still take DDL after replay"); + assert_eq!(row_count(&node.client, "recreated").await, 4); + + node.shutdown().await; +} + +/// CREATE MATERIALIZED VIEW, DROP, and CREATE again on the same name, all +/// before a graceful restart of the single-node metadata group. The replayed +/// `DeleteMaterializedView` must fence to the first incarnation: the second +/// incarnation's descriptor and refreshed rows must survive replay. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn replayed_drop_materialized_view_keeps_recreated_view() { + let data_dir = tempfile::tempdir().expect("tempdir"); + let data_path = data_dir.path().to_path_buf(); + let node = TestClusterNode::spawn_single_node_calvin_on_path(11, data_path.clone()) + .await + .expect("spawn single-node metadata group"); + wait_for_single_node_ready(&node).await; + + node.client + .simple_query("CREATE COLLECTION mv_src (id BIGINT PRIMARY KEY, amount BIGINT)") + .await + .expect("create source collection"); + node.client + .simple_query("INSERT INTO mv_src (id, amount) VALUES (1, 10)") + .await + .expect("seed source row"); + + node.client + .simple_query( + "CREATE MATERIALIZED VIEW mv_recreated ON mv_src AS SELECT id, amount FROM mv_src", + ) + .await + .expect("create first incarnation of the view"); + node.client + .simple_query("DROP MATERIALIZED VIEW mv_recreated") + .await + .expect("drop first incarnation"); + + node.client + .simple_query( + "CREATE MATERIALIZED VIEW mv_recreated ON mv_src AS SELECT id, amount FROM mv_src", + ) + .await + .expect("create second incarnation of the view"); + node.client + .simple_query("REFRESH MATERIALIZED VIEW mv_recreated") + .await + .expect("refresh second incarnation"); + assert_eq!( + row_count(&node.client, "mv_recreated").await, + 1, + "second incarnation must hold the refreshed row before restart" + ); + + node.graceful_shutdown_wal_only().await; + let node = TestClusterNode::spawn_single_node_calvin_on_path(11, data_path) + .await + .expect("restart against the persisted metadata log"); + wait_for_single_node_ready(&node).await; + + assert!( + node.has_materialized_view(TENANT, "mv_recreated"), + "the second incarnation's descriptor must survive replay of the group-0 log" + ); + assert_eq!( + row_count(&node.client, "mv_recreated").await, + 1, + "replaying the first incarnation's DeleteMaterializedView must not reclaim the \ + second incarnation's rows" + ); + + node.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/descriptor_versioning_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/descriptor_versioning_cross_node.rs index a684ad000..da7afd64f 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/descriptor_versioning_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/descriptor_versioning_cross_node.rs @@ -19,7 +19,7 @@ use crate::common; use std::time::Duration; -use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; +use common::cluster_harness::{TestCluster, wait_for}; const TENANT: u64 = 1; @@ -222,113 +222,3 @@ async fn distinct_collections_get_independent_versions() { cluster.shutdown().await; } - -#[tokio::test(flavor = "multi_thread", worker_threads = 6)] -async fn historical_descriptor_entries_replay_without_regressing_the_latest_version() { - let data_dir = tempfile::tempdir().expect("tempdir"); - let data_path = data_dir.path().to_path_buf(); - let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path.clone()) - .await - .expect("spawn single-node metadata group"); - wait_for( - "single-node sequencer leader elected", - Duration::from_secs(10), - Duration::from_millis(50), - || node.sequencer_leader() == node.node_id, - ) - .await; - wait_for( - "single-node metadata leader elected", - Duration::from_secs(10), - Duration::from_millis(50), - || node.shared.is_metadata_leader(), - ) - .await; - - node.client - .simple_query( - "CREATE COLLECTION replay_graph (id TEXT PRIMARY KEY, name TEXT) \ - WITH (engine='document_strict')", - ) - .await - .expect("create graph-bearing collection"); - wait_for( - "collection descriptor reaches version 1", - Duration::from_secs(10), - Duration::from_millis(50), - || { - node.collection_descriptor(TENANT, "replay_graph") - .map(|v| v.0) - == Some(1) - }, - ) - .await; - - node.client - .simple_query( - "GRAPH INSERT EDGE IN replay_graph FROM 'a' TO 'b' \ - TYPE 'knows' PROPERTIES '{}'", - ) - .await - .expect("insert edge and mark collection edge-bearing"); - wait_for( - "edge-bearing descriptor reaches version 2", - Duration::from_secs(10), - Duration::from_millis(50), - || { - node.collection_descriptor(TENANT, "replay_graph") - .map(|v| v.0) - == Some(2) - }, - ) - .await; - - node.graceful_shutdown_wal_only().await; - let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path) - .await - .expect("restart against the persisted catalog and full metadata log"); - wait_for( - "single-node sequencer leader re-elected after restart", - Duration::from_secs(10), - Duration::from_millis(50), - || node.sequencer_leader() == node.node_id, - ) - .await; - wait_for( - "single-node metadata leader re-elected after restart", - Duration::from_secs(10), - Duration::from_millis(50), - || node.shared.is_metadata_leader(), - ) - .await; - - wait_for( - "latest collection descriptor remains visible after replay", - Duration::from_secs(10), - Duration::from_millis(50), - || { - node.collection_descriptor(TENANT, "replay_graph") - .map(|v| v.0) - == Some(2) - }, - ) - .await; - - node.client - .simple_query( - "CREATE COLLECTION ddl_after_descriptor_replay \ - (id TEXT PRIMARY KEY) WITH (engine='document_strict')", - ) - .await - .expect( - "historical metadata replay must advance its watermark so later DDL remains usable", - ); - assert_eq!( - node.collection_descriptor(TENANT, "replay_graph") - .map(|version| version.0), - Some(2), - "replaying historical version 1 must not overwrite the persisted latest version 2" - ); - - node.shutdown().await; -} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/gateway_execute.rs b/nodedb-cluster-tests/tests/common_suite/cases/gateway_execute.rs index 91c32d7c0..d5c3f5365 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/gateway_execute.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/gateway_execute.rs @@ -34,6 +34,7 @@ fn test_ctx() -> QueryContext { trace_id: nodedb_types::TraceId::ZERO, database_id: nodedb_types::id::DatabaseId::DEFAULT, txn_id: None, + linearizable: false, } } @@ -73,7 +74,7 @@ async fn gateway_execute_kv_put_get_single_node() { key: b"smoke-key".to_vec(), value: mp_string("smoke-value"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"smoke-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, diff --git a/nodedb-cluster-tests/tests/common_suite/cases/graph_match_ryow_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/graph_match_ryow_cross_node.rs index c5af3092b..e61738039 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/graph_match_ryow_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/graph_match_ryow_cross_node.rs @@ -104,9 +104,8 @@ async fn staged_varlen_edge_is_read_your_own_writes_across_cores() { .await .expect("spawn standalone single-node-calvin server"); - // The lone sequencer voter self-elects; wait for it so `calvin_available` is - // genuinely operational (a cross-shard edge is dual-home, not forced - // single-home). + // The lone sequencer voter self-elects; wait for it so the sequencer is + // operational before the cross-shard edge write. wait_for( "single-node sequencer leader elected", Duration::from_secs(10), @@ -116,7 +115,7 @@ async fn staged_varlen_edge_is_read_your_own_writes_across_cores() { .await; assert!( node.shared.cluster_transport.is_some() && node.shared.sequencer_inbox.get().is_some(), - "single-node calvin must wire calvin_available (cluster_transport + sequencer_inbox)" + "single-node calvin must wire cluster_transport and sequencer_inbox" ); node.client diff --git a/nodedb-cluster-tests/tests/common_suite/cases/graph_wcc_core_error_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/graph_wcc_core_error_cross_node.rs new file mode 100644 index 000000000..eb64b42a3 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/graph_wcc_core_error_cross_node.rs @@ -0,0 +1,126 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A graph algorithm fails when one Data-Plane core fails its round. +//! +//! Each node runs 2 cores, and each core holds only the graph nodes its own +//! vShards home to. The fail point `graph::wcc_superstep::core1` fails the +//! WCC round on core 1 of every node. The all-core gather must fail the query +//! with that core's error. Merging core 0 alone returns a partial component +//! set as a complete answer. +//! +//! Requires `--features failpoints`. + +#![cfg(feature = "failpoints")] + +use std::collections::HashSet; +use std::time::Duration; + +use nodedb_types::fail_point::FailGuard; + +use crate::common::cluster_harness::{TestCluster, wait_for}; + +const WCC_SQL: &str = "GRAPH ALGO WCC ON 'gwcc_core_err'"; +const CORE1_FAULT: &str = "graph::wcc_superstep::core1"; +const FAULT_DETAIL: &str = "injected wcc core fault"; +const CHAIN_LEN: usize = 12; + +/// The `node_id` set of a WCC result, or the query's error text. +async fn wcc_nodes(client: &tokio_postgres::Client) -> Result, String> { + let msgs = client + .simple_query(WCC_SQL) + .await + .map_err(|e| match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + })?; + let mut out = HashSet::new(); + for m in &msgs { + if let tokio_postgres::SimpleQueryMessage::Row(r) = m { + out.insert(r.get("node_id").unwrap_or("").to_string()); + } + } + Ok(out) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn wcc_fails_when_one_core_fails() { + let cluster = TestCluster::spawn_three_with_cores(2) + .await + .expect("3-node 2-core cluster"); + + cluster + .exec_ddl_on_any_leader("CREATE COLLECTION gwcc_core_err") + .await + .expect("CREATE COLLECTION gwcc_core_err"); + + wait_for( + "all 3 nodes see gwcc_core_err", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 1) + }, + ) + .await; + + // The chain's names hash to vShards on every node and on both cores. + for i in 0..CHAIN_LEN - 1 { + let src = format!("c_{i}"); + let dst = format!("c_{}", i + 1); + cluster.nodes[0] + .client + .simple_query(&format!( + "GRAPH INSERT EDGE IN 'gwcc_core_err' FROM '{src}' TO '{dst}' TYPE 'K'" + )) + .await + .unwrap_or_else(|e| panic!("insert {src} -> {dst}: {e}")); + } + let chain: HashSet = (0..CHAIN_LEN).map(|i| format!("c_{i}")).collect(); + + for idx in 0..cluster.nodes.len() { + wait_for( + &format!("node {idx} sees all {CHAIN_LEN} wcc nodes"), + Duration::from_secs(30), + Duration::from_millis(200), + || { + tokio::task::block_in_place(|| { + tokio::runtime::Handle::current().block_on(async { + wcc_nodes(&cluster.nodes[idx].client).await.ok() == Some(chain.clone()) + }) + }) + }, + ) + .await; + } + + { + let _fault = FailGuard::fail(CORE1_FAULT, FAULT_DETAIL); + for idx in 0..cluster.nodes.len() { + match wcc_nodes(&cluster.nodes[idx].client).await { + Ok(nodes) => panic!( + "node {idx}: WCC returned {} of {CHAIN_LEN} nodes while core 1 failed; \ + a failed core must fail the query", + nodes.len() + ), + Err(error) => assert!( + error.contains(FAULT_DETAIL), + "node {idx}: the query must fail with the core's own error, got {error}" + ), + } + } + } + + // With the fault cleared, every node returns the full component again. + for idx in 0..cluster.nodes.len() { + assert_eq!( + wcc_nodes(&cluster.nodes[idx].client).await, + Ok(chain.clone()), + "node {idx}: WCC after the fault clears" + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/hilo_surrogate_uniqueness.rs b/nodedb-cluster-tests/tests/common_suite/cases/hilo_surrogate_uniqueness.rs index c72cca998..ff46220b5 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/hilo_surrogate_uniqueness.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/hilo_surrogate_uniqueness.rs @@ -147,10 +147,9 @@ async fn hilo_surrogate_globally_unique_and_disjoint_across_nodes() { insert_batch(&cluster.nodes[1].client, COLLECTION, "n1", KEYS_PER_NODE).await; // ── Step 4: Wait for catalogs to reflect both batches ───────────────────── - // The catalog is written on the inserting node synchronously (before the - // INSERT response is returned), so both nodes should already have their own - // entries. We wait up to 10 s for each node's count to reach the expected - // value to absorb any scheduling jitter. + // Every replica of the collection's home group binds each key when the + // INSERT applies there. We wait up to 10 s for each node's count to reach + // the expected value to absorb apply lag. for (idx, node) in cluster.nodes[..2].iter().enumerate() { let node_prefix = if idx == 0 { "n0" } else { "n1" }; wait_for( @@ -173,11 +172,10 @@ async fn hilo_surrogate_globally_unique_and_disjoint_across_nodes() { let node0_bindings = read_catalog_surrogates(&cluster.nodes[0].shared, COLLECTION); let node1_bindings = read_catalog_surrogates(&cluster.nodes[1].shared, COLLECTION); - // Each node only has its own prefix keys in its local catalog. The surrogate - // assigner writes a catalog row on the node that received the INSERT; binding - // replication is WAL-replay-based, not an immediate Raft write. So node 0 - // holds all `n0_*` bindings and node 1 holds all `n1_*` bindings; we union - // them to form the cluster-wide view. + // The inserting node draws each key's surrogate from its own reserved + // batch and the write carries it; the collection home binds it. The `n0_*` + // keys therefore hold node 0's values and the `n1_*` keys node 1's; we + // union them to form the cluster-wide view. let node0_set: HashSet = node0_bindings .iter() .filter(|(pk, _)| pk.starts_with("n0")) diff --git a/nodedb-cluster-tests/tests/common_suite/cases/http_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/http_gateway_migration.rs index 286c119a4..69fca7e32 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/http_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/http_gateway_migration.rs @@ -32,6 +32,7 @@ fn test_ctx() -> QueryContext { trace_id: nodedb_types::TraceId::ZERO, database_id: nodedb_types::id::DatabaseId::DEFAULT, txn_id: None, + linearizable: false, } } @@ -74,7 +75,7 @@ async fn http_gateway_migration_single_node_query() { key: b"row-1".to_vec(), value: mp_string("hello-http"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"row-1".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -154,7 +155,7 @@ async fn http_gateway_migration_cross_node_query() { key: b"cross-key".to_vec(), value: mp_string("cross-value"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"cross-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -257,6 +258,7 @@ fn http_gateway_error_mapping_not_leader_is_503() { vshard_id: VShardId::new(1), leader_node: 2, leader_addr: "10.0.0.2:9000".into(), + leader_term: 1, }; let (status, _) = GatewayErrorMap::to_http(&err); assert_eq!(status, 503, "NotLeader should map to 503, got {status}"); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/idle_lease_retention.rs b/nodedb-cluster-tests/tests/common_suite/cases/idle_lease_retention.rs new file mode 100644 index 000000000..ea74316ec --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/idle_lease_retention.rs @@ -0,0 +1,116 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! An idle descriptor lease stays granted, and a drain still ends promptly. +//! +//! A statement's lease stays granted after its last holder finishes, so +//! sequential writes to one collection acquire it through Raft once: the +//! lease's grant, and so its expiry, does not change across the writes. A +//! drain start releases the idle lease on the node that holds it, so the +//! drain returns well inside its budget instead of waiting for the expiry. + +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use nodedb_cluster::{DescriptorId, DescriptorKind}; + +use crate::common; +use common::cluster_harness::TestCluster; + +const TENANT: u64 = 1; +const COLLECTION: &str = "idle_lease_docs"; + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn sequential_writes_reuse_one_lease_and_a_drain_releases_it() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLLECTION} (id TEXT PRIMARY KEY, body TEXT) \ + WITH (engine='document_strict')" + )) + .await + .expect("CREATE COLLECTION"); + let node = &cluster.nodes[0]; + + let own_lease = || { + node.leases_for_descriptor(DescriptorKind::Collection, TENANT, COLLECTION) + .into_iter() + .find(|lease| lease.node_id == node.node_id) + }; + + let mut granted = None; + for i in 0..5 { + node.client + .simple_query(&format!( + "INSERT INTO {COLLECTION} (id, body) VALUES ('k{i}', 'v{i}')" + )) + .await + .unwrap_or_else(|e| panic!("INSERT k{i}: {e}")); + let lease = own_lease().expect("the statement's lease stays granted"); + match granted { + None => granted = Some(lease.expires_at), + Some(first) => assert_eq!( + lease.expires_at, first, + "write {i} re-acquired the lease instead of reusing the idle grant" + ), + } + } + assert_eq!( + node.shared.lease_refcount.current(&DescriptorId::new( + 0, + TENANT, + DescriptorKind::Collection, + COLLECTION.to_string(), + )), + 0, + "no statement holds the lease between writes" + ); + + // A drain on the collection releases the idle lease at its start. + let version = node + .shared + .credentials + .catalog() + .get_collection(nodedb_types::DatabaseId::DEFAULT, TENANT, COLLECTION) + .expect("catalog read") + .expect("collection exists") + .descriptor_version + .max(1); + let shared = Arc::clone(&node.shared); + let id = DescriptorId::new( + 0, + TENANT, + DescriptorKind::Collection, + COLLECTION.to_string(), + ); + let drained = { + let started = Instant::now(); + let result = nodedb::control::lease::drain_for_ddl_async( + &shared, + id.clone(), + version, + Duration::from_secs(5), + 0, + ) + .await; + let elapsed = started.elapsed(); + nodedb::control::lease::end_drain_async(&shared, id, nodedb_cluster::DrainOwner::Ddl) + .await + .expect("end the drain"); + (result, elapsed) + }; + assert!( + drained.0.is_ok(), + "the drain must pass the idle lease: {:?}", + drained.0 + ); + assert!( + drained.1 < Duration::from_secs(5), + "the drain waited {:?} for an idle lease", + drained.1 + ); + assert!( + own_lease().is_none(), + "the drain start released the idle lease" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/ilp_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/ilp_gateway_migration.rs index 302dc8842..3eaf974bd 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/ilp_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/ilp_gateway_migration.rs @@ -30,6 +30,7 @@ fn test_ctx() -> QueryContext { trace_id: nodedb_types::TraceId::ZERO, database_id: nodedb_types::id::DatabaseId::DEFAULT, txn_id: None, + linearizable: false, } } @@ -209,6 +210,7 @@ fn ilp_gateway_error_not_leader_is_moved() { vshard_id: VShardId::new(1), leader_node: 2, leader_addr: "10.0.0.2:9000".into(), + leader_term: 1, }; let msg = GatewayErrorMap::to_resp(&err); assert!( diff --git a/nodedb-cluster-tests/tests/common_suite/cases/install_snapshot_edge_endpoints.rs b/nodedb-cluster-tests/tests/common_suite/cases/install_snapshot_edge_endpoints.rs new file mode 100644 index 000000000..6a76aea4e --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/install_snapshot_edge_endpoints.rs @@ -0,0 +1,329 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A learner caught up by InstallSnapshot resolves every edge endpoint. +//! +//! A key's surrogate is minted once, at its collection home `G_c`, and every +//! `G_c` replica binds it. In a collection that holds edges the bind also +//! lives on the group of `from_key(endpoint)`: a live edge write carries each +//! endpoint's surrogate, and every replica of the endpoint's group binds it on +//! apply. The snapshot of an endpoint's group must carry that bind, or the +//! caught-up node holds edges whose endpoints it cannot resolve. +//! +//! The test fails a snapshot that ships binds by collection home only. On 3 nodes with RF 2 and [`GROUPS`] data groups it picks a collection +//! whose group `G_c` the learner's 4-node placement leaves out, and an +//! endpoint group `G_e` that placement names the learner in. It writes edges +//! whose endpoints all home on `G_e`, then adds the learner. The learner hosts +//! no `G_c` replica. +//! +//! An edge write runs as a Calvin transaction: the sequencer log orders it, +//! and each endpoint home applies it from its scheduler, outside its data +//! group's log. Before the learner joins, the test therefore compacts two +//! logs past the edge writes: `G_e`'s, with filler documents of a collection +//! homed on `G_e`, and the sequencer's, with filler edges. The learner then +//! can learn the edges and binds only from a `G_e` snapshot, which a +//! collection-home-only snapshot ships without any of the binds. The learner's local `G_e` snapshot +//! index at or above `G_e`'s index after the edge writes proves it caught up +//! by snapshot install. + +use std::time::Duration; + +use nodedb::types::{DatabaseId, TenantId}; +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_types::CollectionKey; + +use crate::common::cluster_harness::shared_steps::{ + db_detail, group_members, group_of_key, group_status, +}; +use crate::common::cluster_harness::{TestCluster, TestClusterNode, wait_for, wait_for_async}; + +const COMPACTION_THRESHOLD: u64 = 4; +const RF: usize = 2; +/// Data groups of the cluster. +const GROUPS: u64 = 4; +/// Node ids after the learner joins. +const NODES_WITH_LEARNER: [u64; 4] = [1, 2, 3, 4]; +/// Prefix of the candidate collection names. The test picks the first one +/// whose group the learner never joins. +const COLLECTION_PREFIX: &str = "snap_edge_ep_"; +/// Prefix of the candidate filler collection names. The test picks the first +/// one homed on the endpoint group. +const FILL_PREFIX: &str = "snap_edge_fill_"; +const PAIRS: usize = 6; +const MAX_FILLER: usize = 400; +const TENANT: u64 = 1; + +/// The surrogate `node` binds `endpoint` to in its own catalog. +fn local_bind(node: &TestClusterNode, collection: &str, endpoint: &str) -> Option { + node.shared + .surrogate_assigner + .lookup_bound( + CollectionKey::from_bare(DatabaseId::DEFAULT, collection), + TenantId::new(TENANT), + endpoint.as_bytes(), + ) + .ok() + .flatten() + .map(|s| s.as_u32()) +} + +/// `count` node keys homed on `group_id`, named `{prefix}{i}`. +fn keys_on_group(node: &TestClusterNode, group_id: u64, prefix: &str, count: usize) -> Vec { + (0..1_000_000) + .map(|i| format!("{prefix}{i}")) + .filter(|k| group_of_key(node, k) == group_id) + .take(count) + .collect() +} + +async fn insert_edge(node: &TestClusterNode, collection: &str, src: &str, dst: &str) { + node.client + .simple_query(&format!( + "GRAPH INSERT EDGE IN '{collection}' FROM '{src}' TO '{dst}' TYPE 'l'" + )) + .await + .unwrap_or_else(|e| panic!("insert {src} -> {dst}: {}", db_detail(&e))); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn learner_caught_up_by_snapshot_resolves_every_edge_endpoint() { + let mut cluster = TestCluster::spawn_three_with_groups_compaction_threshold_and_rf( + GROUPS, + COMPACTION_THRESHOLD, + RF, + ) + .await + .expect("3-node cluster with 4 groups, low compaction threshold and rf=2"); + + // The 4-node placement the learner's join leads to, from the same rules + // the cluster runs. + let view = &cluster.nodes[0]; + let data_groups: Vec = (1..=GROUPS).collect(); + let placement_with_learner = nodedb_cluster::rebalancer::placement::compute_placement( + &NODES_WITH_LEARNER, + &data_groups, + RF as u32, + ); + let learner_id = NODES_WITH_LEARNER[3]; + let learner_joins = |g: u64| { + placement_with_learner + .get(&g) + .is_some_and(|p| p.contains(&learner_id)) + }; + // G_c must not place the learner: a member of G_c binds every endpoint + // of the collection through G_c, and the test cannot fail on a + // collection-home-only snapshot. G_e must place it, so it catches up on G_e by snapshot. + let (collection, collection_group) = (0..10_000) + .map(|i| format!("{COLLECTION_PREFIX}{i}")) + .find_map(|name| { + let group = view.group_id_for_collection(&name)?; + (!learner_joins(group)).then_some((name, group)) + }) + .expect("a collection name whose group the learner never joins"); + let endpoint_group = data_groups + .iter() + .copied() + .find(|g| *g != collection_group && learner_joins(*g)) + .expect("placement offers an endpoint group the learner joins"); + let endpoint_members = group_members(view, endpoint_group); + // A document collection homed on G_e. Its single-shard autocommit + // inserts go through G_e's Raft log and grow it. + let fill_collection = (0..10_000) + .map(|i| format!("{FILL_PREFIX}{i}")) + .find(|name| view.group_id_for_collection(name) == Some(endpoint_group)) + .expect("a filler collection name homed on the endpoint group"); + + for name in [&collection, &fill_collection] { + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {name}")) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION {name}: {e}")); + } + wait_for( + "all nodes see both collections", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 2) + }, + ) + .await; + + // Edges whose endpoints all home on G_e, written from a G_e member. + let writer = cluster + .nodes + .iter() + .find(|n| endpoint_members.contains(&n.node_id)) + .expect("a member of the endpoint group"); + let endpoints = keys_on_group(view, endpoint_group, "ep_", PAIRS * 2); + assert_eq!( + endpoints.len(), + PAIRS * 2, + "enough keys home on the endpoint group" + ); + for pair in endpoints.chunks(2) { + insert_edge(writer, &collection, &pair[0], &pair[1]).await; + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + let edges_applied = cluster + .nodes + .iter() + .filter_map(|n| group_status(n, endpoint_group)) + .map(|s| s.last_applied) + .max() + .expect("the endpoint group is hosted"); + // The collection home mints every key's surrogate, so each G_c replica + // binds every endpoint. + let collection_members = group_members(view, collection_group); + for ep in &endpoints { + for node in cluster + .nodes + .iter() + .filter(|n| collection_members.contains(&n.node_id)) + { + assert!( + local_bind(node, &collection, ep).is_some(), + "G_c replica {} binds {ep}: the collection home mints it", + node.node_id + ); + } + } + + // An edge write runs as a Calvin transaction: it lands in the sequencer + // log, and each endpoint home applies it from its scheduler, not from + // its data group's log. A node that later joins G_e can learn the edges + // two ways: from a G_e snapshot, or by replaying the sequencer log for + // G_e's vShards. The test must leave only the snapshot, so it compacts + // both logs past the edge writes: + // - G_e's log, with filler documents homed on G_e (single-shard + // autocommit writes, which go through the data group); + // - the sequencer log, with filler edges (Calvin transactions). + let sequencer_after_edges = cluster + .nodes + .iter() + .filter_map(|n| group_status(n, SEQUENCER_GROUP_ID)) + .map(|s| s.last_applied) + .max() + .expect("the sequencer group is hosted"); + let filler = keys_on_group(view, endpoint_group, "fill_", MAX_FILLER * 2); + // Every voter of `group_id` compacted past `index`. A node that left the + // group can still host its replica until it unmounts it. That replica + // gets no entries and never compacts, and it never sends the learner a + // snapshot, so it is not part of the premise. + let voters_compacted = |cluster: &TestCluster, group_id: u64, index: u64| { + let voters: Vec = match group_members(&cluster.nodes[0], group_id) { + voters if !voters.is_empty() => voters, + // A group the routing table does not list: every host votes. + _ => cluster + .nodes + .iter() + .filter(|n| group_status(n, group_id).is_some()) + .map(|n| n.node_id) + .collect(), + }; + !voters.is_empty() + && voters.iter().all(|voter| { + cluster + .nodes + .iter() + .find(|n| n.node_id == *voter) + .and_then(|n| group_status(n, group_id)) + .is_some_and(|s| s.snapshot_index >= index) + }) + }; + let compacted = |cluster: &TestCluster| { + voters_compacted(cluster, endpoint_group, edges_applied) + && voters_compacted(cluster, SEQUENCER_GROUP_ID, sequencer_after_edges) + }; + // Every node's view of a group, for the failure message. + let replicas = |cluster: &TestCluster, group_id: u64| -> Vec<(u64, u64, u64)> { + cluster + .nodes + .iter() + .filter_map(|n| { + group_status(n, group_id).map(|s| (n.node_id, s.snapshot_index, s.last_applied)) + }) + .collect() + }; + let mut written = 0; + for (i, pair) in filler.chunks(2).enumerate() { + if compacted(&cluster) { + break; + } + writer + .client + .simple_query(&format!( + "INSERT INTO {fill_collection} {{ id: 'f{i}', n: {i} }}" + )) + .await + .unwrap_or_else(|e| panic!("filler document f{i}: {}", db_detail(&e))); + insert_edge(writer, &collection, &pair[0], &pair[1]).await; + written += 1; + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + assert!( + compacted(&cluster), + "G_e voters compacted past {edges_applied} and sequencer voters past \ + {sequencer_after_edges} after {written} filler rounds; (node, snapshot_index, \ + last_applied) of G_e: {:?}, of the sequencer: {:?}", + replicas(&cluster, endpoint_group), + replicas(&cluster, SEQUENCER_GROUP_ID) + ); + let expected: Vec<(String, u32)> = endpoints + .iter() + .map(|ep| { + let s = cluster + .nodes + .iter() + .find_map(|n| local_bind(n, &collection, ep)) + .unwrap_or_else(|| panic!("a G_e replica binds {ep}")); + (ep.clone(), s) + }) + .collect(); + + let learner_id = { + let joined = cluster + .add_learner_node() + .await + .expect("add learner node") + .node_id; + assert_eq!(joined, learner_id, "the learner joins as the placed node"); + joined + }; + let learner = cluster + .nodes + .iter() + .find(|n| n.node_id == learner_id) + .expect("learner present"); + + wait_for_async( + "the learner installs a G_e snapshot covering the edge writes", + Duration::from_secs(30), + Duration::from_millis(100), + || async move { + group_status(learner, endpoint_group).is_some_and(|s| s.snapshot_index >= edges_applied) + }, + ) + .await; + assert!( + group_status(learner, collection_group).is_none(), + "the learner hosts G_c {collection_group}; its binds could come from G_c and the old \ + rule would pass" + ); + + for (ep, s) in &expected { + assert_eq!( + local_bind(learner, &collection, ep), + Some(*s), + "learner {learner_id} resolves endpoint {ep} from the G_e snapshot" + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/install_snapshot_multi_core.rs b/nodedb-cluster-tests/tests/common_suite/cases/install_snapshot_multi_core.rs new file mode 100644 index 000000000..a8f682664 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/install_snapshot_multi_core.rs @@ -0,0 +1,291 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A learner caught up by a real Raft `InstallSnapshot` on a multi-core node. +//! +//! - It holds every row on the core its collection routes to. Reads route to +//! one owning core per collection, so a snapshot installed onto a single +//! core leaves every collection homed elsewhere invisible to reads. +//! - Its install survives a restart together with the writes made after it. +//! A boot that re-applies the installed snapshot erases the later +//! writes and brings back the rows they deleted. +//! +//! Both tests spread rows over collections homed on several cores, and force +//! the learner onto the snapshot path by compacting the leader's logs. + +use std::collections::{HashMap, HashSet}; +use std::time::{Duration, Instant}; + +use nodedb::types::TenantId; + +use crate::common::cluster_harness::shared_steps::key_collection; +use crate::common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +const COMPACTION_THRESHOLD: u64 = 4; +const CORES: usize = 4; +const COLLECTIONS: usize = 6; +const ROWS: usize = 12; +const POST_ROWS: usize = 6; + +fn collection(i: usize) -> String { + format!("snap_mc_{i}") +} + +async fn count_rows(client: &tokio_postgres::Client, table: &str) -> Option { + let rows = client + .simple_query(&format!("SELECT COUNT(*) FROM {table}")) + .await + .ok()?; + rows.iter().find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => r.get(0).and_then(|s| s.parse().ok()), + _ => None, + }) +} + +async fn exec_on_any(cluster: &TestCluster, sql: &str) { + cluster.nodes[0] + .client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); +} + +/// A 3-node cluster whose collections' groups compacted, plus a learner caught +/// up by snapshot. Returns the cluster, the learner's node id, and the groups. +async fn cluster_with_snapshot_learner() -> (TestCluster, u64, HashSet) { + let mut cluster = TestCluster::spawn_three_with_compaction_threshold_rf_and_cores( + COMPACTION_THRESHOLD, + 4, + CORES, + ) + .await + .expect("3-node cluster, rf=4, 4 cores per node"); + + for i in 0..COLLECTIONS { + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {} (id TEXT PRIMARY KEY, payload TEXT) \ + WITH (engine='document_strict')", + collection(i) + )) + .await + .expect("CREATE COLLECTION"); + } + wait_for( + "all nodes see the collections", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= COLLECTIONS) + }, + ) + .await; + + for i in 0..COLLECTIONS { + for r in 0..ROWS { + exec_on_any( + &cluster, + &format!( + "INSERT INTO {} (id, payload) VALUES ('r{r}', 'v{r}')", + collection(i) + ), + ) + .await; + } + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + + // Every group the collections use compacted, so the learner can only be + // caught up by a snapshot. + let groups: HashSet = (0..COLLECTIONS) + .map(|i| { + cluster.nodes[0] + .group_id_for_collection(&collection(i)) + .expect("collection maps to a data group") + }) + .collect(); + for gid in &groups { + let compacted = cluster + .nodes + .iter() + .map(|n| n.group_snapshot_index(*gid)) + .max() + .unwrap_or(0); + assert!(compacted > 0, "group {gid} must compact before the join"); + } + + let learner_id = cluster + .add_learner_node() + .await + .expect("add learner node") + .node_id; + { + let learner = node(&cluster, learner_id); + wait_for( + "the learner installs a snapshot of every group", + Duration::from_secs(30), + Duration::from_millis(100), + || { + groups.iter().all(|gid| { + learner.hosts_data_group(*gid) + && learner.local_snapshot_index_for_group(*gid) > 0 + }) + }, + ) + .await; + assert_eq!(learner.num_cores(), CORES); + } + (cluster, learner_id, groups) +} + +fn node(cluster: &TestCluster, node_id: u64) -> &TestClusterNode { + cluster + .nodes + .iter() + .find(|n| n.node_id == node_id) + .expect("node present") +} + +/// Each test collection's home core on `node`, checked to span a core other +/// than the one a single-core install would pick. +fn homes(node: &TestClusterNode) -> HashMap { + let system_core = node.home_core_of("__system"); + let homes: HashMap = (0..COLLECTIONS) + .map(|i| (collection(i), node.home_core_of(&collection(i)))) + .collect(); + assert!( + homes.values().any(|core| *core != system_core), + "test collections must home on a core other than core {system_core}" + ); + homes +} + +fn tenants(node: &TestClusterNode, homes: &HashMap) -> HashSet { + node.shared + .credentials + .catalog() + .load_all_collections_across_databases() + .expect("catalog read") + .into_iter() + .filter(|c| homes.contains_key(&c.name)) + .map(|c| c.tenant_id) + .collect() +} + +/// Rows per test collection on its owning core of `node`. Panics on a row +/// stored on any other core. +async fn owned_rows( + node: &TestClusterNode, + homes: &HashMap, + tenants: &HashSet, +) -> HashMap { + let mut per_collection: HashMap = HashMap::new(); + let mut misplaced: Vec = Vec::new(); + for core in 0..CORES { + for tenant in tenants { + for key in node + .document_keys_on_core(core, TenantId::new(*tenant)) + .await + { + let Some(coll) = key_collection(&key) else { + continue; + }; + let Some(home) = homes.get(coll) else { + continue; + }; + if *home == core { + *per_collection.entry(coll.to_string()).or_default() += 1; + } else { + misplaced.push(format!("{coll} on core {core}, home {home}")); + } + } + } + } + assert!( + misplaced.is_empty(), + "rows off their owning core: {misplaced:?}" + ); + per_collection +} + +/// Poll until every test collection holds exactly `expected` rows on its +/// owning core of `node`. +async fn await_owned_rows(node: &TestClusterNode, expected: usize, what: &str) { + let homes = homes(node); + let tenants = tenants(node, &homes); + let deadline = Instant::now() + Duration::from_secs(30); + loop { + let per_collection = owned_rows(node, &homes, &tenants).await; + if homes + .keys() + .all(|c| per_collection.get(c).copied().unwrap_or(0) == expected) + { + return; + } + assert!( + Instant::now() < deadline, + "{what}: owning cores never held {expected} rows per collection: {per_collection:?}" + ); + tokio::time::sleep(Duration::from_millis(200)).await; + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn learner_snapshot_install_places_rows_on_owning_cores() { + let (cluster, learner_id, _) = cluster_with_snapshot_learner().await; + let learner = node(&cluster, learner_id); + + await_owned_rows(learner, ROWS, "after the install").await; + + // A routed read on the learner sees every row of every collection. + for i in 0..COLLECTIONS { + let coll = collection(i); + let deadline = Instant::now() + Duration::from_secs(30); + loop { + if count_rows(&learner.client, &coll).await == Some(ROWS) { + break; + } + assert!( + Instant::now() < deadline, + "learner COUNT(*) of {coll} never reached {ROWS}" + ); + tokio::time::sleep(Duration::from_millis(100)).await; + } + } + + cluster.shutdown().await; +} + +/// Writes after the install survive a restart of the installed node, and a +/// row deleted after the install stays deleted. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn post_install_writes_survive_restart() { + let (cluster, learner_id, _) = cluster_with_snapshot_learner().await; + await_owned_rows(node(&cluster, learner_id), ROWS, "after the install").await; + + for i in 0..COLLECTIONS { + let coll = collection(i); + for p in 0..POST_ROWS { + exec_on_any( + &cluster, + &format!("INSERT INTO {coll} (id, payload) VALUES ('p{p}', 'w{p}')"), + ) + .await; + } + exec_on_any(&cluster, &format!("DELETE FROM {coll} WHERE id = 'r0'")).await; + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + let expected = ROWS + POST_ROWS - 1; + await_owned_rows(node(&cluster, learner_id), expected, "before the restart").await; + + let cluster = cluster.restart_all().await.expect("restart every node"); + await_owned_rows(node(&cluster, learner_id), expected, "after the restart").await; + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/install_snapshot_multi_core_fault.rs b/nodedb-cluster-tests/tests/common_suite/cases/install_snapshot_multi_core_fault.rs new file mode 100644 index 000000000..0f73d47cc --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/install_snapshot_multi_core_fault.rs @@ -0,0 +1,159 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A snapshot install that one core fails reports a typed, retryable error, +//! and a re-install of the same bytes converges to every row on its owning +//! core. +//! +//! The fail point `snapshot_install::node::core1` makes core 1 of node N +//! report its share as failed after it installed it: the other core holds the +//! snapshot, this one does not count as settled. The re-install clears every +//! core before it installs, so no row lands twice or off its core. +//! +//! Requires `--features failpoints`. + +#![cfg(feature = "failpoints")] + +use std::collections::HashMap; +use std::time::Duration; + +use nodedb::control::cluster::snapshot_applier::DataPlaneSnapshotApplier; +use nodedb::control::cluster::snapshot_builder::DataPlaneSnapshotBuilder; +use nodedb::control::cluster::snapshot_install::SnapshotInstallError; +use nodedb::types::TenantId; +use nodedb_cluster::SnapshotBuilder; +use nodedb_types::fail_point::FailGuard; + +use crate::common::cluster_harness::shared_steps::key_collection; +use crate::common::cluster_harness::{TestCluster, wait_for}; + +const CORES: usize = 2; +const COLLECTIONS: usize = 8; +const ROWS: usize = 5; + +fn collection(i: usize) -> String { + format!("snap_fault_{i}") +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn failed_core_install_is_retryable_and_reinstall_converges() { + let cluster = TestCluster::spawn_three_with_cores(CORES) + .await + .expect("3-node 2-core cluster"); + + for i in 0..COLLECTIONS { + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {} (id TEXT PRIMARY KEY, payload TEXT) \ + WITH (engine='document_strict')", + collection(i) + )) + .await + .expect("CREATE COLLECTION"); + } + wait_for( + "all nodes see the collections", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= COLLECTIONS) + }, + ) + .await; + for i in 0..COLLECTIONS { + for r in 0..ROWS { + cluster.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {} (id, payload) VALUES ('r{r}', 'v{r}')", + collection(i) + )) + .await + .unwrap_or_else(|e| panic!("insert {} r{r}: {e}", collection(i))); + } + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + + let source = &cluster.nodes[0]; + let target = &cluster.nodes[1]; + let gid = source + .group_id_for_collection(&collection(0)) + .expect("collection maps to a data group"); + let in_group: Vec = (0..COLLECTIONS) + .map(collection) + .filter(|c| source.group_id_for_collection(c) == Some(gid)) + .collect(); + + let bytes = DataPlaneSnapshotBuilder::new(source.shared.clone()) + .build_group_snapshot(gid, 0, 0) + .await + .expect("build the group snapshot") + .bytes; + assert!(!bytes.is_empty(), "group {gid} snapshot carries data"); + + let applier = DataPlaneSnapshotApplier::new(target.shared.clone()); + { + let _fault = FailGuard::fail( + &format!("snapshot_install::node{}::core1", target.node_id), + "injected core install fault", + ); + let err = applier + .install(gid, &bytes) + .await + .expect_err("a failed core must fail the install"); + assert!( + matches!(err, SnapshotInstallError::CoreInstall { core_id: 1, .. }), + "unexpected error: {err}" + ); + assert!(err.is_retryable(), "a core install error is retryable"); + } + + applier + .install(gid, &bytes) + .await + .expect("the re-install converges"); + + let homes: HashMap = in_group + .iter() + .map(|c| (c.clone(), target.home_core_of(c))) + .collect(); + let tenant = target + .shared + .credentials + .catalog() + .load_all_collections_across_databases() + .expect("catalog read") + .into_iter() + .find(|c| c.name == in_group[0]) + .expect("collection in catalog") + .tenant_id; + + let mut per_collection: HashMap = HashMap::new(); + for core in 0..CORES { + for key in target + .document_keys_on_core(core, TenantId::new(tenant)) + .await + { + let Some(coll) = key_collection(&key) else { + continue; + }; + if let Some(home) = homes.get(coll) { + assert_eq!(*home, core, "{coll} row on core {core}, home {home}"); + *per_collection.entry(coll.to_string()).or_default() += 1; + } + } + } + for coll in &in_group { + assert_eq!( + per_collection.get(coll).copied().unwrap_or(0), + ROWS, + "{coll} rows after the re-install" + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/kafka_sink_single_owner.rs b/nodedb-cluster-tests/tests/common_suite/cases/kafka_sink_single_owner.rs new file mode 100644 index 000000000..756949e5f --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/kafka_sink_single_owner.rs @@ -0,0 +1,206 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A Kafka sink publishes every event once, from one node, across a +//! failover. +//! +//! The broker is librdkafka's in-process mock cluster (`rdkafka::mocking`). +//! Every node runs a producer task for the stream, but only the node that +//! holds the leader lease of the stream's owning group publishes, and it +//! commits the group's offsets through the metadata log. So: +//! +//! - every event reaches the topic exactly once while the cluster is stable; +//! - after the publishing node dies, the next lease holder resumes from the +//! committed offsets: every later event arrives once, and no earlier one +//! arrives again; +//! - every record carries the owner's fencing token, and the next owner's +//! token is higher. + +use crate::common; +use common::cluster_harness::{TestCluster, wait_for}; + +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use rdkafka::config::ClientConfig; +use rdkafka::consumer::{Consumer, StreamConsumer}; +use rdkafka::message::{Headers, Message}; +use rdkafka::mocking::MockCluster; + +use nodedb::event::cdc::sink_owner::owning_group; +use nodedb_types::DatabaseId; + +use super::webhook_sink_single_owner::{Deliveries, FencingTokens, distinct, duplicates}; + +const COLLECTION: &str = "kafka_owner_rows"; +const STREAM: &str = "kafka_owner_feed"; +const TOPIC: &str = "kafka_owner_topic"; +const TENANT: u64 = 1; +const BEFORE: usize = 5; +const AFTER: usize = 4; +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(100); + +/// Read `TOPIC` from the start, counting each record's deliveries by key and +/// recording each record's fencing token. +async fn consume(consumer: StreamConsumer, deliveries: Deliveries, tokens: FencingTokens) { + loop { + let Ok(message) = consumer.recv().await else { + continue; + }; + if let Some(token) = message.headers().and_then(|headers| { + headers + .iter() + .find(|header| header.key == "fencing-token") + .and_then(|header| header.value) + .and_then(|value| std::str::from_utf8(value).ok()) + .and_then(|value| value.parse::().ok()) + }) { + tokens.lock().unwrap_or_else(|p| p.into_inner()).push(token); + } + if let Some(key) = message.key().and_then(|key| std::str::from_utf8(key).ok()) { + *deliveries + .lock() + .unwrap_or_else(|p| p.into_inner()) + .entry(key.to_owned()) + .or_default() += 1; + } + } +} + +async fn insert(cluster: &TestCluster, row: usize) { + let sql = format!("INSERT INTO {COLLECTION} {{ id: 'row-{row}', n: {row} }}"); + let deadline = Instant::now() + CONVERGE; + loop { + match cluster.nodes[0].client.simple_query(&sql).await { + Ok(_) => return, + Err(error) if Instant::now() < deadline => { + tracing::debug!(row, %error, "insert not accepted yet; retrying"); + tokio::time::sleep(Duration::from_millis(200)).await; + } + Err(error) => panic!("insert row-{row}: {error}"), + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_kafka_sink_publishes_every_event_once_across_a_failover() { + let broker = MockCluster::new(1).expect("start the in-process Kafka mock broker"); + broker + .create_topic(TOPIC, 1, 1) + .expect("create the mock topic"); + let bootstrap = broker.bootstrap_servers(); + + let consumer: StreamConsumer = ClientConfig::new() + .set("bootstrap.servers", &bootstrap) + .set("group.id", "kafka_owner_test") + .set("auto.offset.reset", "earliest") + .create() + .expect("create the test consumer"); + consumer + .subscribe(&[TOPIC]) + .expect("subscribe to the topic"); + let deliveries: Deliveries = Arc::default(); + let tokens: FencingTokens = Arc::default(); + let reader = tokio::spawn(consume( + consumer, + Arc::clone(&deliveries), + Arc::clone(&tokens), + )); + + let mut cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {COLLECTION}")) + .await + .expect("create collection"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE CHANGE STREAM {STREAM} ON {COLLECTION} \ + WITH (DELIVERY = 'kafka', BROKERS = '{bootstrap}', TOPIC = '{TOPIC}')" + )) + .await + .expect("create change stream"); + wait_for("every node registers the stream", CONVERGE, STEP, || { + cluster + .nodes + .iter() + .all(|node| node.has_change_stream(DatabaseId::DEFAULT, TENANT, STREAM)) + }) + .await; + + for row in 0..BEFORE { + insert(&cluster, row).await; + } + wait_for("the topic receives every event", CONVERGE, STEP, || { + distinct(&deliveries) == BEFORE + }) + .await; + // Let a stray second publish surface before counting. + tokio::time::sleep(Duration::from_secs(2)).await; + assert!( + duplicates(&deliveries).is_empty(), + "an event was published twice while the cluster was stable: {:?}", + duplicates(&deliveries) + ); + let first_owner_tokens = tokens.lock().unwrap_or_else(|p| p.into_inner()).clone(); + assert_eq!( + first_owner_tokens.len(), + BEFORE, + "every record carries a fencing token" + ); + + // Kill the node that publishes: the leader of the stream's owning group. + let group = owning_group(&cluster.nodes[0].shared, DatabaseId::DEFAULT, STREAM) + .expect("the stream's name maps to a data group"); + let leader = cluster.nodes[0] + .all_group_leaders() + .into_iter() + .find_map(|(id, leader)| (id == group).then_some(leader)) + .unwrap_or(0); + let owner = cluster + .nodes + .iter() + .position(|node| node.node_id == leader) + .unwrap_or_else(|| panic!("no live node leads the owning group {group}")); + let dead = cluster.nodes.remove(owner); + let dead_id = dead.node_id; + dead.shutdown().await; + wait_for("the survivors elect a new owner", CONVERGE, STEP, || { + cluster.nodes.iter().all(|node| { + node.all_group_leaders() + .into_iter() + .any(|(id, leader)| id == group && leader != 0 && leader != dead_id) + }) + }) + .await; + + for row in BEFORE..BEFORE + AFTER { + insert(&cluster, row).await; + } + wait_for( + "the new owner publishes the later events", + CONVERGE, + STEP, + || distinct(&deliveries) == BEFORE + AFTER, + ) + .await; + tokio::time::sleep(Duration::from_secs(2)).await; + assert!( + duplicates(&deliveries).is_empty(), + "the failover published an event twice: {:?}", + duplicates(&deliveries) + ); + let all_tokens = tokens.lock().unwrap_or_else(|p| p.into_inner()).clone(); + let first_owner_max = first_owner_tokens.iter().copied().max().unwrap_or(0); + assert!( + all_tokens[first_owner_tokens.len()..] + .iter() + .all(|token| *token > first_owner_max), + "the new owner's fencing token rises above the old owner's: {all_tokens:?}" + ); + + reader.abort(); + cluster.shutdown().await; + drop(broker); +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/linearizable_read_leadership.rs b/nodedb-cluster-tests/tests/common_suite/cases/linearizable_read_leadership.rs index b16f78fb3..b40165ab3 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/linearizable_read_leadership.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/linearizable_read_leadership.rs @@ -12,7 +12,11 @@ //! one that can show how far behind the leader it is. `eventual` accepts any //! replica at all, so it keeps being served when nothing else is. All three //! are asserted: one alone cannot tell a working guarantee apart from a read -//! path that is simply broken. +//! path that is broken. +//! +//! A leader can answer from its lease without asking the quorum. The lease +//! ends `election_timeout_min` minus a drift margin after the quorum last +//! answered, so an isolated leader must refuse once that has passed. use crate::common; use common::cluster_harness::{TestCluster, wait::wait_for}; @@ -21,6 +25,15 @@ use std::time::Duration; const COLLECTION: &str = "linread"; +/// `election_timeout_min` of the harness cluster tuning. A leader lease never +/// outlives it. +const ELECTION_TIMEOUT_MIN: Duration = Duration::from_millis(500); + +/// How long an isolated leader waits before the strong read that must be +/// refused. A lease ends within one `ELECTION_TIMEOUT_MIN`. Three give a loaded +/// test host room for scheduling delay. +const LEASE_EXPIRY_MARGIN: u32 = 3; + /// Bring up three nodes holding one row. async fn seeded_cluster() -> TestCluster { let cluster = TestCluster::spawn_three() @@ -186,3 +199,120 @@ async fn a_replica_in_contact_serves_a_bounded_staleness_read() { cluster.shutdown().await; } + +/// Index of the node that leads the most Raft groups. +fn busiest_leader_index(cluster: &TestCluster) -> usize { + (0..cluster.nodes.len()) + .max_by_key(|&i| { + let node = &cluster.nodes[i]; + node.all_group_leaders() + .iter() + .filter(|&&(_, leader)| leader == node.node_id) + .count() + }) + .unwrap_or(0) +} + +/// `val` of the first row in `messages`. +fn first_val(messages: &[tokio_postgres::SimpleQueryMessage]) -> Option { + messages.iter().find_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get("val").map(str::to_string), + _ => None, + }) +} + +/// A leader cut off from its quorum serves from its lease only until the +/// lease ends. Past it the followers can already have elected a successor, so +/// the next strong read must be refused, with no retry. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn an_isolated_leader_refuses_strong_reads_once_its_lease_expires() { + let mut cluster = seeded_cluster().await; + let old = busiest_leader_index(&cluster); + + let select = format!("SELECT val FROM {COLLECTION} WHERE id = 'a'"); + wait_for( + "strong read served by the leader while the quorum is healthy", + Duration::from_secs(15), + Duration::from_millis(200), + || { + tokio::task::block_in_place(|| { + tokio::runtime::Handle::current() + .block_on(cluster.nodes[old].client.simple_query(&select)) + }) + .is_ok() + }, + ) + .await; + + let isolated = cluster.nodes.remove(old); + for node in cluster.nodes.drain(..) { + node.shutdown().await; + } + // No quorum has answered since the shutdowns finished, so every lease + // anchor predates this point. + tokio::time::sleep(ELECTION_TIMEOUT_MIN * LEASE_EXPIRY_MARGIN).await; + + let read = isolated.client.simple_query(&select).await; + assert!( + read.is_err(), + "an isolated leader served a strong read past its lease: {read:?}" + ); + + isolated.shutdown().await; +} + +/// A write the new leader acknowledged is visible to every strong read that +/// is served after it. A read can be refused while routing settles on the new +/// leader. It must never return the value from before the write. +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_write_acknowledged_by_the_new_leader_is_visible_to_the_next_strong_read() { + let mut cluster = seeded_cluster().await; + let old = busiest_leader_index(&cluster); + let old_leader = cluster.nodes.remove(old); + old_leader.shutdown().await; + + let update = format!("UPDATE {COLLECTION} SET val = 'w' WHERE id = 'a'"); + wait_for( + "write acknowledged by the new leader", + Duration::from_secs(20), + Duration::from_millis(250), + || { + tokio::task::block_in_place(|| { + tokio::runtime::Handle::current() + .block_on(cluster.nodes[0].client.simple_query(&update)) + }) + .is_ok() + }, + ) + .await; + + let select = format!("SELECT val FROM {COLLECTION} WHERE id = 'a'"); + for node in &cluster.nodes { + let deadline = tokio::time::Instant::now() + Duration::from_secs(15); + loop { + match node.client.simple_query(&select).await { + Ok(messages) => { + assert_eq!( + first_val(&messages).as_deref(), + Some("w"), + "node {} served a strong read that misses an acknowledged write", + node.node_id + ); + break; + } + Err(error) => { + assert!( + tokio::time::Instant::now() < deadline, + "node {} never served the strong read: {error}", + node.node_id + ); + tokio::time::sleep(Duration::from_millis(200)).await; + } + } + } + } + + for node in cluster.nodes.drain(..) { + node.shutdown().await; + } +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/listeners_gateway_smoke.rs b/nodedb-cluster-tests/tests/common_suite/cases/listeners_gateway_smoke.rs index 74c3a0104..e5c1f8c56 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/listeners_gateway_smoke.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/listeners_gateway_smoke.rs @@ -37,6 +37,7 @@ fn test_ctx(_trace_id: u64) -> QueryContext { trace_id: nodedb_types::TraceId::ZERO, database_id: nodedb_types::id::DatabaseId::DEFAULT, txn_id: None, + linearizable: false, } } @@ -75,7 +76,7 @@ async fn pgwire_gateway_smoke_cache_hit() { key: b"pgwire-smoke-key".to_vec(), value: mp_string("pgwire-smoke-val"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"pgwire-smoke-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -146,7 +147,7 @@ async fn http_gateway_smoke_cache_hit() { key: b"http-smoke-key".to_vec(), value: mp_string("http-smoke-val"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"http-smoke-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -211,7 +212,7 @@ async fn resp_gateway_smoke_cache_hit() { key: b"resp-smoke-key".to_vec(), value: mp_string("resp-smoke-val"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"resp-smoke-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -279,7 +280,7 @@ async fn ilp_gateway_smoke_cache_hit() { key: b"ilp-smoke-key".to_vec(), value: mp_string("ilp-smoke-val"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"ilp-smoke-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -344,7 +345,7 @@ async fn native_gateway_smoke_cache_hit() { key: b"native-smoke-key".to_vec(), value: mp_string("native-smoke-val"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"native-smoke-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, diff --git a/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs b/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs index 9ab4585ae..21819ef65 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/listeners_typed_not_leader.rs @@ -75,6 +75,7 @@ fn test_ctx() -> QueryContext { trace_id: nodedb_types::TraceId::ZERO, database_id: nodedb_types::id::DatabaseId::DEFAULT, txn_id: None, + linearizable: false, } } @@ -176,7 +177,7 @@ async fn pgwire_not_leader_retry_uses_shared_gateway() { key: b"pgwire-key".to_vec(), value: mp_string("val"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"pgwire-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -242,7 +243,7 @@ async fn http_not_leader_gateway_error_mapping() { key: b"http-key".to_vec(), value: mp_string("v"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"http-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -261,6 +262,7 @@ async fn http_not_leader_gateway_error_mapping() { vshard_id: VShardId::new(0), leader_node: 2, leader_addr: "10.0.0.2:9400".into(), + leader_term: 1, }; let (status, _body) = GatewayErrorMap::to_http(¬_leader); assert_eq!( @@ -314,7 +316,7 @@ async fn resp_not_leader_gateway_error_mapping() { key: b"resp-key".to_vec(), value: mp_string("v"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"resp-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -334,6 +336,7 @@ async fn resp_not_leader_gateway_error_mapping() { vshard_id: VShardId::new(0), leader_node: 3, leader_addr: "10.0.0.3:9400".into(), + leader_term: 1, }; let resp_err = GatewayErrorMap::to_resp(¬_leader); assert!( @@ -395,6 +398,7 @@ async fn ilp_not_leader_gateway_error_mapping() { vshard_id: VShardId::new(0), leader_node: 2, leader_addr: "10.0.0.2:9400".into(), + leader_term: 1, }; let err_str = GatewayErrorMap::to_resp(¬_leader); assert!( @@ -449,7 +453,7 @@ async fn native_not_leader_gateway_error_mapping() { key: b"native-key".to_vec(), value: mp_string("v"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"native-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -469,6 +473,7 @@ async fn native_not_leader_gateway_error_mapping() { vshard_id: VShardId::new(0), leader_node: 1, leader_addr: "127.0.0.1:9400".into(), + leader_term: 1, }; let (native_code, _native_msg) = GatewayErrorMap::to_native(¬_leader); assert_eq!( @@ -509,6 +514,7 @@ async fn not_leader_counter_increments_per_retry_attempt() { vshard_id: VShardId::new(0), leader_node: 0, leader_addr: String::new(), + leader_term: 0, }) } else { Ok::<(), Error>(()) diff --git a/nodedb-cluster-tests/tests/common_suite/cases/metadata_floor_restart.rs b/nodedb-cluster-tests/tests/common_suite/cases/metadata_floor_restart.rs new file mode 100644 index 000000000..d421c8906 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/metadata_floor_restart.rs @@ -0,0 +1,134 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! Group 0 durable applied floor across a restart. +//! +//! A graceful WAL-only restart of a single-node metadata group resumes +//! delivery above the saved floor. The catalog, descriptor leases, drains, +//! and pending DDL records come back from their `SystemCatalog` rows, and no +//! catalog entry below the floor is applied again. + +use crate::common; + +use common::cluster_harness::TestClusterNode; +use common::cluster_harness::shared_steps::{propose_and_apply, wait_for_single_node_ready}; + +use nodedb_cluster::{DescriptorId, DescriptorKind, DescriptorLease, DrainOwner, MetadataEntry}; +use nodedb_types::{DatabaseId, Hlc}; + +const TENANT: u64 = 1; +const COLLECTION: &str = "mfr_orders"; +const PENDING_TOKEN: u64 = 9_001; + +fn far_future() -> Hlc { + let now_ns = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_nanos() as u64) + .unwrap_or(0); + Hlc::new(now_ns + 3_600_000_000_000, 0) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn host_state_survives_restart_above_the_floor() { + let data_dir = tempfile::tempdir().expect("tempdir"); + let data_path = data_dir.path().to_path_buf(); + let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path.clone()) + .await + .expect("spawn single-node metadata group"); + wait_for_single_node_ready(&node).await; + + node.client + .simple_query(&format!( + "CREATE COLLECTION {COLLECTION} (id TEXT PRIMARY KEY, val BIGINT) \ + WITH (engine='document_strict')" + )) + .await + .expect("create collection"); + + let leased = DescriptorId::new(0, TENANT, DescriptorKind::Collection, "mfr_leased"); + let drained = DescriptorId::new(0, TENANT, DescriptorKind::Collection, "mfr_drained"); + let lease = DescriptorLease { + descriptor_id: leased.clone(), + version: 1, + node_id: node.node_id, + expires_at: far_future(), + }; + let owner = DrainOwner::MoveTenant { + tenant_id: TENANT, + source_db_id: 0, + }; + propose_and_apply(&node, &MetadataEntry::DescriptorLeaseGrant(lease.clone())).await; + propose_and_apply( + &node, + &MetadataEntry::DescriptorDrainStart { + descriptor_id: drained.clone(), + up_to_version: 5, + expires_at: far_future(), + proposer_node_id: node.node_id, + owner: owner.clone(), + }, + ) + .await; + // A pending propose reserves only under the DDL preparation owner's token. + propose_and_apply( + &node, + &MetadataEntry::DdlPrepareAcquire { + token: PENDING_TOKEN, + node_id: node.node_id, + }, + ) + .await; + propose_and_apply( + &node, + &MetadataEntry::DdlPendingPropose { + token: PENDING_TOKEN, + objects: Vec::new(), + proposed_at: far_future(), + }, + ) + .await; + + node.graceful_shutdown_wal_only().await; + let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path) + .await + .expect("restart against the persisted metadata log"); + wait_for_single_node_ready(&node).await; + + assert!( + node.shared + .credentials + .catalog() + .get_collection(DatabaseId::DEFAULT, TENANT, COLLECTION) + .expect("read collection") + .is_some(), + "the collection row survives the restart" + ); + { + let cache = node + .shared + .metadata_cache + .read() + .unwrap_or_else(|p| p.into_inner()); + assert_eq!( + cache.leases.get(&(leased.clone(), node.node_id)), + Some(&lease), + "the lease is seeded from its row" + ); + assert_eq!( + cache.catalog_entries_applied, 0, + "no catalog entry below the floor is applied again" + ); + } + assert!( + node.shared + .lease_drain + .snapshot() + .iter() + .any(|(id, held_by, _)| *id == drained && *held_by == owner), + "the drain is seeded from its row" + ); + assert!( + node.shared.pending_ddl.contains(PENDING_TOKEN), + "the pending DDL record is seeded from its row" + ); + + node.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/metadata_replay.rs b/nodedb-cluster-tests/tests/common_suite/cases/metadata_replay.rs new file mode 100644 index 000000000..8d3ce08b5 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/metadata_replay.rs @@ -0,0 +1,466 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! Regression coverage for group-0 metadata replay damaging a later +//! incarnation of the same name. +//! +//! Each test drives a create/drop/recreate (or clone/move) sequence, then a +//! graceful WAL-only restart of a single-node metadata group, and asserts the +//! post-restart state still matches the pre-restart state, not the historical +//! entry replay reasserts. +//! +//! - `DeleteConsumerGroup` and `DeleteChangeStream` are fenced by the HLC +//! incarnation stamp (`nodedb/src/control/catalog_entry/incarnation/`). +//! - `CloneDatabase` applies once: its lineage edge marks it applied. +//! - `MoveTenantCutover` writes only the snapshot it carries. Every later +//! change to the rows it touches is a later log entry, so replay restores it. + +use crate::common; + +use common::cluster_harness::TestClusterNode; +use common::cluster_harness::shared_steps::{database_id, wait_for_single_node_ready}; + +use nodedb::event::cdc::CdcOffset; +use nodedb_types::DatabaseId; + +const TENANT: u64 = 1; + +/// Whether `name` is an active collection under `database_id` for +/// `tenant_id`, read through the local `SystemCatalog` redb. +fn has_collection( + node: &TestClusterNode, + database_id: DatabaseId, + tenant_id: u64, + name: &str, +) -> bool { + node.shared + .credentials + .catalog() + .load_collections_for_tenant(database_id, tenant_id) + .expect("load collections") + .iter() + .any(|c| c.name == name) +} + +/// CREATE, DROP, and CREATE again on the same consumer-group name, each +/// incarnation committing its own offset, all before a graceful restart of +/// the single-node metadata group. `forget_offsets` +/// (`nodedb/src/control/catalog_entry/post_apply/consumer_group.rs`) +/// deletes the node-local `offset_store` row by stream+group name alone, so +/// replaying the first incarnation's `DeleteConsumerGroup` must not erase +/// the second incarnation's already-persisted offset. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn replayed_consumer_group_drop_keeps_recreated_group_offsets() { + const TOPIC: &str = "mrt_cg_topic"; + const GROUP: &str = "mrt_cg_group"; + const RECREATED_OFFSET: u64 = 42; + + let data_dir = tempfile::tempdir().expect("tempdir"); + let data_path = data_dir.path().to_path_buf(); + let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path.clone()) + .await + .expect("spawn single-node metadata group"); + wait_for_single_node_ready(&node).await; + + node.client + .simple_query(&format!( + "CREATE TOPIC {TOPIC} WITH (RETENTION = '2 hours')" + )) + .await + .expect("create topic"); + node.client + .simple_query(&format!("CREATE CONSUMER GROUP {GROUP} ON {TOPIC}")) + .await + .expect("create first incarnation of the group"); + node.client + .simple_query(&format!( + "COMMIT OFFSET PARTITION 0 AT 0:5 ON {TOPIC} CONSUMER GROUP {GROUP}" + )) + .await + .expect("commit an offset on the first incarnation"); + + node.client + .simple_query(&format!("DROP CONSUMER GROUP {GROUP} ON {TOPIC}")) + .await + .expect("drop the first incarnation"); + + node.client + .simple_query(&format!("CREATE CONSUMER GROUP {GROUP} ON {TOPIC}")) + .await + .expect("create the second incarnation of the group"); + node.client + .simple_query(&format!( + "COMMIT OFFSET PARTITION 0 AT 0:{RECREATED_OFFSET} ON {TOPIC} CONSUMER GROUP {GROUP}" + )) + .await + .expect("commit an offset on the second incarnation"); + + let stream = format!("topic:{TOPIC}"); + assert_eq!( + node.shared + .offset_store + .get_offset(DatabaseId::DEFAULT, TENANT, &stream, GROUP, 0), + CdcOffset::whole_index(RECREATED_OFFSET), + "the second incarnation must hold its committed offset before restart" + ); + + node.graceful_shutdown_wal_only().await; + let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path) + .await + .expect("restart against the persisted metadata log"); + wait_for_single_node_ready(&node).await; + + assert_eq!( + node.shared + .offset_store + .get_offset(DatabaseId::DEFAULT, TENANT, &stream, GROUP, 0), + CdcOffset::whole_index(RECREATED_OFFSET), + "replaying the first incarnation's DeleteConsumerGroup must not wipe the second \ + incarnation's committed offset" + ); + + node.shutdown().await; +} + +/// CREATE, DROP, and CREATE again on the same change-stream name, the +/// second incarnation carrying its own consumer group and committed offset, +/// all before a graceful restart of the single-node metadata group. +/// `DeleteChangeStream`'s post-apply +/// (`nodedb/src/control/catalog_entry/post_apply/change_stream.rs`) +/// unregisters the stream by name and cascades offset deletion over every +/// group currently attached to that name, so replaying the first +/// incarnation's drop must not unregister the second incarnation or wipe +/// its group's offset. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn replayed_change_stream_drop_keeps_recreated_stream() { + const SRC: &str = "mrt_cs_src"; + const STREAM: &str = "mrt_cs_stream"; + const GROUP: &str = "mrt_cs_group"; + const OFFSET: u64 = 7; + + let data_dir = tempfile::tempdir().expect("tempdir"); + let data_path = data_dir.path().to_path_buf(); + let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path.clone()) + .await + .expect("spawn single-node metadata group"); + wait_for_single_node_ready(&node).await; + + node.client + .simple_query(&format!( + "CREATE COLLECTION {SRC} (id TEXT PRIMARY KEY, val BIGINT) WITH (engine='document_strict')" + )) + .await + .expect("create the collection the stream watches"); + node.client + .simple_query(&format!("CREATE CHANGE STREAM {STREAM} ON {SRC}")) + .await + .expect("create the first incarnation of the stream"); + + node.client + .simple_query(&format!("DROP CHANGE STREAM {STREAM}")) + .await + .expect("drop the first incarnation"); + + node.client + .simple_query(&format!("CREATE CHANGE STREAM {STREAM} ON {SRC}")) + .await + .expect("create the second incarnation of the stream"); + node.client + .simple_query(&format!("CREATE CONSUMER GROUP {GROUP} ON {STREAM}")) + .await + .expect("create a group on the second incarnation"); + // Stands in for consumption progress: the offset a faulty replay wipes lives in + // the same node-local store regardless of how it was produced. + node.client + .simple_query(&format!( + "COMMIT OFFSET PARTITION 0 AT 0:{OFFSET} ON {STREAM} CONSUMER GROUP {GROUP}" + )) + .await + .expect("commit an offset on the second incarnation's group"); + + assert!( + node.has_change_stream(DatabaseId::DEFAULT, TENANT, STREAM), + "the second incarnation must be registered before restart" + ); + assert_eq!( + node.shared + .offset_store + .get_offset(DatabaseId::DEFAULT, TENANT, STREAM, GROUP, 0), + CdcOffset::whole_index(OFFSET), + "the second incarnation's group must hold its committed offset before restart" + ); + + node.graceful_shutdown_wal_only().await; + let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path) + .await + .expect("restart against the persisted metadata log"); + wait_for_single_node_ready(&node).await; + + assert!( + node.has_change_stream(DatabaseId::DEFAULT, TENANT, STREAM), + "replaying the first incarnation's DeleteChangeStream must not unregister the \ + second incarnation's stream" + ); + assert_eq!( + node.shared + .offset_store + .get_offset(DatabaseId::DEFAULT, TENANT, STREAM, GROUP, 0), + CdcOffset::whole_index(OFFSET), + "replaying the first incarnation's DeleteChangeStream must not wipe the second \ + incarnation's group offset" + ); + + node.shutdown().await; +} + +/// CLONE a database, drop one collection from the child, and create a new +/// collection in the source, all before a graceful restart of the +/// single-node metadata group. `clone_apply` +/// (`nodedb/src/control/catalog_entry/apply/database.rs`) enumerates the +/// source's currently active collections when it applies, so replay must not +/// re-apply it: that resurrects the dropped child collection and pull in +/// the source's later collection. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn replayed_clone_keeps_child_changes() { + const SRC_DB: &str = "mrt_clone_src"; + const CHILD_DB: &str = "mrt_clone_child"; + const DROPPED: &str = "mrt_will_drop"; + const STAYS: &str = "mrt_stays"; + const NEW_IN_SRC: &str = "mrt_new_in_src"; + + let data_dir = tempfile::tempdir().expect("tempdir"); + let data_path = data_dir.path().to_path_buf(); + let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path.clone()) + .await + .expect("spawn single-node metadata group"); + wait_for_single_node_ready(&node).await; + + node.client + .simple_query(&format!("CREATE DATABASE {SRC_DB}")) + .await + .expect("create the source database"); + node.client + .simple_query(&format!("USE DATABASE {SRC_DB}")) + .await + .expect("use the source database"); + node.client + .simple_query(&format!( + "CREATE COLLECTION {DROPPED} (id TEXT PRIMARY KEY, val TEXT)" + )) + .await + .expect("create the collection the child will drop"); + node.client + .simple_query(&format!( + "CREATE COLLECTION {STAYS} (id TEXT PRIMARY KEY, val TEXT)" + )) + .await + .expect("create the collection that stays in both databases"); + + node.client + .simple_query("USE DATABASE default") + .await + .expect("use the default database"); + node.client + .simple_query(&format!("CLONE DATABASE {CHILD_DB} FROM {SRC_DB}")) + .await + .expect("clone the database"); + + let child_db = database_id(&node, CHILD_DB); + let src_db = database_id(&node, SRC_DB); + assert!( + has_collection(&node, child_db, TENANT, DROPPED), + "the clone must carry the source's active collections at clone time" + ); + + node.client + .simple_query(&format!("USE DATABASE {CHILD_DB}")) + .await + .expect("use the child database"); + node.client + .simple_query(&format!("DROP COLLECTION {DROPPED}")) + .await + .expect("drop the collection from the child after cloning"); + + node.client + .simple_query(&format!("USE DATABASE {SRC_DB}")) + .await + .expect("use the source database"); + node.client + .simple_query(&format!( + "CREATE COLLECTION {NEW_IN_SRC} (id TEXT PRIMARY KEY, val TEXT)" + )) + .await + .expect("create a collection in the source after cloning"); + + node.client + .simple_query("USE DATABASE default") + .await + .expect("use the default database"); + + assert!( + !has_collection(&node, child_db, TENANT, DROPPED), + "the child must not see the dropped collection before restart" + ); + assert!( + !has_collection(&node, child_db, TENANT, NEW_IN_SRC), + "the child must not see the source's later collection before restart" + ); + assert!(has_collection(&node, src_db, TENANT, STAYS)); + + node.graceful_shutdown_wal_only().await; + let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path) + .await + .expect("restart against the persisted metadata log"); + wait_for_single_node_ready(&node).await; + + assert!( + !has_collection(&node, child_db, TENANT, DROPPED), + "replaying CloneDatabase must not resurrect a collection dropped from the child" + ); + assert!( + !has_collection(&node, child_db, TENANT, NEW_IN_SRC), + "replaying CloneDatabase must not pull in a collection created in the source \ + after the clone" + ); + + node.shutdown().await; +} + +/// MOVE a tenant's database to a target database, ALTER the moved +/// collection's owner in the target, and create a same-name collection in +/// the source, all before a graceful restart of the single-node metadata +/// group. `move_cutover` (`nodedb/src/control/catalog_entry/apply/tenant.rs`) +/// rewrites the target from the collection snapshot carried at propose +/// time and deletes source rows by name, so replaying it must not revert +/// the target's later ALTER or delete the source's later same-name +/// collection. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn replayed_move_tenant_keeps_later_changes() { + const SRC_DB: &str = "mrt_move_src"; + const TGT_DB: &str = "mrt_move_tgt"; + const MOVED: &str = "mrt_moved"; + const NEW_OWNER: &str = "mrt_new_owner"; + + let data_dir = tempfile::tempdir().expect("tempdir"); + let data_path = data_dir.path().to_path_buf(); + let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path.clone()) + .await + .expect("spawn single-node metadata group"); + wait_for_single_node_ready(&node).await; + + node.client + .simple_query(&format!( + "CREATE USER {NEW_OWNER} WITH PASSWORD 'pw' ROLE READWRITE" + )) + .await + .expect("create the user the target ALTER assigns"); + + node.client + .simple_query(&format!("CREATE DATABASE {SRC_DB}")) + .await + .expect("create the source database"); + node.client + .simple_query(&format!("USE DATABASE {SRC_DB}")) + .await + .expect("use the source database"); + node.client + .simple_query("CREATE TENANT mrt_tenant ID 50") + .await + .expect("create the tenant to move"); + node.client + .simple_query(&format!( + "CREATE COLLECTION {MOVED} (id STRING PRIMARY KEY, val STRING) WITH (engine='kv')" + )) + .await + .expect("create the collection the tenant moves"); + node.client + .simple_query(&format!( + "INSERT INTO {MOVED} (id, val) VALUES ('k1', 'v1')" + )) + .await + .expect("seed a row"); + + node.client + .simple_query("USE DATABASE default") + .await + .expect("use the default database"); + node.client + .simple_query(&format!("CREATE DATABASE {TGT_DB}")) + .await + .expect("create the target database"); + node.client + .simple_query(&format!("USE DATABASE {TGT_DB}")) + .await + .expect("use the target database"); + node.client + .simple_query(&format!( + "CREATE COLLECTION {MOVED} (id STRING PRIMARY KEY, val STRING) WITH (engine='kv')" + )) + .await + .expect("pre-create the matching target collection MOVE TENANT requires"); + + node.client + .simple_query("USE DATABASE default") + .await + .expect("use the default database"); + node.client + .simple_query(&format!("MOVE TENANT mrt_tenant FROM {SRC_DB} TO {TGT_DB}")) + .await + .expect("move the tenant"); + + let target_db = database_id(&node, TGT_DB); + let source_db = database_id(&node, SRC_DB); + + node.client + .simple_query(&format!("USE DATABASE {TGT_DB}")) + .await + .expect("use the target database"); + node.client + .simple_query(&format!("ALTER COLLECTION {MOVED} OWNER TO {NEW_OWNER}")) + .await + .expect("alter the moved collection's owner in the target"); + + node.client + .simple_query(&format!("USE DATABASE {SRC_DB}")) + .await + .expect("use the source database"); + node.client + .simple_query(&format!( + "CREATE COLLECTION {MOVED} (id STRING PRIMARY KEY, val STRING) WITH (engine='kv')" + )) + .await + .expect("create a same-name collection in the source after the move"); + + node.client + .simple_query("USE DATABASE default") + .await + .expect("use the default database"); + + assert_eq!( + node.owner_of("collection", target_db.as_u64(), TENANT, MOVED) + .as_deref(), + Some(NEW_OWNER), + "the target must carry the ALTER before restart" + ); + assert!( + has_collection(&node, source_db, TENANT, MOVED), + "the source must carry its later same-name collection before restart" + ); + + node.graceful_shutdown_wal_only().await; + let node = TestClusterNode::spawn_single_node_calvin_on_path(4, data_path) + .await + .expect("restart against the persisted metadata log"); + wait_for_single_node_ready(&node).await; + + assert_eq!( + node.owner_of("collection", target_db.as_u64(), TENANT, MOVED) + .as_deref(), + Some(NEW_OWNER), + "replaying MoveTenantCutover must not revert the target's later ALTER OWNER" + ); + assert!( + has_collection(&node, source_db, TENANT, MOVED), + "replaying MoveTenantCutover must not delete the source's later same-name collection" + ); + + node.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/metadata_snapshot_catch_up.rs b/nodedb-cluster-tests/tests/common_suite/cases/metadata_snapshot_catch_up.rs new file mode 100644 index 000000000..cf4faad71 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/metadata_snapshot_catch_up.rs @@ -0,0 +1,247 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Catch-up of metadata Raft group 0 by snapshot after its log compacted. +//! +//! A cluster with a low compaction threshold stops one member, runs enough +//! DDL to compact group 0 past that member's log, and brings it back. A +//! fresh node then joins. Neither can be caught up by `AppendEntries`: each +//! installs a group 0 image. Afterwards their replicated catalog rows equal +//! the leader's, a collection purged while the member was down is gone from +//! its storage, and a collection created meanwhile takes writes on it. + +use std::time::Duration; + +use crate::common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use nodedb_types::{DatabaseId, TenantId}; + +const COMPACTION_THRESHOLD: u64 = 4; +const TENANT: u64 = 1; +const PURGED: &str = "mss_purged"; +const KEPT_COUNT: usize = 12; + +fn kept(i: usize) -> String { + format!("mss_kept_{i}") +} + +/// Replicated tables that legitimately differ across nodes. Each also takes +/// writes on one node outside group 0 apply, and an install keeps this +/// node's higher counter or its own pending rows: +/// - `surrogate_hwm`: each node flushes its own surrogate assigner. +/// - `sync_producer_hwm`: each node flushes its own producer allocator. +/// - `sync_producers`: a registration or fence lands locally before it +/// commits. +/// - `sync_peer_bindings`: a peer-id claim lands locally before it commits. +/// - `tenant_id_hwm`: tenant creation allocates the next id locally. +/// - `topics_ep`: a publish advances the topic's sequence on its node. +/// - `pending_leave_cleanup`: each node removes a row once its own view of +/// the leaver's leases and drains is clear. +const LOCAL_WRITE_TABLES: &[&str] = &[ + "surrogate_hwm", + "sync_producer_hwm", + "sync_producers", + "sync_peer_bindings", + "tenant_id_hwm", + "topics_ep", + "pending_leave_cleanup", +]; + +/// Whether a row of `label` is node-local and left out of the comparison: +/// the `metadata` counter the credential store writes, and the GAP_FREE log +/// rows each node writes into `sequence_state`. +fn node_local_row(label: &str, key: &[u8]) -> bool { + match label { + "metadata" => key == b"next_user_id", + "sequence_state" => key + .splitn(3, |byte| *byte == b':') + .nth(2) + .is_some_and(|name| name.starts_with(b"log:")), + _ => false, + } +} + +/// A replicated table's label paired with its `(key, value)` rows. +type LabeledRows = (String, Vec<(Vec, Vec)>); + +/// Every replicated table except [`LOCAL_WRITE_TABLES`], without node-local +/// rows. Every node holds the same rows once it applied the same entries. +fn compared_rows(node: &TestClusterNode) -> Vec { + node.shared + .credentials + .catalog() + .begin_replicated_read() + .expect("begin replicated read") + .dump() + .expect("dump replicated tables") + .into_iter() + .filter(|(label, _)| !LOCAL_WRITE_TABLES.contains(&label.as_str())) + .map(|(label, rows)| { + let rows = rows + .into_iter() + .filter(|(key, _)| !node_local_row(&label, key)) + .collect(); + (label, rows) + }) + .collect() +} + +async fn keys_of(node: &TestClusterNode, collection: &str) -> Vec { + let core = node.home_core_of(collection); + node.document_keys_on_core(core, TenantId::new(TENANT)) + .await + .into_iter() + .filter(|key| key.contains(collection)) + .collect() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn lagging_member_and_joiner_catch_up_by_metadata_snapshot() { + let mut cluster = + TestCluster::spawn_three_with_compaction_threshold_and_rf(COMPACTION_THRESHOLD, 4) + .await + .expect("3-node cluster with low compaction threshold and rf=4"); + + // A collection every member stores, purged later while one member is down. + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {PURGED} (id TEXT PRIMARY KEY, payload TEXT) \ + WITH (engine='document_strict')" + )) + .await + .expect("create the collection to purge"); + cluster.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {PURGED} (id, payload) VALUES ('p1', 'v1')" + )) + .await + .expect("insert into the collection to purge"); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + let lagging_id = cluster.nodes[2].node_id; + assert!( + !keys_of(&cluster.nodes[2], PURGED).await.is_empty(), + "the member to stop stores the collection before it stops" + ); + + let stopped = cluster + .stop_member(2) + .await + .expect("stop the lagging member"); + + cluster + .exec_ddl_on_any_leader(&format!("DROP COLLECTION {PURGED} PURGE")) + .await + .expect("purge while the member is down"); + for i in 0..KEPT_COUNT { + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {} (id TEXT PRIMARY KEY, payload TEXT) \ + WITH (engine='document_strict')", + kept(i) + )) + .await + .expect("create a collection while the member is down"); + } + wait_for( + "group 0 compacted on a running member", + Duration::from_secs(30), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .any(|n| n.local_snapshot_index_for_group(0) > 0) + }, + ) + .await; + + cluster + .restart_member(stopped) + .await + .expect("restart the lagging member"); + let joiner_id = cluster + .add_learner_node() + .await + .expect("add a joining node") + .node_id; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + + for caught_up in [lagging_id, joiner_id] { + let node = cluster + .nodes + .iter() + .find(|n| n.node_id == caught_up) + .expect("caught-up node present"); + wait_for( + "the caught-up node installed a group 0 snapshot", + Duration::from_secs(30), + Duration::from_millis(50), + || node.local_snapshot_index_for_group(0) > 0, + ) + .await; + let leader = &cluster.nodes[0]; + wait_for( + "the caught-up node's replicated rows equal the leader's", + Duration::from_secs(30), + Duration::from_millis(100), + || compared_rows(node) == compared_rows(leader), + ) + .await; + let catalog = node.shared.credentials.catalog(); + assert!( + catalog + .get_collection(DatabaseId::DEFAULT, TENANT, PURGED) + .expect("read the purged collection") + .is_none(), + "node {caught_up}: the purged collection is gone from the catalog" + ); + for i in 0..KEPT_COUNT { + assert!( + catalog + .get_collection(DatabaseId::DEFAULT, TENANT, &kept(i)) + .expect("read a kept collection") + .is_some(), + "node {caught_up}: collection {} arrived with the snapshot", + kept(i) + ); + } + } + + let lagging = cluster + .nodes + .iter() + .find(|n| n.node_id == lagging_id) + .expect("lagging member present"); + assert!( + keys_of(lagging, PURGED).await.is_empty(), + "the purged collection's storage is reclaimed on the lagging member" + ); + + // A collection created while the member was down takes writes there. + let written = kept(0); + cluster.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {written} (id, payload) VALUES ('w1', 'v1')" + )) + .await + .expect("insert into a collection created while the member was down"); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + let lagging = cluster + .nodes + .iter() + .find(|n| n.node_id == lagging_id) + .expect("lagging member present"); + assert!( + !keys_of(lagging, &written).await.is_empty(), + "the lagging member stores writes to a collection it learned by snapshot" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/mod.rs b/nodedb-cluster-tests/tests/common_suite/cases/mod.rs index 9fc78978a..d40f209e2 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/mod.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/mod.rs @@ -3,13 +3,17 @@ mod a4_placement_convergence; mod alert_rule_cross_node; mod alter_quota_cross_node; +mod array_install_snapshot_cluster; mod array_raft_partition_recovery; mod array_raft_replication; mod array_raft_snapshot_install; mod array_vshard_migration; mod assign_surrogate_cross_node; +mod backup_schedule_failover; mod bitemporal_array_cluster; +mod calvin_cdc_net_kinds; mod calvin_cluster_pgwire_e2e; +mod calvin_joiner_after_sequencer_compaction; mod calvin_multi_shard_bitemporal_best_effort_restart; mod calvin_multi_shard_bitemporal_restart; mod calvin_multi_shard_redo_restart; @@ -20,45 +24,69 @@ mod calvin_multishard_txn_staging_cross_node; mod calvin_ollp_cross_node; mod calvin_ollp_pk_delete; mod calvin_ollp_update; +mod calvin_remote_participant_report; mod calvin_submit_routed_cross_node; +mod calvin_superseded_collection_cluster; mod catalog_put_if_absent; +mod cdc_consume_across_leader_change; mod checkpoint_cross_node; +mod clone_materialize_cross_node; mod cluster_array; +mod cluster_array_body_read; mod cluster_array_cell_raft_replication; +mod cluster_array_ws_rpc; mod cluster_backup_remote_cut; mod cluster_backup_restore; mod cluster_backup_restore_databases; mod cluster_backup_restore_engines; +mod cluster_backup_restore_graph_edges; mod cluster_cdc_publish_once; +mod cluster_cdc_transaction_events; mod cluster_collection_hard_delete; mod cluster_crdt_replication; mod cluster_epoch_self_fence; mod cluster_execute_request; mod cluster_partition_strategy_replication; +mod cluster_pitr_restore_point; mod cluster_post_apply_follower_dispatch; mod cluster_restore_documents_restart; mod cluster_restore_refuses_on_non_replica; +mod cluster_restore_retry; +mod cluster_restore_surrogate_conflict; +mod cluster_restore_surrogate_floor; mod cluster_surrogate_replication; mod column_stats_cross_node; +mod committed_publish_failover; +mod committed_publish_metadata_lag; mod constraint_delivery; mod crdt_compact_cross_node; mod cross_node_pk_lookup; mod cross_node_pk_write; mod cross_node_read_occ_abort; +mod database_lifecycle_cross_node; +mod ddl_prepare_lease_reclaim; mod descriptor_lease_cross_node; +mod descriptor_lease_dead_holder; mod descriptor_lease_drain; mod descriptor_lease_forwarding_and_renewal; mod descriptor_lease_planner_integration; +mod descriptor_replay_incarnation; mod descriptor_versioning_cross_node; mod gateway_execute; mod gather_join_in_txn_occ; mod graph_index_edges_replicate_to_followers; mod graph_match_ryow_cross_node; +mod graph_wcc_core_error_cross_node; mod hilo_surrogate_uniqueness; mod http_gateway_migration; +mod idle_lease_retention; mod ilp_gateway_migration; mod install_snapshot_crdt_constraints_cluster; mod install_snapshot_e2e_cluster; +mod install_snapshot_edge_endpoints; +mod install_snapshot_multi_core; +mod install_snapshot_multi_core_fault; +mod kafka_sink_single_owner; mod kv_atomic_autocommit_replicates; mod learner_cleanup; mod linearizable_read_leadership; @@ -67,16 +95,26 @@ mod listeners_typed_not_leader; mod materialized_sum_cross_core; mod materialized_sum_cross_shard; mod materialized_sum_replication; +mod metadata_floor_restart; +mod metadata_replay; +mod metadata_snapshot_catch_up; +mod move_tenant_array_cross_node; +mod move_tenant_cross_node; +mod move_tenant_reissue_retry; mod multi_replica_data_groups; mod native_gateway_migration; mod native_gather_join_in_txn_occ; +mod native_write_drain_gate; mod node_labels_replicate_to_followers; +mod owner_read_on_non_replica; mod pgwire_gateway_migration; mod planner_local_only; mod prepared_cache_invalidation; mod proposal_committed_twice_applies_once; mod resp_gateway_migration; mod retention_policy_cross_node; +mod routing_leader_hint_follows_election; +mod routing_redirect_hint_unhosted_group; mod scope_quota_cross_node; mod shuffle_aggregate_cost_model; mod shuffle_aggregate_cross_node; @@ -97,6 +135,7 @@ mod single_node_calvin_returning; mod single_node_calvin_two_phase; mod single_node_calvin_write_versions; mod sql_ddl_cluster; +mod sql_schedule_lease_gating; mod surrogate_raft; mod surrogate_vshard_transfer; mod sync_collection_schema; @@ -106,8 +145,21 @@ mod sync_delta_reject_unique_violation; mod sync_failover; mod sync_peer_id_collision; mod sync_retryable_delta_refusal; +mod timeseries_calvin_read_your_writes; +mod timeseries_conflicting_types_replicas; +mod timeseries_resolved_replicas; +mod timeseries_ring_drop_catch_up; +mod timeseries_snapshot_schema_follower; mod topic_consumer_group_cross_node; +mod topic_publish_across_home_change; +mod trigger_body_atomic_cross_node; +mod trigger_cross_shard_origination; +mod trigger_firing_failover; +mod trigger_shipped_block_atomic_cross_node; +mod ts_native_ingest; mod vector_alter_params_cross_node; mod vector_index_dispatch_cross_node; mod vshard_names; +mod webhook_sink_partial_replication; +mod webhook_sink_single_owner; mod write_admission_concurrent_same_key_replay; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/move_tenant_array_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/move_tenant_array_cross_node.rs new file mode 100644 index 000000000..8f4790616 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/move_tenant_array_cross_node.rs @@ -0,0 +1,193 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! `MOVE TENANT` of a database that holds an array, on a three-node cluster +//! with two Data-Plane cores per node. +//! +//! A cell routes to its vShard by Hilbert prefix alone, so the move copies no +//! cell: the cutover rekeys the catalog row, the surrogate bindings, and each +//! core's store. The test writes cells, flushes some and leaves the rest in +//! memtables, moves the database, and reads every cell through the target +//! from every node. No node keeps the array under the source. + +use std::time::Duration; + +use nodedb_types::DatabaseId; + +use crate::common; +use common::cluster_harness::TestCluster; +use common::cluster_harness::shared_steps::{database_id, db_detail, use_database}; + +const SOURCE: &str = "mt_arr_src"; +const TARGET: &str = "mt_arr_tgt"; +const MOVED_TENANT: &str = "mt_arr_owner"; +const ARRAY: &str = "mt_grid"; +/// `qual` of every cell written: 1 + 2 + 10 flushed, 100 + 200 in memtables. +const TOTAL: f64 = 313.0; +const CELLS: usize = 5; + +/// Every row of `sql` on `node_idx`, as `column -> text`. +async fn rows( + cluster: &TestCluster, + node_idx: usize, + sql: &str, +) -> Vec> { + let messages = cluster.nodes[node_idx] + .client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql} on node {node_idx}: {}", db_detail(&e))); + messages + .into_iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => Some( + row.columns() + .iter() + .enumerate() + .map(|(i, c)| (c.name().to_string(), row.get(i).unwrap_or("").to_string())) + .collect(), + ), + _ => None, + }) + .collect() +} + +/// Whether node `node_idx` holds the array under `database`, in the durable +/// catalog and in the in-memory mirror. +fn holds_array(cluster: &TestCluster, node_idx: usize, database: DatabaseId) -> (bool, bool) { + let shared = &cluster.nodes[node_idx].shared; + let durable = shared + .credentials + .catalog() + .load_all_arrays() + .expect("catalog read") + .iter() + .any(|a| a.array_id.database_id == database && a.name == ARRAY); + let mirror = shared + .array_catalog + .read() + .expect("array mirror") + .all_entries() + .iter() + .any(|a| a.array_id.database_id == database && a.name == ARRAY); + (durable, mirror) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn move_tenant_array_cells_are_readable_through_the_target_from_every_node() { + let cluster = TestCluster::spawn_three_with_cores(2) + .await + .expect("cluster"); + + for database in [SOURCE, TARGET] { + cluster + .exec_ddl_on_any_leader(&format!("CREATE DATABASE {database}")) + .await + .unwrap_or_else(|e| panic!("CREATE DATABASE {database}: {e}")); + } + use_database(&cluster, SOURCE).await; + let ddl_idx = cluster + .exec_ddl_on_any_leader(&format!( + "CREATE ARRAY {ARRAY} \ + DIMS (chr INT64 [0..9], pos INT64 [0..99]) \ + ATTRS (qual FLOAT64) \ + TILE_EXTENTS (1, 100) \ + CELL_ORDER HILBERT" + )) + .await + .unwrap_or_else(|e| panic!("CREATE ARRAY {ARRAY}: {e}")); + + // Flushed cells through one node, then memtable-only cells through + // another: the rekey must carry both. + let writer = (ddl_idx + 1) % cluster.nodes.len(); + for sql in [ + format!( + "INSERT INTO ARRAY {ARRAY} COORDS (0, 10) VALUES (1.0), \ + COORDS (0, 20) VALUES (2.0), COORDS (1, 10) VALUES (10.0)" + ), + format!("SELECT ARRAY_FLUSH('{ARRAY}')"), + ] { + cluster.nodes[writer] + .exec(&sql) + .await + .unwrap_or_else(|e| panic!("{sql} on node {writer}: {e}")); + } + let late_writer = (writer + 1) % cluster.nodes.len(); + let sql = format!( + "INSERT INTO ARRAY {ARRAY} COORDS (2, 10) VALUES (100.0), COORDS (2, 20) VALUES (200.0)" + ); + cluster.nodes[late_writer] + .exec(&sql) + .await + .unwrap_or_else(|e| panic!("{sql} on node {late_writer}: {e}")); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + + use_database(&cluster, "default").await; + cluster + .exec_ddl_on_any_leader(&format!("CREATE TENANT {MOVED_TENANT} ID 78")) + .await + .unwrap_or_else(|e| panic!("CREATE TENANT {MOVED_TENANT}: {e}")); + cluster + .exec_ddl_on_any_leader(&format!( + "MOVE TENANT {MOVED_TENANT} FROM {SOURCE} TO {TARGET}" + )) + .await + .unwrap_or_else(|e| panic!("MOVE TENANT {MOVED_TENANT}: {e}")); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + + let source_id = database_id(&cluster.nodes[0], SOURCE); + let target_id = database_id(&cluster.nodes[0], TARGET); + use_database(&cluster, TARGET).await; + for node_idx in 0..cluster.nodes.len() { + assert_eq!( + holds_array(&cluster, node_idx, source_id), + (false, false), + "node {node_idx} must hold no array under the source" + ); + assert_eq!( + holds_array(&cluster, node_idx, target_id), + (true, true), + "node {node_idx} must hold the array under the target" + ); + + let sum = rows( + &cluster, + node_idx, + &format!("SELECT * FROM ARRAY_AGG('{ARRAY}', 'qual', 'sum')"), + ) + .await; + let total: f64 = sum + .first() + .and_then(|row| row.get("result")) + .and_then(|text| text.parse().ok()) + .unwrap_or_else(|| panic!("node {node_idx}: no sum in {sum:?}")); + assert!( + (total - TOTAL).abs() < 1e-4, + "node {node_idx}: sum over the moved array must be {TOTAL}, got {total}" + ); + + let cells = rows( + &cluster, + node_idx, + &format!( + "SELECT * FROM ARRAY_SLICE('{ARRAY}', '{{chr: [0, 9], pos: [0, 99]}}', ['qual'], 100)" + ), + ) + .await; + assert_eq!(cells.len(), CELLS, "node {node_idx}: cells {cells:?}"); + } + + // The source database names the array on no node. + use_database(&cluster, SOURCE).await; + for (node_idx, node) in cluster.nodes.iter().enumerate() { + let sql = format!("SELECT * FROM ARRAY_AGG('{ARRAY}', 'qual', 'sum')"); + assert!( + node.client.simple_query(&sql).await.is_err(), + "node {node_idx}: {SOURCE}.{ARRAY} must not resolve after the move" + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/move_tenant_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/move_tenant_cross_node.rs new file mode 100644 index 000000000..57bcb0dc7 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/move_tenant_cross_node.rs @@ -0,0 +1,232 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! `MOVE TENANT` on a three-node cluster with two Data-Plane cores per node. +//! +//! A collection's home vShard hashes its database. Moving a collection to +//! another database therefore moves its rows to another vShard: in general +//! another core and another Raft group. The test seeds several collections, +//! so the source homes spread across vShards, moves them, and reads every row +//! back through the target database from each node. Each node must then hold +//! nothing under the source key and no pending reclaim of it. + +use std::collections::BTreeSet; +use std::time::Duration; + +use nodedb_types::id::VShardId; +use nodedb_types::{CollectionKey, DatabaseId}; + +use crate::common; +use common::cluster_harness::TestCluster; +use common::cluster_harness::shared_steps::{database_id, db_detail, use_database}; + +const SOURCE: &str = "mt_cl_src"; +const TARGET: &str = "mt_cl_tgt"; +const MOVED_TENANT: &str = "mt_cl_owner"; +const ROWS: usize = 5; +const DOC_COLLECTIONS: [&str; 3] = ["mt_docs_a", "mt_docs_b", "mt_docs_c"]; +const KV_COLLECTIONS: [&str; 3] = ["mt_kv_a", "mt_kv_b", "mt_kv_c"]; + +/// Create every collection in `database`, with the same schema in each. +async fn create_collections(cluster: &TestCluster, database: &str) { + use_database(cluster, database).await; + for name in DOC_COLLECTIONS { + let ddl = format!( + "CREATE COLLECTION {name} (id TEXT PRIMARY KEY, content TEXT) \ + WITH (engine='document_strict')" + ); + cluster + .exec_ddl_on_any_leader(&ddl) + .await + .unwrap_or_else(|e| panic!("{ddl} in {database}: {e}")); + } + for name in KV_COLLECTIONS { + let ddl = format!( + "CREATE COLLECTION {name} (key STRING PRIMARY KEY, value STRING) WITH (engine='kv')" + ); + cluster + .exec_ddl_on_any_leader(&ddl) + .await + .unwrap_or_else(|e| panic!("{ddl} in {database}: {e}")); + } +} + +/// Fill every source collection through node 0. Each row's text names its +/// collection and its key. +async fn fill_source(cluster: &TestCluster) { + use_database(cluster, SOURCE).await; + for i in 0..ROWS { + for name in DOC_COLLECTIONS { + let sql = format!("INSERT INTO {name} (id, content) VALUES ('k{i}', '{name}-{i}')"); + cluster.nodes[0] + .client + .simple_query(&sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {}", db_detail(&e))); + } + for name in KV_COLLECTIONS { + let sql = format!("INSERT INTO {name} (key, value) VALUES ('k{i}', '{name}-{i}')"); + cluster.nodes[0] + .client + .simple_query(&sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {}", db_detail(&e))); + } + } +} + +/// The first column of every row `sql` returns on node `node_idx`. +async fn first_column(cluster: &TestCluster, node_idx: usize, sql: &str) -> Vec { + let messages = cluster.nodes[node_idx] + .client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql} on node {node_idx}: {}", db_detail(&e))); + messages + .into_iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .collect() +} + +fn every_collection() -> impl Iterator { + DOC_COLLECTIONS.into_iter().chain(KV_COLLECTIONS) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn move_tenant_rows_are_readable_through_the_target_from_every_node() { + let cluster = TestCluster::spawn_three_with_cores(2) + .await + .expect("cluster"); + + for database in [SOURCE, TARGET] { + cluster + .exec_ddl_on_any_leader(&format!("CREATE DATABASE {database}")) + .await + .unwrap_or_else(|e| panic!("CREATE DATABASE {database}: {e}")); + } + create_collections(&cluster, SOURCE).await; + create_collections(&cluster, TARGET).await; + fill_source(&cluster).await; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + + // The source homes spread across vShards, and the moved rows change + // vShard: they must travel to the target homes. + let source_id = database_id(&cluster.nodes[0], SOURCE); + let target_id = database_id(&cluster.nodes[0], TARGET); + let home = |db: DatabaseId, name: &str| { + VShardId::from_collection(CollectionKey::from_bare(db, name)).as_u32() + }; + let source_homes: BTreeSet = every_collection().map(|n| home(source_id, n)).collect(); + assert!( + source_homes.len() > 1, + "the source collections must home to more than one vShard: {source_homes:?}" + ); + assert!( + every_collection().any(|n| home(source_id, n) != home(target_id, n)), + "at least one collection must change vShard when it moves" + ); + + use_database(&cluster, "default").await; + cluster + .exec_ddl_on_any_leader(&format!("CREATE TENANT {MOVED_TENANT} ID 77")) + .await + .unwrap_or_else(|e| panic!("CREATE TENANT {MOVED_TENANT}: {e}")); + cluster + .exec_ddl_on_any_leader(&format!( + "MOVE TENANT {MOVED_TENANT} FROM {SOURCE} TO {TARGET}" + )) + .await + .unwrap_or_else(|e| panic!("MOVE TENANT {MOVED_TENANT}: {e}")); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + + // Every row, read through the target from every node. + use_database(&cluster, TARGET).await; + let expected_keys: Vec = (0..ROWS).map(|i| format!("k{i}")).collect(); + for node_idx in 0..cluster.nodes.len() { + for name in DOC_COLLECTIONS { + let mut ids = first_column(&cluster, node_idx, &format!("SELECT id FROM {name}")).await; + ids.sort(); + assert_eq!(ids, expected_keys, "{TARGET}.{name} on node {node_idx}"); + for i in 0..ROWS { + let sql = format!("SELECT content FROM {name} WHERE id = 'k{i}'"); + assert_eq!( + first_column(&cluster, node_idx, &sql).await, + vec![format!("{name}-{i}")], + "{sql} on node {node_idx}" + ); + } + } + for name in KV_COLLECTIONS { + let count = + first_column(&cluster, node_idx, &format!("SELECT COUNT(*) FROM {name}")).await; + assert_eq!( + count, + vec![ROWS.to_string()], + "{TARGET}.{name} on node {node_idx}" + ); + for i in 0..ROWS { + let sql = format!("SELECT value FROM {name} WHERE key = 'k{i}'"); + assert_eq!( + first_column(&cluster, node_idx, &sql).await, + vec![format!("{name}-{i}")], + "{sql} on node {node_idx}" + ); + } + } + } + + // The source namespace is gone, and every node reclaimed its storage. + use_database(&cluster, SOURCE).await; + for (node_idx, node) in cluster.nodes.iter().enumerate() { + for name in every_collection() { + let sql = format!("SELECT COUNT(*) FROM {name}"); + if let Ok(messages) = node.client.simple_query(&sql).await { + let count: Vec = messages + .into_iter() + .filter_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(row) => { + row.get(0).map(str::to_owned) + } + _ => None, + }) + .collect(); + assert!( + count.is_empty() || count == vec!["0".to_string()], + "{SOURCE}.{name} on node {node_idx} must hold no rows, got {count:?}" + ); + } + } + let pending = node + .shared + .credentials + .catalog() + .load_pending_reclaim_queue() + .expect("pending-reclaim read"); + let owed: Vec<&str> = pending + .iter() + .filter(|entry| entry.database_id == source_id.as_u64()) + .map(|entry| entry.name.as_str()) + .collect(); + assert!( + owed.is_empty(), + "node {node_idx} must reclaim the source storage without a pending retry: {owed:?}" + ); + let source_rows = node + .shared + .credentials + .catalog() + .load_all_collections(source_id) + .expect("catalog read"); + assert!( + source_rows.iter().all(|c| !c.is_active), + "node {node_idx} must hold no active source collection" + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/move_tenant_reissue_retry.rs b/nodedb-cluster-tests/tests/common_suite/cases/move_tenant_reissue_retry.rs new file mode 100644 index 000000000..c21a882c7 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/move_tenant_reissue_retry.rs @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A `MOVE TENANT` retried after its cutover failed past the re-issue leaves +//! each moved row in the target exactly once. +//! +//! Columnar and timeseries ingests append. The fail point +//! `move_tenant::cutover::before_proposal` fails the first cutover after +//! every row is re-issued into the target and before the catalog proposal. +//! The retry re-issues the same rows again. The re-issue clears each +//! append-only target collection first, so the counts stay exact. +//! +//! Requires `--features failpoints`. + +#![cfg(feature = "failpoints")] + +use std::time::Duration; + +use nodedb_types::fail_point::FailGuard; + +use crate::common; +use common::cluster_harness::TestCluster; +use common::cluster_harness::shared_steps::{db_detail, use_database}; + +const SOURCE: &str = "mt_rt_src"; +const TARGET: &str = "mt_rt_tgt"; +const MOVED_TENANT: &str = "mt_rt_owner"; +const COLUMNAR: &str = "mt_rt_cols"; +const TIMESERIES: &str = "mt_rt_ts"; +const ROWS: usize = 4; +const FAULT: &str = "move_tenant::cutover::before_proposal"; + +async fn create_collections(cluster: &TestCluster, database: &str) { + use_database(cluster, database).await; + for ddl in [ + format!( + "CREATE COLLECTION {COLUMNAR} COLUMNS (id TEXT, region TEXT, ts BIGINT) WITH (engine='columnar')" + ), + format!( + "CREATE COLLECTION {TIMESERIES} \ + COLUMNS (id TEXT, ts BIGINT TIME_KEY, metric TEXT, value FLOAT) \ + WITH (engine='timeseries')" + ), + ] { + cluster + .exec_ddl_on_any_leader(&ddl) + .await + .unwrap_or_else(|e| panic!("{ddl} in {database}: {e}")); + } +} + +async fn count(cluster: &TestCluster, node_idx: usize, collection: &str) -> String { + let sql = format!("SELECT COUNT(*) FROM {collection}"); + let messages = cluster.nodes[node_idx] + .client + .simple_query(&sql) + .await + .unwrap_or_else(|e| panic!("{sql} on node {node_idx}: {}", db_detail(&e))); + messages + .into_iter() + .find_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .unwrap_or_else(|| panic!("{sql} on node {node_idx} returned no row")) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_retried_move_holds_each_append_only_row_once() { + let cluster = TestCluster::spawn_three().await.expect("cluster"); + for database in [SOURCE, TARGET] { + cluster + .exec_ddl_on_any_leader(&format!("CREATE DATABASE {database}")) + .await + .unwrap_or_else(|e| panic!("CREATE DATABASE {database}: {e}")); + } + create_collections(&cluster, SOURCE).await; + create_collections(&cluster, TARGET).await; + + use_database(&cluster, SOURCE).await; + for i in 0..ROWS { + let ts = (i + 1) * 1000; + for sql in [ + format!("INSERT INTO {COLUMNAR} (id, region, ts) VALUES ('c{i}', 'r{i}', {ts})"), + format!( + "INSERT INTO {TIMESERIES} (id, ts, metric, value) VALUES ('t{i}', {ts}, 'cpu', 1.0)" + ), + ] { + cluster.nodes[0] + .client + .simple_query(&sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {}", db_detail(&e))); + } + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + + use_database(&cluster, "default").await; + cluster + .exec_ddl_on_any_leader(&format!("CREATE TENANT {MOVED_TENANT} ID 78")) + .await + .unwrap_or_else(|e| panic!("CREATE TENANT {MOVED_TENANT}: {e}")); + let move_sql = format!("MOVE TENANT {MOVED_TENANT} FROM {SOURCE} TO {TARGET}"); + + // First attempt: every row reaches the target, then the cutover fails. + { + let _fault = FailGuard::fail(FAULT, "injected before the cutover proposal"); + let refused = cluster.nodes[0].exec(&move_sql).await; + assert!( + refused.is_err(), + "the first MOVE TENANT must fail at {FAULT}" + ); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + use_database(&cluster, TARGET).await; + for collection in [COLUMNAR, TIMESERIES] { + assert_eq!( + count(&cluster, 0, collection).await, + ROWS.to_string(), + "{TARGET}.{collection} must hold the rows the failed attempt re-issued" + ); + } + + // Retry: the re-issue replaces each target collection's rows. + use_database(&cluster, "default").await; + cluster + .exec_ddl_on_any_leader(&move_sql) + .await + .unwrap_or_else(|e| panic!("retried {move_sql}: {e}")); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + + use_database(&cluster, TARGET).await; + for node_idx in 0..cluster.nodes.len() { + for collection in [COLUMNAR, TIMESERIES] { + assert_eq!( + count(&cluster, node_idx, collection).await, + ROWS.to_string(), + "{TARGET}.{collection} on node {node_idx} must hold each row once" + ); + } + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs index ecb671784..c2c3fc55e 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/native_gateway_migration.rs @@ -32,6 +32,7 @@ fn test_ctx() -> QueryContext { trace_id: nodedb_types::TraceId::ZERO, database_id: nodedb_types::id::DatabaseId::DEFAULT, txn_id: None, + linearizable: false, } } @@ -75,7 +76,7 @@ async fn native_gateway_migration_single_node_select() { key: b"native-key".to_vec(), value: mp_string("native-value"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"native-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -146,7 +147,7 @@ async fn native_gateway_migration_cross_node_select() { key: b"cross-native-key".to_vec(), value: mp_string("cross-native-value"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"cross-native-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -212,6 +213,7 @@ fn native_gateway_error_not_leader_code() { vshard_id: VShardId::new(1), leader_node: 2, leader_addr: "10.0.0.1:9000".into(), + leader_term: 1, }; let (code, msg) = GatewayErrorMap::to_native(&err); assert_eq!(code, ErrorCode::NOT_LEADER, "got {code}"); diff --git a/nodedb-cluster-tests/tests/common_suite/cases/native_write_drain_gate.rs b/nodedb-cluster-tests/tests/common_suite/cases/native_write_drain_gate.rs new file mode 100644 index 000000000..4f9d409c5 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/native_write_drain_gate.rs @@ -0,0 +1,131 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! A native-protocol write, which never runs the SQL planner, is refused while +//! its collection is under a descriptor drain, and succeeds once the drain +//! ends. +//! +//! Every write entry point takes a descriptor lease on the collections it +//! touches. The lease acquire refuses a drained descriptor, so the drain gates +//! non-SQL writes the same way it gates planned ones. + +use std::sync::Arc; +use std::time::Duration; + +use nodedb_client::NodeDb; +use nodedb_cluster::{DescriptorId, DescriptorKind}; +use nodedb_types::document::Document; + +use crate::common; +use common::cluster_harness::{TestCluster, wait_for}; + +const TENANT: u64 = 1; +const COLLECTION: &str = "drain_gate_docs"; +const WAIT_BUDGET: Duration = Duration::from_secs(5); +const POLL: Duration = Duration::from_millis(20); + +fn doc(id: &str) -> Document { + let mut doc = Document::new(id); + doc.set("body", nodedb_types::Value::String(id.to_string())); + doc +} + +async fn count(cluster: &TestCluster) -> String { + let messages = cluster.nodes[0] + .client + .simple_query(&format!("SELECT COUNT(*) FROM {COLLECTION}")) + .await + .expect("count"); + messages + .into_iter() + .find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .expect("a count row") +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn native_write_is_refused_during_a_drain() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLLECTION} WITH (engine='document_schemaless')" + )) + .await + .expect("CREATE COLLECTION"); + + let version = cluster.nodes[0] + .shared + .credentials + .catalog() + .get_collection(nodedb_types::DatabaseId::DEFAULT, TENANT, COLLECTION) + .expect("catalog read") + .expect("collection exists") + .descriptor_version + .max(1); + let id = DescriptorId::new( + 0, + TENANT, + DescriptorKind::Collection, + COLLECTION.to_string(), + ); + + // Start the drain through the metadata group, as a DDL does. + let shared = Arc::clone(&cluster.nodes[0].shared); + let start_id = id.clone(); + tokio::spawn(async move { + let now = shared.hlc_clock.now(); + let entry = nodedb_cluster::MetadataEntry::DescriptorDrainStart { + descriptor_id: start_id, + up_to_version: version, + expires_at: nodedb_types::Hlc::new(now.wall_ns.saturating_add(60_000_000_000), 0), + proposer_node_id: shared.node_id, + owner: nodedb_cluster::DrainOwner::Ddl, + }; + let handle = shared.metadata_raft.get().expect("metadata raft handle"); + handle + .propose_async(nodedb_cluster::encode_entry(&entry).expect("encode")) + .await + .expect("propose drain start"); + }) + .await + .expect("join"); + wait_for("every node observes the drain", WAIT_BUDGET, POLL, || { + cluster.nodes.iter().all(|n| n.has_drain_for(&id, version)) + }) + .await; + + let native = cluster.nodes[1].native_client(); + let refused = native.document_put(COLLECTION, doc("during")).await; + assert!( + refused.is_err(), + "a native write to a drained collection must be refused" + ); + // A read takes a lease on the collection too, so the drain refuses it as + // well. The count after the drain ends shows the refused write never + // landed. + + // End the drain. The same write lands. + let shared = Arc::clone(&cluster.nodes[0].shared); + nodedb::control::lease::end_drain_async(&shared, id.clone(), nodedb_cluster::DrainOwner::Ddl) + .await + .expect("end drain"); + wait_for("every node clears the drain", WAIT_BUDGET, POLL, || { + cluster.nodes.iter().all(|n| !n.has_drain_for(&id, version)) + }) + .await; + + native + .document_put(COLLECTION, doc("after")) + .await + .expect("a native write after the drain must land"); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(10)) + .await; + assert_eq!( + count(&cluster).await, + "1", + "only the write after the drain lands" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/owner_read_on_non_replica.rs b/nodedb-cluster-tests/tests/common_suite/cases/owner_read_on_non_replica.rs new file mode 100644 index 000000000..96a9a7a7f --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/owner_read_on_non_replica.rs @@ -0,0 +1,126 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A read issued on a node that does not replicate the collection's home +//! group reads the rows from the group's owner. +//! +//! With a replication factor of 1, each data group lives on one node. The +//! other nodes hold no rows of the collection. `CREATE GRAPH INDEX` scans the +//! collection, and `COPY ... TO` reads it through the internal dispatch +//! funnel. Both run on a node outside the group and must see every row. + +use std::time::Duration; + +use crate::common; +use common::cluster_harness::TestCluster; +use common::cluster_harness::shared_steps::db_detail; +use common::cluster_harness::wait::wait_for; + +const COLLECTION: &str = "owner_read_docs"; +const INDEX: &str = "owner_read_reports"; +const ROOT: &str = "root"; +const CHILDREN: [&str; 3] = ["c0", "c1", "c2"]; + +/// The first row's `column` value from a simple query. +async fn first_value(client: &tokio_postgres::Client, sql: &str, column: &str) -> String { + let messages = client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {}", db_detail(&e))); + messages + .iter() + .find_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(column).map(str::to_owned), + _ => None, + }) + .unwrap_or_else(|| panic!("{sql}: no row with column `{column}`")) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn reads_on_a_node_outside_the_home_group_see_every_row() { + let cluster = TestCluster::spawn_three_with_replication_factor(1) + .await + .expect("cluster"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {COLLECTION}")) + .await + .expect("CREATE COLLECTION"); + let group_id = cluster.nodes[0] + .group_id_for_collection(COLLECTION) + .expect("the collection's data group"); + // Placement convergence removes the nodes outside the group's placement, + // so exactly one node ends up replicating it. + wait_for( + "exactly one node replicates the collection's group", + Duration::from_secs(30), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .filter(|node| node.replicates_data_group(group_id)) + .count() + == 1 + }, + ) + .await; + let reader = cluster + .nodes + .iter() + .find(|node| !node.replicates_data_group(group_id)) + .expect("a node that does not replicate the group"); + + reader + .client + .simple_query(&format!( + "INSERT INTO {COLLECTION} (id, parent) VALUES ('{ROOT}', '')" + )) + .await + .unwrap_or_else(|e| panic!("insert {ROOT}: {}", db_detail(&e))); + for child in CHILDREN { + reader + .client + .simple_query(&format!( + "INSERT INTO {COLLECTION} (id, parent) VALUES ('{child}', '{ROOT}')" + )) + .await + .unwrap_or_else(|e| panic!("insert {child}: {}", db_detail(&e))); + } + + // The index scan reads the collection's home group from its owner. A + // local-only scan on this node finds no rows and creates no edges. + let edges_created = first_value( + &reader.client, + &format!("CREATE GRAPH INDEX {INDEX} ON {COLLECTION} (parent -> id)"), + "edges_created", + ) + .await; + assert_eq!( + edges_created, + CHILDREN.len().to_string(), + "CREATE GRAPH INDEX on a node outside the home group must index every row" + ); + + // COPY TO reads through the internal dispatch funnel on this node. + let directory = tempfile::tempdir().expect("temporary directory"); + let path = directory.path().join("owner_read.ndjson"); + let path_text = path.to_str().expect("UTF-8 temporary path"); + reader + .client + .simple_query(&format!( + "COPY {COLLECTION} TO '{path_text}' WITH (FORMAT ndjson)" + )) + .await + .unwrap_or_else(|e| panic!("COPY TO: {}", db_detail(&e))); + let exported = std::fs::read_to_string(&path).expect("read the exported file"); + let rows = exported + .lines() + .filter(|line| !line.trim().is_empty()) + .count(); + assert_eq!( + rows, + CHILDREN.len() + 1, + "a funnel read on a node outside the home group must return every row, got: {exported}" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/pgwire_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/pgwire_gateway_migration.rs index 8a3b46cdb..0406d231d 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/pgwire_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/pgwire_gateway_migration.rs @@ -35,6 +35,7 @@ fn test_ctx() -> QueryContext { trace_id: nodedb_types::TraceId::ZERO, database_id: nodedb_types::id::DatabaseId::DEFAULT, txn_id: None, + linearizable: false, } } @@ -188,7 +189,7 @@ async fn pgwire_gateway_migration_plan_cache_hits() { key: b"cache-key".to_vec(), value: mp_string("cache-val"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"cache-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, diff --git a/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs b/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs index cd1a628f3..740a0be21 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/proposal_committed_twice_applies_once.rs @@ -104,6 +104,7 @@ async fn a_proposal_committed_twice_moves_the_counter_once() { TenantId::new(TENANT), b"ctr", ) + .await .expect("the seeded key has a surrogate"); let plan = PhysicalPlan::Kv(KvOp::Incr { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, COLL), @@ -122,7 +123,7 @@ async fn a_proposal_committed_twice_moves_the_counter_once() { to_replicated_entry(TenantId::new(TENANT), DatabaseId::DEFAULT, vshard, &write) .expect("encode the proposal") .expect("KV_INCR encodes to a replicated entry"); - let bytes = entry.to_bytes(); + let bytes = entry.encode().expect("encode the proposal bytes"); entry_bytes = Some(bytes.clone()); bytes } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/resp_gateway_migration.rs b/nodedb-cluster-tests/tests/common_suite/cases/resp_gateway_migration.rs index 2a77900cc..12e7841ef 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/resp_gateway_migration.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/resp_gateway_migration.rs @@ -28,6 +28,7 @@ fn test_ctx() -> QueryContext { trace_id: nodedb_types::TraceId::ZERO, database_id: nodedb_types::id::DatabaseId::DEFAULT, txn_id: None, + linearizable: false, } } @@ -71,7 +72,7 @@ async fn resp_gateway_migration_single_node_set_get() { key: b"mykey".to_vec(), value: mp_string("myvalue"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"mykey".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -144,7 +145,7 @@ async fn resp_gateway_migration_cross_node_get() { key: b"cross-key".to_vec(), value: mp_string("cross-value"), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_test_support::kv_rows::kv_row_surrogate(b"cross-key".as_ref()), returning: None, rls_filters: Vec::new(), provenance: None, @@ -213,6 +214,7 @@ fn resp_gateway_error_not_leader_is_moved() { vshard_id: VShardId::new(1), leader_node: 2, leader_addr: "10.0.0.2:9000".into(), + leader_term: 1, }; let msg = GatewayErrorMap::to_resp(&err); assert!( diff --git a/nodedb-cluster-tests/tests/common_suite/cases/routing_leader_hint_follows_election.rs b/nodedb-cluster-tests/tests/common_suite/cases/routing_leader_hint_follows_election.rs new file mode 100644 index 000000000..5e925b2df --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/routing_leader_hint_follows_election.rs @@ -0,0 +1,138 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Every node's routing leader hint follows its own Raft's elections. +//! +//! Killing a group leader makes SWIM clear the hint, and nothing in the +//! metadata log names the new leader. Each surviving node's Raft tick +//! writes the leader it observes, at its term, into its routing table. So +//! within one election timeout of the new leader's election, every survivor's +//! hint names it. Backup source selection, change-bus forwarding and the +//! gateway read that hint. + +use crate::common; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for, wait_for_report}; + +use std::collections::BTreeMap; +use std::time::{Duration, Instant}; + +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(20); +/// The harness's fast election tuning: `election_timeout_max_ms`. +const ELECTION_TIMEOUT: Duration = Duration::from_millis(1_000); + +/// Every hosted group's leader as `node`'s Raft reports it. +fn raft_leaders(node: &TestClusterNode) -> BTreeMap { + node.all_group_leaders().into_iter().collect() +} + +/// The routing hint of `group_id` on `node`. +fn hinted_leader(node: &TestClusterNode, group_id: u64) -> u64 { + node.shared + .cluster_routing + .as_ref() + .expect("cluster routing") + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_info(group_id) + .map_or(0, |info| info.leader) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn every_survivor_hints_the_new_leader_within_an_election_timeout() { + let mut cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + + // A data group, and the node leading it. + let (group_id, leader) = { + let leaders = raft_leaders(&cluster.nodes[0]); + leaders + .into_iter() + .find(|&(group_id, leader)| group_id != 0 && leader != 0) + .expect("a data group with a leader") + }; + wait_for( + "every node hints the current leader", + CONVERGE, + STEP, + || { + cluster + .nodes + .iter() + .all(|node| hinted_leader(node, group_id) == leader) + }, + ) + .await; + + let idx = cluster + .nodes + .iter() + .position(|node| node.node_id == leader) + .expect("the leader is a member"); + let dead = cluster.nodes.remove(idx); + dead.shutdown().await; + + // Wait for the survivors' Raft to agree on a new leader. + let nodes = &cluster.nodes; + wait_for("the survivors elect a new leader", CONVERGE, STEP, || { + let first = raft_leaders(&nodes[0]).get(&group_id).copied().unwrap_or(0); + first != 0 + && first != leader + && nodes + .iter() + .all(|node| raft_leaders(node).get(&group_id).copied() == Some(first)) + }) + .await; + let elected = raft_leaders(&nodes[0])[&group_id]; + + // Within one election timeout, every survivor's hint names it. + let deadline = Instant::now() + ELECTION_TIMEOUT; + loop { + let hints: Vec = nodes + .iter() + .map(|node| hinted_leader(node, group_id)) + .collect(); + if hints.iter().all(|&hint| hint == elected) { + break; + } + assert!( + Instant::now() < deadline, + "group {group_id}: survivors hint {hints:?}, the new leader is {elected}" + ); + tokio::time::sleep(STEP).await; + } + + // Every other routed group the survivors host hints its Raft leader too. + // The Calvin sequencer group is not part of the routing topology, so no + // node holds a hint for it. + for node in nodes { + for (group, raft_leader) in raft_leaders(node) { + if group == nodedb_cluster::calvin::SEQUENCER_GROUP_ID + || raft_leader == 0 + || raft_leader == leader + { + continue; + } + wait_for_report( + "each hint follows its group's Raft leader", + CONVERGE, + STEP, + || { + let current = raft_leaders(node).get(&group).copied(); + let hint = hinted_leader(node, group); + if current != Some(raft_leader) || hint == raft_leader { + Ok(()) + } else { + Err(format!( + "node {}: group {group}: raft leader {raft_leader}, hint {hint}", + node.node_id + )) + } + }, + ) + .await; + } + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/routing_redirect_hint_unhosted_group.rs b/nodedb-cluster-tests/tests/common_suite/cases/routing_redirect_hint_unhosted_group.rs new file mode 100644 index 000000000..6fe3b9af1 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/routing_redirect_hint_unhosted_group.rs @@ -0,0 +1,220 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A node that hosts no replica of a group follows leader redirects by term. +//! +//! With a replication factor of 2, one of the three nodes, the outsider, +//! hosts no replica of a data group. Its Raft never observes that group's +//! leader, so its routing hint moves by redirects. After the group elects a +//! new leader, the outsider's leader probe asks the old leader its hint +//! names, which redirects it with the new leader and the new term. The hint +//! then names the new leader at the new term, and the next read-index +//! request reaches it. A later redirect at the old term, through the +//! gateway's retry path, does not move the hint back. + +use crate::common; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use std::time::Duration; + +use nodedb::control::gateway::retry::retry_not_leader; +use nodedb::types::VShardId; + +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(20); +const READ_INDEX_TIMEOUT: Duration = Duration::from_secs(5); +/// Each bounce elects either replica. Ten bounces that all re-elect the +/// same replica mean the election is not moving at all. +const MAX_BOUNCES: usize = 10; + +/// The routing hint of `group_id` on `node`: `(leader, leader_term)`. +fn hint(node: &TestClusterNode, group_id: u64) -> (u64, u64) { + node.shared + .cluster_routing + .as_ref() + .expect("cluster routing") + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_info(group_id) + .map_or((0, 0), |info| (info.leader, info.leader_term)) +} + +/// The leader of `group_id` as `node`'s Raft reports it, `0` when none. +fn raft_leader(node: &TestClusterNode, group_id: u64) -> u64 { + node.all_group_leaders() + .into_iter() + .find(|&(group, _)| group == group_id) + .map_or(0, |(_, leader)| leader) +} + +fn node_by_id(cluster: &TestCluster, node_id: u64) -> &TestClusterNode { + cluster + .nodes + .iter() + .find(|node| node.node_id == node_id) + .expect("a cluster member") +} + +/// Wait until both replicas' Raft agree on a leader of `group_id`, and both +/// replicas' routing hints name it at the same term. Returns that +/// `(leader, term)`. +async fn settled_leader(cluster: &TestCluster, group_id: u64, replicas: &[u64]) -> (u64, u64) { + wait_for("both replicas settle on one leader", CONVERGE, STEP, || { + let hints: Vec<(u64, u64)> = replicas + .iter() + .map(|&id| hint(node_by_id(cluster, id), group_id)) + .collect(); + let (leader, term) = hints[0]; + leader != 0 + && term != 0 + && hints.iter().all(|&h| h == (leader, term)) + && replicas + .iter() + .all(|&id| raft_leader(node_by_id(cluster, id), group_id) == leader) + }) + .await; + hint(node_by_id(cluster, replicas[0]), group_id) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn an_outsider_follows_a_redirect_to_the_new_leader_and_ignores_a_stale_one() { + let mut cluster = TestCluster::spawn_three_with_replication_factor(2) + .await + .expect("spawn 3-node cluster with replication factor 2"); + + // A vShard-owning data group with two replicas, and the node outside it. + let (group_id, replicas, vshard_id) = { + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + routing + .group_ids() + .into_iter() + .filter(|&group| group != nodedb_cluster::METADATA_GROUP_ID) + .find_map(|group| { + let info = routing.group_info(group)?; + let vshard = routing.vshards_for_group(group).first().copied()?; + (info.members.len() == 2).then(|| (group, info.members.clone(), vshard)) + }) + .expect("a data group with two replicas") + }; + let outsider_id = cluster + .nodes + .iter() + .map(|node| node.node_id) + .find(|id| !replicas.contains(id)) + .expect("a node outside the group"); + assert!( + !node_by_id(&cluster, outsider_id).hosts_data_group(group_id), + "node {outsider_id} must host no replica of group {group_id}" + ); + + // Bounce the follower until the group elects the other replica. With its + // follower down, the leader loses quorum contact and steps down, so the + // follower's return starts a new election at a higher term. + let (mut old_leader, mut old_term) = settled_leader(&cluster, group_id, &replicas).await; + let mut bounces = 0; + let (new_leader, new_term) = loop { + bounces += 1; + assert!( + bounces <= MAX_BOUNCES, + "group {group_id} re-elected node {old_leader} {MAX_BOUNCES} times" + ); + let follower = replicas + .iter() + .copied() + .find(|&id| id != old_leader) + .expect("the group's other replica"); + let index = cluster + .nodes + .iter() + .position(|node| node.node_id == follower) + .expect("the follower is a member"); + let stopped = cluster.stop_member(index).await.expect("stop the follower"); + { + let leader_node = node_by_id(&cluster, old_leader); + wait_for( + "the leader steps down without its quorum", + CONVERGE, + STEP, + || raft_leader(leader_node, group_id) != old_leader, + ) + .await; + } + cluster + .restart_member(stopped) + .await + .expect("restart the follower"); + let (leader, term) = settled_leader(&cluster, group_id, &replicas).await; + assert!(term > old_term, "the new election runs at a higher term"); + if leader != old_leader { + break (leader, term); + } + old_term = term; + old_leader = leader; + }; + + let outsider = node_by_id(&cluster, outsider_id); + wait_for( + "the outsider sees every node active", + CONVERGE, + STEP, + || outsider.active_topology_size() == 3, + ) + .await; + let routing = outsider + .shared + .cluster_routing + .as_ref() + .expect("cluster routing"); + + // The outsider's leader probe asks the node its hint names. The old + // leader redirects it, so the hint moves to the new leader at the new + // term. + wait_for( + "the outsider's hint follows the new leader", + CONVERGE, + STEP, + || hint(outsider, group_id) == (new_leader, new_term), + ) + .await; + + // The next request reaches the new leader. + let gate = outsider + .shared + .raft_read_gate + .get() + .expect("raft read gate") + .clone(); + let read = gate.read_index(group_id, READ_INDEX_TIMEOUT).await; + assert!( + read.is_ok(), + "the new leader {new_leader} answers the read index: {read:?}" + ); + + // A stale redirect naming the old leader at the old term, through the + // gateway's retry path, leaves the hint on the new leader. + let _ = retry_not_leader(Some(&**routing), |attempt| async move { + if attempt == 0 { + Err(nodedb::Error::NotLeader { + vshard_id: VShardId::new(vshard_id), + leader_node: old_leader, + leader_addr: String::new(), + leader_term: old_term, + }) + } else { + Ok::<(), nodedb::Error>(()) + } + }) + .await; + assert_eq!( + hint(outsider, group_id), + (new_leader, new_term), + "a redirect at the old term never moves the hint back" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_cross_node.rs index 08beb3d1a..df1cb124c 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_aggregate_cross_node.rs @@ -183,6 +183,7 @@ async fn produce( deadline_remaining_ms: 15_000, trace_id: [0u8; 16], descriptor_versions: vec![], + read_groups: vec![], }; let resp = transport .send_rpc_to_addr(producer_addr, RaftRpc::ShuffleProduceRequest(req)) diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_consume_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_consume_cross_node.rs index f343e19fe..4085e4d72 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_consume_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_consume_cross_node.rs @@ -150,6 +150,7 @@ async fn produce_side( deadline_remaining_ms: 15_000, trace_id: [0u8; 16], descriptor_versions: vec![], + read_groups: vec![], }; let resp = transport .send_rpc_to_addr(producer_addr, RaftRpc::ShuffleProduceRequest(req)) diff --git a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_produce_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_produce_cross_node.rs index 49f8bf830..f4351274d 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/shuffle_produce_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/shuffle_produce_cross_node.rs @@ -189,6 +189,7 @@ async fn shuffle_produce_partitions_and_fans_out_across_nodes() { deadline_remaining_ms: 10_000, trace_id: [0u8; 16], descriptor_versions: vec![], // ProviderScan touches no catalog collection + read_groups: vec![], }; // The coordinator drives the producer node directly via its QUIC address and diff --git a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin.rs b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin.rs index 08ce5de9d..b27c3269c 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin.rs @@ -1,28 +1,18 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Single-node Calvin always-on: a STANDALONE (non-cluster) server can run the -//! full Calvin stack — sequencer Raft group + per-vShard schedulers — so a +//! Single-node Calvin is always on: a server with no `[cluster]` runs the full +//! Calvin stack — sequencer Raft group + per-vShard schedulers — so a //! cross-core (cross-vShard) transaction traverses the SAME deterministic -//! Calvin path it would in a multi-node cluster. -//! -//! Gated behind `server.single_node_calvin` (default `false`). When the flag is -//! off, the standalone server starts no Calvin stack and every write stays on -//! the existing single-node path; when it is on, the server synthesizes a +//! Calvin path as in a multi-node cluster. The server synthesizes a //! one-node cluster (self-seeded, replication factor 1) via //! `init_single_node_calvin`. //! -//! ## What each test proves -//! -//! - `flag_on`: with the flag set, `calvin_available` becomes true -//! (`cluster_transport` + `sequencer_inbox` are both wired) and an AUTOCOMMIT -//! cross-shard write — a `GRAPH INSERT EDGE` whose endpoints home to DISTINCT -//! vShards — is admitted to a Calvin epoch (the sequencer's `admitted_total` -//! advances) and dual-homes atomically (a reverse/IN traversal from the -//! destination reaches the source). That is the sequencer→scheduler path, not -//! the single-home fast path and not a `SequencerUnavailable` error. -//! - `flag_off`: a standalone server with no Calvin stack reports -//! `calvin_available == false` and single-shard writes still commit on the -//! fast path. +//! The test proves that `cluster_transport` and `sequencer_inbox` are both +//! wired, and that an AUTOCOMMIT cross-shard write — a `GRAPH INSERT EDGE` +//! whose endpoints home to DISTINCT vShards — is admitted to a Calvin epoch +//! (the sequencer's `admitted_total` advances) and dual-homes atomically (a +//! reverse/IN traversal from the destination reaches the source). That is the +//! sequencer→scheduler path, not the single-home fast path. use crate::common; @@ -34,7 +24,6 @@ use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; use nodedb_types::id::VShardId; use common::cluster_harness::{TestClusterNode, wait_for}; -use common::pgwire_harness::TestServer; /// Observed sequencer-group leader id from a node's local Raft status, or `0` /// if no leader is known yet. @@ -104,11 +93,10 @@ fn traversed_node_ids(v: &serde_json::Value) -> HashSet { .collect() } -/// Flag ON: a standalone server with `single_node_calvin = true` runs the -/// Calvin stack and routes an autocommit cross-shard edge insert through the -/// single-node sequencer→scheduler path. +/// A server with no `[cluster]` runs the Calvin stack and routes an autocommit +/// cross-shard edge insert through the single-node sequencer→scheduler path. #[tokio::test(flavor = "multi_thread", worker_threads = 4)] -async fn single_node_calvin_flag_on_routes_cross_shard_write_through_sequencer() { +async fn single_node_calvin_routes_cross_shard_write_through_sequencer() { // 4 Data-Plane cores so distinct vShards land on distinct cores — a genuine // cross-core transaction. let node = TestClusterNode::spawn_single_node_calvin(4) @@ -124,14 +112,13 @@ async fn single_node_calvin_flag_on_routes_cross_shard_write_through_sequencer() ) .await; - // calvin_available == true: both fields the neutral gate checks are wired. assert!( node.shared.cluster_transport.is_some(), "single-node calvin must install cluster_transport" ); assert!( node.shared.sequencer_inbox.get().is_some(), - "single-node calvin must install sequencer_inbox → calvin_available" + "single-node calvin must install sequencer_inbox" ); // A graph-capable collection. @@ -150,8 +137,8 @@ async fn single_node_calvin_flag_on_routes_cross_shard_write_through_sequencer() let (src, dst) = distinct_vshard_node_keys(); let admitted_before = sequencer_admitted(&node); - // AUTOCOMMIT cross-shard edge insert. Because `calvin_available` is true and - // the endpoints home to distinct vShards, `insert_edge` dual-homes the edge + // AUTOCOMMIT cross-shard edge insert. Because the endpoints home to + // distinct vShards, `insert_edge` dual-homes the edge // atomically through Calvin (`submit_calvin_routed`) instead of taking the // single-home fast path. If the single-node sequencer stack were not // operational this would error (`SequencerUnavailable`) or time out. @@ -188,46 +175,3 @@ async fn single_node_calvin_flag_on_routes_cross_shard_write_through_sequencer() node.shutdown().await; } - -/// Flag OFF: a standalone server starts no Calvin stack, so `calvin_available` -/// is false and single-shard writes commit on the existing fast path. -#[tokio::test(flavor = "multi_thread", worker_threads = 2)] -async fn single_node_calvin_flag_off_keeps_fast_path() { - let server = TestServer::start().await; - - // No cluster/Calvin stack was started on a standalone server. - assert!( - server.shared.cluster_transport.is_none(), - "standalone server (flag off) must not install cluster_transport" - ); - assert!( - server.shared.sequencer_inbox.get().is_none(), - "standalone server (flag off) must leave sequencer_inbox unset → calvin_available false" - ); - - // A single-shard write commits and reads back on the fast path. - server - .client - .simple_query("CREATE COLLECTION sncalvin_fastpath (id TEXT PRIMARY KEY, v TEXT)") - .await - .expect("CREATE COLLECTION sncalvin_fastpath"); - server - .client - .simple_query("INSERT INTO sncalvin_fastpath (id, v) VALUES ('k1', 'hello')") - .await - .expect("single-shard INSERT on the fast path"); - - let rows = server - .client - .simple_query("SELECT v FROM sncalvin_fastpath WHERE id = 'k1'") - .await - .expect("SELECT sncalvin_fastpath"); - let count = rows - .iter() - .filter(|m| matches!(m, tokio_postgres::SimpleQueryMessage::Row(_))) - .count(); - assert_eq!( - count, 1, - "single-shard row must be present after the fast-path write" - ); -} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_graph_txn.rs b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_graph_txn.rs index babc516ea..df6e70c11 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_graph_txn.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/single_node_calvin_graph_txn.rs @@ -13,8 +13,8 @@ //! endpoint the edge must be staged into BOTH overlays. //! //! `dual_home_edge_stages_both_overlays_and_rollback_tears_down` proves the new -//! behavior end to end on a `single_node_calvin` server (where `calvin_available` -//! is true so a cross-shard edge is genuinely dual-home, not forced single-home): +//! behavior end to end on a `single_node_calvin` server, where a cross-shard +//! edge is dual-home: //! //! 1. `BEGIN`; `GRAPH INSERT EDGE` across two distinct vShards is ACCEPTED (no //! `CrossShardInExplicitTransaction`) and staged. @@ -115,9 +115,8 @@ async fn dual_home_edge_stages_both_overlays_and_rollback_tears_down() { .await .expect("spawn standalone single-node-calvin server"); - // The lone sequencer voter self-elects; wait for it so `calvin_available` is - // genuinely operational (a cross-shard edge is dual-home, not forced - // single-home). + // The lone sequencer voter self-elects; wait for it so the sequencer is + // operational before the cross-shard edge write. wait_for( "single-node sequencer leader elected", Duration::from_secs(10), @@ -127,7 +126,7 @@ async fn dual_home_edge_stages_both_overlays_and_rollback_tears_down() { .await; assert!( node.shared.cluster_transport.is_some() && node.shared.sequencer_inbox.get().is_some(), - "single-node calvin must wire calvin_available (cluster_transport + sequencer_inbox)" + "single-node calvin must wire cluster_transport and sequencer_inbox" ); node.client diff --git a/nodedb-cluster-tests/tests/common_suite/cases/sql_schedule_lease_gating.rs b/nodedb-cluster-tests/tests/common_suite/cases/sql_schedule_lease_gating.rs new file mode 100644 index 000000000..b32e2e6d4 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/sql_schedule_lease_gating.rs @@ -0,0 +1,177 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A cross-collection SQL schedule fires only on the vShard 0 lease holder. +//! +//! A schedule whose body names no collection runs on the `_system` +//! coordinator: the node that leads vShard 0's group under a leader lease +//! valid now. The schedule fires every minute on the real scheduler, and +//! each node records its own runs in its job history. +//! +//! - While one node holds the lease, only that node records runs. +//! - Once that node is cut off from both peers, its lease lapses and it +//! records no further run. A peer takes the lease and records runs. +//! - The leader balancer can move the lease between the live peers, so +//! either peer can fire. No two nodes ever fire the same minute. +//! +//! The test waits on minute boundaries, so it runs for a few minutes. + +use crate::common; +use common::cluster_harness::shared_steps::holds_vshard0_lease; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use std::sync::Arc; +use std::time::Duration; + +const SCHEDULE: &str = "slg_every_minute"; +const CONVERGE: Duration = Duration::from_secs(30); +/// Long enough for at least one minute boundary and its fire. +const FIRE_WAIT: Duration = Duration::from_secs(150); +const STEP: Duration = Duration::from_millis(250); + +/// The minute of each run `node` recorded for the schedule. +fn run_minutes(node: &TestClusterNode) -> Vec { + let Some(def) = node + .shared + .schedule_registry + .list_all() + .into_iter() + .find(|def| def.name == SCHEDULE) + else { + return Vec::new(); + }; + node.shared + .job_history + .last_runs(def.database_id, def.tenant_id, SCHEDULE, 100) + .iter() + .map(|run| run.started_at / 60_000) + .collect() +} + +/// Runs `node` recorded for the schedule. +fn runs(node: &TestClusterNode) -> usize { + run_minutes(node).len() +} + +/// Cut `node_id` off from every other node, both ways. +fn isolate(cluster: &TestCluster, node_id: u64) { + let transport = |node: &TestClusterNode| { + Arc::clone( + node.shared + .cluster_transport + .as_ref() + .expect("cluster transport"), + ) + }; + let cut = cluster + .nodes + .iter() + .find(|node| node.node_id == node_id) + .map(transport) + .expect("the node is a member"); + for node in cluster.nodes.iter().filter(|node| node.node_id != node_id) { + transport(node).sever(node_id); + cut.sever(node.node_id); + } +} + +fn holder(cluster: &TestCluster) -> Option { + let holders: Vec = cluster + .nodes + .iter() + .filter(|node| holds_vshard0_lease(node)) + .map(|node| node.node_id) + .collect(); + match holders.as_slice() { + [only] => Some(*only), + _ => None, + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_cross_collection_schedule_fires_only_on_the_lease_holder() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + wait_for("one node holds the vShard 0 lease", CONVERGE, STEP, || { + holder(&cluster).is_some() + }) + .await; + let first = holder(&cluster).expect("a lease holder"); + + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE SCHEDULE {SCHEDULE} CRON '* * * * *' AS BEGIN RETURN 1; END" + )) + .await + .expect("create schedule"); + + // Only the holder fires. + wait_for( + "the lease holder fires the schedule", + FIRE_WAIT, + STEP, + || { + cluster + .nodes + .iter() + .any(|node| node.node_id == first && runs(node) > 0) + }, + ) + .await; + assert_eq!(holder(&cluster), Some(first), "the lease stayed put"); + for node in cluster.nodes.iter().filter(|node| node.node_id != first) { + assert_eq!( + runs(node), + 0, + "node {} holds no lease and fires nothing", + node.node_id + ); + } + + // Cut the holder off. Its lease lapses, and a peer takes it. + isolate(&cluster, first); + let cut_off = cluster + .nodes + .iter() + .find(|node| node.node_id == first) + .expect("the cut-off node"); + wait_for("the cut-off node's lease lapses", CONVERGE, STEP, || { + !holds_vshard0_lease(cut_off) + }) + .await; + let lapsed_runs = runs(cut_off); + let peers: Vec<&TestClusterNode> = cluster + .nodes + .iter() + .filter(|node| node.node_id != first) + .collect(); + wait_for("a peer takes the vShard 0 lease", CONVERGE, STEP, || { + peers + .iter() + .filter(|node| holds_vshard0_lease(node)) + .count() + == 1 + }) + .await; + + // A peer fires. The cut-off node, through the same minute boundaries, + // fires nothing more. + wait_for("a peer fires the schedule", FIRE_WAIT, STEP, || { + peers.iter().any(|node| runs(node) > 0) + }) + .await; + assert_eq!( + runs(cut_off), + lapsed_runs, + "a node without its lease fires nothing" + ); + + // One coordinator fires each minute, whichever peer holds the lease. + let mut minutes: Vec = cluster.nodes.iter().flat_map(run_minutes).collect(); + let fired = minutes.len(); + minutes.sort_unstable(); + minutes.dedup(); + assert_eq!(minutes.len(), fired, "two nodes fired the same minute"); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/surrogate_vshard_transfer.rs b/nodedb-cluster-tests/tests/common_suite/cases/surrogate_vshard_transfer.rs index 7e4d96001..7d69d7b8e 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/surrogate_vshard_transfer.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/surrogate_vshard_transfer.rs @@ -67,18 +67,17 @@ fn pg_detail(e: &tokio_postgres::Error) -> String { } } -/// Reassign vshard group 1's leader to `new_leader_node_id` in the routing -/// table on every node. This mirrors the atomic cut-over that -/// `MigrationExecutor::phase3_cutover` achieves via `RoutingChange::LeadershipTransfer` -/// once the vShard migration executor path is fully wired — we reproduce its -/// externally-observable effect (routing table update) directly so the test -/// can assert surrogate durability without requiring the executor. +/// Point vShard group 1's routing hint at `new_leader_node_id` on every +/// node, at one term above the hint's term. A real cut-over transfers +/// leadership, and the transfer's election names the target at a higher +/// term. The test reproduces that routing effect directly, so it can assert +/// surrogate durability without the migration executor. fn simulate_cutover(cluster: &TestCluster, new_leader_node_id: u64) { for node in &cluster.nodes { if let Some(ref routing) = node.shared.cluster_routing { let mut table = routing.write().unwrap_or_else(|p| p.into_inner()); - // Data group (group 1) leader is reassigned to the target node. - table.set_leader(1, new_leader_node_id); + let term = table.group_info(1).map_or(0, |info| info.leader_term); + table.observe_leader(1, new_leader_node_id, term + 1); } } } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/sync_failover.rs b/nodedb-cluster-tests/tests/common_suite/cases/sync_failover.rs index 30c8d8043..7bd9ce2bc 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/sync_failover.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/sync_failover.rs @@ -78,6 +78,7 @@ async fn cluster_sync_producer_fence_survives_failover() { registration.current_epoch, CREATED_MS, ) + .await .expect("propose producer register"); registration.producer_id }; @@ -96,6 +97,7 @@ async fn cluster_sync_producer_fence_survives_failover() { LITE_ID, 5, ) + .await .expect("propose producer fence"); } diff --git a/nodedb-cluster-tests/tests/common_suite/cases/timeseries_calvin_read_your_writes.rs b/nodedb-cluster-tests/tests/common_suite/cases/timeseries_calvin_read_your_writes.rs new file mode 100644 index 000000000..cc9753ad0 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/timeseries_calvin_read_your_writes.rs @@ -0,0 +1,142 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A transaction that commits through Calvin reads its own timeseries rows. +//! +//! The transaction inserts a timeseries row and a document on another +//! vShard, so its COMMIT is sequenced through the single-node Calvin stack. +//! Before COMMIT, a `SELECT` in the same transaction reads the timeseries +//! row. After COMMIT, which resolves the ingest to its rows before it is +//! sequenced and stages those rows on the scheduler, the row reads back +//! committed, with the value the transaction wrote. + +use crate::common; + +use std::time::Duration; + +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_types::DatabaseId; + +use common::cluster_harness::shared_steps::sequencer_admitted; +use common::cluster_harness::{TestClusterNode, wait_for}; + +const TIMESERIES: &str = "sncalvin_ts_ryw"; + +fn sequencer_leader(node: &TestClusterNode) -> u64 { + let Some(status_fn) = node.shared.raft_status_fn.get() else { + return 0; + }; + status_fn() + .into_iter() + .find(|g| g.group_id == SEQUENCER_GROUP_ID) + .map(|g| g.leader_id) + .unwrap_or(0) +} + +fn vshard_of(collection: &str) -> u32 { + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection) + .vshard() + .as_u32() +} + +/// A document collection name whose vShard differs from the timeseries +/// collection's, so a transaction writing both is cross-shard. +fn other_vshard_collection() -> String { + let home = vshard_of(TIMESERIES); + (0u32..4096) + .map(|i| format!("sncalvin_ryw_doc_{i}")) + .find(|name| vshard_of(name) != home) + .expect("a collection name on another vShard within 4096 tries") +} + +/// The `(id, value)` cells of every row `sql` returns. +async fn id_value_rows(node: &TestClusterNode, sql: &str) -> Vec<(String, String)> { + let messages = node + .client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + messages + .into_iter() + .filter_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(row) => Some(( + row.get("id").unwrap_or_default().to_string(), + row.get("value").unwrap_or_default().to_string(), + )), + _ => None, + }) + .collect() +} + +async fn exec(node: &TestClusterNode, sql: &str) { + node.client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_calvin_committed_transaction_reads_its_own_timeseries_row() { + let node = TestClusterNode::spawn_single_node_calvin(4) + .await + .expect("spawn standalone single-node-calvin server"); + wait_for( + "single-node sequencer leader elected", + Duration::from_secs(10), + Duration::from_millis(50), + || sequencer_leader(&node) == node.node_id, + ) + .await; + + let documents = other_vshard_collection(); + exec( + &node, + &format!( + "CREATE COLLECTION {TIMESERIES} \ + COLUMNS (id TEXT, ts BIGINT TIME_KEY, value FLOAT) \ + WITH (engine='timeseries')" + ), + ) + .await; + exec(&node, &format!("CREATE COLLECTION {documents}")).await; + wait_for( + "both collections visible on the node", + Duration::from_secs(10), + Duration::from_millis(50), + || node.cached_collection_count() >= 2, + ) + .await; + + let select = format!("SELECT id, value FROM {TIMESERIES}"); + let expected = vec![("r-1".to_string(), "2.5".to_string())]; + let admitted_before = sequencer_admitted(&node); + + exec(&node, "BEGIN").await; + exec( + &node, + &format!("INSERT INTO {TIMESERIES} (id, ts, value) VALUES ('r-1', 1000, 2.5)"), + ) + .await; + exec( + &node, + &format!("INSERT INTO {documents} {{ id: 'd-1', n: 1 }}"), + ) + .await; + assert_eq!( + id_value_rows(&node, &select).await, + expected, + "the transaction reads its own timeseries row before COMMIT" + ); + exec(&node, "COMMIT").await; + + assert!( + sequencer_admitted(&node) > admitted_before, + "the cross-shard COMMIT is sequenced through Calvin" + ); + assert_eq!( + id_value_rows(&node, &select).await, + expected, + "the committed row reads back with the value the transaction wrote" + ); + + node.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/timeseries_conflicting_types_replicas.rs b/nodedb-cluster-tests/tests/common_suite/cases/timeseries_conflicting_types_replicas.rs new file mode 100644 index 000000000..e23eb72cb --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/timeseries_conflicting_types_replicas.rs @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Two sessions on different nodes give a new timeseries column conflicting +//! types at the same time. The cluster stays up and every replica agrees. +//! +//! Each proposer resolves its ingest against its own replica. Both can +//! resolve before either entry applies, so both entries can reach the log +//! with the column under different types. The log orders them. The later +//! entry's conflicting rows are rejected at its log position on every +//! replica, since every replica holds the same schema there. Its client +//! learns the count as a warning. When the later ingest resolves after the +//! earlier one applied, its resolve rejects the same rows instead, with the +//! same warning. +//! +//! ## Test shape +//! +//! Bring up 3 nodes. Create a timeseries collection that declares only its +//! `ts` time key, with a change stream on it. For each round, open a native +//! session on node 0 and one on node 1, and ingest at once two raw ILP lines +//! each that give a fresh column a float from node 0 and a string from +//! node 1. A fresh column comes only from a raw ILP line (see +//! `ts_native_ingest`). Then: +//! +//! - no node fail-stopped a core; +//! - in each round exactly one of the two ingests reports its two lines +//! rejected, and the other reports none; +//! - every replica, read from its own Data Plane, holds the same rows: two +//! per round, plus the warm-up rows; +//! - every replica's change stream serves one event per stored row. + +use crate::common; +use common::cluster_harness::shared_steps::{fail_stopped, local_timeseries_rows}; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use std::time::{Duration, Instant}; + +use nodedb::event::cdc::consume::{ConsumeError, ConsumeParams, consume_local}; +use nodedb_types::DatabaseId; + +use super::ts_native_ingest::{ + assert_native_ok, ingest_native, ingest_until_accepted, native_session, rejection_warnings, +}; + +const COLLECTION: &str = "ts_conflicting_types"; +const STREAM: &str = "ts_conflicting_types_feed"; +const GROUP: &str = "ts_conflicting_types_readers"; +const TENANT: u64 = 1; + +/// Rounds of concurrent conflicting inserts, each on a fresh column. +const ROUNDS: usize = 8; +/// Rows each insert carries. +const ROWS_PER_INSERT: usize = 2; +/// Rows the warm-up inserts store, one per session node. +const WARM_UP_ROWS: usize = 2; +const TOTAL_ROWS: usize = WARM_UP_ROWS + ROUNDS * ROWS_PER_INSERT; + +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(100); + +/// The ILP lines of `round` from session `session`: two lines that give +/// column `c{round}` a float from session 0 and a string from session 1. +/// Every line has its own timestamp, in nanoseconds. +fn round_lines(round: usize, session: usize) -> String { + let lines: Vec = (0..ROWS_PER_INSERT) + .map(|row| { + let ts_ms = 10_000 * (round as u64 + 1) + 100 * session as u64 + row as u64; + let column_value = if session == 0 { + format!("{}.5", round + row) + } else { + format!("\"s{round}-{row}\"") + }; + format!( + "{COLLECTION} value={value},c{round}={column_value} {ts_ns}", + value = row as f64 + 0.25, + ts_ns = ts_ms * 1_000_000 + ) + }) + .collect(); + lines.join("\n") +} + +/// Whether some core of `node` fail-stopped. +/// The rows `node`'s own replica stores, each rendered canonically, sorted. +/// The number of change events `node` serves after the group's committed +/// offsets. +fn event_count(node: &TestClusterNode) -> usize { + let params = ConsumeParams { + database_id: DatabaseId::DEFAULT, + tenant_id: TENANT, + stream_name: STREAM, + group_name: GROUP, + partition: None, + limit: 10 * TOTAL_ROWS, + }; + match consume_local(&node.shared, ¶ms) { + Ok(result) => result.events.len(), + Err(ConsumeError::BufferEmpty(_)) => 0, + Err(error) => panic!("node {}: consume failed: {error}", node.node_id), + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn concurrent_conflicting_types_stay_up_and_agree_on_every_replica() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLLECTION} (ts BIGINT TIME_KEY) WITH (engine='timeseries')" + )) + .await + .expect("create the timeseries collection"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE CHANGE STREAM {STREAM} ON {COLLECTION}")) + .await + .expect("create the change stream"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE CONSUMER GROUP {GROUP} ON {STREAM}")) + .await + .expect("create the consumer group"); + wait_for( + "every node registers the stream and the group", + CONVERGE, + STEP, + || { + cluster.nodes.iter().all(|node| { + node.has_change_stream(DatabaseId::DEFAULT, TENANT, STREAM) + && node + .shared + .group_registry + .get(DatabaseId::DEFAULT, TENANT, STREAM, GROUP) + .is_some() + }) + }, + ) + .await; + + // Both session nodes accept writes before the rounds start. + for (session, node) in cluster.nodes.iter().take(2).enumerate() { + ingest_until_accepted( + node, + COLLECTION, + &format!( + "{COLLECTION} value=0.5 {}", + (session as u64 + 1) * 1_000_000 + ), + ) + .await; + } + + for round in 0..ROUNDS { + let mut float_session = native_session(&cluster.nodes[0]).await; + let mut string_session = native_session(&cluster.nodes[1]).await; + let float_lines = round_lines(round, 0); + let string_lines = round_lines(round, 1); + let (float_reply, string_reply) = tokio::join!( + ingest_native(&mut float_session, 1, COLLECTION, &float_lines), + ingest_native(&mut string_session, 1, COLLECTION, &string_lines), + ); + assert_native_ok(&float_reply, "the float ingest"); + assert_native_ok(&string_reply, "the string ingest"); + let float_notices = rejection_warnings(&float_reply, COLLECTION); + let string_notices = rejection_warnings(&string_reply, COLLECTION); + let expected = format!("{ROWS_PER_INSERT} line(s)"); + let reported: Vec<&String> = float_notices.iter().chain(&string_notices).collect(); + assert_eq!( + reported.len(), + 1, + "round {round}: exactly the later statement reports its rows rejected; \ + float session {float_notices:?}, string session {string_notices:?}" + ); + assert!( + reported.iter().all(|notice| notice.contains(&expected)), + "round {round}: the later statement reports all {ROWS_PER_INSERT} rows \ + rejected, got {reported:?}" + ); + } + cluster.wait_for_full_apply_convergence(CONVERGE).await; + + for node in &cluster.nodes { + assert!( + !fail_stopped(node), + "node {} fail-stopped a core on a type conflict", + node.node_id + ); + } + + // Every replica stores the same rows: the warm-up rows and the earlier + // insert of each round. + let deadline = Instant::now() + CONVERGE; + let mut per_node: Vec<(u64, Vec)> = Vec::new(); + while Instant::now() < deadline { + per_node.clear(); + for node in &cluster.nodes { + per_node.push(( + node.node_id, + local_timeseries_rows(node, TENANT, COLLECTION).await, + )); + } + if per_node.iter().all(|(_, rows)| rows.len() == TOTAL_ROWS) { + break; + } + tokio::time::sleep(STEP).await; + } + for (node_id, rows) in &per_node { + assert_eq!( + rows.len(), + TOTAL_ROWS, + "node {node_id} stores {} of {TOTAL_ROWS} rows", + rows.len() + ); + } + let (reference_node, reference) = &per_node[0]; + for (node_id, rows) in &per_node[1..] { + assert_eq!( + rows, reference, + "node {node_id} stores other rows than node {reference_node}" + ); + } + + // Every replica serves one event per stored row, and none for a + // rejected row. + wait_for( + "every replica serves one event per stored row", + CONVERGE, + STEP, + || { + cluster + .nodes + .iter() + .all(|node| event_count(node) == TOTAL_ROWS) + }, + ) + .await; + for node in &cluster.nodes { + assert_eq!( + event_count(node), + TOTAL_ROWS, + "node {} serves another event count than it stores rows", + node.node_id + ); + assert!( + !fail_stopped(node), + "node {} fail-stopped a core", + node.node_id + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/timeseries_resolved_replicas.rs b/nodedb-cluster-tests/tests/common_suite/cases/timeseries_resolved_replicas.rs new file mode 100644 index 000000000..56ac3c50d --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/timeseries_resolved_replicas.rs @@ -0,0 +1,235 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A timeseries ingest the leader accepts lands identically on every replica. +//! +//! The proposer resolves each ingest to its rows before the entry exists. +//! Every replica then installs exactly the rows the entry carries. A replica +//! whose memtable budget is far below the others flushes around every +//! record, and still stores every row: a memory limit at apply makes room +//! first and never drops a row. +//! +//! ## Test shape +//! +//! Bring up 3 nodes, node 3 with a memtable budget of a few hundred bytes. +//! Create a timeseries collection with a change stream on it. Through node +//! 0, insert several multi-row batches. Then, on every node: +//! +//! - its own replica, read from its own Data Plane, holds every row, with +//! the same values as every other replica; +//! - its change stream serves one event per row, at the same positions as +//! every other replica. + +use crate::common; +use common::cluster_harness::shared_steps::local_timeseries_rows; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use std::collections::BTreeMap; +use std::time::{Duration, Instant}; + +use nodedb::event::cdc::CdcOffset; +use nodedb::event::cdc::consume::{ConsumeError, ConsumeParams, consume_local}; +use nodedb_types::DatabaseId; +use nodedb_types::config::tuning::TimeseriesToning; + +const COLLECTION: &str = "ts_resolved_replicas"; +const STREAM: &str = "ts_resolved_replicas_feed"; +const GROUP: &str = "ts_resolved_replicas_readers"; +const TENANT: u64 = 1; + +/// The node whose memtable budget is set low. +const LOW_BUDGET_NODE: u64 = 3; +/// Insert statements, each one ingest. +const BATCHES: usize = 5; +/// Rows per insert statement. +const ROWS_PER_BATCH: usize = 20; +const TOTAL_ROWS: usize = BATCHES * ROWS_PER_BATCH; + +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(100); + +fn pg_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +/// A budget every batch passes, so node 3 flushes around every record. +fn low_budget() -> TimeseriesToning { + TimeseriesToning { + memtable_budget_bytes: 256, + memtable_hard_limit_bytes: 512, + ..TimeseriesToning::default() + } +} + +/// The `INSERT` of batch `batch`: `ROWS_PER_BATCH` rows with distinct +/// timestamps, two hosts, and values that differ by row. +fn batch_insert(batch: usize) -> String { + let rows: Vec = (0..ROWS_PER_BATCH) + .map(|row| { + let n = batch * ROWS_PER_BATCH + row; + let host = if n.is_multiple_of(2) { "alpha" } else { "beta" }; + format!( + "('r-{n}', {ts}, '{host}', {value})", + ts = 1_000 * (n as u64 + 1), + value = n as f64 + 0.25 + ) + }) + .collect(); + format!( + "INSERT INTO {COLLECTION} (id, ts, host, value) VALUES {}", + rows.join(", ") + ) +} + +/// Run `sql` through `node`, retrying while the cluster elects a leader. +async fn exec_retrying(node: &TestClusterNode, sql: &str) { + let deadline = Instant::now() + CONVERGE; + loop { + match node.client.simple_query(sql).await { + Ok(_) => return, + Err(error) if Instant::now() < deadline => { + tracing::debug!(error = %pg_detail(&error), "statement not accepted yet; retrying"); + tokio::time::sleep(Duration::from_millis(200)).await; + } + Err(error) => panic!("{sql}: {}", pg_detail(&error)), + } + } +} + +/// The rows `node`'s own replica stores, each rendered canonically, sorted. +/// The change events `node` serves after the group's committed offsets, by +/// partition, each partition in the order it was served. +fn events(node: &TestClusterNode) -> BTreeMap> { + let params = ConsumeParams { + database_id: DatabaseId::DEFAULT, + tenant_id: TENANT, + stream_name: STREAM, + group_name: GROUP, + partition: None, + limit: 10 * TOTAL_ROWS, + }; + let served = match consume_local(&node.shared, ¶ms) { + Ok(result) => result.events, + Err(ConsumeError::BufferEmpty(_)) => Vec::new(), + Err(error) => panic!("node {}: consume failed: {error}", node.node_id), + }; + let mut out: BTreeMap> = BTreeMap::new(); + for event in &served { + out.entry(event.partition) + .or_default() + .push((event.position(), event.row_id.clone())); + } + out +} + +fn event_count(events: &BTreeMap>) -> usize { + events.values().map(Vec::len).sum() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_leader_accepted_ingest_lands_identically_on_every_replica() { + let cluster = + TestCluster::spawn_three_with_node_timeseries_tuning(LOW_BUDGET_NODE, low_budget()) + .await + .expect("spawn 3-node cluster"); + + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLLECTION} \ + COLUMNS (id TEXT, ts BIGINT TIME_KEY, host TEXT, value FLOAT) \ + WITH (engine='timeseries')" + )) + .await + .expect("create the timeseries collection"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE CHANGE STREAM {STREAM} ON {COLLECTION}")) + .await + .expect("create the change stream"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE CONSUMER GROUP {GROUP} ON {STREAM}")) + .await + .expect("create the consumer group"); + wait_for( + "every node registers the stream and the group", + CONVERGE, + STEP, + || { + cluster.nodes.iter().all(|node| { + node.has_change_stream(DatabaseId::DEFAULT, TENANT, STREAM) + && node + .shared + .group_registry + .get(DatabaseId::DEFAULT, TENANT, STREAM, GROUP) + .is_some() + }) + }, + ) + .await; + + for batch in 0..BATCHES { + exec_retrying(&cluster.nodes[0], &batch_insert(batch)).await; + } + cluster.wait_for_full_apply_convergence(CONVERGE).await; + + // Every replica, the low-budget one included, stores every row. + let deadline = Instant::now() + CONVERGE; + let mut per_node: Vec<(u64, Vec)> = Vec::new(); + while Instant::now() < deadline { + per_node.clear(); + for node in &cluster.nodes { + per_node.push(( + node.node_id, + local_timeseries_rows(node, TENANT, COLLECTION).await, + )); + } + if per_node.iter().all(|(_, rows)| rows.len() == TOTAL_ROWS) { + break; + } + tokio::time::sleep(STEP).await; + } + for (node_id, rows) in &per_node { + assert_eq!( + rows.len(), + TOTAL_ROWS, + "node {node_id} stores {} of {TOTAL_ROWS} rows; a replica that resolves or \ + drops rows on its own diverges", + rows.len() + ); + } + let (reference_node, reference) = &per_node[0]; + for (node_id, rows) in &per_node[1..] { + assert_eq!( + rows, reference, + "node {node_id} stores other values than node {reference_node}" + ); + } + + // Every replica serves one event per row, at the same positions. + wait_for( + "every replica serves one event per row", + CONVERGE, + STEP, + || { + cluster + .nodes + .iter() + .all(|node| event_count(&events(node)) == TOTAL_ROWS) + }, + ) + .await; + let reference_events = events(&cluster.nodes[0]); + assert_eq!(event_count(&reference_events), TOTAL_ROWS); + for node in &cluster.nodes[1..] { + assert_eq!( + events(node), + reference_events, + "node {} serves a different change sequence than node {}", + node.node_id, + cluster.nodes[0].node_id + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/timeseries_ring_drop_catch_up.rs b/nodedb-cluster-tests/tests/common_suite/cases/timeseries_ring_drop_catch_up.rs new file mode 100644 index 000000000..c71430df4 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/timeseries_ring_drop_catch_up.rs @@ -0,0 +1,347 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Every replica drops the events of timeseries installs that store by +//! column name and reject a row at apply. WAL catch-up rebuilds exactly the +//! events of the rows that landed, with the images they were stored with. +//! +//! ## Test shape +//! +//! Bring up 3 nodes with a timeseries collection that declares only its +//! `ts` time key, and a change stream on it, and land one warm-up row. Park +//! every apply of the collection at a per-replica gate in the write funnel, +//! then send three raw ILP lines, one per node, through the native +//! `TimeseriesIngest` opcode: a fresh column comes only from a raw ILP line +//! (see `ts_native_ingest`). A parked apply holds every later entry of its +//! group, so only A reaches the gates. B goes out once every replica's gate +//! holds A, and C once every replica committed B, so the log orders them as +//! sent: +//! +//! - A gives the new column `extra` a float. +//! - B gives a new column `y` a float and names no `extra`. +//! - C gives `extra` a string. +//! +//! Each proposer resolves against a replica that applied none of them, so +//! the log carries all three as resolved. With the event ring dropping the +//! collection's events, release the applies. Every replica stores A exactly, +//! stores B by column name with `extra` empty, and rejects C's row. Then: +//! +//! - C's client reports its row rejected, and A's and B's report none; +//! - every replica holds the same three rows, and no core fail-stopped; +//! - every replica's change stream serves one event per stored row, none for +//! C, and B's event carries `extra` as stored, empty, which the resolve's +//! image of B never named. + +#![cfg(feature = "failpoints")] + +use crate::common; +use common::cluster_harness::shared_steps::{fail_stopped, local_timeseries_rows}; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use std::path::{Path, PathBuf}; +use std::time::{Duration, Instant}; + +use nodedb::event::cdc::consume::{ConsumeError, ConsumeParams, consume_local}; +use nodedb_types::DatabaseId; +use nodedb_types::fail_point::{FailAction, FailGuard}; + +use super::ts_native_ingest::{ + ingest_native, ingest_until_accepted, native_session, rejection_warnings, +}; + +const COLLECTION: &str = "ts_ring_drop"; +const STREAM: &str = "ts_ring_drop_feed"; +const GROUP: &str = "ts_ring_drop_readers"; +const TENANT: u64 = 1; + +/// The warm-up row, A's row and B's row. C's row never lands. +const STORED_ROWS: usize = 3; + +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(100); +const GATE_STEP: Duration = Duration::from_millis(20); + +/// One replica's apply gate on the collection. +/// +/// Every apply that reaches the gate parks until the release file exists. +/// Each arrival creates the arrival marker, so a test that removes the +/// marker learns when the next apply arrives. +struct ApplyGate { + release: PathBuf, + arrival: PathBuf, + _guard: FailGuard, +} + +impl ApplyGate { + fn install(directory: &Path, node_id: u64) -> Self { + let release = directory.join(format!("release-node{node_id}")); + let arrival = directory.join(format!("release-node{node_id}.parked")); + let guard = FailGuard::install( + &format!("funnel::before_dispatch::node{node_id}::{COLLECTION}"), + FailAction::WaitForFile(release.clone()), + ); + Self { + release, + arrival, + _guard: guard, + } + } + + fn clear_arrival(&self) { + match std::fs::remove_file(&self.arrival) { + Ok(()) => {} + Err(error) if error.kind() == std::io::ErrorKind::NotFound => {} + Err(error) => panic!("remove {}: {error}", self.arrival.display()), + } + } + + fn arrived(&self) -> bool { + self.arrival.exists() + } + + fn release(&self) { + std::fs::write(&self.release, b"release").expect("release the held applies"); + } +} + +/// The commit index `node`'s Raft holds for `group_id`, or `0` when the +/// group is not hosted there. +fn group_commit_index(node: &TestClusterNode, group_id: u64) -> u64 { + node.shared + .cluster_observer + .get() + .and_then(|observer| observer.group_status.upgrade()) + .map(|status| status.group_statuses()) + .unwrap_or_default() + .into_iter() + .find(|group| group.group_id == group_id) + .map_or(0, |group| group.commit_index) +} + +/// The new values of the change events `node` serves. +fn event_values(node: &TestClusterNode) -> Vec { + let params = ConsumeParams { + database_id: DatabaseId::DEFAULT, + tenant_id: TENANT, + stream_name: STREAM, + group_name: GROUP, + partition: None, + limit: 100, + }; + match consume_local(&node.shared, ¶ms) { + Ok(result) => result + .events + .into_iter() + .map(|event| event.new_value.clone().unwrap_or(serde_json::Value::Null)) + .collect(), + Err(ConsumeError::BufferEmpty(_)) => Vec::new(), + Err(error) => panic!("node {}: consume failed: {error}", node.node_id), + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn catch_up_after_an_apply_time_rejection_rebuilds_only_landed_rows() { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLLECTION} (ts BIGINT TIME_KEY) WITH (engine='timeseries')" + )) + .await + .expect("create the timeseries collection"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE CHANGE STREAM {STREAM} ON {COLLECTION}")) + .await + .expect("create the change stream"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE CONSUMER GROUP {GROUP} ON {STREAM}")) + .await + .expect("create the consumer group"); + wait_for( + "every node registers the stream and the group", + CONVERGE, + STEP, + || { + cluster.nodes.iter().all(|node| { + node.has_change_stream(DatabaseId::DEFAULT, TENANT, STREAM) + && node + .shared + .group_registry + .get(DatabaseId::DEFAULT, TENANT, STREAM, GROUP) + .is_some() + }) + }, + ) + .await; + ingest_until_accepted( + &cluster.nodes[0], + COLLECTION, + &format!("{COLLECTION} value=0.5 1000000000"), + ) + .await; + cluster.wait_for_full_apply_convergence(CONVERGE).await; + + // Each replica holds each apply of the collection at its own gate until + // released. A gate marks each arrival, so the test sends a statement only + // once every replica holds the one before it: the log then orders A, B + // and C as sent. + let gate_dir = tempfile::tempdir().expect("gate tempdir"); + let gates: Vec = cluster + .nodes + .iter() + .map(|node| ApplyGate::install(gate_dir.path(), node.node_id)) + .collect(); + + let lines = [ + format!("{COLLECTION} value=1.0,extra=1.5 2000000000"), + format!("{COLLECTION} value=2.0,y=7.5 3000000000"), + format!("{COLLECTION} value=3.0,extra=\"text\" 4000000000"), + ]; + // A parked apply holds every later entry of its group, so only A reaches + // the gates. B and C each go out once every replica committed the line + // before it. + let group_id = cluster.nodes[0] + .group_id_for_collection(COLLECTION) + .expect("the collection maps to a data group"); + let mut sent = Vec::new(); + for (label, (node, line)) in ["A", "B", "C"] + .into_iter() + .zip(cluster.nodes.iter().zip(lines)) + { + for gate in &gates { + gate.clear_arrival(); + } + let committed_before: Vec = cluster + .nodes + .iter() + .map(|node| group_commit_index(node, group_id)) + .collect(); + let mut session = native_session(node).await; + sent.push(tokio::spawn(async move { + ingest_native(&mut session, 1, COLLECTION, &line).await + })); + if label == "A" { + wait_for("every replica holds A's apply", CONVERGE, GATE_STEP, || { + gates.iter().all(ApplyGate::arrived) + }) + .await; + } else { + wait_for( + &format!("every replica commits {label}'s line"), + CONVERGE, + GATE_STEP, + || { + cluster + .nodes + .iter() + .zip(&committed_before) + .all(|(node, before)| group_commit_index(node, group_id) > *before) + }, + ) + .await; + } + } + + // The ring loses every event of the collection while the applies run. + let dropped = FailGuard::fail(&format!("event::bus::drop::{COLLECTION}"), "ring full"); + for gate in &gates { + gate.release(); + } + let mut replies = Vec::new(); + for statement in sent { + replies.push(statement.await.expect("statement task")); + } + + let deadline = Instant::now() + CONVERGE; + let mut per_node: Vec<(u64, Vec)> = Vec::new(); + while Instant::now() < deadline { + per_node.clear(); + for node in &cluster.nodes { + per_node.push(( + node.node_id, + local_timeseries_rows(node, TENANT, COLLECTION).await, + )); + } + if per_node.iter().all(|(_, rows)| rows.len() == STORED_ROWS) { + break; + } + tokio::time::sleep(STEP).await; + } + drop(dropped); + drop(gates); + + assert!( + rejection_warnings(&replies[0], COLLECTION).is_empty(), + "A stored its row: {:?}", + replies[0] + ); + assert!( + rejection_warnings(&replies[1], COLLECTION).is_empty(), + "B stored its row: {:?}", + replies[1] + ); + let c_notices = rejection_warnings(&replies[2], COLLECTION); + assert!( + c_notices.iter().any(|notice| notice.contains("1 line(s)")), + "C's client learns its row was rejected, got {c_notices:?}" + ); + + for (node_id, rows) in &per_node { + assert_eq!(rows.len(), STORED_ROWS, "node {node_id} stores {rows:?}"); + } + let (reference_node, reference) = &per_node[0]; + for (node_id, rows) in &per_node[1..] { + assert_eq!( + rows, reference, + "node {node_id} stores other rows than node {reference_node}" + ); + } + for node in &cluster.nodes { + assert!( + !fail_stopped(node), + "node {} fail-stopped a core", + node.node_id + ); + } + + wait_for( + "every replica's catch-up serves one event per stored row", + CONVERGE, + STEP, + || { + cluster + .nodes + .iter() + .all(|node| event_values(node).len() == STORED_ROWS) + }, + ) + .await; + for node in &cluster.nodes { + let values = event_values(node); + assert_eq!( + values.len(), + STORED_ROWS, + "node {}: {values:?}", + node.node_id + ); + assert!( + values + .iter() + .all(|value| !value.get("extra").is_some_and(serde_json::Value::is_string)), + "node {}: an event carries C's rejected row: {values:?}", + node.node_id + ); + let b = values + .iter() + .find(|value| value.get("y").is_some_and(|y| !y.is_null())) + .unwrap_or_else(|| panic!("node {}: no event for B: {values:?}", node.node_id)); + assert_eq!( + b.get("extra"), + Some(&serde_json::Value::Null), + "node {}: B's event is not the row as stored: {b}", + node.node_id + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/timeseries_snapshot_schema_follower.rs b/nodedb-cluster-tests/tests/common_suite/cases/timeseries_snapshot_schema_follower.rs new file mode 100644 index 000000000..fdcfda14a --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/timeseries_snapshot_schema_follower.rs @@ -0,0 +1,188 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A replica that catches up by a real Raft `InstallSnapshot` holds the +//! leader's timeseries schema, before and after it restarts, and rejects a +//! conflicting row exactly as every other replica does. +//! +//! ## Test shape +//! +//! Bring up 3 nodes with a low log-compaction threshold and a replication +//! factor of 4, so a fourth node lands on every data group. Give a +//! timeseries collection that declares only its `ts` time key a column +//! `extra` of floats, and write enough rows that the data group's log +//! compacts past the start. Every row is a raw ILP line through the native +//! `TimeseriesIngest` opcode: a fresh column comes only from a raw ILP line +//! (see `ts_native_ingest`). Add a +//! fresh fourth node: only an `InstallSnapshot` can make it whole. Once it +//! mounted the group from a snapshot, restart it, so its schema comes from +//! its own disk. Then send a row that gives `extra` a string through the +//! restarted node, the node that resolves it: +//! +//! - its client reports the row rejected; +//! - every replica holds the same rows, none of them the rejected one; +//! - no core fail-stopped. +//! +//! A replica that lost the schema resolves the row whole and stores it +//! while the others reject it. + +use crate::common; +use common::cluster_harness::shared_steps::{fail_stopped, local_timeseries_rows}; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use std::time::{Duration, Instant}; + +use super::ts_native_ingest::{ + assert_native_ok, ingest_native, ingest_until_accepted, native_session, rejection_warnings, +}; + +const COMPACTION_THRESHOLD: u64 = 4; +const ROW_COUNT: usize = 30; +const COLLECTION: &str = "ts_snapshot_schema"; +const TENANT: u64 = 1; + +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(100); + +/// Poll every replica until each holds `expected` rows, and return them. +async fn converged_rows(cluster: &TestCluster, expected: usize) -> Vec<(u64, Vec)> { + let deadline = Instant::now() + CONVERGE; + loop { + let mut per_node = Vec::new(); + for node in &cluster.nodes { + per_node.push(( + node.node_id, + local_timeseries_rows(node, TENANT, COLLECTION).await, + )); + } + if per_node.iter().all(|(_, rows)| rows.len() == expected) || Instant::now() >= deadline { + return per_node; + } + tokio::time::sleep(STEP).await; + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_follower_restored_by_install_snapshot_rejects_as_the_leader() { + let mut cluster = + TestCluster::spawn_three_with_compaction_threshold_and_rf(COMPACTION_THRESHOLD, 4) + .await + .expect("3-node cluster with low compaction threshold and rf=4"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {COLLECTION} (ts BIGINT TIME_KEY) WITH (engine='timeseries')" + )) + .await + .expect("create the timeseries collection"); + + // Each ingest is one data-group entry; together they compact the log. + for i in 0..ROW_COUNT { + ingest_until_accepted( + &cluster.nodes[0], + COLLECTION, + &format!( + "{COLLECTION} value={i}.0,extra={i}.5 {ts_ns}", + ts_ns = 1_000_000_000 * (i as u64 + 1) + ), + ) + .await; + } + cluster.wait_for_full_apply_convergence(CONVERGE).await; + let compacted = cluster + .nodes + .iter() + .map(TestClusterNode::max_data_group_snapshot_index) + .max() + .unwrap_or(0); + assert!( + compacted > 0, + "the data group's log compacted before the new node joins" + ); + let gid = cluster.nodes[0] + .group_id_for_collection(COLLECTION) + .expect("the collection maps to a data group"); + + // A fresh node is made whole only by an InstallSnapshot. + let learner_id = cluster + .add_learner_node() + .await + .expect("add a fourth node") + .node_id; + let learner_index = cluster + .nodes + .iter() + .position(|node| node.node_id == learner_id) + .expect("the fourth node is a member"); + wait_for( + "the fourth node mounts the data group from a snapshot", + CONVERGE, + STEP, + || { + let learner = &cluster.nodes[learner_index]; + learner.hosts_data_group(gid) && learner.local_snapshot_index_for_group(gid) > 0 + }, + ) + .await; + let before_restart = converged_rows(&cluster, ROW_COUNT).await; + for (node_id, rows) in &before_restart { + assert_eq!( + rows.len(), + ROW_COUNT, + "node {node_id} stores {} rows", + rows.len() + ); + } + + // After the restart its schema comes from its own disk. + let stopped = cluster + .stop_member(learner_index) + .await + .expect("stop the fourth node"); + cluster + .restart_member(stopped) + .await + .expect("restart the fourth node"); + + let mut session = native_session(&cluster.nodes[learner_index]).await; + let reply = ingest_native( + &mut session, + 1, + COLLECTION, + &format!("{COLLECTION} value=1.0,extra=\"text\" 999000000000"), + ) + .await; + assert_native_ok(&reply, "the conflicting ingest"); + let notices = rejection_warnings(&reply, COLLECTION); + assert!( + notices.iter().any(|notice| notice.contains("1 line(s)")), + "the restarted node reports the conflicting row rejected, got {notices:?}" + ); + + cluster.wait_for_full_apply_convergence(CONVERGE).await; + let after = converged_rows(&cluster, ROW_COUNT).await; + let (reference_node, reference) = &after[0]; + for (node_id, rows) in &after { + assert_eq!( + rows.len(), + ROW_COUNT, + "node {node_id} stores {} rows", + rows.len() + ); + assert_eq!( + rows, reference, + "node {node_id} stores other rows than node {reference_node}" + ); + assert!( + rows.iter().all(|row| !row.contains("\"text\"")), + "node {node_id} stored the rejected row" + ); + } + for node in &cluster.nodes { + assert!( + !fail_stopped(node), + "node {} fail-stopped a core", + node.node_id + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/topic_consumer_group_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/topic_consumer_group_cross_node.rs index a7ea17304..e94cec440 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/topic_consumer_group_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/topic_consumer_group_cross_node.rs @@ -11,7 +11,8 @@ //! follower. Each test asserts the follower's durable row AND its live //! registry entry separately. //! -//! Message publication is out of scope: it is a node-local data path. +//! Message publication is out of scope here; `topic_publish_across_home_change` +//! covers it. use crate::common; diff --git a/nodedb-cluster-tests/tests/common_suite/cases/topic_publish_across_home_change.rs b/nodedb-cluster-tests/tests/common_suite/cases/topic_publish_across_home_change.rs new file mode 100644 index 000000000..5fa6e0d11 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/topic_publish_across_home_change.rs @@ -0,0 +1,258 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A topic keeps every message across a change of its home leader. +//! +//! `PUBLISH TO` proposes a `TopicPublish` entry to the data group of the +//! topic's home vShard. Every replica appends the message at apply, at the +//! entry's log position, and `COMMIT OFFSET` is a replicated catalog entry. +//! So a consumer that reads part of the topic from the home leader, commits, +//! and resumes on another node after the leader dies receives every message +//! exactly once, in order. + +use crate::common; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +use std::collections::BTreeSet; +use std::time::{Duration, Instant}; + +use nodedb::event::cdc::CdcOffset; +use nodedb::event::cdc::consume::{ConsumeError, ConsumeParams, consume_local}; +use nodedb::event::topic::publish::topic_vshard; +use nodedb_types::DatabaseId; + +const TOPIC: &str = "topic_home_failover"; +const GROUP: &str = "topic_home_readers"; +const TENANT: u64 = 1; + +/// Messages published before the home leader dies. +const BEFORE: usize = 6; +/// Messages published after the home leader dies. +const AFTER: usize = 3; +/// Messages the consumer reads and commits before the leader dies. +const FIRST_READ: usize = 3; + +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(100); + +/// The buffer and consumer-group key of [`TOPIC`]. +fn stream() -> String { + format!("topic:{TOPIC}") +} + +/// One consumed message: its position and its topic sequence. +type Seen = (CdcOffset, u64); + +/// The messages `node` serves after the group's committed offset. +fn read(node: &TestClusterNode, limit: usize) -> Vec { + let stream = stream(); + let params = ConsumeParams { + database_id: DatabaseId::DEFAULT, + tenant_id: TENANT, + stream_name: &stream, + group_name: GROUP, + partition: None, + limit, + }; + match consume_local(&node.shared, ¶ms) { + Ok(result) => result + .events + .iter() + .map(|event| (event.position(), event.sequence)) + .collect(), + Err(ConsumeError::BufferEmpty(_)) => Vec::new(), + Err(error) => panic!("node {}: consume failed: {error}", node.node_id), + } +} + +/// Publish message `n` through `node`, retrying while the home group elects +/// a leader. +async fn publish(node: &TestClusterNode, n: usize) { + let sql = format!("PUBLISH TO {TOPIC} 'message-{n}'"); + let deadline = Instant::now() + CONVERGE; + loop { + match node.client.simple_query(&sql).await { + Ok(_) => return, + Err(error) if Instant::now() < deadline => { + tracing::debug!(n, %error, "publish not accepted yet; retrying"); + tokio::time::sleep(Duration::from_millis(200)).await; + } + Err(error) => panic!("publish message-{n}: {error}"), + } + } +} + +/// The topic's home data group and its leader, as `node` sees them. +fn home_leader(node: &TestClusterNode) -> (u64, u64) { + let group = node + .shared + .cluster_routing + .as_ref() + .expect("cluster routing") + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(topic_vshard(DatabaseId::DEFAULT, TOPIC)) + .expect("the home vShard maps to a data group"); + let leader = node + .all_group_leaders() + .into_iter() + .find_map(|(id, leader)| (id == group).then_some(leader)) + .unwrap_or(0); + (group, leader) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn topic_consumer_receives_every_message_once_after_the_home_leader_dies() { + let mut cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + + cluster + .exec_ddl_on_any_leader(&format!("CREATE TOPIC {TOPIC}")) + .await + .expect("create topic"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE CONSUMER GROUP {GROUP} ON {TOPIC}")) + .await + .expect("create consumer group"); + let stream = stream(); + wait_for( + "every node registers the topic and the group", + CONVERGE, + STEP, + || { + cluster.nodes.iter().all(|node| { + node.shared + .ep_topic_registry + .get(DatabaseId::DEFAULT, TENANT, TOPIC) + .is_some() + && node + .shared + .group_registry + .get(DatabaseId::DEFAULT, TENANT, &stream, GROUP) + .is_some() + }) + }, + ) + .await; + + for n in 0..BEFORE { + publish(&cluster.nodes[0], n).await; + } + wait_for("every replica holds every message", CONVERGE, STEP, || { + cluster + .nodes + .iter() + .all(|node| read(node, 1_000).len() == BEFORE) + }) + .await; + + // Every replica serves the same messages at the same positions. + let reference = read(&cluster.nodes[0], 1_000); + for node in &cluster.nodes { + assert_eq!( + read(node, 1_000), + reference, + "node {} serves a different message sequence", + node.node_id + ); + } + + // Consume part of the topic from its home leader, and commit. + let (group, leader) = home_leader(&cluster.nodes[0]); + let leader_idx = cluster + .nodes + .iter() + .position(|node| node.node_id == leader) + .unwrap_or_else(|| panic!("no live node leads the home group {group}")); + let first = read(&cluster.nodes[leader_idx], FIRST_READ); + assert_eq!(first, reference[..FIRST_READ].to_vec()); + let (committed, _) = *first.last().expect("a first read"); + cluster.nodes[leader_idx] + .client + .simple_query(&format!( + "COMMIT OFFSET PARTITION 0 AT {committed} ON {TOPIC} CONSUMER GROUP {GROUP}" + )) + .await + .unwrap_or_else(|e| panic!("commit offset {committed}: {e}")); + wait_for( + "every node holds the committed offset", + CONVERGE, + STEP, + || { + cluster.nodes.iter().all(|node| { + node.shared + .offset_store + .get_offset(DatabaseId::DEFAULT, TENANT, &stream, GROUP, 0) + == committed + }) + }, + ) + .await; + + // Kill the home leader. + let dead = cluster.nodes.remove(leader_idx); + let dead_id = dead.node_id; + dead.shutdown().await; + wait_for( + "the survivors elect a new home leader", + CONVERGE, + STEP, + || { + cluster.nodes.iter().all(|node| { + node.all_group_leaders() + .into_iter() + .any(|(id, leader)| id == group && leader != 0 && leader != dead_id) + }) + }, + ) + .await; + + for n in BEFORE..BEFORE + AFTER { + publish(&cluster.nodes[0], n).await; + } + let remaining = BEFORE - FIRST_READ + AFTER; + wait_for( + "every survivor holds the new messages", + CONVERGE, + STEP, + || { + cluster + .nodes + .iter() + .all(|node| read(node, 1_000).len() == remaining) + }, + ) + .await; + + // Resume on a survivor. Both survivors serve the same continuation. + let resumed = read(&cluster.nodes[0], 1_000); + for node in &cluster.nodes { + assert_eq!( + read(node, 1_000), + resumed, + "survivor {} resumes with a different sequence", + node.node_id + ); + } + + // Every message arrives exactly once, in order. + let delivered: Vec = first.iter().chain(resumed.iter()).copied().collect(); + let sequences: Vec = delivered.iter().map(|(_, sequence)| *sequence).collect(); + let distinct: BTreeSet = sequences.iter().copied().collect(); + assert_eq!( + sequences.len(), + BEFORE + AFTER, + "duplicate or missing messages: {sequences:?}" + ); + assert_eq!( + distinct, + (1..=(BEFORE + AFTER) as u64).collect::>(), + "missing messages: {sequences:?}" + ); + assert!( + delivered.windows(2).all(|pair| pair[0].0 < pair[1].0), + "positions must rise with publication order: {delivered:?}" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/trigger_body_atomic_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/trigger_body_atomic_cross_node.rs new file mode 100644 index 000000000..4b66d14a7 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/trigger_body_atomic_cross_node.rs @@ -0,0 +1,173 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! An AFTER trigger body runs as one transaction across a local and a +//! cross-node write. +//! +//! The body first writes a collection led by another node, then a +//! collection led by the firing node. The local write collides with a +//! blocker row: a `document_strict` duplicate primary key is refused with +//! 23505 at the statement, inside a transaction as outside one. The first +//! attempt therefore fails. The test then: +//! +//! 1. checks the failed attempt sent nothing to the other node; +//! 2. removes the blocker so the queued retry succeeds; +//! 3. checks each row landed exactly once on both collections. +//! +//! The firing node is the leader of the source collection's group. The +//! cross-node collection is chosen from candidates by reading the routing +//! table and the group leaders: its group is led by another node. + +use crate::common; + +use std::time::Duration; + +use common::cluster_harness::shared_steps::{group_of, leader_of, name_where, row_count}; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for, wait_for_async}; + +const SRC: &str = "atom_src"; + +async fn has_row(node: &TestClusterNode, collection: &str, id: &str, val: &str) -> bool { + let Ok(rows) = node + .client + .simple_query(&format!("SELECT val FROM {collection} WHERE id = '{id}'")) + .await + else { + return false; + }; + rows.iter().any(|msg| match msg { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0) == Some(val), + _ => false, + }) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn trigger_body_retry_lands_local_and_cross_node_rows_once() { + let cluster = TestCluster::spawn_three().await.expect("cluster"); + let probe = &cluster.nodes[0]; + + wait_for( + "every data group has a leader", + Duration::from_secs(10), + Duration::from_millis(50), + || { + probe + .all_group_leaders() + .iter() + .all(|(_, leader)| *leader != 0) + }, + ) + .await; + + let src_group = group_of(probe, SRC); + let firing_node = leader_of(probe, src_group); + let reader = cluster + .nodes + .iter() + .find(|n| n.node_id != firing_node) + .expect("a second node"); + let local = name_where("atom_local", |name| group_of(probe, name) == src_group); + let remote = (0..4096u32) + .map(|i| format!("atom_remote_{i}")) + .find(|name| { + let group = group_of(probe, name); + group != src_group && leader_of(probe, group) != firing_node + }) + .unwrap_or_else(|| { + panic!( + "no data group is led by a node other than {firing_node}: leaders {:?}", + probe.all_group_leaders() + ) + }); + let remote_group = group_of(probe, &remote); + + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {SRC} (id TEXT PRIMARY KEY, val BIGINT) \ + WITH (engine='document_strict')" + )) + .await + .expect("create source"); + for name in [&local, &remote] { + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {name} (id TEXT PRIMARY KEY, val BIGINT) \ + WITH (engine='document_strict')" + )) + .await + .expect("create target"); + } + + // The blocker makes the body's local write collide on its first attempt. + reader + .exec(&format!( + "INSERT INTO {local} (id, val) VALUES ('o1-log', 0)" + )) + .await + .expect("insert blocker"); + + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE TRIGGER atom_trig AFTER INSERT ON {SRC} FOR EACH ROW \ + BEGIN \ + INSERT INTO {remote} (id, val) VALUES (NEW.id || '-r', 1); \ + INSERT INTO {local} (id, val) VALUES (NEW.id || '-log', 1); \ + END" + )) + .await + .expect("create trigger"); + wait_for( + "trigger visible on all nodes", + Duration::from_secs(10), + Duration::from_millis(50), + || cluster.nodes.iter().all(|n| n.has_trigger(1, "atom_trig")), + ) + .await; + + assert_ne!( + leader_of(probe, remote_group), + firing_node, + "the cross-node collection's group must stay led by another node" + ); + reader + .exec(&format!("INSERT INTO {SRC} (id, val) VALUES ('o1', 1)")) + .await + .expect("insert source row"); + + // The first attempt collides on the blocker. It must send nothing. + tokio::time::sleep(Duration::from_millis(400)).await; + assert_eq!( + row_count(reader, &remote).await, + 0, + "a failed trigger body must not deliver its cross-node write" + ); + assert!( + has_row(reader, &local, "o1-log", "0").await, + "a failed trigger body must not overwrite the blocker" + ); + + reader + .exec(&format!("DELETE FROM {local} WHERE id = 'o1-log'")) + .await + .expect("delete blocker"); + + wait_for_async( + "the retried body lands both rows", + Duration::from_secs(20), + Duration::from_millis(100), + || async { + has_row(reader, &local, "o1-log", "1").await + && has_row(reader, &remote, "o1-r", "1").await + }, + ) + .await; + + // Let any duplicate delivery or retry that lands arrive first. + tokio::time::sleep(Duration::from_secs(2)).await; + assert_eq!(row_count(reader, &local).await, 1, "local row lands once"); + assert_eq!( + row_count(reader, &remote).await, + 1, + "cross-node row lands once" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/trigger_cross_shard_origination.rs b/nodedb-cluster-tests/tests/common_suite/cases/trigger_cross_shard_origination.rs new file mode 100644 index 000000000..c38942587 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/trigger_cross_shard_origination.rs @@ -0,0 +1,156 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! A trigger body whose writes all home to other nodes sends them only when +//! the body commits, and each statement lands once. +//! +//! An AFTER trigger on a source collection led by the firing node writes two +//! collections led by other nodes. The second collection does not exist when +//! the trigger first fires, so the body fails at its second statement. The +//! test: +//! +//! 1. checks the failed body sent neither write, so the first collection +//! stays empty; +//! 2. creates the second collection, so the queued retry succeeds; +//! 3. checks each collection holds the body's row exactly once. +//! +//! The body with a local write that fails its commit is covered by +//! `trigger_body_atomic_cross_node`. + +use crate::common; + +use std::time::Duration; + +use common::cluster_harness::shared_steps::{group_of, leader_of, row_count}; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for, wait_for_async}; + +const SRC: &str = "origin_src"; + +/// First `{prefix}_` whose group is led by a node other than +/// `firing_node`. +fn remote_name(probe: &TestClusterNode, prefix: &str, firing_node: u64) -> String { + (0..4096u32) + .map(|i| format!("{prefix}_{i}")) + .find(|name| leader_of(probe, group_of(probe, name)) != firing_node) + .unwrap_or_else(|| { + panic!( + "no data group is led by a node other than {firing_node}: leaders {:?}", + probe.all_group_leaders() + ) + }) +} + +async fn has_row(node: &TestClusterNode, collection: &str, id: &str) -> bool { + let Ok(rows) = node + .client + .simple_query(&format!("SELECT id FROM {collection} WHERE id = '{id}'")) + .await + else { + return false; + }; + rows.iter() + .any(|msg| matches!(msg, tokio_postgres::SimpleQueryMessage::Row(_))) +} + +async fn create_strict(cluster: &TestCluster, name: &str) { + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {name} (id TEXT PRIMARY KEY, val BIGINT) \ + WITH (engine='document_strict')" + )) + .await + .unwrap_or_else(|e| panic!("create {name}: {e}")); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_remote_only_body_sends_its_writes_at_commit_and_each_lands_once() { + let cluster = TestCluster::spawn_three().await.expect("cluster"); + let probe = &cluster.nodes[0]; + + wait_for( + "every data group has a leader", + Duration::from_secs(10), + Duration::from_millis(50), + || { + probe + .all_group_leaders() + .iter() + .all(|(_, leader)| *leader != 0) + }, + ) + .await; + + let firing_node = leader_of(probe, group_of(probe, SRC)); + let reader = cluster + .nodes + .iter() + .find(|n| n.node_id != firing_node) + .expect("a second node"); + let first = remote_name(probe, "origin_first", firing_node); + let second = remote_name(probe, "origin_second", firing_node); + + create_strict(&cluster, SRC).await; + create_strict(&cluster, &first).await; + + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE TRIGGER origin_trig AFTER INSERT ON {SRC} FOR EACH ROW \ + BEGIN \ + INSERT INTO {first} (id, val) VALUES (NEW.id || '-a', 1); \ + INSERT INTO {second} (id, val) VALUES (NEW.id || '-b', 2); \ + END" + )) + .await + .expect("create trigger"); + wait_for( + "trigger visible on all nodes", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.has_trigger(1, "origin_trig")) + }, + ) + .await; + + reader + .exec(&format!("INSERT INTO {SRC} (id, val) VALUES ('o1', 1)")) + .await + .expect("insert source row"); + + // The body fails at its second statement. Its first write is held until + // the body commits, so nothing reaches the other node. + tokio::time::sleep(Duration::from_millis(400)).await; + assert_eq!( + row_count(reader, &first).await, + 0, + "a failed body must not send a write it held for another node" + ); + + create_strict(&cluster, &second).await; + + wait_for_async( + "the retried body lands both rows", + Duration::from_secs(20), + Duration::from_millis(100), + || async { + has_row(reader, &first, "o1-a").await && has_row(reader, &second, "o1-b").await + }, + ) + .await; + + // Let any duplicate delivery or retry that lands arrive first. + tokio::time::sleep(Duration::from_secs(2)).await; + assert_eq!( + row_count(reader, &first).await, + 1, + "the first write lands once" + ); + assert_eq!( + row_count(reader, &second).await, + 1, + "the second write lands once" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/trigger_firing_failover.rs b/nodedb-cluster-tests/tests/common_suite/cases/trigger_firing_failover.rs new file mode 100644 index 000000000..ca37ff11e --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/trigger_firing_failover.rs @@ -0,0 +1,396 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! An AFTER trigger and a DEFINE EVENT action run exactly once per committed +//! row across a leader kill. +//! +//! Every replica of the source collection's group holds each inserted row's +//! action. Only the node that holds that group's leader lease fires, from a +//! replicated cursor. The action adds one to the row's tally, so an action +//! that ran twice leaves a tally of 2. Each scenario runs once with a trigger +//! body and once with a DEFINE EVENT THEN clause. +//! +//! Every node's firing is gated, so no election that moves the group's +//! leadership lets an unplanned node fire early. +//! +//! - `a_trigger_held_before_a_leader_kill_fires_once` keeps every node's +//! firing parked while the rows apply everywhere, kills the source's +//! leader, then releases the survivors. The new owner fires every row once +//! from the cursor. +//! - `a_trigger_fired_before_a_leader_kill_does_not_fire_twice` lets only +//! the leader fire, parks it before it commits the cursor, kills it, then +//! releases the survivors. The new owner fires the rows again from the old +//! cursor. Each body committed its event's key, so none runs twice. + +#![cfg(feature = "failpoints")] + +use crate::common; +use common::cluster_harness::shared_steps::{group_of, kill_node, leader_index_of}; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for, wait_for_async}; + +use std::path::{Path, PathBuf}; +use std::time::Duration; + +use nodedb::event::cdc::CdcOffset; +use nodedb_types::fail_point::{FailAction, FailGuard}; + +const SRC: &str = "firing_failover_src"; +const ROWS: [&str; 3] = ["r1", "r2", "r3"]; + +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(100); + +/// How many of the source's trigger actions `node` holds for firing. +fn held(node: &TestClusterNode) -> usize { + let Some(ledgers) = node.shared.sink_ledgers.get() else { + return 0; + }; + let ledger = &ledgers.actions.ledger; + ledger + .partitions() + .unwrap_or_default() + .into_iter() + .map(|partition| { + ledger + .held_after(partition, CdcOffset::ZERO, 10_000) + .map(|held| held.iter().filter(|(_, a)| a.collection == SRC).count()) + .unwrap_or(0) + }) + .sum() +} + +/// A tally collection in the source's group, so the body writes on the +/// firing node and commits its key there. +fn tally_name(cluster: &TestCluster) -> String { + let probe = &cluster.nodes[0]; + let group = group_of(probe, SRC); + (0..4096u32) + .map(|i| format!("firing_failover_tally_{i}")) + .find(|name| group_of(probe, name) == group) + .unwrap_or_else(|| panic!("no tally name maps to {SRC}'s group {group}")) +} + +/// One node's firing gate: its release file and the marker its first +/// arrival writes. +struct Gate { + release: PathBuf, + parked: PathBuf, + _guard: FailGuard, +} + +impl Gate { + fn install(dir: &Path, point: &str, node_id: u64) -> Self { + let release = dir.join(format!("{point}-node{node_id}")); + let parked = dir.join(format!("{point}-node{node_id}.parked")); + let guard = FailGuard::install( + &format!("trigger::{point}::node{node_id}"), + FailAction::WaitForFile(release.clone()), + ); + Self { + release, + parked, + _guard: guard, + } + } + + fn release(&self) { + std::fs::write(&self.release, b"release").expect("release a firing gate"); + } +} + +/// Park every node's firing before anything is written. +async fn park_every_node(cluster: &TestCluster, dir: &Path) -> Vec<(u64, Gate)> { + let gates: Vec<(u64, Gate)> = cluster + .nodes + .iter() + .map(|node| { + ( + node.node_id, + Gate::install(dir, "before_firing", node.node_id), + ) + }) + .collect(); + wait_for("every node's firing parks", CONVERGE, STEP, || { + gates.iter().all(|(_, gate)| gate.parked.exists()) + }) + .await; + gates +} + +/// The action that adds one to a row's tally. +#[derive(Clone, Copy)] +enum Action { + /// An AFTER trigger whose body runs the update. + Trigger, + /// A DEFINE EVENT whose THEN clause runs the update. + Event, +} + +/// Whether `node`'s catalog holds `action` on the source. +fn has_action(node: &TestClusterNode, action: Action) -> bool { + match action { + Action::Trigger => node.has_trigger(1, &format!("{SRC}_tally")), + Action::Event => node + .shared + .credentials + .catalog() + .event_definitions(nodedb_types::DatabaseId::DEFAULT, 1, SRC) + .is_some_and(|defs| !defs.is_empty()), + } +} + +/// A three-node cluster with the source, the tally seeded at 0 per row, and +/// `action`, which adds one to a row's tally. +async fn cluster_with_tally(action: Action) -> (TestCluster, String) { + let cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + wait_for("every data group has a leader", CONVERGE, STEP, || { + cluster.nodes[0] + .all_group_leaders() + .iter() + .all(|(_, leader)| *leader != 0) + }) + .await; + let tally = tally_name(&cluster); + for ddl in [ + format!( + "CREATE COLLECTION {SRC} (id TEXT PRIMARY KEY, v BIGINT) \ + WITH (engine='document_strict')" + ), + format!( + "CREATE COLLECTION {tally} (id TEXT PRIMARY KEY, n BIGINT) \ + WITH (engine='document_strict')" + ), + ] { + cluster + .exec_ddl_on_any_leader(&ddl) + .await + .unwrap_or_else(|e| panic!("{ddl}: {e}")); + } + for row in ROWS { + cluster.nodes[0] + .exec(&format!("INSERT INTO {tally} (id, n) VALUES ('{row}', 0)")) + .await + .unwrap_or_else(|e| panic!("seed tally {row}: {e}")); + } + let ddl = match action { + Action::Trigger => format!( + "CREATE TRIGGER {SRC}_tally AFTER INSERT ON {SRC} FOR EACH ROW \ + BEGIN UPDATE {tally} SET n = n + 1 WHERE id = NEW.id; END" + ), + Action::Event => format!( + "DEFINE EVENT {SRC}_tally ON {SRC} WHEN INSERT \ + THEN UPDATE {tally} SET n = n + 1 WHERE id = $document_id" + ), + }; + cluster + .exec_ddl_on_any_leader(&ddl) + .await + .unwrap_or_else(|e| panic!("{ddl}: {e}")); + wait_for("every node registers the action", CONVERGE, STEP, || { + cluster.nodes.iter().all(|node| has_action(node, action)) + }) + .await; + (cluster, tally) +} + +/// Insert every row through `node`. +async fn insert_rows(node: &TestClusterNode) { + for (n, row) in ROWS.iter().enumerate() { + node.exec(&format!("INSERT INTO {SRC} (id, v) VALUES ('{row}', {n})")) + .await + .unwrap_or_else(|e| panic!("insert {row}: {e}")); + } +} + +/// The text of a query error, with its SQLSTATE and message. +fn pg_detail(error: &tokio_postgres::Error) -> String { + match error.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{error}"), + } +} + +/// The tally of every row as `node` reads it, in row order. +async fn read_tallies(node: &TestClusterNode, tally: &str) -> Result, String> { + let mut out = Vec::with_capacity(ROWS.len()); + for row in ROWS { + let messages = node + .client + .simple_query(&format!("SELECT n FROM {tally} WHERE id = '{row}'")) + .await + .map_err(|e| format!("node {}: read tally {row}: {}", node.node_id, pg_detail(&e)))?; + let n = messages + .iter() + .find_map(|message| match message { + tokio_postgres::SimpleQueryMessage::Row(r) => r.get(0).map(str::to_owned), + _ => None, + }) + .unwrap_or_default(); + out.push(n); + } + Ok(out) +} + +/// The tallies `node` reads once a read succeeds. A read that races a +/// leader change is retried; one that keeps failing fails the test with its +/// error. +async fn tallies(node: &TestClusterNode, tally: &str) -> Vec { + let deadline = tokio::time::Instant::now() + CONVERGE; + loop { + match read_tallies(node, tally).await { + Ok(tallies) => return tallies, + Err(error) if tokio::time::Instant::now() >= deadline => panic!("{error}"), + Err(_) => tokio::time::sleep(STEP).await, + } + } +} + +fn every_row(value: &str) -> Vec { + ROWS.iter().map(|_| value.to_owned()).collect() +} + +/// Release the gates of the nodes still in `cluster`. +fn release_survivors(cluster: &TestCluster, gates: &[(u64, Gate)]) { + for (node_id, gate) in gates { + if cluster.nodes.iter().any(|node| node.node_id == *node_id) { + gate.release(); + } + } +} + +/// The survivors read every row's tally at 1, and the firing cursor passes +/// every action on every survivor. Several firing passes later the tallies +/// still read 1. +async fn survivors_fired_every_row_once(cluster: &TestCluster, tally: &str) { + let reader = &cluster.nodes[0]; + wait_for_async( + "every row's trigger fired on the new owner", + CONVERGE, + STEP, + || async { read_tallies(reader, tally).await.ok() == Some(every_row("1")) }, + ) + .await; + wait_for( + "the firing cursor passes every action on every survivor", + CONVERGE, + STEP, + || cluster.nodes.iter().all(|node| held(node) == 0), + ) + .await; + tokio::time::sleep(Duration::from_millis(1_000)).await; + for node in &cluster.nodes { + assert_eq!( + tallies(node, tally).await, + every_row("1"), + "node {}: a row's trigger fired twice or not at all", + node.node_id + ); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_trigger_held_before_a_leader_kill_fires_once() { + held_before_a_leader_kill_fires_once(Action::Trigger).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_trigger_fired_before_a_leader_kill_does_not_fire_twice() { + fired_before_a_leader_kill_does_not_fire_twice(Action::Trigger).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn an_event_action_held_before_a_leader_kill_runs_once() { + held_before_a_leader_kill_fires_once(Action::Event).await; +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn an_event_action_run_before_a_leader_kill_does_not_run_twice() { + fired_before_a_leader_kill_does_not_fire_twice(Action::Event).await; +} + +async fn held_before_a_leader_kill_fires_once(action: Action) { + let (mut cluster, tally) = cluster_with_tally(action).await; + let gate_dir = tempfile::tempdir().expect("gate directory"); + let gates = park_every_node(&cluster, gate_dir.path()).await; + + let leader = leader_index_of(&cluster, SRC); + insert_rows(&cluster.nodes[(leader + 1) % cluster.nodes.len()]).await; + wait_for( + "every replica holds every row's trigger action", + CONVERGE, + STEP, + || cluster.nodes.iter().all(|node| held(node) == ROWS.len()), + ) + .await; + for node in &cluster.nodes { + assert_eq!( + tallies(node, &tally).await, + every_row("0"), + "node {}: no trigger fires while every owner is parked", + node.node_id + ); + } + + // Leadership can have moved since the insert: kill whoever leads now. + let leader = leader_index_of(&cluster, SRC); + kill_node(&mut cluster, leader).await; + release_survivors(&cluster, &gates); + survivors_fired_every_row_once(&cluster, &tally).await; + cluster.shutdown().await; +} + +async fn fired_before_a_leader_kill_does_not_fire_twice(action: Action) { + let (mut cluster, tally) = cluster_with_tally(action).await; + let gate_dir = tempfile::tempdir().expect("gate directory"); + let gates = park_every_node(&cluster, gate_dir.path()).await; + + let writer = (leader_index_of(&cluster, SRC) + 1) % cluster.nodes.len(); + insert_rows(&cluster.nodes[writer]).await; + wait_for( + "every replica holds every row's trigger action", + CONVERGE, + STEP, + || cluster.nodes.iter().all(|node| held(node) == ROWS.len()), + ) + .await; + + // Only the current leader fires, and it parks before its cursor commit. + // That gate's release file is never created: the leader dies parked. + let leader = leader_index_of(&cluster, SRC); + let leader_id = cluster.nodes[leader].node_id; + let commit_gate = Gate::install(gate_dir.path(), "before_cursor_commit", leader_id); + let (_, leader_gate) = gates + .iter() + .find(|(node_id, _)| *node_id == leader_id) + .expect("the leader's firing gate"); + leader_gate.release(); + wait_for( + "the leader parks before its cursor commit", + CONVERGE, + Duration::from_millis(20), + || commit_gate.parked.exists(), + ) + .await; + let reader = &cluster.nodes[(leader + 1) % cluster.nodes.len()]; + wait_for_async( + "the leader's bodies committed every tally", + CONVERGE, + STEP, + || async { read_tallies(reader, &tally).await.ok() == Some(every_row("1")) }, + ) + .await; + for node in &cluster.nodes { + assert!( + held(node) > 0, + "node {}: the cursor never passed the fired actions", + node.node_id + ); + } + + kill_node(&mut cluster, leader).await; + release_survivors(&cluster, &gates); + survivors_fired_every_row_once(&cluster, &tally).await; + drop(commit_gate); + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/trigger_shipped_block_atomic_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/trigger_shipped_block_atomic_cross_node.rs new file mode 100644 index 000000000..01a9e7f88 --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/trigger_shipped_block_atomic_cross_node.rs @@ -0,0 +1,214 @@ +// SPDX-License-Identifier: BUSL-1.1 +//! A trigger write shipped to another node commits with every write it +//! derives, on every vShard, or not at all. +//! +//! An AFTER trigger on a source collection led by the firing node inserts +//! into `entries`, a collection led by another node. `entries` is the source +//! of a materialized sum whose target, `accounts`, homes to a third vShard. +//! The insert therefore derives a balance write on the `accounts` vShard. The +//! firing node ships the statement whole, and the receiving node commits the +//! entry and the balance as one Calvin transaction. +//! +//! The test: +//! +//! 1. forces the shipped insert to fail on a duplicate primary key, and +//! checks neither the entry nor the balance change landed; +//! 2. fires the trigger for a fresh row, and checks both landed once. + +use crate::common; + +use std::time::Duration; + +use common::cluster_harness::shared_steps::{group_of, leader_of, name_where, row_count}; +use common::cluster_harness::{TestCluster, TestClusterNode, wait_for, wait_for_async}; + +const SRC: &str = "ship_src"; + +fn vshard_of(collection: &str) -> u32 { + nodedb_types::CollectionKey::from_bare(nodedb_types::DatabaseId::DEFAULT, collection) + .vshard() + .as_u32() +} + +async fn balance(node: &TestClusterNode, accounts: &str) -> Option { + let rows = node + .client + .simple_query(&format!( + "SELECT balance FROM {accounts} WHERE id = 'acc-1'" + )) + .await + .ok()?; + rows.into_iter().find_map(|msg| match msg { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_string), + _ => None, + }) +} + +async fn has_entry(node: &TestClusterNode, entries: &str, id: &str) -> bool { + let Ok(rows) = node + .client + .simple_query(&format!("SELECT id FROM {entries} WHERE id = '{id}'")) + .await + else { + return false; + }; + rows.iter() + .any(|msg| matches!(msg, tokio_postgres::SimpleQueryMessage::Row(_))) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_shipped_write_commits_with_its_derived_writes_or_not_at_all() { + let cluster = TestCluster::spawn_three().await.expect("cluster"); + let probe = &cluster.nodes[0]; + + wait_for( + "every data group has a leader", + Duration::from_secs(10), + Duration::from_millis(50), + || { + probe + .all_group_leaders() + .iter() + .all(|(_, leader)| *leader != 0) + }, + ) + .await; + + let src_group = group_of(probe, SRC); + let firing_node = leader_of(probe, src_group); + let reader = cluster + .nodes + .iter() + .find(|n| n.node_id != firing_node) + .expect("a second node"); + let entries = name_where("ship_entries", |name| { + let group = group_of(probe, name); + group != src_group && leader_of(probe, group) != firing_node + }); + let entries_group = group_of(probe, &entries); + // The harness runs two data groups, so the third participant is a third + // vShard, not a third group. + let accounts = name_where("ship_accounts", |name| { + let vshard = vshard_of(name); + vshard != vshard_of(&entries) && vshard != vshard_of(SRC) + }); + + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {SRC} (id TEXT PRIMARY KEY, val BIGINT) \ + WITH (engine='document_strict')" + )) + .await + .expect("create source"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {accounts} (id TEXT PRIMARY KEY, owner TEXT) \ + WITH (engine='document_strict')" + )) + .await + .expect("create accounts"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {entries} (id TEXT PRIMARY KEY, account_id TEXT, amount TEXT) \ + WITH (engine='document_strict')" + )) + .await + .expect("create entries"); + cluster + .exec_ddl_on_any_leader(&format!( + "ALTER COLLECTION {accounts} ADD COLUMN balance TEXT \ + MATERIALIZED_SUM SOURCE {entries} \ + ON {entries}.account_id = {accounts}.id VALUE {entries}.amount" + )) + .await + .expect("declare materialized sum"); + + reader + .exec(&format!( + "INSERT INTO {accounts} (id, owner, balance) VALUES ('acc-1', 'alice', '100')" + )) + .await + .expect("seed account"); + // A materialized sum requires its target row, so the account the blocker + // credits exists too. + reader + .exec(&format!( + "INSERT INTO {accounts} (id, owner, balance) VALUES ('acc-2', 'bob', '0')" + )) + .await + .expect("seed the blocker's account"); + // The blocker holds the id the first shipped insert writes. It credits + // another account, so acc-1's balance counts only the shipped entries. + reader + .exec(&format!( + "INSERT INTO {entries} (id, account_id, amount) VALUES ('o1', 'acc-2', '0')" + )) + .await + .expect("insert blocker"); + + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE TRIGGER ship_trig AFTER INSERT ON {SRC} FOR EACH ROW \ + BEGIN \ + INSERT INTO {entries} (id, account_id, amount) VALUES (NEW.id, 'acc-1', '5'); \ + END" + )) + .await + .expect("create trigger"); + wait_for( + "trigger visible on all nodes", + Duration::from_secs(10), + Duration::from_millis(50), + || cluster.nodes.iter().all(|n| n.has_trigger(1, "ship_trig")), + ) + .await; + assert_ne!( + leader_of(probe, entries_group), + firing_node, + "the entries group must stay led by another node" + ); + + // The shipped insert collides with the blocker on every attempt. + reader + .exec(&format!("INSERT INTO {SRC} (id, val) VALUES ('o1', 1)")) + .await + .expect("insert failing source row"); + tokio::time::sleep(Duration::from_secs(2)).await; + assert_eq!( + row_count(reader, &entries).await, + 1, + "a failed shipped insert leaves only the blocker" + ); + assert_eq!( + balance(reader, &accounts).await.as_deref(), + Some("100"), + "a failed shipped insert commits none of its derived writes" + ); + + // A fresh row ships an insert that commits with its balance write. + reader + .exec(&format!("INSERT INTO {SRC} (id, val) VALUES ('o2', 2)")) + .await + .expect("insert source row"); + wait_for_async( + "the shipped insert and its balance land", + Duration::from_secs(20), + Duration::from_millis(100), + || async { + has_entry(reader, &entries, "o2").await + && balance(reader, &accounts).await.as_deref() == Some("105") + }, + ) + .await; + + // Let any duplicate delivery or retry that lands arrive first. + tokio::time::sleep(Duration::from_secs(2)).await; + assert_eq!(row_count(reader, &entries).await, 2, "the entry lands once"); + assert_eq!( + balance(reader, &accounts).await.as_deref(), + Some("105"), + "the balance counts the entry once" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/ts_native_ingest.rs b/nodedb-cluster-tests/tests/common_suite/cases/ts_native_ingest.rs new file mode 100644 index 000000000..41ff198bb --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/ts_native_ingest.rs @@ -0,0 +1,99 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Raw ILP ingest into a timeseries collection over the native protocol. +//! +//! A timeseries collection takes no undeclared column over SQL: the planner +//! closes its schema to the catalog's columns. A fresh column, and so a type +//! conflict on one, comes only from a raw ILP line. The native +//! `TimeseriesIngest` opcode carries raw ILP lines on a session, inside or +//! outside a transaction. It changes the live schema only. The ILP listener +//! also projects each new column into the catalog, and that catalog change +//! drains every descriptor lease an open transaction holds on the collection. + +use std::time::Duration; + +use nodedb_test_support::native_harness::{do_handshake, send_request}; +use nodedb_types::protocol::opcodes::ResponseStatus; +use nodedb_types::protocol::text_fields::TextFields; +use nodedb_types::protocol::{HelloFrame, NativeResponse, OpCode}; +use tokio::net::TcpStream; + +use crate::common::cluster_harness::TestClusterNode; + +/// How long a retried ingest waits for the collection's group to serve it. +const ACCEPT_WAIT: Duration = Duration::from_secs(20); + +/// A native protocol session on `node`, as the harness superuser. +pub(super) async fn native_session(node: &TestClusterNode) -> TcpStream { + let addr = std::net::SocketAddr::new(std::net::Ipv4Addr::LOCALHOST.into(), node.native_port); + let (stream, _ack) = do_handshake(addr, &HelloFrame::current()) + .await + .unwrap_or_else(|e| panic!("native handshake with node {}: {e:?}", node.node_id)); + stream +} + +/// Ingest the raw ILP `lines` into `collection` through the native +/// `TimeseriesIngest` opcode. +pub(super) async fn ingest_native( + stream: &mut TcpStream, + seq: u64, + collection: &str, + lines: &str, +) -> NativeResponse { + send_request( + stream, + seq, + OpCode::TimeseriesIngest, + TextFields { + collection: Some(collection.to_string()), + payload: Some(lines.as_bytes().to_vec()), + format: Some("ilp".to_string()), + ..Default::default() + }, + ) + .await +} + +/// Ingest `lines` through a fresh session on `node`, retrying while the +/// collection's group elects or mounts. Panics once [`ACCEPT_WAIT`] passes. +pub(super) async fn ingest_until_accepted( + node: &TestClusterNode, + collection: &str, + lines: &str, +) -> NativeResponse { + let deadline = tokio::time::Instant::now() + ACCEPT_WAIT; + let mut seq = 0; + loop { + let mut stream = native_session(node).await; + seq += 1; + let response = ingest_native(&mut stream, seq, collection, lines).await; + if response.status != ResponseStatus::Error { + return response; + } + assert!( + tokio::time::Instant::now() < deadline, + "node {} never accepted `{lines}`: {response:?}", + node.node_id + ); + tokio::time::sleep(Duration::from_millis(200)).await; + } +} + +/// Panic unless `response` succeeded. +pub(super) fn assert_native_ok(response: &NativeResponse, what: &str) { + assert_ne!( + response.status, + ResponseStatus::Error, + "{what} must succeed: {response:?}" + ); +} + +/// The warnings of `response` that report rejected lines of `collection`. +pub(super) fn rejection_warnings(response: &NativeResponse, collection: &str) -> Vec { + response + .warnings + .iter() + .filter(|warning| warning.contains(collection) && warning.contains("rejected")) + .cloned() + .collect() +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/vector_index_dispatch_cross_node.rs b/nodedb-cluster-tests/tests/common_suite/cases/vector_index_dispatch_cross_node.rs index db9ae3cc3..50773c9b9 100644 --- a/nodedb-cluster-tests/tests/common_suite/cases/vector_index_dispatch_cross_node.rs +++ b/nodedb-cluster-tests/tests/common_suite/cases/vector_index_dispatch_cross_node.rs @@ -121,6 +121,8 @@ async fn execute_on_node(node: &TestClusterNode, plan: PhysicalPlan) -> ExecuteR // Empty: the probe binds no descriptor version, so nothing is fenced. descriptor_versions: Vec::new(), txn_id: None, + vshard_id: None, + read_groups: Vec::new(), }; match transport .send_rpc_to_addr(node.listen_addr, RaftRpc::ExecuteRequest(request)) diff --git a/nodedb-cluster-tests/tests/common_suite/cases/webhook_sink_partial_replication.rs b/nodedb-cluster-tests/tests/common_suite/cases/webhook_sink_partial_replication.rs new file mode 100644 index 000000000..56209bb4e --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/webhook_sink_partial_replication.rs @@ -0,0 +1,153 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A webhook sink delivers every event once when the stream's rows live on +//! a node that does not run the sink. +//! +//! With replication factor 1 on 3 nodes, each data group has one member. The +//! test picks a collection whose group lives on a different node than the +//! stream's owning group. The sink owner holds none of the stream's events +//! in its own buffer, so it reads them from the collection's group member +//! over the remote consume path, and commits the replicated offsets. Every +//! event must reach the endpoint exactly once. + +use crate::common; +use common::cluster_harness::{TestCluster, wait_for}; + +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use tokio::net::TcpListener; + +use nodedb::event::cdc::sink_owner::owning_group; +use nodedb_types::{CollectionKey, DatabaseId}; + +use super::webhook_sink_single_owner::{Deliveries, FencingTokens, distinct, duplicates, serve}; + +const STREAM: &str = "webhook_rf1_feed"; +const TENANT: u64 = 1; +const ROWS: usize = 6; +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(100); + +/// The sole member of data group `group_id`. +fn sole_member(cluster: &TestCluster, group_id: u64) -> u64 { + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("a cluster node has a routing table"); + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let members = routing + .group_info(group_id) + .map(|info| info.members.clone()) + .unwrap_or_default(); + assert_eq!( + members.len(), + 1, + "replication factor 1 gives group {group_id} one member: {members:?}" + ); + members[0] +} + +/// The data group a collection's rows route to. +fn collection_group(cluster: &TestCluster, collection: &str) -> u64 { + let vshard = nodedb_cluster::routing::vshard_for_collection(CollectionKey::from_bare( + DatabaseId::DEFAULT, + collection, + )); + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("a cluster node has a routing table"); + routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(vshard) + .expect("every vShard maps to a group") +} + +/// A collection name whose group lives on another node than `sink_node`. +fn remote_collection(cluster: &TestCluster, sink_node: u64) -> String { + (0..256) + .map(|i| format!("webhook_rf1_rows_{i}")) + .find(|name| sole_member(cluster, collection_group(cluster, name)) != sink_node) + .expect("some collection name routes to another node's group") +} + +async fn insert(cluster: &TestCluster, collection: &str, row: usize) { + let sql = format!("INSERT INTO {collection} {{ id: 'row-{row}', n: {row} }}"); + let deadline = Instant::now() + CONVERGE; + loop { + match cluster.nodes[0].client.simple_query(&sql).await { + Ok(_) => return, + Err(error) if Instant::now() < deadline => { + tracing::debug!(row, %error, "insert not accepted yet; retrying"); + tokio::time::sleep(Duration::from_millis(200)).await; + } + Err(error) => panic!("insert row-{row}: {error}"), + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_webhook_reads_events_from_the_node_that_holds_them() { + let listener = TcpListener::bind("127.0.0.1:0").await.expect("bind"); + let url = format!("http://{}/hook", listener.local_addr().expect("addr")); + let deliveries: Deliveries = Arc::default(); + let tokens: FencingTokens = Arc::default(); + let server = tokio::spawn(serve( + listener, + Arc::clone(&deliveries), + Arc::clone(&tokens), + )); + + let cluster = TestCluster::spawn_three_with_replication_factor(1) + .await + .expect("spawn 3-node cluster with replication factor 1"); + let stream_group = owning_group(&cluster.nodes[0].shared, DatabaseId::DEFAULT, STREAM) + .expect("the stream's name maps to a data group"); + let sink_node = sole_member(&cluster, stream_group); + let collection = remote_collection(&cluster, sink_node); + + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {collection}")) + .await + .expect("create collection"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE CHANGE STREAM {STREAM} ON {collection} WITH (URL = '{url}')" + )) + .await + .expect("create change stream"); + wait_for("every node registers the stream", CONVERGE, STEP, || { + cluster + .nodes + .iter() + .all(|node| node.has_change_stream(DatabaseId::DEFAULT, TENANT, STREAM)) + }) + .await; + + for row in 0..ROWS { + insert(&cluster, &collection, row).await; + } + wait_for("the endpoint receives every event", CONVERGE, STEP, || { + distinct(&deliveries) == ROWS + }) + .await; + // Let a stray second delivery surface before counting. + tokio::time::sleep(Duration::from_secs(2)).await; + assert_eq!( + distinct(&deliveries), + ROWS, + "the endpoint receives exactly the stream's events" + ); + assert!( + duplicates(&deliveries).is_empty(), + "an event was delivered twice: {:?}", + duplicates(&deliveries) + ); + + server.abort(); + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/common_suite/cases/webhook_sink_single_owner.rs b/nodedb-cluster-tests/tests/common_suite/cases/webhook_sink_single_owner.rs new file mode 100644 index 000000000..61832b72a --- /dev/null +++ b/nodedb-cluster-tests/tests/common_suite/cases/webhook_sink_single_owner.rs @@ -0,0 +1,272 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A webhook sink delivers every event once, from one node, across a +//! failover. +//! +//! Every node runs a delivery task for the stream, but only the node that +//! holds the leader lease of the stream's owning group delivers, and it +//! commits the group's offsets through the metadata log. So: +//! +//! - every event reaches the endpoint exactly once while the cluster is +//! stable; +//! - after the delivering node dies, the next lease holder resumes from the +//! committed offsets: every later event arrives once, and no earlier one +//! arrives again; +//! - every delivery carries the owner's fencing token, and the next owner's +//! token is higher. + +use crate::common; +use common::cluster_harness::{TestCluster, wait_for}; + +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; +use std::time::{Duration, Instant}; + +use tokio::io::{AsyncBufReadExt, AsyncReadExt, AsyncWriteExt, BufReader}; +use tokio::net::TcpListener; + +use nodedb::event::cdc::consume::{ConsumeParams, consume_local}; +use nodedb::event::cdc::sink_owner::owning_group; +use nodedb_types::DatabaseId; + +const COLLECTION: &str = "webhook_owner_rows"; +const STREAM: &str = "webhook_owner_feed"; +const TENANT: u64 = 1; +const BEFORE: usize = 5; +const AFTER: usize = 4; +const CONVERGE: Duration = Duration::from_secs(30); +const STEP: Duration = Duration::from_millis(100); + +/// Deliveries per `X-Idempotency-Key`, which names one event. +pub(super) type Deliveries = Arc>>; + +/// The `X-Fencing-Token` of every delivery, in arrival order. +pub(super) type FencingTokens = Arc>>; + +/// Serve HTTP/1.1 POSTs on `listener`, counting each event's deliveries and +/// recording each delivery's fencing token. +pub(super) async fn serve(listener: TcpListener, deliveries: Deliveries, tokens: FencingTokens) { + loop { + let Ok((stream, _)) = listener.accept().await else { + return; + }; + let deliveries = Arc::clone(&deliveries); + let tokens = Arc::clone(&tokens); + tokio::spawn(async move { + let (read, mut write) = stream.into_split(); + let mut reader = BufReader::new(read); + loop { + let mut key = None; + let mut token = None; + let mut length = 0usize; + let mut line = String::new(); + loop { + line.clear(); + match reader.read_line(&mut line).await { + Ok(0) | Err(_) => return, + Ok(_) => {} + } + let header = line.trim_end(); + if header.is_empty() { + break; + } + let lower = header.to_ascii_lowercase(); + if let Some(value) = lower.strip_prefix("content-length:") { + length = value.trim().parse().unwrap_or(0); + } + if lower.starts_with("x-idempotency-key:") { + key = header.split_once(':').map(|(_, v)| v.trim().to_owned()); + } + if let Some(value) = lower.strip_prefix("x-fencing-token:") { + token = value.trim().parse::().ok(); + } + } + let mut body = vec![0u8; length]; + if reader.read_exact(&mut body).await.is_err() { + return; + } + if let Some(token) = token { + tokens.lock().unwrap_or_else(|p| p.into_inner()).push(token); + } + if let Some(key) = key { + *deliveries + .lock() + .unwrap_or_else(|p| p.into_inner()) + .entry(key) + .or_default() += 1; + } + if write + .write_all(b"HTTP/1.1 200 OK\r\ncontent-length: 0\r\n\r\n") + .await + .is_err() + { + return; + } + } + }); + } +} + +pub(super) fn distinct(deliveries: &Deliveries) -> usize { + deliveries.lock().unwrap_or_else(|p| p.into_inner()).len() +} + +pub(super) fn duplicates(deliveries: &Deliveries) -> Vec<(String, usize)> { + deliveries + .lock() + .unwrap_or_else(|p| p.into_inner()) + .iter() + .filter(|(_, count)| **count > 1) + .map(|(key, count)| (key.clone(), *count)) + .collect() +} + +async fn insert(cluster: &TestCluster, node: usize, row: usize) { + let sql = format!("INSERT INTO {COLLECTION} {{ id: 'row-{row}', n: {row} }}"); + let deadline = Instant::now() + CONVERGE; + loop { + match cluster.nodes[node].client.simple_query(&sql).await { + Ok(_) => return, + Err(error) if Instant::now() < deadline => { + tracing::debug!(row, %error, "insert not accepted yet; retrying"); + tokio::time::sleep(Duration::from_millis(200)).await; + } + Err(error) => panic!("insert row-{row}: {error}"), + } + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 4)] +async fn a_webhook_delivers_every_event_once_across_a_failover() { + let listener = TcpListener::bind("127.0.0.1:0").await.expect("bind"); + let url = format!("http://{}/hook", listener.local_addr().expect("addr")); + let deliveries: Deliveries = Arc::default(); + let tokens: FencingTokens = Arc::default(); + let server = tokio::spawn(serve( + listener, + Arc::clone(&deliveries), + Arc::clone(&tokens), + )); + + let mut cluster = TestCluster::spawn_three() + .await + .expect("spawn 3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {COLLECTION}")) + .await + .expect("create collection"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE CHANGE STREAM {STREAM} ON {COLLECTION} WITH (URL = '{url}')" + )) + .await + .expect("create change stream"); + wait_for("every node registers the stream", CONVERGE, STEP, || { + cluster + .nodes + .iter() + .all(|node| node.has_change_stream(DatabaseId::DEFAULT, TENANT, STREAM)) + }) + .await; + + for row in 0..BEFORE { + insert(&cluster, 0, row).await; + } + wait_for("the endpoint receives every event", CONVERGE, STEP, || { + distinct(&deliveries) == BEFORE + }) + .await; + + // The delivering node commits after each batch. Once every node's + // committed offsets cover every buffered event, no survivor redelivers. + let group_name = format!("_webhook:{STREAM}"); + wait_for( + "every node holds offsets past every event", + CONVERGE, + STEP, + || { + cluster.nodes.iter().all(|node| { + let params = ConsumeParams { + database_id: DatabaseId::DEFAULT, + tenant_id: TENANT, + stream_name: STREAM, + group_name: &group_name, + partition: None, + limit: 1, + }; + matches!( + consume_local(&node.shared, ¶ms), + Ok(result) if result.events.is_empty() + ) + }) + }, + ) + .await; + // Let a stray second delivery surface before counting. + tokio::time::sleep(Duration::from_secs(1)).await; + assert!( + duplicates(&deliveries).is_empty(), + "an event was delivered twice while the cluster was stable: {:?}", + duplicates(&deliveries) + ); + + let first_owner_tokens = tokens.lock().unwrap_or_else(|p| p.into_inner()).clone(); + assert_eq!( + first_owner_tokens.len(), + BEFORE, + "every delivery carries a fencing token" + ); + + // Kill the node that delivers: the leader of the stream's owning group. + let group = owning_group(&cluster.nodes[0].shared, DatabaseId::DEFAULT, STREAM) + .expect("the stream's name maps to a data group"); + let leader = cluster.nodes[0] + .all_group_leaders() + .into_iter() + .find_map(|(id, leader)| (id == group).then_some(leader)) + .unwrap_or(0); + let owner = cluster + .nodes + .iter() + .position(|node| node.node_id == leader) + .unwrap_or_else(|| panic!("no live node leads the owning group {group}")); + let dead = cluster.nodes.remove(owner); + let dead_id = dead.node_id; + dead.shutdown().await; + wait_for("the survivors elect a new owner", CONVERGE, STEP, || { + cluster.nodes.iter().all(|node| { + node.all_group_leaders() + .into_iter() + .any(|(id, leader)| id == group && leader != 0 && leader != dead_id) + }) + }) + .await; + + for row in BEFORE..BEFORE + AFTER { + insert(&cluster, 0, row).await; + } + wait_for( + "the new owner delivers the later events", + CONVERGE, + STEP, + || distinct(&deliveries) == BEFORE + AFTER, + ) + .await; + tokio::time::sleep(Duration::from_secs(1)).await; + assert!( + duplicates(&deliveries).is_empty(), + "the failover delivered an event twice: {:?}", + duplicates(&deliveries) + ); + let all_tokens = tokens.lock().unwrap_or_else(|p| p.into_inner()).clone(); + let first_owner_max = first_owner_tokens.iter().copied().max().unwrap_or(0); + assert!( + all_tokens[first_owner_tokens.len()..] + .iter() + .all(|token| *token > first_owner_max), + "the new owner's fencing token rises above the old owner's: {all_tokens:?}" + ); + + server.abort(); + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/dml_suite/cases/sql_cluster_cross_node_dml.rs b/nodedb-cluster-tests/tests/dml_suite/cases/sql_cluster_cross_node_dml.rs index 388047548..f4470ae60 100644 --- a/nodedb-cluster-tests/tests/dml_suite/cases/sql_cluster_cross_node_dml.rs +++ b/nodedb-cluster-tests/tests/dml_suite/cases/sql_cluster_cross_node_dml.rs @@ -11,6 +11,8 @@ #[path = "../../sql_cluster_cross_node_dml_tests/auth_objects.rs"] mod auth_objects; +#[path = "../../sql_cluster_cross_node_dml_tests/calvin_read_failover_cross_node.rs"] +mod calvin_read_failover_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/cluster_boot.rs"] mod cluster_boot; #[path = "../../sql_cluster_cross_node_dml_tests/ddl_objects.rs"] @@ -27,16 +29,32 @@ mod graph_algo_pagerank_personalized_cross_node; mod graph_algo_wcc_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/graph_delete_reverse_cross_node.rs"] mod graph_delete_reverse_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/graph_homed_read_failover.rs"] +mod graph_homed_read_failover; +#[path = "../../sql_cluster_cross_node_dml_tests/graph_homed_read_per_core.rs"] +mod graph_homed_read_per_core; #[path = "../../sql_cluster_cross_node_dml_tests/graph_implicit_reverse_cross_node.rs"] mod graph_implicit_reverse_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/graph_match_cross_node.rs"] mod graph_match_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/graph_match_read_set_cross_node.rs"] +mod graph_match_read_set_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/graph_match_ryow_cross_node.rs"] mod graph_match_ryow_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/graph_match_varlen_truncation_recovery_cross_node.rs"] mod graph_match_varlen_truncation_recovery_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/graph_multicore_cross_node.rs"] mod graph_multicore_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/graph_native_capped_walks_cross_node.rs"] +mod graph_native_capped_walks_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/graph_native_walks_cross_node.rs"] +mod graph_native_walks_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/graph_native_walks_support.rs"] +mod graph_native_walks_support; +#[path = "../../sql_cluster_cross_node_dml_tests/graph_owner_reads_cross_node.rs"] +mod graph_owner_reads_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/graph_rag_fusion_cross_node.rs"] +mod graph_rag_fusion_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/graph_traverse_cross_node.rs"] mod graph_traverse_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/graph_traverse_reverse_cross_node.rs"] @@ -45,6 +63,8 @@ mod graph_traverse_reverse_cross_node; mod join_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/native_implicit_edge_delete_cross_node.rs"] mod native_implicit_edge_delete_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/native_remote_home_surrogate_cross_node.rs"] +mod native_remote_home_surrogate_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/permission_tree_cross_node.rs"] mod permission_tree_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/permission_tree_lease_partition.rs"] @@ -55,3 +75,9 @@ mod schema_objects; mod select_remote_stream_cross_node; #[path = "../../sql_cluster_cross_node_dml_tests/select_streaming_cross_node.rs"] mod select_streaming_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/surrogate_identity_cross_node.rs"] +mod surrogate_identity_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/txn_resolve_on_follower_cross_node.rs"] +mod txn_resolve_on_follower_cross_node; +#[path = "../../sql_cluster_cross_node_dml_tests/txn_stage_forward_cross_node.rs"] +mod txn_stage_forward_cross_node; diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/cluster_procedures.rs b/nodedb-cluster-tests/tests/misc_suite/cases/cluster_procedures.rs index 7584066b5..86d4b879e 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/cluster_procedures.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/cluster_procedures.rs @@ -9,7 +9,6 @@ //! - Procedure body parsing with transaction control statements use nodedb::control::planner::procedural::ast::*; -use nodedb::control::planner::procedural::executor::transaction::ProcedureTransactionCtx; use nodedb::control::planner::procedural::parse_block; use nodedb::types::{DatabaseId, TenantId, VShardId}; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; @@ -41,7 +40,7 @@ fn procedure_dml_creates_replicated_entry() { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "archive"), document_id: "a-1".into(), value: b"{}".to_vec(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), pk_bytes: Vec::new(), returning: None, rls_filters: Vec::new(), @@ -84,54 +83,49 @@ fn procedure_reads_not_replicated() { } // --------------------------------------------------------------------------- -// Mid-procedure COMMIT: buffered tasks flushed independently +// Procedure DML: each planned write replicates through Raft // --------------------------------------------------------------------------- #[test] -fn tx_ctx_commit_yields_independent_tasks() { - let mut ctx = ProcedureTransactionCtx::new(); +fn procedure_writes_replicate_independently() { + // Two DML statements of a procedure body, as planned tasks. + let tasks = [ + PhysicalTask { + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(0), + database_id: DatabaseId::DEFAULT, + plan: PhysicalPlan::Document(DocumentOp::PointPut { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "orders"), + document_id: "o-1".into(), + value: b"{}".to_vec(), + surrogate: nodedb_types::Surrogate::new(2), + pk_bytes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + }), + post_set_op: PostSetOp::None, + txn_id: None, + }, + PhysicalTask { + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(1), + database_id: DatabaseId::DEFAULT, + plan: PhysicalPlan::Document(DocumentOp::PointDelete { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "temp"), + document_id: "t-1".into(), + surrogate: Some(nodedb_types::Surrogate::new(1)), + pk_bytes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, + resolved_sum_targets: Vec::new(), + }), + post_set_op: PostSetOp::None, + txn_id: None, + }, + ]; - // Simulate two DML statements in a procedure body. - ctx.buffer_task(PhysicalTask { - tenant_id: TenantId::new(1), - vshard_id: VShardId::new(0), - database_id: DatabaseId::DEFAULT, - plan: PhysicalPlan::Document(DocumentOp::PointPut { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "orders"), - document_id: "o-1".into(), - value: b"{}".to_vec(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - post_set_op: PostSetOp::None, - txn_id: None, - }); - ctx.buffer_task(PhysicalTask { - tenant_id: TenantId::new(1), - vshard_id: VShardId::new(0), - database_id: DatabaseId::DEFAULT, - plan: PhysicalPlan::Document(DocumentOp::PointDelete { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "temp"), - document_id: "t-1".into(), - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - returning: None, - rls_filters: Vec::new(), - rls_write_check: nodedb_types::RlsWriteCheck::NoPolicyApplies, - resolved_sum_targets: Vec::new(), - }), - post_set_op: PostSetOp::None, - txn_id: None, - }); - - // COMMIT flushes both tasks. - let tasks = ctx.take_buffered_tasks(); - assert_eq!(tasks.len(), 2); - - // Each task is an independent write that replicates via Raft. for task in &tasks { assert!( encode_entry(task.tenant_id, task.database_id, task.vshard_id, &task.plan,) @@ -141,56 +135,6 @@ fn tx_ctx_commit_yields_independent_tasks() { } } -// --------------------------------------------------------------------------- -// Cross-shard DML: different vshards in same procedure -// --------------------------------------------------------------------------- - -#[test] -fn procedure_can_target_multiple_vshards() { - let mut ctx = ProcedureTransactionCtx::new(); - - // Task on vshard 0 - ctx.buffer_task(PhysicalTask { - tenant_id: TenantId::new(1), - vshard_id: VShardId::new(0), - database_id: DatabaseId::DEFAULT, - plan: PhysicalPlan::Document(DocumentOp::PointPut { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "a"), - document_id: "d1".into(), - value: vec![], - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - post_set_op: PostSetOp::None, - txn_id: None, - }); - // Task on vshard 1 (different shard) - ctx.buffer_task(PhysicalTask { - tenant_id: TenantId::new(1), - vshard_id: VShardId::new(1), - database_id: DatabaseId::DEFAULT, - plan: PhysicalPlan::Document(DocumentOp::PointPut { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "b"), - document_id: "d2".into(), - value: vec![], - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - post_set_op: PostSetOp::None, - txn_id: None, - }); - - let tasks = ctx.take_buffered_tasks(); - assert_eq!(tasks.len(), 2); - assert_ne!(tasks[0].vshard_id, tasks[1].vshard_id); -} - // --------------------------------------------------------------------------- // Replicated constraint violations → DeltaReject with CompensationHint // --------------------------------------------------------------------------- @@ -243,21 +187,6 @@ fn delta_reject_with_custom_hint() { )); } -// --------------------------------------------------------------------------- -// Procedure execution: coordinator-only (not migrated to shard leader) -// --------------------------------------------------------------------------- - -#[test] -fn procedure_executes_on_coordinator() { - // Procedures execute on the coordinator that received CALL, never - // migrating to a shard leader; body DML dispatches to shard leaders via - // the normal query path (ProcedureTransactionCtx buffers locally, then - // flushes via dispatch_to_data_plane). - let mut ctx = ProcedureTransactionCtx::new(); - // Empty context = no migration state, executes locally. - assert!(ctx.take_buffered_tasks().is_empty()); -} - // --------------------------------------------------------------------------- // Procedure body with COMMIT parses correctly // --------------------------------------------------------------------------- diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs b/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs index 5ef85e352..213dc9d7e 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/cluster_triggers.rs @@ -44,7 +44,7 @@ fn async_trigger_not_in_raft_log() { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "orders"), document_id: "doc-1".into(), value: b"{}".to_vec(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), pk_bytes: Vec::new(), returning: None, rls_filters: Vec::new(), @@ -63,9 +63,12 @@ fn async_trigger_not_in_raft_log() { assert!(entry.is_some()); // The entry serializes to bytes that can be deserialized back. - let bytes = entry.unwrap().to_bytes(); + let bytes = entry + .expect("a PointPut has a replicated encoding") + .encode() + .expect("encode replicated entry bytes"); let (tid, vsid, restored, _resolved_now_ms) = - nodedb::control::wal_replication::from_replicated_entry(&bytes, None) + nodedb::control::wal_replication::decode_replicated_entry(&bytes) .unwrap() .unwrap(); assert_eq!(tid, TenantId::new(1)); @@ -92,6 +95,7 @@ fn cross_shard_request_carries_cascade_depth() { target_vshard: 1, source_lsn: 100, source_sequence: 1, + origin: String::new(), cascade_depth: 3, source_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "orders").to_string(), }; @@ -111,6 +115,7 @@ fn cross_shard_request_serialization_roundtrip() { target_vshard: 2, source_lsn: 500, source_sequence: 42, + origin: String::new(), cascade_depth: 1, source_collection: "articles".into(), }; @@ -193,6 +198,11 @@ fn event_source_preserved_through_write_event() { valid_time_ms: None, user_id: None, statement_digest: None, + // The write committed now, so the event stays in age retention. + commit_hlc: std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .ok() + .and_then(|elapsed| u64::try_from(elapsed.as_nanos()).ok()), }; // After leader failover, new leader's Event Plane replays from WAL. // The replayed events have source: User → triggers fire. @@ -208,7 +218,7 @@ fn replicated_entry_roundtrip_point_delete() { let plan = PhysicalPlan::Document(DocumentOp::PointDelete { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "orders"), document_id: "doc-99".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: Some(nodedb_types::Surrogate::new(99)), pk_bytes: Vec::new(), returning: None, rls_filters: Vec::new(), @@ -223,8 +233,8 @@ fn replicated_entry_roundtrip_point_delete() { ) .expect("encode replicated entry") .expect("a write plan encodes to an entry"); - let bytes = entry.to_bytes(); - let (_, _, restored, _) = nodedb::control::wal_replication::from_replicated_entry(&bytes, None) + let bytes = entry.encode().expect("encode the replicated entry"); + let (_, _, restored, _) = nodedb::control::wal_replication::decode_replicated_entry(&bytes) .unwrap() .unwrap(); assert!(matches!( @@ -242,7 +252,7 @@ fn read_ops_not_replicated() { rls_filters: vec![], system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: None, pk_bytes: Vec::new(), }); let entry = encode_entry( @@ -268,7 +278,7 @@ fn procedure_dml_is_normal_write() { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "archive"), document_id: "a-1".into(), value: b"{}".to_vec(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), pk_bytes: Vec::new(), returning: None, rls_filters: Vec::new(), diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/decommission_flow.rs b/nodedb-cluster-tests/tests/misc_suite/cases/decommission_flow.rs index 2e9342df3..fd980798d 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/decommission_flow.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/decommission_flow.rs @@ -62,7 +62,7 @@ impl MetadataProposer for DirectProposer { async fn propose_and_wait(&self, entry: MetadataEntry) -> Result { let idx = self.next_index.fetch_add(1, Ordering::SeqCst); let bytes = encode_entry(&entry).expect("encode metadata entry"); - self.applier.apply(&[(idx, bytes)]); + self.applier.apply(&[(idx, bytes)]).await; self.proposed.lock().unwrap().push(entry); Ok(idx) } diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/install_snapshot_basic.rs b/nodedb-cluster-tests/tests/misc_suite/cases/install_snapshot_basic.rs index 7fda467a5..722372890 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/install_snapshot_basic.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/install_snapshot_basic.rs @@ -42,6 +42,8 @@ fn snapshot_req(term: u64, index: u64, done: bool, data: Vec) -> InstallSnap done, group_id: 0, total_size: 0, + voters: Vec::new(), + learners: Vec::new(), } } diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/install_snapshot_chunked.rs b/nodedb-cluster-tests/tests/misc_suite/cases/install_snapshot_chunked.rs index df323b844..218ac9b5a 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/install_snapshot_chunked.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/install_snapshot_chunked.rs @@ -38,6 +38,8 @@ fn make_req( done, group_id, total_size: 0, + voters: Vec::new(), + learners: Vec::new(), } } @@ -106,16 +108,18 @@ async fn chunked_happy_path() { } } - // After commit the `.partial` file must be gone and `.snap` must exist. + // A committed install keeps no file: no partial, no staged file, no copy. let recv_dir = data_dir.join("recv_snapshots"); - assert!( - !recv_dir.join("42.partial").exists(), - "partial file must be removed after commit" - ); - assert!( - recv_dir.join("42.snap").exists(), - "snap file must exist after commit" - ); + let left: Vec = std::fs::read_dir(&recv_dir) + .expect("read recv_snapshots") + .map(|e| { + e.expect("dir entry") + .file_name() + .to_string_lossy() + .into_owned() + }) + .collect(); + assert!(left.is_empty(), "files left after commit: {left:?}"); } // --------------------------------------------------------------------------- @@ -234,11 +238,17 @@ async fn corrupt_chunk_crc() { "unexpected error: {err}" ); - // The partial file must still exist after CRC failure (not renamed to .snap). + // The partial file stays for inspection, and nothing is staged. let recv_dir = data_dir.join("recv_snapshots"); assert!( - !recv_dir.join("7.snap").exists(), - "snap must NOT exist after CRC failure" + recv_dir.join("7.partial").exists(), + "partial must remain after CRC failure" + ); + assert!( + nodedb_cluster::install_snapshot::staged::list_staged(&recv_dir) + .expect("list staged") + .is_empty(), + "nothing may be staged after CRC failure" ); } diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/reachability_loop.rs b/nodedb-cluster-tests/tests/misc_suite/cases/reachability_loop.rs index 7d5f6debc..290c0f556 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/reachability_loop.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/reachability_loop.rs @@ -14,7 +14,9 @@ use std::time::{Duration, Instant}; use async_trait::async_trait; -use nodedb_cluster::circuit_breaker::{CircuitBreaker, CircuitBreakerConfig, CircuitState}; +use nodedb_cluster::circuit_breaker::{ + Admission, CircuitBreaker, CircuitBreakerConfig, CircuitState, +}; use nodedb_cluster::error::{ClusterError, Result}; use nodedb_cluster::reachability::{ ReachabilityDriver, ReachabilityDriverConfig, ReachabilityProber, @@ -61,7 +63,7 @@ async fn reachability_loop_recovers_open_breaker_without_user_traffic() { // driver still needs to drive the actual transition. cooldown: Duration::from_millis(100), })); - breaker.record_failure(42); + breaker.record_failure(42, Admission::Normal); assert_eq!(breaker.state(42), CircuitState::Open); // --- Flappy prober starts "unhealthy". --- @@ -81,19 +83,19 @@ async fn reachability_loop_recovers_open_breaker_without_user_traffic() { impl ReachabilityProber for RelayProber { async fn probe(&self, peer: u64) -> Result<()> { // Mirror send_rpc: check → probe → record outcome. - if self.breaker.check(peer).is_err() { + let Ok(admission) = self.breaker.check(peer) else { return Err(ClusterError::CircuitOpen { node_id: peer, failures: self.breaker.failure_count(peer), }); - } + }; match self.inner.probe(peer).await { Ok(()) => { - self.breaker.record_success(peer); + self.breaker.record_success(peer, admission); Ok(()) } Err(e) => { - self.breaker.record_failure(peer); + self.breaker.record_failure(peer, admission); Err(e) } } diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/swim_routing_invalidation.rs b/nodedb-cluster-tests/tests/misc_suite/cases/swim_routing_invalidation.rs index bf7e0e7e4..e64358ac7 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/swim_routing_invalidation.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/swim_routing_invalidation.rs @@ -77,9 +77,9 @@ async fn swim_dead_leader_clears_routing_hint() { guard.set_leader(3, 3); } - // --- Hook node A to the routing table. --- + // --- Hook node A (id 1, a voter of every group) to the routing table. --- let hook: Arc = - Arc::new(RoutingLivenessHook::new(rt.clone(), resolver_static())); + Arc::new(RoutingLivenessHook::new(rt.clone(), resolver_static(), 1)); let h_a: SwimHandle = spawn_with_subscribers( fast_cfg(), diff --git a/nodedb-cluster-tests/tests/misc_suite/cases/transport_security.rs b/nodedb-cluster-tests/tests/misc_suite/cases/transport_security.rs index 056045d46..95910aa4c 100644 --- a/nodedb-cluster-tests/tests/misc_suite/cases/transport_security.rs +++ b/nodedb-cluster-tests/tests/misc_suite/cases/transport_security.rs @@ -1,19 +1,18 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Integration tests for Phase L cluster transport security. +//! Integration tests for cluster transport security. //! //! These tests exercise the full Raft-over-QUIC and SWIM-over-UDP paths -//! end-to-end against the security properties L.1 / L.2 / L.3 / L.5 are -//! supposed to enforce: +//! end-to-end against the security properties the transports enforce: //! -//! - **Raft mTLS (L.1):** transports built with different `TlsCredentials` +//! - **Raft mTLS:** transports built with different `TlsCredentials` //! (distinct CAs) cannot handshake and outbound RPCs fail. Mixing mTLS //! and Insecure modes likewise fails. -//! - **Raft frame MAC + anti-replay (L.2):** a forged request arriving +//! - **Raft frame MAC + anti-replay:** a forged request arriving //! on an authenticated QUIC stream with the wrong cluster MAC key is //! rejected; a replayed frame from a captured envelope is rejected by //! the per-peer seq window. -//! - **SWIM UDP MAC (L.3):** a datagram signed with the wrong cluster +//! - **SWIM UDP MAC:** a datagram signed with the wrong cluster //! key is rejected; a datagram with a spoofed source address is //! rejected. //! - **Counter observability:** `insecure_transport_count()` bumps for @@ -65,6 +64,8 @@ impl RaftRpcHandler for EchoHandler { term: req.term, success: true, last_log_index: req.prev_log_index + req.entries.len() as u64, + round: req.round, + needs_snapshot: false, })) } RaftRpc::RequestVoteRequest(req) => { @@ -202,6 +203,8 @@ fn sample_append(term: u64) -> AppendEntriesRequest { entries: vec![], leader_commit: 0, group_id: 0, + round: 1, + replicated_floor: 0, } } @@ -258,7 +261,7 @@ fn install_shared_identity(nodes: &[(u64, [u8; 32])], transports: &[&Arc Err(ClusterError::Transport { @@ -337,6 +339,8 @@ fn sample_append(term: u64) -> AppendEntriesRequest { entries: vec![], leader_commit: 0, group_id: 0, + round: 1, + replicated_floor: 0, } } @@ -441,6 +445,7 @@ fn handshake_capabilities_roundtrip() { let hs = VersionHandshake { range: (r.min.0, r.max.0), capabilities: caps, + build_id: "test-build".to_owned(), }; let bytes = zerompk::to_msgpack_vec(&hs).unwrap(); let decoded: VersionHandshake = zerompk::from_msgpack(&bytes).unwrap(); @@ -455,6 +460,7 @@ fn ack_capabilities_roundtrip() { let ack = VersionHandshakeAck { agreed: 2, capabilities: caps, + build_id: "test-build".to_owned(), }; let bytes = zerompk::to_msgpack_vec(&ack).unwrap(); let decoded: VersionHandshakeAck = zerompk::from_msgpack(&bytes).unwrap(); @@ -469,6 +475,7 @@ fn capabilities_non_zero_accepted_in_negotiation() { let hs = VersionHandshake { range: (r.min.0, r.max.0), capabilities: u64::MAX, + build_id: "test-build".to_owned(), }; let bytes = zerompk::to_msgpack_vec(&hs).unwrap(); let decoded: VersionHandshake = zerompk::from_msgpack(&bytes).unwrap(); @@ -571,6 +578,7 @@ async fn handshake_rejects_incompatible_versions() { let ack = VersionHandshakeAck { agreed: u16::MAX, capabilities: 0, + build_id: "fake-server-build".to_owned(), }; let payload = zerompk::to_msgpack_vec(&ack).unwrap(); write_framed_raw(&mut send, &payload).await; @@ -623,8 +631,10 @@ fn handle_join_request_rejects_incompatible_wire_version() { node_id: 2, listen_addr: "10.0.0.2:9400".into(), wire_version: 0, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), spiffe_id: None, spki_pin: None, + swim_addr: None, }; let resp = handle_join_request(&req, &mut topology, &routing, 42); @@ -674,6 +684,7 @@ async fn mismatch_does_not_dispatch_rpc() { let hs = VersionHandshake { range: (u16::MAX - 1, u16::MAX), capabilities: 0, + build_id: "raw-client-build".to_owned(), }; let payload = zerompk::to_msgpack_vec(&hs).unwrap(); write_framed_raw(&mut send, &payload).await; @@ -709,6 +720,7 @@ async fn mismatch_closes_quic_connection_with_app_error() { let hs = VersionHandshake { range: (u16::MAX - 1, u16::MAX), capabilities: 0, + build_id: "raw-client-build".to_owned(), }; let payload = zerompk::to_msgpack_vec(&hs).unwrap(); write_framed_raw(&mut send, &payload).await; diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/calvin_read_failover_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/calvin_read_failover_cross_node.rs new file mode 100644 index 000000000..7686335ea --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/calvin_read_failover_cross_node.rs @@ -0,0 +1,173 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A read-write transaction's read is never validated against another node's +//! WAL. +//! +//! A transaction reads a row of a collection whose home group leader is `A`, +//! so `A` serves the read and its version is a position in `A`'s WAL. `A` is +//! then cut off from both other nodes, they elect `B`, and a write to the read +//! row commits at `B`. The partition heals. The transaction also writes, so +//! its COMMIT goes through Calvin, and `B` validates the read. `B`'s versions +//! do not compare with `A`'s, so the commit must abort with a retryable +//! serialization error. + +use std::time::Duration; + +use nodedb_client::NativeClient; +use nodedb_client::native::pool::PoolConfig; +use nodedb_types::id::VShardId; +use nodedb_types::{CollectionKey, DatabaseId}; + +use crate::common::cluster_harness::{TestCluster, TestClusterNode, wait_for, wait_for_async}; +use crate::common::occ_shuffle::pg_detail; + +const DOCS: &str = "crf_docs"; +const LOG: &str = "crf_log"; +/// The native transport's message for a serialization abort. +const SERIALIZATION_ABORT: &str = "could not serialize access due to concurrent update"; + +fn pinned_native_client(node: &TestClusterNode) -> NativeClient { + node.native_client_with(|base| PoolConfig { + max_size: 1, + ..base + }) +} + +/// Sever node `a` and node `b` from each other, both ways, or heal the link. +fn link(cluster: &TestCluster, a: usize, b: usize, severed: bool) { + let (a_node, b_node) = (&cluster.nodes[a], &cluster.nodes[b]); + let a_transport = a_node + .shared + .cluster_transport + .as_ref() + .expect("cluster transport"); + let b_transport = b_node + .shared + .cluster_transport + .as_ref() + .expect("cluster transport"); + if severed { + a_transport.sever(b_node.node_id); + b_transport.sever(a_node.node_id); + } else { + a_transport.heal(b_node.node_id); + b_transport.heal(a_node.node_id); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_read_served_by_a_deposed_leader_aborts_the_calvin_commit() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + for coll in [DOCS, LOG] { + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {coll}")) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION {coll}: {e}")); + } + wait_for( + "all 3 nodes see the collections", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 2) + }, + ) + .await; + cluster.nodes[0] + .client + .simple_query(&format!("INSERT INTO {DOCS} (id, value) VALUES ('a', '1')")) + .await + .unwrap_or_else(|e| panic!("seed row: {}", pg_detail(&e))); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + + let group = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard( + VShardId::from_collection(CollectionKey::from_bare(DatabaseId::DEFAULT, DOCS)).as_u32(), + ) + .expect("the collection's group"); + let leader_of = |observer: usize| -> u64 { + cluster.nodes[observer] + .all_group_leaders() + .into_iter() + .find(|(g, _)| *g == group) + .map(|(_, leader)| leader) + .unwrap_or(0) + }; + let old_leader_id = leader_of(0); + let old = cluster + .nodes + .iter() + .position(|n| n.node_id == old_leader_id) + .expect("the collection's group has a leader"); + let reader = (old + 1) % 3; + let writer = (old + 2) % 3; + + let driver = pinned_native_client(&cluster.nodes[reader]); + driver.begin().await.expect("native BEGIN"); + driver + .query(&format!("SELECT value FROM {DOCS} WHERE id = 'a'")) + .await + .expect("in-txn read served by the old leader"); + driver + .query(&format!( + "INSERT INTO {LOG} (id, value) VALUES ('entry', '1')" + )) + .await + .expect("buffer a write so the commit goes through Calvin"); + + // Move the group's leadership: cut the old leader off until the other two + // elect a new one, then commit a write to the read row there. + link(&cluster, old, reader, true); + link(&cluster, old, writer, true); + wait_for( + "the reader and the writer elect a new leader of the collection's group", + Duration::from_secs(30), + Duration::from_millis(100), + || { + let leader = leader_of(writer); + leader != 0 && leader != old_leader_id + }, + ) + .await; + let conflicting = format!("UPDATE {DOCS} SET value = '2' WHERE id = 'a'"); + wait_for_async( + "the conflicting write commits at the new leader", + Duration::from_secs(30), + Duration::from_millis(200), + || async { + cluster.nodes[writer] + .client + .simple_query(&conflicting) + .await + .is_ok() + }, + ) + .await; + link(&cluster, old, reader, false); + link(&cluster, old, writer, false); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(30)) + .await; + + let err = driver + .commit() + .await + .expect_err("a read the deposed leader served must not validate at the new leader"); + assert!( + err.message().contains(SERIALIZATION_ABORT), + "expected a retryable serialization abort, got: {err}" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_homed_read_failover.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_homed_read_failover.rs new file mode 100644 index 000000000..bcf89abbc --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_homed_read_failover.rs @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A read-only transaction's homed read is never validated by a node that lost +//! leadership. +//! +//! A transaction reads the neighbors of `r0` through a node that does not +//! lead `r0`'s key vShard, so the read is homed on that vShard's leader `L`. +//! `L` is then cut off from both other nodes. They elect a new leader, and an +//! edge out of `r0` commits there. `L` is reconnected to the reader's node +//! only, still cut off from the rest of the cluster. The reader's COMMIT must +//! fail: `L` holds no leader lease, and the new leader's versions do +//! not compare with the read `L` served. + +use std::time::Duration; + +use nodedb_types::id::VShardId; + +use crate::common::cluster_harness::{TestCluster, wait_for, wait_for_async}; +use crate::common::occ_shuffle::pg_detail; + +const EDGES: &str = "hrf_edges"; +const READ_KEY: &str = "r0"; + +/// Sever node `a` and node `b` from each other, both ways, or heal the link. +fn link(cluster: &TestCluster, a: usize, b: usize, severed: bool) { + let (a_node, b_node) = (&cluster.nodes[a], &cluster.nodes[b]); + let a_transport = a_node + .shared + .cluster_transport + .as_ref() + .expect("cluster transport"); + let b_transport = b_node + .shared + .cluster_transport + .as_ref() + .expect("cluster transport"); + if severed { + a_transport.sever(b_node.node_id); + b_transport.sever(a_node.node_id); + } else { + a_transport.heal(b_node.node_id); + b_transport.heal(a_node.node_id); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_deposed_leader_never_validates_a_homed_read() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {EDGES}")) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION {EDGES}: {e}")); + wait_for( + "all 3 nodes see the collection", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 1) + }, + ) + .await; + cluster.nodes[0] + .client + .simple_query(&format!( + "GRAPH INSERT EDGE IN '{EDGES}' FROM '{READ_KEY}' TO 's0' TYPE 'K'" + )) + .await + .unwrap_or_else(|e| panic!("seed edge: {}", pg_detail(&e))); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + + let group = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(VShardId::from_key(READ_KEY.as_bytes()).as_u32()) + .expect("the read key's group"); + let leader_of = |observer: usize| -> u64 { + cluster.nodes[observer] + .all_group_leaders() + .into_iter() + .find(|(g, _)| *g == group) + .map(|(_, leader)| leader) + .unwrap_or(0) + }; + let old_leader_id = leader_of(0); + let old = cluster + .nodes + .iter() + .position(|n| n.node_id == old_leader_id) + .expect("the read key's group has a leader"); + let reader = (old + 1) % 3; + let writer = (old + 2) % 3; + let read_sql = format!("GRAPH NEIGHBORS IN '{EDGES}' OF '{READ_KEY}' DIRECTION out"); + + let client = &cluster.nodes[reader].client; + client.simple_query("BEGIN").await.expect("BEGIN"); + client + .simple_query(&read_sql) + .await + .unwrap_or_else(|e| panic!("in-txn NEIGHBORS: {}", pg_detail(&e))); + + // Cut the leader off. The other two elect a new leader of the group. + link(&cluster, old, reader, true); + link(&cluster, old, writer, true); + wait_for( + "the reader and the writer elect a new leader of the read key's group", + Duration::from_secs(30), + Duration::from_millis(100), + || { + let leader = leader_of(writer); + leader != 0 && leader != old_leader_id + }, + ) + .await; + let conflicting = + format!("GRAPH INSERT EDGE IN '{EDGES}' FROM '{READ_KEY}' TO 'fresh' TYPE 'K'"); + wait_for_async( + "the conflicting edge commits at the new leader", + Duration::from_secs(30), + Duration::from_millis(200), + || async { + cluster.nodes[writer] + .client + .simple_query(&conflicting) + .await + .is_ok() + }, + ) + .await; + + // The deposed leader can reach the reader again, never the writer. + link(&cluster, old, reader, false); + let commit = client.simple_query("COMMIT").await; + assert!( + commit.is_err(), + "a homed read the deposed leader served committed past a conflicting write: {commit:?}" + ); + + link(&cluster, old, writer, false); + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_homed_read_per_core.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_homed_read_per_core.rs new file mode 100644 index 000000000..c9834b803 --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_homed_read_per_core.rs @@ -0,0 +1,209 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A read-only transaction's homed graph read is checked on the one core that +//! owns the read vShard. +//! +//! With a replication factor of 1 and two cores per node, a group's leader +//! runs its vShards on both of its cores. vShard `v` belongs to group +//! `1 + v % GROUPS` and runs on core `v % CORES`. With an even group count +//! every group's vShards share one parity, so a group sits on one core, and +//! the leader balance gives each node one group. [`GROUPS`] is odd, so every +//! group spans both cores. A transaction reads the neighbors of `r` +//! through another node, so the read is homed on `r`'s key vShard. A +//! concurrent edge between two keys that the same leader owns on its other +//! core must not abort the COMMIT: it changed nothing the read saw. A +//! concurrent edge out of `r` must abort it. + +use std::collections::HashMap; +use std::time::Duration; + +use nodedb_types::id::VShardId; + +use crate::common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; +use crate::common::occ_shuffle::{pg_detail, pg_sqlstate}; + +const EDGES: &str = "hpc_edges"; +const CORES: usize = 2; +/// Data groups of the cluster. Odd, so every group spans both cores. +const GROUPS: u64 = 3; + +/// The node that leads `key`'s key vShard, and the core it runs it on. +fn owner<'a>( + cluster: &'a TestCluster, + group_of: &HashMap, + leaders: &HashMap, + key: &str, +) -> (&'a TestClusterNode, usize) { + let vshard = VShardId::from_key(key.as_bytes()); + let leader = group_of + .get(&vshard.as_u32()) + .and_then(|g| leaders.get(g)) + .copied() + .unwrap_or_else(|| panic!("vShard {} of '{key}' has a leader", vshard.as_u32())); + let node = cluster + .nodes + .iter() + .find(|n| n.node_id == leader) + .unwrap_or_else(|| panic!("leader {leader} is a cluster node")); + let core = node + .shared + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()) + .router() + .resolve(vshard) + .unwrap_or_else(|| panic!("a core owns vShard {}", vshard.as_u32())); + (node, core) +} + +async fn insert_edge(node: &TestClusterNode, src: &str, dst: &str) { + node.client + .simple_query(&format!( + "GRAPH INSERT EDGE IN '{EDGES}' FROM '{src}' TO '{dst}' TYPE 'K'" + )) + .await + .unwrap_or_else(|e| panic!("edge {src}->{dst}: {}", pg_detail(&e))); +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_write_on_another_core_of_the_leader_keeps_the_read() { + let cluster = TestCluster::spawn_three_with_groups_rf_and_cores(GROUPS, 1, CORES) + .await + .expect("3-node RF1 cluster"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {EDGES}")) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION {EDGES}: {e}")); + wait_for( + "all 3 nodes see the collection", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 1) + }, + ) + .await; + let group_of: HashMap = { + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + let mut map = HashMap::new(); + for g in routing.group_ids().into_iter().filter(|g| *g != 0) { + for vs in routing.vshards_for_group(g) { + map.insert(vs, g); + } + } + map + }; + // At replication factor 1 each node hosts only its own groups, so the + // leaders come from every group's replica. + wait_for( + "every data group has a leader", + Duration::from_secs(30), + Duration::from_millis(50), + || { + let leaders = cluster.data_group_leaders(); + group_of.values().all(|g| leaders.contains_key(g)) + }, + ) + .await; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + let leaders: HashMap = cluster.data_group_leaders(); + + // `r` is read. `x` and `y` live on `r`'s leader, on the other core. + let reader_key = "r0"; + let (leader, read_core) = owner(&cluster, &group_of, &leaders, reader_key); + let other_core: Vec = (0..10_000) + .map(|i| format!("oc{i}")) + .filter(|key| { + let (node, core) = owner(&cluster, &group_of, &leaders, key); + node.node_id == leader.node_id && core != read_core + }) + .take(2) + .collect(); + assert_eq!( + other_core.len(), + 2, + "the leader of '{reader_key}' owns keys on its other core" + ); + let coordinator = cluster + .nodes + .iter() + .find(|n| n.node_id != leader.node_id) + .expect("a node that does not lead the read vShard"); + let writer = cluster + .nodes + .iter() + .find(|n| n.node_id != coordinator.node_id) + .expect("a second node"); + + insert_edge(writer, reader_key, "r_seed").await; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + let read_sql = format!("GRAPH NEIGHBORS IN '{EDGES}' OF '{reader_key}' DIRECTION out"); + + // A write on the leader's other core leaves the read current. + coordinator + .client + .simple_query("BEGIN") + .await + .expect("BEGIN"); + coordinator + .client + .simple_query(&read_sql) + .await + .unwrap_or_else(|e| panic!("in-txn NEIGHBORS: {}", pg_detail(&e))); + insert_edge(writer, &other_core[0], &other_core[1]).await; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + coordinator + .client + .simple_query("COMMIT") + .await + .unwrap_or_else(|e| { + panic!( + "a write on another core of the read vShard's leader must not abort: {}", + pg_detail(&e) + ) + }); + + // A write on the read vShard aborts. + coordinator + .client + .simple_query("BEGIN") + .await + .expect("BEGIN"); + coordinator + .client + .simple_query(&read_sql) + .await + .unwrap_or_else(|e| panic!("in-txn NEIGHBORS: {}", pg_detail(&e))); + insert_edge(writer, reader_key, "r_fresh").await; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + let err = coordinator + .client + .simple_query("COMMIT") + .await + .expect_err("a write on the read vShard must abort the read-only transaction"); + assert_eq!( + pg_sqlstate(&err).as_deref(), + Some("40001"), + "expected serialization_failure (40001), got: {}", + pg_detail(&err) + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_match_read_set_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_match_read_set_cross_node.rs new file mode 100644 index 000000000..d726ea532 --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_match_read_set_cross_node.rs @@ -0,0 +1,270 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A cluster MATCH inside a transaction puts every vShard it read on the +//! transaction's read-set, each at the watermark that vShard served. +//! +//! With a replication factor of 1, a pattern over a graph whose node keys +//! cover every data group reads edges on every node. Another transaction then +//! writes an edge on a vShard led by a different node than the reader's +//! coordinator. The reader's COMMIT must fail with a serialization error: +//! +//! - over pgwire, read-only: the commit checks each home vShard's write floor +//! on its leader; +//! - over the native protocol, with a write: the commit goes through Calvin, +//! whose participants include every vShard the MATCH read. +//! +//! A control transaction with no concurrent write commits, so the check does +//! not abort everything. + +use std::collections::{BTreeSet, HashMap}; +use std::time::Duration; + +use nodedb_client::NativeClient; +use nodedb_client::native::pool::PoolConfig; +use nodedb_types::id::VShardId; + +use crate::common::cluster_harness::{TestCluster, TestClusterNode, wait_for, wait_for_async}; +use crate::common::occ_shuffle::{pg_detail, pg_sqlstate}; + +const EDGES: &str = "mrs_edges"; +const LOG: &str = "mrs_log"; +const MIN_NODES: usize = 24; +/// The native transport's message for a serialization abort. +const SERIALIZATION_ABORT: &str = "could not serialize access due to concurrent update"; + +/// Names `n0, n1, …`, enough that their key vShards cover every data group. +fn node_names(groups: &BTreeSet, group_of: &HashMap) -> Vec { + let mut out = Vec::new(); + let mut covered = BTreeSet::new(); + let mut i = 0usize; + while out.len() < MIN_NODES || covered != *groups { + let name = format!("n{i}"); + let vshard = VShardId::from_key(name.as_bytes()).as_u32(); + covered.insert(group_of.get(&vshard).copied().unwrap_or(0)); + out.push(name); + i += 1; + } + out +} + +fn row_count(msgs: &[tokio_postgres::SimpleQueryMessage]) -> usize { + msgs.iter() + .filter(|m| matches!(m, tokio_postgres::SimpleQueryMessage::Row(_))) + .count() +} + +/// Insert `src -> dst` from `writer` and wait until `probe` sees it. +async fn concurrent_edge(writer: &TestClusterNode, probe: &TestClusterNode, src: &str, dst: &str) { + writer + .client + .simple_query(&format!( + "GRAPH INSERT EDGE IN '{EDGES}' FROM '{src}' TO '{dst}' TYPE 'K'" + )) + .await + .unwrap_or_else(|e| panic!("concurrent edge {src}->{dst}: {}", pg_detail(&e))); + wait_for_async( + &format!("edge {src}->{dst} visible before COMMIT"), + Duration::from_secs(15), + Duration::from_millis(50), + || async { + let msgs = probe + .client + .simple_query(&format!( + "GRAPH NEIGHBORS IN '{EDGES}' OF '{src}' DIRECTION out" + )) + .await + .expect("probe neighbors"); + msgs.iter().any(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(row) => { + (0..row.len()).any(|i| row.get(i).is_some_and(|v| v.contains(dst))) + } + _ => false, + }) + }, + ) + .await; +} + +fn pinned_native_client(node: &TestClusterNode) -> NativeClient { + node.native_client_with(|base| PoolConfig { + max_size: 1, + ..base + }) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn a_concurrent_edge_on_a_read_vshard_aborts_the_match_transaction() { + let cluster = TestCluster::spawn_three_with_replication_factor(1) + .await + .expect("3-node RF1 cluster"); + for coll in [EDGES, LOG] { + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {coll}")) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION {coll}: {e}")); + } + wait_for( + "all 3 nodes see the collections", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 2) + }, + ) + .await; + + let (groups, group_of): (BTreeSet, HashMap) = { + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + let groups: BTreeSet = routing + .group_ids() + .into_iter() + .filter(|g| *g != 0) + .collect(); + let mut map = HashMap::new(); + for &g in &groups { + for vs in routing.vshards_for_group(g) { + map.insert(vs, g); + } + } + (groups, map) + }; + wait_for( + "each data group has exactly one replica and a leader", + Duration::from_secs(30), + Duration::from_millis(50), + || { + groups.iter().all(|&g| { + cluster + .nodes + .iter() + .filter(|node| node.replicates_data_group(g)) + .count() + == 1 + }) && { + let leaders = cluster.data_group_leaders(); + groups.iter().all(|g| leaders.contains_key(g)) + } + }, + ) + .await; + + let names = node_names(&groups, &group_of); + let n = names.len(); + for i in 0..n { + cluster.nodes[0] + .client + .simple_query(&format!( + "GRAPH INSERT EDGE IN '{EDGES}' FROM '{}' TO '{}' TYPE 'K'", + names[i], + names[(i + 1) % n] + )) + .await + .unwrap_or_else(|e| panic!("insert ring edge {i}: {}", pg_detail(&e))); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + + // The reader's coordinator, and ring nodes whose key vShard another node + // leads: a write there lands on a vShard the pattern read on a remote node. + let coordinator = &cluster.nodes[0]; + // Each node hosts only the groups placed on it, so the leaders come from + // every group's replica. + let leaders: HashMap = cluster.data_group_leaders(); + let remote_sources: Vec<&String> = names + .iter() + .filter(|name| { + let vshard = VShardId::from_key(name.as_bytes()).as_u32(); + group_of + .get(&vshard) + .and_then(|g| leaders.get(g)) + .is_some_and(|leader| *leader != coordinator.node_id) + }) + .collect(); + assert!( + remote_sources.len() >= 2, + "the ring must have nodes led by other cluster nodes" + ); + let writer = &cluster.nodes[1]; + let probe = &cluster.nodes[2]; + let match_sql = format!("MATCH (a)-[:K]->(b) IN '{EDGES}' RETURN a, b"); + + // Control: nothing writes the graph, so the read-set still holds. + coordinator + .client + .simple_query("BEGIN") + .await + .expect("BEGIN"); + let rows = coordinator + .client + .simple_query(&match_sql) + .await + .unwrap_or_else(|e| panic!("in-txn MATCH: {}", pg_detail(&e))); + assert_eq!(row_count(&rows), n, "the MATCH returns every ring edge"); + coordinator + .client + .simple_query("COMMIT") + .await + .unwrap_or_else(|e| { + panic!( + "a MATCH transaction with no concurrent write must commit: {}", + pg_detail(&e) + ) + }); + + // pgwire, read-only: a concurrent edge on a remote vShard aborts COMMIT. + coordinator + .client + .simple_query("BEGIN") + .await + .expect("BEGIN"); + coordinator + .client + .simple_query(&match_sql) + .await + .unwrap_or_else(|e| panic!("in-txn MATCH: {}", pg_detail(&e))); + concurrent_edge(writer, probe, remote_sources[0], "fresh_pg").await; + let err = coordinator + .client + .simple_query("COMMIT") + .await + .expect_err("a MATCH read-set with a concurrently written vShard must not commit"); + assert_eq!( + pg_sqlstate(&err).as_deref(), + Some("40001"), + "expected serialization_failure (40001), got: {}", + pg_detail(&err) + ); + + // Native, with a write: the commit goes through Calvin, and the vShard the + // concurrent edge landed on is a participant that finds its read stale. + let driver = pinned_native_client(coordinator); + driver.begin().await.expect("native BEGIN"); + driver.query(&match_sql).await.expect("in-txn native MATCH"); + driver + .query(&format!( + "INSERT INTO {LOG} (id, value) VALUES ('entry', '1')" + )) + .await + .expect("buffer a write into the log collection"); + concurrent_edge(writer, probe, remote_sources[1], "fresh_native").await; + let err = driver + .commit() + .await + .expect_err("a native MATCH read-set with a concurrently written vShard must not commit"); + assert!( + err.message().contains(SERIALIZATION_ABORT), + "expected a serialization abort, got: {err}" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_native_capped_walks_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_native_capped_walks_cross_node.rs new file mode 100644 index 000000000..f47cdc76f --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_native_capped_walks_cross_node.rs @@ -0,0 +1,260 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Native graph walks and `GRAPH TRAVERSE` under a visit cap admit the same +//! nodes in a cluster as on a single node. +//! +//! With a replication factor of 1 and node keys that cover every data group, +//! a capped walk crosses every node's partitions. The reference is the same +//! graph on a single node with the same cap. + +use std::collections::BTreeSet; +use std::time::Duration; + +use nodedb_test_support::native_harness::send_request; +use nodedb_types::protocol::{OpCode, ResponseStatus, TextFields}; + +use super::graph_native_walks_support::{ + data_groups, names, native_session, wait_one_replica_per_group, walk, walk_names, +}; +use crate::common::cluster_harness::{TestCluster, wait_for}; +use crate::common::pgwire_harness::TestServer; + +/// The visit cap of the capped walks: the hub and its 12 neighbours leave +/// room for 7 of the 24 nodes one level further. +const CAP: usize = 20; +const CAPPED_COLL: &str = "gwalk_capped"; + +/// The node ids of a `GRAPH TRAVERSE` result: `{nodes:[{id,depth}], ...}`. +fn traverse_node_ids(msgs: &[tokio_postgres::SimpleQueryMessage]) -> BTreeSet<(String, u64)> { + let cell = msgs + .iter() + .find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .expect("GRAPH TRAVERSE returns a row"); + let value: serde_json::Value = sonic_rs::from_str(&cell).expect("GRAPH TRAVERSE returns JSON"); + value["nodes"] + .as_array() + .expect("a nodes array") + .iter() + .map(|node| { + ( + node["id"].as_str().expect("a node id").to_string(), + node["depth"].as_u64().expect("a node depth"), + ) + }) + .collect() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn capped_walks_admit_the_single_node_set() { + let graph_tuning = nodedb_types::config::tuning::GraphTuning { + max_visited: CAP, + ..Default::default() + }; + let cluster = + TestCluster::spawn_three_with_replication_factor_and_graph_tuning(1, graph_tuning.clone()) + .await + .expect("3-node RF1 cluster"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {CAPPED_COLL}")) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION {CAPPED_COLL}: {e}")); + wait_for( + "all 3 nodes see the collection", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 1) + }, + ) + .await; + let (groups, group_of) = data_groups(&cluster); + wait_one_replica_per_group(&cluster, &groups).await; + + // The hub points at 12 nodes, inserted in reverse name order, and each of + // them at two more. Edge order differs from name order everywhere. + let first = names("f", &groups, &group_of); + let second = names("s", &groups, &group_of); + let hub = "hub"; + let mut inserts: Vec = Vec::new(); + for mid in first.iter().take(12).rev() { + inserts.push(format!( + "GRAPH INSERT EDGE IN '{CAPPED_COLL}' FROM '{hub}' TO '{mid}' TYPE 'K'" + )); + } + for (i, mid) in first.iter().take(12).enumerate() { + for leaf in [&second[2 * i + 1], &second[2 * i]] { + inserts.push(format!( + "GRAPH INSERT EDGE IN '{CAPPED_COLL}' FROM '{mid}' TO '{leaf}' TYPE 'K'" + )); + } + } + // Two paths of equal length to `x`, stored larger name first, and a tail + // `t3` five hops out that the cap stops the path search short of. + let (tie_low, tie_high) = if first[0] < first[1] { + (&first[0], &first[1]) + } else { + (&first[1], &first[0]) + }; + for (src, dst) in [ + (tie_high.as_str(), "x"), + (tie_low.as_str(), "x"), + (second[0].as_str(), "t1"), + ("t1", "t2"), + ("t2", "t3"), + ] { + inserts.push(format!( + "GRAPH INSERT EDGE IN '{CAPPED_COLL}' FROM '{src}' TO '{dst}' TYPE 'K'" + )); + } + + let reference = TestServer::start_with_graph_tuning(graph_tuning).await; + reference + .exec(&format!("CREATE COLLECTION {CAPPED_COLL}")) + .await + .unwrap_or_else(|e| panic!("reference CREATE COLLECTION {CAPPED_COLL}: {e}")); + for (i, insert) in inserts.iter().enumerate() { + reference.exec(insert).await.expect("reference insert"); + cluster.nodes[i % cluster.nodes.len()] + .client + .simple_query(insert) + .await + .unwrap_or_else(|e| panic!("cluster {insert}: {e:?}")); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + + let traverse_sql = format!("GRAPH TRAVERSE IN '{CAPPED_COLL}' FROM '{hub}' DEPTH 2"); + let expected_traverse = traverse_node_ids( + &reference + .client + .simple_query(&traverse_sql) + .await + .expect("reference GRAPH TRAVERSE"), + ); + assert_eq!( + expected_traverse.len(), + CAP, + "the single-node traverse stops at the cap" + ); + + let mut single = native_session(reference.native_port).await; + let mut seq = 100u64; + let fields = TextFields { + collection: Some(CAPPED_COLL.to_string()), + start_node: Some(hub.to_string()), + depth: Some(2), + direction: Some("out".to_string()), + ..Default::default() + }; + seq += 1; + let expected_hop: BTreeSet = + walk_names(&walk(&mut single, seq, OpCode::GraphHop, fields.clone()).await) + .into_iter() + .collect(); + assert_eq!( + expected_hop.len(), + CAP, + "the single-node hop stops at the cap" + ); + + let path_between = |from: &str, to: &str| TextFields { + collection: Some(CAPPED_COLL.to_string()), + start_node: Some(from.to_string()), + end_node: Some(to.to_string()), + depth: Some(5), + ..Default::default() + }; + let path_fields = |to: &str| path_between(hub, to); + // An endpoint no edge names is absent from the graph: no path, from + // either end. + let absent_ends = [(hub, "ghost"), ("ghost", "x")]; + let mut expected_absent = Vec::new(); + for (from, to) in absent_ends { + seq += 1; + let response = + send_request(&mut single, seq, OpCode::GraphPath, path_between(from, to)).await; + assert_ne!( + response.status, + ResponseStatus::Ok, + "the single-node path from {from} to {to} finds an absent endpoint: {response:?}" + ); + expected_absent.push(response.status); + } + seq += 1; + let expected_tie = + walk_names(&walk(&mut single, seq, OpCode::GraphPath, path_fields("x")).await); + assert_eq!( + expected_tie, + vec![hub.to_string(), tie_low.clone(), "x".to_string()], + "the single-node path takes the smaller-named of two equal paths" + ); + seq += 1; + let capped = send_request(&mut single, seq, OpCode::GraphPath, path_fields("t3")).await; + assert_ne!( + capped.status, + ResponseStatus::Ok, + "the single-node path search stops at the cap before reaching t3: {capped:?}" + ); + + for (idx, node) in cluster.nodes.iter().enumerate() { + let got_traverse = traverse_node_ids( + &node + .client + .simple_query(&traverse_sql) + .await + .unwrap_or_else(|e| panic!("node {idx}: GRAPH TRAVERSE: {e}")), + ); + assert_eq!( + got_traverse, expected_traverse, + "node {idx}: a capped GRAPH TRAVERSE admits the single-node nodes" + ); + let mut clustered = native_session(node.native_port).await; + seq += 1; + let got_hop: BTreeSet = + walk_names(&walk(&mut clustered, seq, OpCode::GraphHop, fields.clone()).await) + .into_iter() + .collect(); + assert_eq!( + got_hop, expected_hop, + "node {idx}: a capped native hop admits the single-node set" + ); + seq += 1; + let got_tie = + walk_names(&walk(&mut clustered, seq, OpCode::GraphPath, path_fields("x")).await); + assert_eq!( + got_tie, expected_tie, + "node {idx}: a native path takes the single-node tie" + ); + seq += 1; + let got_capped = + send_request(&mut clustered, seq, OpCode::GraphPath, path_fields("t3")).await; + assert_eq!( + got_capped.status, capped.status, + "node {idx}: a capped native path answers as the single node does: {got_capped:?}" + ); + for ((from, to), expected) in absent_ends.iter().zip(&expected_absent) { + seq += 1; + let got = send_request( + &mut clustered, + seq, + OpCode::GraphPath, + path_between(from, to), + ) + .await; + assert_eq!( + &got.status, expected, + "node {idx}: a path from {from} to {to} answers an absent endpoint as the \ + single node does: {got:?}" + ); + } + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_native_walks_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_native_walks_cross_node.rs new file mode 100644 index 000000000..311c2305b --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_native_walks_cross_node.rs @@ -0,0 +1,168 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Native graph walks sent as one plan give a single node's answer in a +//! cluster. +//! +//! With a replication factor of 1 and node keys that cover every data group, +//! a walk crosses every node's partitions. The reference is the same graph on +//! a single node, read over the same native opcodes. +//! +//! - `GraphHop` returns the reached node set. A single node's Data Plane +//! collects it in a `HashSet` and returns it in that set's iteration order +//! (`nodedb-graph/src/traversal.rs`), so the order is unspecified and the +//! comparison is by set. +//! - `GraphPath` with no collection walks the edges of every collection. The +//! test graph has one shortest path, laid over two collections, so the +//! comparison is by the exact path. + +use std::collections::BTreeSet; +use std::time::Duration; + +use nodedb_types::protocol::{OpCode, TextFields}; + +use super::graph_native_walks_support::{ + data_groups, names, native_session, spread_chain, wait_one_replica_per_group, walk, walk_names, +}; +use crate::common::cluster_harness::{TestCluster, wait_for}; +use crate::common::pgwire_harness::TestServer; + +const HOP_COLL: &str = "gwalk_hop"; +const PATH_COLL_A: &str = "gwalk_path_a"; +const PATH_COLL_B: &str = "gwalk_path_b"; +/// Nodes on the path chain: short enough for the default depth quota. +const PATH_LEN: usize = 8; + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn native_hop_and_collectionless_path_match_a_single_node() { + let cluster = TestCluster::spawn_three_with_replication_factor(1) + .await + .expect("3-node RF1 cluster"); + for coll in [HOP_COLL, PATH_COLL_A, PATH_COLL_B] { + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {coll}")) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION {coll}: {e}")); + } + wait_for( + "all 3 nodes see the collections", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 3) + }, + ) + .await; + + let (groups, group_of) = data_groups(&cluster); + wait_one_replica_per_group(&cluster, &groups).await; + + // Hop graph: a ring plus a chord from every third node. + let hop_names = names("h", &groups, &group_of); + let n = hop_names.len(); + let mut inserts: Vec = Vec::new(); + for i in 0..n { + inserts.push(format!( + "GRAPH INSERT EDGE IN '{HOP_COLL}' FROM '{}' TO '{}' TYPE 'K'", + hop_names[i], + hop_names[(i + 1) % n] + )); + if i % 3 == 0 { + let j = (i * 7 + 3) % n; + if j != i { + inserts.push(format!( + "GRAPH INSERT EDGE IN '{HOP_COLL}' FROM '{}' TO '{}' TYPE 'C'", + hop_names[i], hop_names[j] + )); + } + } + } + // Path graph: one chain whose edges alternate between two collections, so + // only a walk over every collection reaches its end. + let path_names = spread_chain("p", PATH_LEN, &group_of); + for (i, pair) in path_names.windows(2).enumerate() { + let coll = if i % 2 == 0 { PATH_COLL_A } else { PATH_COLL_B }; + inserts.push(format!( + "GRAPH INSERT EDGE IN '{coll}' FROM '{}' TO '{}' TYPE 'N'", + pair[0], pair[1] + )); + } + + let reference = TestServer::start().await; + for coll in [HOP_COLL, PATH_COLL_A, PATH_COLL_B] { + reference + .exec(&format!("CREATE COLLECTION {coll}")) + .await + .unwrap_or_else(|e| panic!("reference CREATE COLLECTION {coll}: {e}")); + } + for insert in &inserts { + reference.exec(insert).await.expect("reference insert"); + cluster.nodes[0] + .client + .simple_query(insert) + .await + .unwrap_or_else(|e| panic!("cluster {insert}: {e:?}")); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + + let mut single = native_session(reference.native_port).await; + let mut seq = 10u64; + for (idx, node) in cluster.nodes.iter().enumerate() { + let mut clustered = native_session(node.native_port).await; + + for start in hop_names.iter().step_by(5) { + for (depth, direction) in [(1u32, "out"), (2, "out"), (3, "both")] { + seq += 1; + let fields = TextFields { + collection: Some(HOP_COLL.to_string()), + start_node: Some(start.clone()), + depth: Some(depth), + direction: Some(direction.to_string()), + ..Default::default() + }; + let expected: BTreeSet = + walk_names(&walk(&mut single, seq, OpCode::GraphHop, fields.clone()).await) + .into_iter() + .collect(); + assert!( + expected.len() > 1, + "the single-node hop from {start} reaches a neighbour" + ); + let got: BTreeSet = + walk_names(&walk(&mut clustered, seq, OpCode::GraphHop, fields).await) + .into_iter() + .collect(); + assert_eq!( + got, expected, + "node {idx}: native hop from {start}, depth {depth}, {direction}, \ + must reach the single-node set" + ); + } + } + + let last = path_names.last().expect("path names"); + seq += 1; + let fields = TextFields { + start_node: Some(path_names[0].clone()), + end_node: Some(last.clone()), + depth: Some(path_names.len() as u32), + ..Default::default() + }; + let expected = walk_names(&walk(&mut single, seq, OpCode::GraphPath, fields.clone()).await); + assert_eq!( + expected, path_names, + "the single-node path with no collection follows the chain over both collections" + ); + let got = walk_names(&walk(&mut clustered, seq, OpCode::GraphPath, fields).await); + assert_eq!( + got, expected, + "node {idx}: a native path with no collection must equal the single-node path" + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_native_walks_support.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_native_walks_support.rs new file mode 100644 index 000000000..29ce16d00 --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_native_walks_support.rs @@ -0,0 +1,157 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Steps the native graph walk tests share: the data-group map of an RF1 +//! cluster, node names that cover every group, and native walk calls. + +use std::collections::{BTreeSet, HashMap}; +use std::time::Duration; + +use nodedb_test_support::native_harness::{open_trust_session, send_request}; +use nodedb_types::id::VShardId; +use nodedb_types::protocol::{NativeResponse, OpCode, ResponseStatus, TextFields}; +use nodedb_types::value::Value; +use tokio::net::TcpStream; + +use crate::common::cluster_harness::{TestCluster, wait_for}; + +pub(super) const MIN_NODES: usize = 24; +/// The trust superuser both harnesses bootstrap. +const SUPERUSER: &str = "nodedb"; + +/// The data groups of `cluster` (group 0 excluded) and the group each vShard +/// belongs to. +pub(super) fn data_groups(cluster: &TestCluster) -> (BTreeSet, HashMap) { + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + let groups: BTreeSet = routing + .group_ids() + .into_iter() + .filter(|g| *g != 0) + .collect(); + let mut map = HashMap::new(); + for &g in &groups { + for vs in routing.vshards_for_group(g) { + map.insert(vs, g); + } + } + (groups, map) +} + +/// Wait until every group in `groups` has exactly one replica in `cluster`. +pub(super) async fn wait_one_replica_per_group(cluster: &TestCluster, groups: &BTreeSet) { + wait_for( + "each data group has exactly one replica", + Duration::from_secs(30), + Duration::from_millis(50), + || { + groups.iter().all(|&g| { + cluster + .nodes + .iter() + .filter(|node| node.replicates_data_group(g)) + .count() + == 1 + }) + }, + ) + .await; +} + +/// Names `{prefix}0, {prefix}1, …`, enough that their key vShards cover every +/// data group. +pub(super) fn names( + prefix: &str, + groups: &BTreeSet, + group_of: &HashMap, +) -> Vec { + let mut out = Vec::new(); + let mut covered = BTreeSet::new(); + let mut i = 0usize; + while out.len() < MIN_NODES || covered != *groups { + let name = format!("{prefix}{i}"); + let vshard = VShardId::from_key(name.as_bytes()).as_u32(); + covered.insert(group_of.get(&vshard).copied().unwrap_or(0)); + out.push(name); + i += 1; + } + out +} + +/// `len` names `{prefix}0, {prefix}1, …` whose consecutive entries home to +/// different data groups, so every hop of a chain over them crosses groups. +pub(super) fn spread_chain(prefix: &str, len: usize, group_of: &HashMap) -> Vec { + let mut out: Vec = Vec::new(); + let mut last_group = None; + let mut i = 0usize; + while out.len() < len { + let name = format!("{prefix}{i}"); + let group = group_of + .get(&VShardId::from_key(name.as_bytes()).as_u32()) + .copied(); + if group != last_group { + last_group = group; + out.push(name); + } + i += 1; + assert!(i < 100_000, "no spread chain of {len} names for '{prefix}'"); + } + out +} + +/// A native session authenticated as the trust superuser, in JSON framing. +pub(super) async fn native_session(port: u16) -> TcpStream { + open_trust_session(port, SUPERUSER).await +} + +/// Every node name in a walk response, in response order. A cell can hold a +/// name, a JSON text of names, or an array of either. +pub(super) fn walk_names(response: &NativeResponse) -> Vec { + fn from_json(value: &serde_json::Value, out: &mut Vec) { + match value { + serde_json::Value::String(s) => out.push(s.clone()), + serde_json::Value::Array(items) => items.iter().for_each(|v| from_json(v, out)), + serde_json::Value::Object(map) => map.values().for_each(|v| from_json(v, out)), + _ => {} + } + } + fn from_value(value: &Value, out: &mut Vec) { + match value { + Value::String(s) => match sonic_rs::from_str::(s) { + Ok(json @ (serde_json::Value::Array(_) | serde_json::Value::Object(_))) => { + from_json(&json, out) + } + _ => out.push(s.clone()), + }, + Value::Array(items) => items.iter().for_each(|v| from_value(v, out)), + Value::Object(map) => map.values().for_each(|v| from_value(v, out)), + _ => {} + } + } + let mut out = Vec::new(); + for row in response.rows.iter().flatten() { + row.iter().for_each(|cell| from_value(cell, &mut out)); + } + out +} + +/// Send `op` with `fields` as request `seq` and return the response. Panics +/// unless the status is `Ok`. +pub(super) async fn walk( + stream: &mut TcpStream, + seq: u64, + op: OpCode, + fields: TextFields, +) -> NativeResponse { + let response = send_request(stream, seq, op, fields).await; + assert_eq!( + response.status, + ResponseStatus::Ok, + "native {op:?} must succeed: {response:?}" + ); + response +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_owner_reads_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_owner_reads_cross_node.rs new file mode 100644 index 000000000..282c70983 --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_owner_reads_cross_node.rs @@ -0,0 +1,274 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Graph reads that need several partitions return the whole graph's answer. +//! +//! A graph edge lives on the key vShard of each endpoint, so with a +//! replication factor of 1 a graph whose node keys cover every data group has +//! a share of its edges on every node. `GRAPH NEIGHBORS`, `SHOW GRAPH STATS` +//! and every `GRAPH ALGO` must answer from all of them. The reference is the +//! same graph on a single node. + +use std::collections::{BTreeMap, BTreeSet, HashMap}; +use std::time::Duration; + +use nodedb_types::id::VShardId; + +use crate::common::cluster_harness::{TestCluster, wait_for}; +use crate::common::pgwire_harness::TestServer; + +const COLL: &str = "gown_xnode"; +/// The smallest ring the graph uses, even when fewer names cover every group. +const MIN_NODES: usize = 24; +/// Algorithms whose answer is exact and deterministic on both sides. +const EXACT_ALGOS: [&str; 10] = [ + "LABEL_PROPAGATION", + "LCC", + "BETWEENNESS", + "CLOSENESS", + "HARMONIC", + "DEGREE", + "LOUVAIN", + "TRIANGLES", + "DIAMETER", + "KCORE", +]; +const PAGERANK_TOLERANCE: f64 = 1e-4; + +/// Node names `n0, n1, …`, enough that their key vShards cover every group. +fn node_names(groups: &BTreeSet, group_of: impl Fn(u32) -> u64) -> Vec { + let mut names = Vec::new(); + let mut covered = BTreeSet::new(); + let mut i = 0usize; + while names.len() < MIN_NODES || covered != *groups { + let name = format!("n{i}"); + covered.insert(group_of(VShardId::from_key(name.as_bytes()).as_u32())); + names.push(name); + i += 1; + } + names +} + +/// A ring over `names` plus a chord from every third node, so the graph has +/// cycles, triangles and nodes of different degree. +fn edges(names: &[String]) -> Vec<(String, String, &'static str)> { + let n = names.len(); + let mut out = Vec::new(); + for i in 0..n { + out.push((names[i].clone(), names[(i + 1) % n].clone(), "K")); + if i % 3 == 0 { + let j = (i * 7 + 3) % n; + if j != i { + out.push((names[i].clone(), names[j].clone(), "C")); + } + } + } + out +} + +async fn cluster_rows(client: &tokio_postgres::Client, sql: &str) -> Vec> { + let msgs = client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + let mut rows = Vec::new(); + for msg in msgs { + if let tokio_postgres::SimpleQueryMessage::Row(row) = msg { + rows.push( + (0..row.len()) + .map(|i| row.get(i).unwrap_or("").to_string()) + .collect(), + ); + } + } + rows +} + +fn sorted(mut rows: Vec>) -> Vec> { + rows.sort(); + rows +} + +/// `node → value` from a two-column algorithm result. +fn by_node(rows: &[Vec]) -> HashMap { + rows.iter() + .map(|row| (row[0].clone(), row.get(1).cloned().unwrap_or_default())) + .collect() +} + +/// The partition of nodes by component id, independent of the id values. +fn components(rows: &[Vec]) -> BTreeSet> { + let mut by_id: BTreeMap> = BTreeMap::new(); + for (node, id) in by_node(rows) { + by_id.entry(id).or_default().insert(node); + } + by_id.into_values().collect() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn graph_reads_spanning_every_group_match_a_single_node() { + let cluster = TestCluster::spawn_three_with_replication_factor(1) + .await + .expect("3-node RF1 cluster"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {COLL}")) + .await + .expect("CREATE COLLECTION"); + wait_for( + "all 3 nodes see the collection", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 1) + }, + ) + .await; + + let (groups, group_of_vshard): (BTreeSet, HashMap) = { + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + let groups: BTreeSet = routing + .group_ids() + .into_iter() + .filter(|g| *g != 0) + .collect(); + let mut map = HashMap::new(); + for &g in &groups { + for vs in routing.vshards_for_group(g) { + map.insert(vs, g); + } + } + (groups, map) + }; + // Placement convergence leaves one replica per group, so each node holds + // only the groups it replicates. + wait_for( + "each data group has exactly one replica", + Duration::from_secs(30), + Duration::from_millis(50), + || { + groups.iter().all(|&g| { + cluster + .nodes + .iter() + .filter(|node| node.replicates_data_group(g)) + .count() + == 1 + }) + }, + ) + .await; + + let names = node_names(&groups, |vs| group_of_vshard.get(&vs).copied().unwrap_or(0)); + let edges = edges(&names); + + let reference = TestServer::start().await; + reference + .exec(&format!("CREATE COLLECTION {COLL}")) + .await + .expect("reference CREATE COLLECTION"); + for (src, dst, label) in &edges { + let insert = + format!("GRAPH INSERT EDGE IN '{COLL}' FROM '{src}' TO '{dst}' TYPE '{label}'"); + reference.exec(&insert).await.expect("reference insert"); + cluster.nodes[0] + .client + .simple_query(&insert) + .await + .unwrap_or_else(|e| panic!("cluster insert {src}->{dst}: {e:?}")); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + + let stats_sql = format!("SHOW GRAPH STATS '{COLL}'"); + let expected_stats = sorted( + reference + .query_rows(&stats_sql) + .await + .expect("reference stats"), + ); + + for (idx, node) in cluster.nodes.iter().enumerate() { + let client = &node.client; + + assert_eq!( + sorted(cluster_rows(client, &stats_sql).await), + expected_stats, + "node {idx}: SHOW GRAPH STATS must count every partition" + ); + + for name in &names { + let sql = format!("GRAPH NEIGHBORS IN '{COLL}' OF '{name}' DIRECTION both"); + let expected = sorted( + reference + .query_rows(&sql) + .await + .expect("reference neighbors"), + ); + assert_eq!( + sorted(cluster_rows(client, &sql).await), + expected, + "node {idx}: GRAPH NEIGHBORS OF '{name}' must read the node's owner" + ); + } + + for algo in EXACT_ALGOS { + let sql = format!("GRAPH ALGO {algo} ON {COLL}"); + let expected = sorted(reference.query_rows(&sql).await.expect("reference algo")); + assert!(!expected.is_empty(), "reference {algo} must return rows"); + assert_eq!( + sorted(cluster_rows(client, &sql).await), + expected, + "node {idx}: GRAPH ALGO {algo} must equal the single-node answer" + ); + } + + let sssp = format!("GRAPH ALGO SSSP ON {COLL} SOURCE '{}'", names[0]); + let expected = sorted(reference.query_rows(&sssp).await.expect("reference sssp")); + assert_eq!( + sorted(cluster_rows(client, &sssp).await), + expected, + "node {idx}: GRAPH ALGO SSSP must equal the single-node answer" + ); + + let wcc = format!("GRAPH ALGO WCC ON {COLL}"); + let expected = reference.query_rows(&wcc).await.expect("reference wcc"); + assert_eq!( + components(&cluster_rows(client, &wcc).await), + components(&expected), + "node {idx}: GRAPH ALGO WCC must find the single-node components" + ); + + let pagerank = format!("GRAPH ALGO PAGERANK ON {COLL}"); + let expected = by_node( + &reference + .query_rows(&pagerank) + .await + .expect("reference pagerank"), + ); + let got = by_node(&cluster_rows(client, &pagerank).await); + assert_eq!( + got.keys().collect::>(), + expected.keys().collect::>(), + "node {idx}: GRAPH ALGO PAGERANK must rank every node" + ); + for (node_name, rank) in &expected { + let want: f64 = rank.parse().expect("reference rank"); + let have: f64 = got[node_name].parse().expect("cluster rank"); + assert!( + (want - have).abs() <= PAGERANK_TOLERANCE, + "node {idx}: PAGERANK of {node_name} is {have}, single node gives {want}" + ); + } + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_rag_fusion_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_rag_fusion_cross_node.rs new file mode 100644 index 000000000..8cc92d9cd --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/graph_rag_fusion_cross_node.rs @@ -0,0 +1,348 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! GraphRAG fusion in a cluster answers what a single node answers. +//! +//! With a replication factor of 1 the collection's vector and text indexes +//! sit on its owner, and graph edges spread over every data group, because +//! the node keys cover every group. The same collection on a single node is +//! the reference. Each surface is compared on every cluster node: +//! +//! - `GRAPH RAG FUSION` over pgwire, two-source and three-source (BM25); +//! - the same SQL over the native protocol; +//! - the native `GraphRagFusion` opcode (two-source). +//! +//! Results compare exactly: node, RRF score, vector rank and distance, hop +//! distance, and the metadata counts. Only the watermark differs by design. + +use std::collections::{BTreeSet, HashMap}; +use std::time::Duration; + +use nodedb_test_support::native_harness::{open_trust_session, send_request}; +use nodedb_types::id::VShardId; +use nodedb_types::protocol::{NativeResponse, OpCode, ResponseStatus, TextFields}; +use tokio::net::TcpStream; + +use crate::common::cluster_harness::{TestCluster, wait_for}; +use crate::common::pgwire_harness::TestServer; + +const COLL: &str = "rag_xnode"; +const MIN_NODES: usize = 24; +const SUPERUSER: &str = "nodedb"; +const QUERY: [f64; 3] = [1.0, 0.0, 0.0]; + +/// Names `d0, d1, …`, enough that their key vShards cover every data group. +fn node_names(groups: &BTreeSet, group_of: &HashMap) -> Vec { + let mut out = Vec::new(); + let mut covered = BTreeSet::new(); + let mut i = 0usize; + while out.len() < MIN_NODES || covered != *groups { + let name = format!("d{i}"); + let vshard = VShardId::from_key(name.as_bytes()).as_u32(); + covered.insert(group_of.get(&vshard).copied().unwrap_or(0)); + out.push(name); + i += 1; + } + out +} + +/// Every statement that builds the collection: indexes, one document per +/// name with a distinct embedding, and a `hop` ring plus `skip` chords. +fn setup_statements(names: &[String]) -> Vec { + let mut out = vec![ + format!("CREATE COLLECTION {COLL}"), + format!("CREATE VECTOR INDEX idx_{COLL}_emb ON {COLL} METRIC cosine DIM 3"), + format!("CREATE SEARCH INDEX idx_{COLL}_fts ON {COLL} FIELDS body ANALYZER 'standard'"), + ]; + let n = names.len(); + for (i, name) in names.iter().enumerate() { + let angle = i as f64 * 0.11; + let body = if i % 3 == 0 { + "alpha omega" + } else { + "beta gamma" + }; + out.push(format!( + "INSERT INTO {COLL} (id, body, embedding) VALUES ('{name}', '{body}', \ + ARRAY[{:.6}, {:.6}, {:.6}])", + angle.cos(), + angle.sin(), + 0.01 * i as f64 + )); + } + for i in 0..n { + out.push(format!( + "GRAPH INSERT EDGE IN '{COLL}' FROM '{}' TO '{}' TYPE 'hop'", + names[i], + names[(i + 1) % n] + )); + if i % 4 == 0 { + out.push(format!( + "GRAPH INSERT EDGE IN '{COLL}' FROM '{}' TO '{}' TYPE 'skip'", + names[i], + names[(i + 5) % n] + )); + } + } + out +} + +fn query_array() -> String { + format!("ARRAY[{:.1}, {:.1}, {:.1}]", QUERY[0], QUERY[1], QUERY[2]) +} + +/// A fusion whose walk hits its visit cap: the admitted nodes must match. +const CAPPED: usize = 3; + +/// The SQL fusions compared: two-source over one label and over every label, +/// three-source with BM25, and a two-source walk cut by `MAX_VISITED` +/// (entry [`CAPPED`]). +fn fusion_sql() -> Vec { + let q = query_array(); + vec![ + format!( + "GRAPH RAG FUSION ON {COLL} QUERY {q} VECTOR_FIELD 'embedding' VECTOR_TOP_K 3 \ + EXPANSION_DEPTH 2 EDGE_LABEL 'hop' FINAL_TOP_K 10 RRF_K (60.0, 10.0)" + ), + format!( + "GRAPH RAG FUSION ON {COLL} QUERY {q} VECTOR_FIELD 'embedding' VECTOR_TOP_K 4 \ + EXPANSION_DEPTH 3 FINAL_TOP_K 20 RRF_K (40.0, 20.0)" + ), + format!( + "GRAPH RAG FUSION ON {COLL} QUERY {q} VECTOR_FIELD 'embedding' VECTOR_TOP_K 3 \ + BM25 'alpha' ON 'body' EXPANSION_DEPTH 2 EDGE_LABEL 'hop' FINAL_TOP_K 15 \ + RRF_K (60.0, 35.0, 50.0)" + ), + format!( + "GRAPH RAG FUSION ON {COLL} QUERY {q} VECTOR_FIELD 'embedding' VECTOR_TOP_K 3 \ + EXPANSION_DEPTH 4 MAX_VISITED 7 FINAL_TOP_K 30 RRF_K (60.0, 10.0)" + ), + ] +} + +/// A fusion answer with every `watermark_lsn` removed and every JSON text +/// cell parsed, so two answers compare by content. +fn normalize(value: serde_json::Value) -> serde_json::Value { + match value { + serde_json::Value::String(s) => match sonic_rs::from_str::(&s) { + Ok(parsed @ (serde_json::Value::Object(_) | serde_json::Value::Array(_))) => { + normalize(parsed) + } + _ => serde_json::Value::String(s), + }, + serde_json::Value::Array(items) => { + serde_json::Value::Array(items.into_iter().map(normalize).collect()) + } + serde_json::Value::Object(map) => serde_json::Value::Object( + map.into_iter() + .filter(|(key, _)| key != "watermark_lsn") + .map(|(key, v)| (key, normalize(v))) + .collect(), + ), + other => other, + } +} + +/// The single `result` cell of a pgwire fusion answer. +async fn pgwire_fusion(client: &tokio_postgres::Client, sql: &str) -> serde_json::Value { + let msgs = client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + let cell = msgs + .iter() + .find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .unwrap_or_else(|| panic!("{sql}: no result row")); + normalize(serde_json::Value::String(cell)) +} + +fn native_rows(response: &NativeResponse) -> serde_json::Value { + let rows: Vec = response + .rows + .iter() + .flatten() + .map(|row| { + serde_json::Value::Array(row.iter().cloned().map(serde_json::Value::from).collect()) + }) + .collect(); + normalize(serde_json::Value::Array(rows)) +} + +/// A native session authenticated as the trust superuser, in JSON framing. +async fn native_session(port: u16) -> TcpStream { + open_trust_session(port, SUPERUSER).await +} + +async fn native_call( + stream: &mut TcpStream, + seq: u64, + op: OpCode, + fields: TextFields, +) -> serde_json::Value { + let response = send_request(stream, seq, op, fields).await; + assert_eq!( + response.status, + ResponseStatus::Ok, + "native {op:?} must succeed: {response:?}" + ); + native_rows(&response) +} + +fn opcode_fields() -> TextFields { + TextFields { + collection: Some(COLL.to_string()), + query_vector: Some(QUERY.iter().map(|v| *v as f32).collect()), + vector_top_k: Some(3), + edge_label: Some("hop".to_string()), + expansion_depth: Some(2), + final_top_k: Some(10), + vector_k: Some(60.0), + graph_k: Some(10.0), + vector_field: Some("embedding".to_string()), + ..Default::default() + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn rag_fusion_across_shards_matches_a_single_node() { + let cluster = TestCluster::spawn_three_with_replication_factor(1) + .await + .expect("3-node RF1 cluster"); + let (groups, group_of): (BTreeSet, HashMap) = { + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + let groups: BTreeSet = routing + .group_ids() + .into_iter() + .filter(|g| *g != 0) + .collect(); + let mut map = HashMap::new(); + for &g in &groups { + for vs in routing.vshards_for_group(g) { + map.insert(vs, g); + } + } + (groups, map) + }; + wait_for( + "each data group has exactly one replica", + Duration::from_secs(30), + Duration::from_millis(50), + || { + groups.iter().all(|&g| { + cluster + .nodes + .iter() + .filter(|node| node.replicates_data_group(g)) + .count() + == 1 + }) + }, + ) + .await; + + let names = node_names(&groups, &group_of); + let statements = setup_statements(&names); + let reference = TestServer::start().await; + for statement in &statements { + reference + .exec(statement) + .await + .unwrap_or_else(|e| panic!("reference {statement}: {e}")); + } + let (ddl, writes) = statements.split_at(3); + for statement in ddl { + cluster + .exec_ddl_on_any_leader(statement) + .await + .unwrap_or_else(|e| panic!("cluster {statement}: {e}")); + } + wait_for( + "all 3 nodes see the collection", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 1) + }, + ) + .await; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + // The writes rotate over the nodes: a key's surrogate comes from its + // collection home whichever node writes it. + for (i, statement) in writes.iter().enumerate() { + cluster.nodes[i % cluster.nodes.len()] + .client + .simple_query(statement) + .await + .unwrap_or_else(|e| panic!("cluster {statement}: {e}")); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + + let sql = fusion_sql(); + let mut expected_sql = Vec::new(); + for statement in &sql { + let answer = pgwire_fusion(&reference.client, statement).await; + assert!( + answer["results"].as_array().is_some_and(|r| !r.is_empty()), + "the reference fusion returns results: {answer}" + ); + expected_sql.push(answer); + } + assert_eq!( + expected_sql[CAPPED]["metadata"]["truncated"], + serde_json::Value::Bool(true), + "the capped reference fusion hits its visit cap: {}", + expected_sql[CAPPED] + ); + let mut single = native_session(reference.native_port).await; + let expected_opcode = + native_call(&mut single, 2, OpCode::GraphRagFusion, opcode_fields()).await; + + for (idx, node) in cluster.nodes.iter().enumerate() { + for (statement, expected) in sql.iter().zip(&expected_sql) { + assert_eq!( + &pgwire_fusion(&node.client, statement).await, + expected, + "node {idx}: pgwire `{statement}` must equal the single-node fusion" + ); + let native = node + .native_client() + .query(statement) + .await + .unwrap_or_else(|e| panic!("node {idx}: native `{statement}`: {e}")); + let cell = native + .rows + .first() + .and_then(|row| row.first()) + .cloned() + .unwrap_or_else(|| panic!("node {idx}: native `{statement}` returned no row")); + assert_eq!( + &normalize(serde_json::Value::from(cell)), + expected, + "node {idx}: native SQL `{statement}` must equal the single-node fusion" + ); + } + let mut clustered = native_session(node.native_port).await; + assert_eq!( + native_call(&mut clustered, 2, OpCode::GraphRagFusion, opcode_fields()).await, + expected_opcode, + "node {idx}: the native fusion opcode must equal the single-node answer" + ); + } + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/native_remote_home_surrogate_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/native_remote_home_surrogate_cross_node.rs new file mode 100644 index 000000000..c648aa48e --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/native_remote_home_surrogate_cross_node.rs @@ -0,0 +1,246 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Native writes issued on a node outside a collection's home group plan with +//! the surrogate the home binds. +//! +//! With a replication factor of 1, each data group lives on one node. Each +//! write below goes over the native protocol to a node that does not replicate +//! the collection's home group: +//! +//! - a document batch insert (`DocumentBatchInsert`) +//! - a KV put (`PointPut` on a KV collection) +//! - a columnar ingest (`ColumnarInsert`) +//! - an edge insert (`EdgePut`), whose endpoints are keys of the edge's +//! collection +//! +//! For every key the home binds one surrogate, and the issuing node keeps that +//! same winner in its own catalog. + +use std::time::Duration; + +use nodedb_test_support::native_harness::{open_trust_session, send_request}; +use nodedb_types::protocol::{BatchDocument, OpCode, ResponseStatus, TextFields}; +use nodedb_types::{CollectionKey, DatabaseId, TenantId}; +use tokio::net::TcpStream; + +use crate::common::cluster_harness::{TestCluster, TestClusterNode, wait_for}; + +const DOCS: &str = "nrh_docs"; +const KV: &str = "nrh_kv"; +const COLS: &str = "nrh_cols"; +const EDGES: &str = "nrh_edges"; +/// The trust superuser the harness bootstraps. +const SUPERUSER: &str = "nodedb"; +/// The tenant the trust superuser writes as. +const TENANT: TenantId = TenantId::new(1); + +/// A native session on `port`, authenticated as the trust superuser. +async fn native_session(port: u16) -> TcpStream { + open_trust_session(port, SUPERUSER).await +} + +/// The collection's home node and a node outside its home group, once +/// exactly one node replicates that group. +async fn home_and_remote(cluster: &TestCluster, collection: &str) -> (usize, usize) { + let group_id = cluster.nodes[0] + .group_id_for_collection(collection) + .unwrap_or_else(|| panic!("the data group of '{collection}'")); + wait_for( + &format!("exactly one node replicates the group of '{collection}'"), + Duration::from_secs(30), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .filter(|node| node.replicates_data_group(group_id)) + .count() + == 1 + }, + ) + .await; + let home = cluster + .nodes + .iter() + .position(|node| node.replicates_data_group(group_id)) + .unwrap_or_else(|| panic!("the home node of '{collection}'")); + let remote = cluster + .nodes + .iter() + .position(|node| !node.replicates_data_group(group_id)) + .unwrap_or_else(|| panic!("a node outside the home group of '{collection}'")); + (home, remote) +} + +async fn write(stream: &mut TcpStream, seq: u64, op: OpCode, fields: TextFields) { + let response = send_request(stream, seq, op, fields).await; + assert_eq!( + response.status, + ResponseStatus::Ok, + "native {op:?} on a remote node must succeed: {response:?}" + ); +} + +/// The surrogate `node`'s catalog binds `pk` to in `collection`. +fn bound(node: &TestClusterNode, collection: &str, pk: &str) -> Option { + node.shared + .surrogate_assigner + .lookup_bound( + CollectionKey::from_bare(DatabaseId::DEFAULT, collection), + TENANT, + pk.as_bytes(), + ) + .unwrap_or_else(|e| panic!("lookup of '{pk}' in '{collection}': {e}")) + .map(|surrogate| surrogate.as_u32()) +} + +/// Every key in `pks` is bound at the home, to a surrogate the issuing node +/// also holds, and distinct keys hold distinct surrogates. +fn assert_home_winners( + cluster: &TestCluster, + collection: &str, + pks: &[&str], + home: usize, + remote: usize, +) { + let mut seen = std::collections::HashSet::new(); + for pk in pks { + let at_home = bound(&cluster.nodes[home], collection, pk).unwrap_or_else(|| { + panic!("the home of '{collection}' must bind '{pk}' written on a remote node") + }); + assert_ne!( + at_home, 0, + "'{pk}' in '{collection}' must not bind the zero surrogate" + ); + assert_eq!( + bound(&cluster.nodes[remote], collection, pk), + Some(at_home), + "the issuing node must keep the home's winner for '{pk}' in '{collection}'" + ); + assert!( + seen.insert(at_home), + "'{pk}' in '{collection}' shares surrogate {at_home} with another key" + ); + } +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn native_writes_on_a_remote_node_plan_with_the_home_surrogate() { + let cluster = TestCluster::spawn_three_with_replication_factor(1) + .await + .expect("3-node RF1 cluster"); + let ddl = [ + format!("CREATE COLLECTION {DOCS}"), + format!("CREATE COLLECTION {KV} (key TEXT PRIMARY KEY, n INT) WITH (engine='kv')"), + format!("CREATE COLLECTION {COLS} COLUMNS (id TEXT, v BIGINT) WITH (engine='columnar')"), + format!("CREATE COLLECTION {EDGES}"), + ]; + for statement in &ddl { + cluster + .exec_ddl_on_any_leader(statement) + .await + .unwrap_or_else(|e| panic!("{statement}: {e}")); + } + wait_for( + "all 3 nodes see the collections", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 4) + }, + ) + .await; + + // Document batch insert. + let (home, remote) = home_and_remote(&cluster, DOCS).await; + let doc_ids = ["d0", "d1", "d2", "d3"]; + let mut session = native_session(cluster.nodes[remote].native_port).await; + write( + &mut session, + 2, + OpCode::DocumentBatchInsert, + TextFields { + collection: Some(DOCS.to_string()), + documents: Some( + doc_ids + .iter() + .map(|id| BatchDocument { + id: (*id).to_string(), + fields: serde_json::json!({ "name": id }), + }) + .collect(), + ), + ..Default::default() + }, + ) + .await; + assert_home_winners(&cluster, DOCS, &doc_ids, home, remote); + + // KV put. + let (home, remote) = home_and_remote(&cluster, KV).await; + let mut session = native_session(cluster.nodes[remote].native_port).await; + write( + &mut session, + 2, + OpCode::PointPut, + TextFields { + collection: Some(KV.to_string()), + document_id: Some("k0".to_string()), + data: Some( + nodedb_types::json_to_msgpack(&serde_json::json!({ "n": 0 })).expect("KV value"), + ), + ..Default::default() + }, + ) + .await; + assert_home_winners(&cluster, KV, &["k0"], home, remote); + + // Columnar ingest. + let (home, remote) = home_and_remote(&cluster, COLS).await; + let col_ids = ["c0", "c1", "c2"]; + let payload = sonic_rs::to_vec( + &col_ids + .iter() + .enumerate() + .map(|(i, id)| serde_json::json!({ "id": id, "v": i })) + .collect::>(), + ) + .expect("columnar payload"); + let mut session = native_session(cluster.nodes[remote].native_port).await; + write( + &mut session, + 2, + OpCode::ColumnarInsert, + TextFields { + collection: Some(COLS.to_string()), + payload: Some(payload), + format: Some("json".to_string()), + ..Default::default() + }, + ) + .await; + assert_home_winners(&cluster, COLS, &col_ids, home, remote); + + // Edge insert: both endpoints are keys of the edge's collection. + let (home, remote) = home_and_remote(&cluster, EDGES).await; + let mut session = native_session(cluster.nodes[remote].native_port).await; + write( + &mut session, + 2, + OpCode::EdgePut, + TextFields { + collection: Some(EDGES.to_string()), + from_node: Some("e_src".to_string()), + to_node: Some("e_dst".to_string()), + edge_type: Some("rel".to_string()), + ..Default::default() + }, + ) + .await; + assert_home_winners(&cluster, EDGES, &["e_src", "e_dst"], home, remote); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/schema_objects.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/schema_objects.rs index 11762a5a4..0af8b7163 100644 --- a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/schema_objects.rs +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/schema_objects.rs @@ -145,7 +145,7 @@ async fn grant_permission_visible_on_every_node() { .await .expect("grant read"); - let target = "collection:1:documents"; + let target = "collection:0:1:documents"; wait_for( "all 3 nodes see the grant", Duration::from_secs(10), diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/surrogate_identity_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/surrogate_identity_cross_node.rs new file mode 100644 index 000000000..e906e5311 --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/surrogate_identity_cross_node.rs @@ -0,0 +1,369 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A key has one surrogate cluster-wide, whichever node writes it. +//! +//! With a replication factor of 1, a document lives on its collection's home +//! vShard and a graph edge on its endpoints' key vShards, often on three +//! different nodes. Each document here is written through one node, and edges +//! to the same keys through the two others. Every node that binds a key must +//! bind it to the one surrogate the collection home minted, so a RAG fusion, +//! which joins vector hits to graph nodes by surrogate, answers what a single +//! node answers. + +use std::collections::{BTreeSet, HashMap}; +use std::time::Duration; + +use nodedb_types::id::VShardId; +use nodedb_types::{CollectionKey, DatabaseId, TenantId}; + +use crate::common::cluster_harness::{TestCluster, wait_for}; +use crate::common::pgwire_harness::TestServer; + +const COLL: &str = "sid_xnode"; +const MIN_NODES: usize = 24; + +/// Names `k0, k1, …`, enough that their key vShards cover every data group. +fn node_names(groups: &BTreeSet, group_of: &HashMap) -> Vec { + let mut out = Vec::new(); + let mut covered = BTreeSet::new(); + let mut i = 0usize; + while out.len() < MIN_NODES || covered != *groups { + let name = format!("k{i}"); + let vshard = VShardId::from_key(name.as_bytes()).as_u32(); + covered.insert(group_of.get(&vshard).copied().unwrap_or(0)); + out.push(name); + i += 1; + } + out +} + +fn doc_insert(i: usize, name: &str) -> String { + let angle = i as f64 * 0.13; + format!( + "INSERT INTO {COLL} (id, embedding) VALUES ('{name}', ARRAY[{:.6}, {:.6}, 0.0])", + angle.cos(), + angle.sin() + ) +} + +fn edge_insert(src: &str, dst: &str) -> String { + format!("GRAPH INSERT EDGE IN '{COLL}' FROM '{src}' TO '{dst}' TYPE 'rel'") +} + +/// The surrogate each node's catalog binds `pk` to, for the nodes that bind it. +fn bindings(cluster: &TestCluster, pk: &str) -> Vec<(usize, u32)> { + cluster + .nodes + .iter() + .enumerate() + .filter_map(|(idx, node)| { + node.shared + .credentials + .catalog() + .get_surrogate_for_pk( + CollectionKey::from_bare(DatabaseId::DEFAULT, COLL), + TenantId::new(1), + pk.as_bytes(), + ) + .ok() + .flatten() + .map(|s| (idx, s.as_u32())) + }) + .collect() +} + +async fn fusion_result(client: &tokio_postgres::Client, sql: &str) -> serde_json::Value { + let msgs = client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + let cell = msgs + .iter() + .find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(row) => row.get(0).map(str::to_owned), + _ => None, + }) + .unwrap_or_else(|| panic!("{sql}: no result row")); + let mut value: serde_json::Value = + sonic_rs::from_str(&cell).unwrap_or_else(|e| panic!("{sql}: result is not JSON: {e}")); + if let Some(metadata) = value + .get_mut("metadata") + .and_then(serde_json::Value::as_object_mut) + { + metadata.remove("watermark_lsn"); + } + value +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn documents_and_edges_through_different_nodes_share_one_surrogate() { + let cluster = TestCluster::spawn_three_with_replication_factor(1) + .await + .expect("3-node RF1 cluster"); + let ddl = [ + format!("CREATE COLLECTION {COLL}"), + format!("CREATE VECTOR INDEX idx_{COLL}_emb ON {COLL} METRIC cosine DIM 3"), + ]; + for statement in &ddl { + cluster + .exec_ddl_on_any_leader(statement) + .await + .unwrap_or_else(|e| panic!("{statement}: {e}")); + } + wait_for( + "all 3 nodes see the collection", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 1) + }, + ) + .await; + let (groups, group_of): (BTreeSet, HashMap) = { + let routing = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()); + let groups: BTreeSet = routing + .group_ids() + .into_iter() + .filter(|g| *g != 0) + .collect(); + let mut map = HashMap::new(); + for &g in &groups { + for vs in routing.vshards_for_group(g) { + map.insert(vs, g); + } + } + (groups, map) + }; + wait_for( + "each data group has exactly one replica", + Duration::from_secs(30), + Duration::from_millis(50), + || { + groups.iter().all(|&g| { + cluster + .nodes + .iter() + .filter(|node| node.replicates_data_group(g)) + .count() + == 1 + }) + }, + ) + .await; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + + // Key i: its document through node i % 3, an edge out of it through node + // (i + 1) % 3, and an edge into it through node (i + 2) % 3. + let names = node_names(&groups, &group_of); + let n = names.len(); + let mut writes: Vec<(usize, String)> = Vec::new(); + for (i, name) in names.iter().enumerate() { + writes.push((i % 3, doc_insert(i, name))); + writes.push(((i + 1) % 3, edge_insert(name, &names[(i + 1) % n]))); + writes.push(((i + 2) % 3, edge_insert(&names[(i + n - 3) % n], name))); + } + + let reference = TestServer::start().await; + for statement in &ddl { + reference + .exec(statement) + .await + .unwrap_or_else(|e| panic!("reference {statement}: {e}")); + } + for (node, statement) in &writes { + reference + .exec(statement) + .await + .unwrap_or_else(|e| panic!("reference {statement}: {e}")); + cluster.nodes[*node] + .client + .simple_query(statement) + .await + .unwrap_or_else(|e| panic!("node {node}: {statement}: {e}")); + } + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + + for name in &names { + let bound = bindings(&cluster, name); + assert!( + !bound.is_empty(), + "some node binds '{name}': its collection home at least" + ); + let distinct: BTreeSet = bound.iter().map(|(_, s)| *s).collect(); + assert_eq!( + distinct.len(), + 1, + "every node binds '{name}' to one surrogate, got {bound:?}" + ); + } + + let sql = format!( + "GRAPH RAG FUSION ON {COLL} QUERY ARRAY[1.0, 0.0, 0.0] VECTOR_FIELD 'embedding' \ + VECTOR_TOP_K 4 EXPANSION_DEPTH 2 EDGE_LABEL 'rel' FINAL_TOP_K 20 RRF_K (60.0, 10.0)" + ); + let expected = fusion_result(&reference.client, &sql).await; + assert!( + expected["results"] + .as_array() + .is_some_and(|r| r.iter().any(|row| row["hop_distance"].is_number())), + "the reference fusion reaches graph nodes from its vector hits: {expected}" + ); + for (idx, node) in cluster.nodes.iter().enumerate() { + assert_eq!( + fusion_result(&node.client, &sql).await, + expected, + "node {idx}: the fusion must equal the single-node answer" + ); + } + + cluster.shutdown().await; +} + +const RACE_COLL: &str = "sid_race"; + +/// The index of the node leading `key`'s collection home vShard. +fn home_index(cluster: &TestCluster, key: CollectionKey<'_>) -> usize { + let leader = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()) + .leader_for_vshard(VShardId::from_collection(key).as_u32()) + .expect("the collection home has a leader"); + cluster + .nodes + .iter() + .position(|node| node.node_id == leader) + .expect("the home leader is a cluster node") +} + +/// Two coordinators that are not the key's home resolve a new key at the same +/// time. Both obtain the one value the home bound, and a third lookup, on a +/// coordinator that never resolved the key, returns it before any write +/// carrying the key applies. +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn racing_coordinators_obtain_the_homes_one_surrogate() { + let cluster = TestCluster::spawn_three_with_replication_factor(1) + .await + .expect("3-node RF1 cluster"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {RACE_COLL}")) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION {RACE_COLL}: {e}")); + wait_for( + "every data group has a leader", + Duration::from_secs(30), + Duration::from_millis(50), + || { + cluster.nodes[0] + .all_group_leaders() + .iter() + .all(|(_, leader)| *leader != 0) + }, + ) + .await; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(20)) + .await; + + let key = CollectionKey::from_bare(DatabaseId::DEFAULT, RACE_COLL); + let tenant = TenantId::new(1); + let home = home_index(&cluster, key); + let a = (home + 1) % 3; + let b = (home + 2) % 3; + + for round in 0..8 { + let pk = format!("race-{round}"); + let spawn_assign = |idx: usize| { + let shared = std::sync::Arc::clone(&cluster.nodes[idx].shared); + let pk = pk.clone(); + tokio::spawn(async move { + shared + .surrogate_assigner + .assign( + CollectionKey::from_bare(DatabaseId::DEFAULT, RACE_COLL), + tenant, + pk.as_bytes(), + ) + .await + }) + }; + let (on_a, on_b) = tokio::join!(spawn_assign(a), spawn_assign(b)); + let on_a = on_a + .expect("join a") + .unwrap_or_else(|e| panic!("assign on node {a}: {e}")); + let on_b = on_b + .expect("join b") + .unwrap_or_else(|e| panic!("assign on node {b}: {e}")); + assert_eq!( + on_a, on_b, + "both coordinators plan '{pk}' with one surrogate" + ); + let at_home = cluster.nodes[home] + .shared + .surrogate_assigner + .lookup_bound(key, tenant, pk.as_bytes()) + .expect("home catalog read"); + assert_eq!( + at_home, + Some(on_a), + "the coordinators plan with the home's bind" + ); + } + + // A key only node `a` resolved: node `b` looks it up before anything + // writes it, and gets the home's value. + let shared = std::sync::Arc::clone(&cluster.nodes[a].shared); + let assigned = tokio::spawn(async move { + shared + .surrogate_assigner + .assign( + CollectionKey::from_bare(DatabaseId::DEFAULT, RACE_COLL), + tenant, + b"lookup-key", + ) + .await + }) + .await + .expect("join") + .expect("assign on a"); + let shared = std::sync::Arc::clone(&cluster.nodes[b].shared); + let looked_up = tokio::spawn(async move { + shared + .surrogate_assigner + .lookup_many( + CollectionKey::from_bare(DatabaseId::DEFAULT, RACE_COLL), + tenant, + &[b"lookup-key".as_slice()], + ) + .await + }) + .await + .expect("join") + .expect("lookup on b") + .into_iter() + .next() + .flatten(); + assert_eq!( + looked_up, + Some(assigned), + "a lookup on another coordinator returns the home's bind before any apply" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/txn_resolve_on_follower_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/txn_resolve_on_follower_cross_node.rs new file mode 100644 index 000000000..7dec65c7a --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/txn_resolve_on_follower_cross_node.rs @@ -0,0 +1,187 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A `MERGE` or `UPDATE ... FROM` inside a transaction, issued on a node that +//! does not lead the target's group, resolves against the rows the same +//! transaction staged. +//! +//! The transaction's staging overlay lives on the target group's leader. The +//! resolve pass is a read (`DocumentOp::ResolveWrite`), so it is routed to the +//! leader with the transaction id, and folds the overlay in there. A resolve on +//! the follower's own replica misses the staged row: `MERGE` inserts +//! a duplicate, and `UPDATE ... FROM` leaves the staged row unchanged. + +use std::collections::BTreeMap; +use std::time::Duration; + +use nodedb_types::CollectionKey; +use nodedb_types::id::DatabaseId; + +use crate::common::cluster_harness::{TestCluster, wait_for}; + +const SOURCE: &str = "txr_source"; +const MERGE_TARGET: &str = "txr_merge_target"; +const UPDATE_TARGET: &str = "txr_update_target"; + +/// The index of a node that does not lead `collection`'s group. +fn non_leader_for(cluster: &TestCluster, collection: &str) -> usize { + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, collection) + .vshard() + .as_u32(); + let group = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(vshard) + .expect("group for the collection's vShard"); + let leader = cluster.nodes[0] + .all_group_leaders() + .into_iter() + .find(|(g, _)| *g == group) + .map(|(_, l)| l) + .expect("leader for the collection's group"); + cluster + .nodes + .iter() + .position(|n| n.node_id != leader) + .expect("a node that does not lead the collection's group") +} + +async fn exec(client: &tokio_postgres::Client, sql: &str) { + client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); +} + +/// `id → (sku, qty)` for every row of `collection`. +async fn rows_by_id( + client: &tokio_postgres::Client, + collection: &str, +) -> BTreeMap { + let sql = format!("SELECT id, sku, qty FROM {collection}"); + let msgs = client + .simple_query(&sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + msgs.iter() + .filter_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => Some(( + r.get("id").unwrap_or("").to_string(), + ( + r.get("sku").unwrap_or("").to_string(), + r.get("qty").unwrap_or("").to_string(), + ), + )), + _ => None, + }) + .collect() +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn in_transaction_resolve_on_a_follower_sees_the_staged_rows() { + let cluster = TestCluster::spawn_three().await.expect("3-node cluster"); + for collection in [SOURCE, MERGE_TARGET, UPDATE_TARGET] { + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {collection} TYPE document")) + .await + .unwrap_or_else(|e| panic!("CREATE COLLECTION {collection}: {e}")); + } + wait_for( + "all 3 nodes see the collections", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 3) + }, + ) + .await; + wait_for( + "all groups have a stable leader", + Duration::from_secs(15), + Duration::from_millis(100), + || { + cluster + .nodes + .iter() + .all(|n| n.all_group_leaders().iter().all(|(_, l)| *l != 0)) + }, + ) + .await; + + exec( + &cluster.nodes[0].client, + &format!("INSERT INTO {SOURCE} (id, sku, qty) VALUES ('s_a', 'a', 5), ('s_b', 'b', 7)"), + ) + .await; + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + // MERGE: the staged 'a' row must match, so only 'b' is inserted. + let coord = non_leader_for(&cluster, MERGE_TARGET); + let client = &cluster.nodes[coord].client; + exec(client, "BEGIN").await; + exec( + client, + &format!("INSERT INTO {MERGE_TARGET} (id, sku, qty) VALUES ('t_a', 'a', 1)"), + ) + .await; + exec( + client, + &format!( + "MERGE INTO {MERGE_TARGET} t USING {SOURCE} s ON t.sku = s.sku \ + WHEN MATCHED THEN UPDATE SET qty = s.qty \ + WHEN NOT MATCHED THEN INSERT (id, sku, qty) VALUES (s.id, s.sku, s.qty)" + ), + ) + .await; + exec(client, "COMMIT").await; + + // UPDATE ... FROM: the staged 'a' row must be updated. + let coord = non_leader_for(&cluster, UPDATE_TARGET); + let client = &cluster.nodes[coord].client; + exec(client, "BEGIN").await; + exec( + client, + &format!("INSERT INTO {UPDATE_TARGET} (id, sku, qty) VALUES ('u_a', 'a', 1)"), + ) + .await; + exec( + client, + &format!( + "UPDATE {UPDATE_TARGET} SET qty = s.qty FROM {SOURCE} s \ + WHERE {UPDATE_TARGET}.sku = s.sku" + ), + ) + .await; + exec(client, "COMMIT").await; + + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + let merged = rows_by_id(&cluster.nodes[0].client, MERGE_TARGET).await; + assert_eq!( + merged, + BTreeMap::from([ + ("t_a".to_string(), ("a".to_string(), "5".to_string())), + ("s_b".to_string(), ("b".to_string(), "7".to_string())), + ]), + "MERGE on a follower must match the row its transaction staged" + ); + + let updated = rows_by_id(&cluster.nodes[0].client, UPDATE_TARGET).await; + assert_eq!( + updated, + BTreeMap::from([("u_a".to_string(), ("a".to_string(), "5".to_string()))]), + "UPDATE ... FROM on a follower must update the row its transaction staged" + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/txn_stage_forward_cross_node.rs b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/txn_stage_forward_cross_node.rs new file mode 100644 index 000000000..59dd93236 --- /dev/null +++ b/nodedb-cluster-tests/tests/sql_cluster_cross_node_dml_tests/txn_stage_forward_cross_node.rs @@ -0,0 +1,195 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! An in-transaction write forwarded to a remote multi-core leader runs on the +//! one core that owns its vShard. +//! +//! A coordinator that does not lead a collection's group forwards each staged +//! write to the leader as a `StageWrite`. The leader runs 2 cores, and only +//! the core owning the collection's vShard holds its rows. Each collection +//! here homes to core 1, so a forward fanned to every core answers with core +//! 0's result: a bulk UPDATE matches no rows there, and `KV_INCR` counts from +//! an absent key. The forward must return the owning core's answer. + +use std::time::Duration; + +use nodedb_types::CollectionKey; +use nodedb_types::id::DatabaseId; + +use crate::common::cluster_harness::{TestCluster, wait_for}; + +const CORES_PER_NODE: usize = 2; +/// The core every test collection homes to on its leader. +const OWNING_CORE: u32 = 1; +const MATCHING_ROWS: usize = 5; + +/// The first `{prefix}_{i}` whose vShard homes to [`OWNING_CORE`]. +fn collection_on_owning_core(prefix: &str) -> String { + (0u32..) + .map(|i| format!("{prefix}_{i}")) + .find(|name| { + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, name).vshard(); + vshard.as_u32() % CORES_PER_NODE as u32 == OWNING_CORE + }) + .unwrap_or_default() +} + +/// The index of a node that does not lead `collection`'s group. +fn non_leader_for(cluster: &TestCluster, collection: &str) -> usize { + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, collection) + .vshard() + .as_u32(); + let group = cluster.nodes[0] + .shared + .cluster_routing + .as_ref() + .expect("cluster_routing") + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(vshard) + .expect("group for the collection's vShard"); + let leader = cluster.nodes[0] + .all_group_leaders() + .into_iter() + .find(|(g, _)| *g == group) + .map(|(_, l)| l) + .expect("leader for the collection's group"); + cluster + .nodes + .iter() + .position(|n| n.node_id != leader) + .expect("a node that does not lead the collection's group") +} + +/// The row count a statement's `CommandComplete` reports. +async fn affected(client: &tokio_postgres::Client, sql: &str) -> u64 { + client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")) + .iter() + .find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::CommandComplete(n) => Some(*n), + _ => None, + }) + .unwrap_or_else(|| panic!("{sql}: no CommandComplete")) +} + +/// The JSON text column a `SELECT KV_*(...)` returns. +async fn kv_result(client: &tokio_postgres::Client, sql: &str) -> serde_json::Value { + let msgs = client + .simple_query(sql) + .await + .unwrap_or_else(|e| panic!("{sql}: {e}")); + let text = msgs + .iter() + .find_map(|m| match m { + tokio_postgres::SimpleQueryMessage::Row(r) => r.get(0).map(str::to_string), + _ => None, + }) + .unwrap_or_else(|| panic!("{sql}: no row")); + serde_json::from_str(&text).unwrap_or_else(|e| panic!("{sql}: {text}: {e}")) +} + +#[tokio::test(flavor = "multi_thread", worker_threads = 6)] +async fn forwarded_staged_writes_answer_from_the_owning_core() { + let cluster = TestCluster::spawn_three_with_cores(CORES_PER_NODE) + .await + .expect("3-node 2-core cluster"); + + let docs = collection_on_owning_core("fwd_docs"); + let kv = collection_on_owning_core("fwd_kv"); + cluster + .exec_ddl_on_any_leader(&format!("CREATE COLLECTION {docs}")) + .await + .expect("CREATE COLLECTION docs"); + cluster + .exec_ddl_on_any_leader(&format!( + "CREATE COLLECTION {kv} (key TEXT PRIMARY KEY, n INT) WITH (engine='kv')" + )) + .await + .expect("CREATE COLLECTION kv"); + + wait_for( + "all 3 nodes see both collections", + Duration::from_secs(10), + Duration::from_millis(50), + || { + cluster + .nodes + .iter() + .all(|n| n.cached_collection_count() >= 2) + }, + ) + .await; + wait_for( + "all groups have a stable leader", + Duration::from_secs(15), + Duration::from_millis(100), + || { + cluster + .nodes + .iter() + .all(|n| n.all_group_leaders().iter().all(|(_, l)| *l != 0)) + }, + ) + .await; + + for i in 0..MATCHING_ROWS { + cluster.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {docs} (id, grp, n) VALUES ('a{i}', 'a', {i})" + )) + .await + .unwrap_or_else(|e| panic!("insert a{i}: {e}")); + } + for i in 0..2 { + cluster.nodes[0] + .client + .simple_query(&format!( + "INSERT INTO {docs} (id, grp, n) VALUES ('b{i}', 'b', {i})" + )) + .await + .unwrap_or_else(|e| panic!("insert b{i}: {e}")); + } + cluster.nodes[0] + .client + .simple_query(&format!("INSERT INTO {kv} (key, n) VALUES ('ctr', 5)")) + .await + .expect("insert ctr"); + cluster + .wait_for_full_apply_convergence(Duration::from_secs(15)) + .await; + + // Bulk UPDATE: the owning core matches every 'a' row. Core 0 matches none. + let coord = non_leader_for(&cluster, &docs); + let client = &cluster.nodes[coord].client; + client.simple_query("BEGIN").await.expect("BEGIN"); + let updated = affected(client, &format!("UPDATE {docs} SET n = 99 WHERE grp = 'a'")).await; + client.simple_query("ROLLBACK").await.expect("ROLLBACK"); + assert_eq!( + updated, MATCHING_ROWS as u64, + "node {}: a forwarded bulk UPDATE must report the owning core's count", + cluster.nodes[coord].node_id + ); + + // KV_INCR: the owning core counts from the stored 5. Core 0 holds no row. + let coord = non_leader_for(&cluster, &kv); + let client = &cluster.nodes[coord].client; + client.simple_query("BEGIN").await.expect("BEGIN"); + let incr = kv_result(client, &format!("SELECT KV_INCR('{kv}', 'ctr', 3)")).await; + let chained = kv_result(client, &format!("SELECT KV_INCR('{kv}', 'ctr', 2)")).await; + client.simple_query("ROLLBACK").await.expect("ROLLBACK"); + assert_eq!( + incr["value"], 8, + "node {}: a forwarded KV_INCR must count from the owning core's row", + cluster.nodes[coord].node_id + ); + assert_eq!( + chained["value"], 10, + "node {}: a second forwarded KV_INCR must chain off the first staged value", + cluster.nodes[coord].node_id + ); + + cluster.shutdown().await; +} diff --git a/nodedb-cluster/Cargo.toml b/nodedb-cluster/Cargo.toml index c3badf550..a3bf676cd 100644 --- a/nodedb-cluster/Cargo.toml +++ b/nodedb-cluster/Cargo.toml @@ -53,6 +53,7 @@ uuid = { workspace = true } # Authenticated frame envelope (origin + anti-replay MAC) for Raft RPCs. hmac = { workspace = true } sha2 = { workspace = true } +hex = { workspace = true } faultbox = { workspace = true, optional = true } diff --git a/nodedb-cluster/src/applied_watcher/watcher.rs b/nodedb-cluster/src/applied_watcher/watcher.rs index 9da1fff63..080c769fa 100644 --- a/nodedb-cluster/src/applied_watcher/watcher.rs +++ b/nodedb-cluster/src/applied_watcher/watcher.rs @@ -31,7 +31,7 @@ impl WaitOutcome { } /// Tracks the highest Raft log index applied on this node for one -/// Raft group. +/// Raft group, and the last index of the batch being applied now. #[derive(Debug, Default)] pub struct AppliedIndexWatcher { state: Mutex, @@ -41,6 +41,10 @@ pub struct AppliedIndexWatcher { #[derive(Debug, Default)] struct State { applied: u64, + /// The last index of the batch the applier is applying now. A batch + /// makes its effects visible entry by entry before `applied` reaches + /// its end. + applying_through: u64, closed: bool, } @@ -67,6 +71,26 @@ impl AppliedIndexWatcher { self.state.lock().unwrap_or_else(|p| p.into_inner()).applied } + /// Declare that the applier starts a batch ending at `last_index`. The + /// applier calls it before the batch mutates any state a reader can see. + /// Idempotent: smaller indices are ignored. + pub fn begin_batch(&self, last_index: u64) { + let mut guard = self.state.lock().unwrap_or_else(|p| p.into_inner()); + guard.applying_through = guard.applying_through.max(last_index); + } + + /// The floor a reader stamps on work planned against what it sees now: + /// `max(applied, applying-through)`. + /// + /// A reader that sees an effect of the batch in progress gets a floor at + /// or above the index of the entry that made it visible. A waiter for + /// that floor waits on [`Self::wait_for`], which the finished batch + /// satisfies. + pub fn floor(&self) -> u64 { + let guard = self.state.lock().unwrap_or_else(|p| p.into_inner()); + guard.applied.max(guard.applying_through) + } + /// True once [`Self::close`] has been called. pub fn is_closed(&self) -> bool { self.state.lock().unwrap_or_else(|p| p.into_inner()).closed @@ -173,4 +197,20 @@ mod tests { w.close(); assert!(w.is_closed()); } + + #[test] + fn the_floor_covers_the_batch_in_progress() { + let w = AppliedIndexWatcher::new(); + w.bump(4); + assert_eq!(w.floor(), 4); + w.begin_batch(9); + assert_eq!(w.current(), 4, "the applied watermark waits for the batch"); + assert_eq!(w.floor(), 9); + w.begin_batch(7); + assert_eq!(w.floor(), 9, "a smaller batch end is ignored"); + w.bump(9); + assert_eq!(w.floor(), 9); + w.bump(12); + assert_eq!(w.floor(), 12); + } } diff --git a/nodedb-cluster/src/auth/join_token.rs b/nodedb-cluster/src/auth/join_token.rs index 90551a160..987709df7 100644 --- a/nodedb-cluster/src/auth/join_token.rs +++ b/nodedb-cluster/src/auth/join_token.rs @@ -105,12 +105,7 @@ pub fn issue_token( /// Encode raw token bytes as a lowercase hex string. pub fn token_to_hex(bytes: &[u8]) -> String { - use std::fmt::Write as _; - let mut out = String::with_capacity(bytes.len() * 2); - for b in bytes { - let _ = write!(out, "{b:02x}"); - } - out + hex::encode(bytes) } /// Compute SHA-256 of the token bytes. Used as the stable identity for @@ -218,25 +213,7 @@ fn parse_layout(bytes: &[u8]) -> Result { } fn hex_decode(s: &str) -> Result, TokenError> { - let mut out = Vec::with_capacity(s.len() / 2); - for chunk in s.as_bytes().chunks(2) { - if chunk.len() != 2 { - return Err(TokenError::InvalidHex); - } - let hi = hex_digit(chunk[0]).ok_or(TokenError::InvalidHex)?; - let lo = hex_digit(chunk[1]).ok_or(TokenError::InvalidHex)?; - out.push((hi << 4) | lo); - } - Ok(out) -} - -fn hex_digit(b: u8) -> Option { - match b { - b'0'..=b'9' => Some(b - b'0'), - b'a'..=b'f' => Some(10 + b - b'a'), - b'A'..=b'F' => Some(10 + b - b'A'), - _ => None, - } + hex::decode(s).map_err(|_| TokenError::InvalidHex) } #[cfg(test)] diff --git a/nodedb-cluster/src/bootstrap/bootstrap_fn.rs b/nodedb-cluster/src/bootstrap/bootstrap_fn.rs index 7f3666a9d..ac3364bb9 100644 --- a/nodedb-cluster/src/bootstrap/bootstrap_fn.rs +++ b/nodedb-cluster/src/bootstrap/bootstrap_fn.rs @@ -36,7 +36,8 @@ pub(super) fn bootstrap( let mut topology = ClusterTopology::new(); topology.add_node( NodeInfo::new(config.node_id, config.listen_addr, NodeState::Active) - .with_spki_pin(local_spki_pin), + .with_spki_pin(local_spki_pin) + .with_swim_addr(config.swim_udp_addr), ); // Create routing table: all groups on this single node. The configured @@ -138,6 +139,7 @@ mod tests { install_snapshot_chunk_bytes: 4 * 1024 * 1024, orphan_partial_max_age_secs: 300, log_compaction_threshold: None, + wire_build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), }; let state = bootstrap(&config, &catalog, None).unwrap(); @@ -182,6 +184,7 @@ mod tests { install_snapshot_chunk_bytes: 4 * 1024 * 1024, orphan_partial_max_age_secs: 300, log_compaction_threshold: None, + wire_build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), }; let state = bootstrap(&config, &catalog, None).unwrap(); diff --git a/nodedb-cluster/src/bootstrap/config.rs b/nodedb-cluster/src/bootstrap/config.rs index 3accc30ec..6e822cab6 100644 --- a/nodedb-cluster/src/bootstrap/config.rs +++ b/nodedb-cluster/src/bootstrap/config.rs @@ -88,12 +88,10 @@ pub struct ClusterConfig { /// (`8` attempts, `32 s` ceiling). Tests override this with a /// faster policy. pub join_retry: JoinRetryPolicy, - /// Optional UDP bind address for the SWIM failure detector. `None` - /// disables SWIM entirely — cluster startup then relies solely on - /// the existing raft transport for membership observations. When - /// `Some`, the operator is expected to spawn SWIM separately via - /// [`crate::spawn_swim`] after the cluster is up and feed the - /// seed list from `seed_nodes`. + /// Bound UDP address of this node's SWIM failure detector, as returned by + /// [`crate::bind_swim_listener`]. Bootstrap, join, and restart advertise + /// it in this node's topology entry, which peers seed SWIM from. `None` + /// for a node that runs no SWIM detector. pub swim_udp_addr: Option, /// Raft election timeout range. Controls how long a follower waits /// before starting an election after losing contact with the leader. @@ -121,6 +119,11 @@ pub struct ClusterConfig { /// naturally require an `InstallSnapshot`. The trigger is gated on /// the data-plane applied watermark, never raft's commit index. pub log_compaction_threshold: Option, + /// Build identity this node advertises in its `JoinRequest`. Production + /// callers set this to `nodedb_types::wire_version::WIRE_BUILD_ID`; + /// never read from the environment. Tests override it to a different + /// string to exercise `handle_join_request`'s build-id rejection path. + pub wire_build_id: String, } /// Result of cluster startup — everything needed to run the Raft loop. diff --git a/nodedb-cluster/src/bootstrap/handle_join.rs b/nodedb-cluster/src/bootstrap/handle_join.rs index adb367a6a..4fb531625 100644 --- a/nodedb-cluster/src/bootstrap/handle_join.rs +++ b/nodedb-cluster/src/bootstrap/handle_join.rs @@ -11,11 +11,13 @@ //! Semantics: //! //! - **New node**: added to topology as `Active`, full wire response returned. -//! - **Known node, same address**: idempotent — no mutation, full wire response returned. +//! - **Known node, same address**: idempotent. The entry changes only to +//! normalize its state to `Active` or to adopt a newly advertised SWIM address. //! - **Known node, different address**: rejected with `success: false`. This //! catches node-id reuse (operator error or a ghost node coming back with a //! stale id on a new address). -//! - **Invalid `listen_addr` in the request**: rejected with `success: false`. +//! - **Invalid `listen_addr` or `swim_addr` in the request**: rejected with +//! `success: false`. use std::net::SocketAddr; @@ -57,11 +59,31 @@ pub fn handle_join_request( ); return reject(format!( "joiner wire_version {} does not match this cluster's wire_version {} — \ - rolling upgrade is required before this node can join", + all nodes must run one build before 1.0; restart every node on the same build", req.wire_version, CLUSTER_WIRE_FORMAT_VERSION )); } + // Wire shapes can change without a `CLUSTER_WIRE_FORMAT_VERSION` bump (see + // `nodedb_types::wire_version`), so the version check above cannot by + // itself prevent two different builds from joining the same cluster and + // misdecoding each other. Build identity is the invariant that actually + // protects this — require an exact match. + let local_build_id = nodedb_types::wire_version::WIRE_BUILD_ID; + if req.build_id != local_build_id { + warn!( + node_id = req.node_id, + joiner_build_id = %req.build_id, + local_build_id, + "join request rejected: joiner build_id mismatch" + ); + return reject(format!( + "joiner build {} does not match this cluster's build {local_build_id} — \ + all nodes must run one build before 1.0; restart every node on the same build", + req.build_id + )); + } + // Validate the listen address early. let addr: SocketAddr = match req.listen_addr.parse() { Ok(a) => a, @@ -70,6 +92,14 @@ pub fn handle_join_request( } }; + let swim_addr: Option = match req.swim_addr.as_deref() { + Some(raw) => match raw.parse() { + Ok(a) => Some(a), + Err(e) => return reject(format!("invalid swim_addr '{raw}': {e}")), + }, + None => None, + }; + let spki_pin: Option<[u8; 32]> = match req.spki_pin.as_deref() { Some(bytes) if bytes.len() == 32 => { let mut pin = [0u8; 32]; @@ -96,12 +126,20 @@ pub fn handle_join_request( req.node_id )); } - // Same id, same address, same identity. If already Active we - // short-circuit; otherwise normalize it to Active. - if existing.state != NodeState::Active - && let Some(entry) = topology.get_node_mut(req.node_id) + // Same id, same address, same identity: normalize to Active and adopt + // the SWIM address the node advertises now. A restarted node may have + // bound a different one. + let needs_active = existing.state != NodeState::Active; + let advertised = swim_addr.map(|a| a.to_string()); + let swim_changed = existing.swim_addr != advertised; + if (needs_active || swim_changed) + && let Some(mut entry) = topology.get_node(req.node_id).cloned() { entry.state = NodeState::Active; + entry.swim_addr = advertised; + // `add_node` replaces the entry and bumps the topology version, + // so the change reaches peers through the topology broadcast. + topology.add_node(entry); } return build_response(topology, routing, cluster_id); } @@ -125,7 +163,8 @@ pub fn handle_join_request( NodeInfo::new(req.node_id, addr, NodeState::Active) .with_wire_version(req.wire_version) .with_spiffe_id(req.spiffe_id.clone()) - .with_spki_pin(spki_pin), + .with_spki_pin(spki_pin) + .with_swim_addr(swim_addr), ); build_response(topology, routing, cluster_id) } @@ -136,18 +175,7 @@ fn build_response( routing: &RoutingTable, cluster_id: u64, ) -> JoinResponse { - let nodes: Vec = topology - .all_nodes() - .map(|n| JoinNodeInfo { - node_id: n.node_id, - addr: n.addr.clone(), - state: n.state.as_u8(), - raft_groups: n.raft_groups.clone(), - wire_version: n.wire_version, - spiffe_id: n.spiffe_id.clone(), - spki_pin: n.spki_pin.map(|arr| arr.to_vec()), - }) - .collect(); + let nodes: Vec = topology.all_nodes().map(NodeInfo::to_wire).collect(); let groups: Vec = routing .group_members() @@ -205,8 +233,10 @@ mod tests { node_id: 2, listen_addr: "10.0.0.2:9400".into(), wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), spiffe_id: None, spki_pin: None, + swim_addr: None, }; let resp = handle_join_request(&req, &mut topology, &routing, 42); @@ -230,8 +260,10 @@ mod tests { node_id: 2, listen_addr: "10.0.0.2:9400".into(), wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), spiffe_id: None, spki_pin: None, + swim_addr: None, }; let _ = handle_join_request(&req, &mut topology, &routing, 42); @@ -254,8 +286,10 @@ mod tests { node_id: 2, listen_addr: "10.0.0.2:9400".into(), wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), spiffe_id: None, spki_pin: None, + swim_addr: None, }; let resp1 = handle_join_request(&req, &mut topology, &routing, 7); @@ -287,8 +321,10 @@ mod tests { node_id: 2, listen_addr: "10.0.0.2:9400".into(), wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), spiffe_id: None, spki_pin: None, + swim_addr: None, }; let resp1 = handle_join_request(&req1, &mut topology, &routing, 11); assert!(resp1.success); @@ -298,8 +334,10 @@ mod tests { node_id: 2, listen_addr: "10.0.0.99:9400".into(), wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), spiffe_id: None, spki_pin: None, + swim_addr: None, }; let resp2 = handle_join_request(&req2, &mut topology, &routing, 11); @@ -325,8 +363,10 @@ mod tests { node_id: 2, listen_addr: "10.0.0.2:9400".into(), wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), spiffe_id: Some("spiffe://nodedb/node/2".into()), spki_pin: Some(pin.to_vec()), + swim_addr: None, }; let response = handle_join_request(&request, &mut topology, &routing, 11); @@ -335,6 +375,37 @@ mod tests { assert!(!topology.contains(2)); } + /// A joiner running a different build is rejected with the + /// operator-facing "restart every node" message, not a rolling-upgrade + /// claim. + #[test] + fn handle_join_rejects_build_id_mismatch() { + let mut topology = topo_with_one_node(); + let routing = RoutingTable::uniform(1, &[1], 1); + + let req = JoinRequest { + node_id: 2, + listen_addr: "10.0.0.2:9400".into(), + wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: "some-other-build".into(), + spiffe_id: None, + spki_pin: None, + swim_addr: None, + }; + + let resp = handle_join_request(&req, &mut topology, &routing, 42); + + assert!(!resp.success); + assert!(resp.error.contains("some-other-build")); + assert!( + resp.error + .contains("all nodes must run one build before 1.0"), + "rejection error must name the fix: {}", + resp.error + ); + assert!(!topology.contains(2)); + } + #[test] fn handle_join_invalid_addr() { let mut topology = ClusterTopology::new(); @@ -344,12 +415,84 @@ mod tests { node_id: 2, listen_addr: "not-a-valid-address".into(), wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), spiffe_id: None, spki_pin: None, + swim_addr: None, }; let resp = handle_join_request(&req, &mut topology, &routing, 42); assert!(!resp.success); assert!(!resp.error.is_empty()); } + + fn join_with_swim(swim_addr: &str) -> JoinRequest { + JoinRequest { + node_id: 2, + listen_addr: "10.0.0.2:9400".into(), + wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), + spiffe_id: None, + spki_pin: None, + swim_addr: Some(swim_addr.into()), + } + } + + /// The joiner's SWIM address lands in its topology entry and in the wire + /// response every peer seeds SWIM from. + #[test] + fn join_carries_the_swim_address() { + let mut topology = topo_with_one_node(); + let routing = RoutingTable::uniform(1, &[1], 1); + + let resp = + handle_join_request(&join_with_swim("10.0.0.2:9401"), &mut topology, &routing, 1); + + assert!(resp.success, "{}", resp.error); + let entry = topology.get_node(2).expect("joiner admitted"); + assert_eq!(entry.swim_socket_addr(), "10.0.0.2:9401".parse().ok()); + let wire = resp + .nodes + .iter() + .find(|n| n.node_id == 2) + .expect("joiner in response"); + assert_eq!(wire.swim_addr.as_deref(), Some("10.0.0.2:9401")); + } + + /// A known node that re-advertises a different SWIM address updates its + /// entry and bumps the topology version so peers pick the change up. + #[test] + fn rejoin_with_a_new_swim_address_updates_the_entry() { + let mut topology = topo_with_one_node(); + let routing = RoutingTable::uniform(1, &[1], 1); + let first = + handle_join_request(&join_with_swim("10.0.0.2:9401"), &mut topology, &routing, 1); + assert!(first.success, "{}", first.error); + let version_before = topology.version(); + + let second = + handle_join_request(&join_with_swim("10.0.0.2:9501"), &mut topology, &routing, 1); + + assert!(second.success, "{}", second.error); + assert_eq!( + topology.get_node(2).and_then(NodeInfo::swim_socket_addr), + "10.0.0.2:9501".parse().ok() + ); + assert!(topology.version() > version_before); + } + + #[test] + fn join_rejects_an_invalid_swim_address() { + let mut topology = topo_with_one_node(); + let routing = RoutingTable::uniform(1, &[1], 1); + let resp = handle_join_request( + &join_with_swim("not-an-address"), + &mut topology, + &routing, + 1, + ); + assert!(!resp.success); + assert!(resp.error.contains("swim_addr"), "{}", resp.error); + assert!(!topology.contains(2)); + } } diff --git a/nodedb-cluster/src/bootstrap/join.rs b/nodedb-cluster/src/bootstrap/join.rs index 228e0f723..7247608c4 100644 --- a/nodedb-cluster/src/bootstrap/join.rs +++ b/nodedb-cluster/src/bootstrap/join.rs @@ -96,8 +96,10 @@ pub(super) async fn join( node_id: config.node_id, listen_addr: config.listen_addr.to_string(), wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: config.wire_build_id.clone(), spiffe_id: None, spki_pin: transport.local_spki_pin().map(|arr| arr.to_vec()), + swim_addr: config.swim_udp_addr.map(|a| a.to_string()), }; let policy = config.join_retry; @@ -263,30 +265,7 @@ fn apply_join_response( // 1. Reconstruct topology. let mut topology = ClusterTopology::new(); for node in &resp.nodes { - let state = NodeState::from_u8(node.state).unwrap_or(NodeState::Active); - let spki_pin: Option<[u8; 32]> = node.spki_pin.as_deref().and_then(|b| { - if b.len() == 32 { - let mut arr = [0u8; 32]; - arr.copy_from_slice(b); - Some(arr) - } else { - None - } - }); - let mut info = NodeInfo::new( - node.node_id, - node.addr.parse().unwrap_or_else(|_| { - "0.0.0.0:0" - .parse() - .expect("invariant: \"0.0.0.0:0\" is a valid SocketAddr literal") - }), - state, - ) - .with_wire_version(node.wire_version) - .with_spiffe_id(node.spiffe_id.clone()) - .with_spki_pin(spki_pin); - // Override raft_groups from wire data (NodeInfo::new starts empty). - info.raft_groups = node.raft_groups.clone(); + let mut info = NodeInfo::from_wire(node); if node.node_id == config.node_id { info.state = NodeState::Active; } @@ -303,6 +282,7 @@ fn apply_join_response( g.group_id, GroupInfo { leader: g.leader, + leader_term: 0, members: g.members.clone(), learners: g.learners.clone(), // Join response carries no placement; it converges via @@ -532,6 +512,7 @@ mod tests { install_snapshot_chunk_bytes: 4 * 1024 * 1024, orphan_partial_max_age_secs: 300, log_compaction_threshold: None, + wire_build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), }; let state1 = bootstrap(&config1, &catalog1, None).unwrap(); @@ -704,6 +685,7 @@ mod tests { install_snapshot_chunk_bytes: 4 * 1024 * 1024, orphan_partial_max_age_secs: 300, log_compaction_threshold: None, + wire_build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), }; let lifecycle = ClusterLifecycleTracker::new(); diff --git a/nodedb-cluster/src/bootstrap/mod.rs b/nodedb-cluster/src/bootstrap/mod.rs index f38c89a13..da2b514a1 100644 --- a/nodedb-cluster/src/bootstrap/mod.rs +++ b/nodedb-cluster/src/bootstrap/mod.rs @@ -23,4 +23,4 @@ pub mod start; pub use config::{ClusterConfig, ClusterState, JoinRetryPolicy}; pub use handle_join::handle_join_request; -pub use start::{start_cluster, start_cluster_subsystems}; +pub use start::{SubsystemHandles, start_cluster, start_cluster_subsystems}; diff --git a/nodedb-cluster/src/bootstrap/probe.rs b/nodedb-cluster/src/bootstrap/probe.rs index ef2d7cb9f..cc1de82fc 100644 --- a/nodedb-cluster/src/bootstrap/probe.rs +++ b/nodedb-cluster/src/bootstrap/probe.rs @@ -230,6 +230,7 @@ mod tests { install_snapshot_chunk_bytes: 4 * 1024 * 1024, orphan_partial_max_age_secs: 300, log_compaction_threshold: None, + wire_build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), } } diff --git a/nodedb-cluster/src/bootstrap/restart.rs b/nodedb-cluster/src/bootstrap/restart.rs index cf590722d..f79407c6d 100644 --- a/nodedb-cluster/src/bootstrap/restart.rs +++ b/nodedb-cluster/src/bootstrap/restart.rs @@ -9,6 +9,7 @@ use tracing::info; use crate::catalog::ClusterCatalog; use crate::error::{ClusterError, Result}; use crate::multi_raft::MultiRaft; +use crate::topology::ClusterTopology; use crate::transport::NexarTransport; use super::config::{ClusterConfig, ClusterState}; @@ -19,11 +20,12 @@ pub(super) fn restart( catalog: &ClusterCatalog, transport: &NexarTransport, ) -> Result { - let topology = catalog + let mut topology = catalog .load_topology()? .ok_or_else(|| ClusterError::Transport { detail: "catalog is bootstrapped but topology is missing".into(), })?; + readvertise_swim_addr(config, catalog, &mut topology)?; // ONE shared routing handle: MultiRaft and ClusterState read/write the // same table so committed Raft conf-changes converge the data-plane view. @@ -105,6 +107,28 @@ pub(super) fn restart( }) } +/// Stamp this node's freshly bound SWIM address on its own topology entry. +/// +/// A changed address bumps the topology version and is persisted. The health +/// monitor's version comparison then pushes it to every peer. +fn readvertise_swim_addr( + config: &ClusterConfig, + catalog: &ClusterCatalog, + topology: &mut ClusterTopology, +) -> Result<()> { + let advertised = config.swim_udp_addr.map(|a| a.to_string()); + let Some(own) = topology.get_node(config.node_id) else { + return Ok(()); + }; + if own.swim_addr == advertised { + return Ok(()); + } + let mut own = own.clone(); + own.swim_addr = advertised; + topology.add_node(own); + catalog.save_topology(topology) +} + #[cfg(test)] mod tests { use super::super::bootstrap_fn::bootstrap; @@ -137,6 +161,7 @@ mod tests { install_snapshot_chunk_bytes: 4 * 1024 * 1024, orphan_partial_max_age_secs: 300, log_compaction_threshold: None, + wire_build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), }; // Bootstrap first. @@ -159,4 +184,51 @@ mod tests { assert_eq!(state.routing.read().unwrap().num_groups(), 5); assert_eq!(state.multi_raft.lock().unwrap().group_count(), 5); } + + /// A restart that binds a different SWIM address stamps it on this + /// node's topology entry and persists it. + #[tokio::test] + async fn restart_readvertises_a_changed_swim_address() { + let (dir, catalog) = temp_catalog(); + let mut config = ClusterConfig { + node_id: 1, + listen_addr: "127.0.0.1:9400".parse().unwrap(), + seed_nodes: vec![], + num_groups: 1, + replication_factor: 1, + data_dir: dir.path().to_path_buf(), + force_bootstrap: false, + join_retry: Default::default(), + swim_udp_addr: "127.0.0.1:9401".parse().ok(), + election_timeout_min: Duration::from_millis(150), + election_timeout_max: Duration::from_millis(300), + install_snapshot_chunk_bytes: 4 * 1024 * 1024, + orphan_partial_max_age_secs: 300, + log_compaction_threshold: None, + wire_build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), + }; + let _ = bootstrap(&config, &catalog, None).unwrap(); + + use crate::transport::credentials::TransportCredentials; + let transport = NexarTransport::new( + 1, + "127.0.0.1:0".parse().unwrap(), + TransportCredentials::Insecure, + ) + .unwrap(); + config.swim_udp_addr = "127.0.0.1:9501".parse().ok(); + let state = restart(&config, &catalog, &transport).unwrap(); + + let expected: Option = "127.0.0.1:9501".parse().ok(); + let live = state.topology.read().unwrap(); + assert_eq!( + live.get_node(1).and_then(|n| n.swim_socket_addr()), + expected + ); + let persisted = catalog.load_topology().unwrap().unwrap(); + assert_eq!( + persisted.get_node(1).and_then(|n| n.swim_socket_addr()), + expected + ); + } } diff --git a/nodedb-cluster/src/bootstrap/start.rs b/nodedb-cluster/src/bootstrap/start.rs index 00d26b1eb..409e4ed3d 100644 --- a/nodedb-cluster/src/bootstrap/start.rs +++ b/nodedb-cluster/src/bootstrap/start.rs @@ -32,7 +32,7 @@ use crate::subsystem::context::BootstrapCtx; use crate::subsystem::health::ClusterHealth; use crate::subsystem::{ DecommissionSubsystem, ReachabilitySubsystem, RebalancerSubsystem, RunningCluster, - SubsystemRegistry, SwimSubsystem, SwimSubsystemConfig, + SubsystemRegistry, SwimSubsystem, SwimSubsystemConfig, SwimWiring, }; use crate::swim::config::SwimConfig; use crate::swim::incarnation::Incarnation; @@ -51,7 +51,8 @@ use super::restart::restart; /// and the `BootstrapCtx` is assembled. The four subsystems are: /// /// 1. `SwimSubsystem` (root, `deps = []`) — failure detector with -/// `RoutingLivenessHook` attached before the UDP socket opens. +/// `RoutingLivenessHook` and the host's `swim.subscribers` attached +/// before the first probe. /// `RoutingLivenessHook` is NOT its own subsystem; it is wired /// inside `SwimSubsystem::start()` as a SWIM subscriber. /// @@ -73,6 +74,7 @@ pub fn register_default_subsystems( ctx: &BootstrapCtx, executor: Arc, catalog: &Arc, + swim: SwimWiring, ) -> crate::error::Result<()> { // Fast-restart rejoin safety: resume SWIM above the last persisted // incarnation. A node that crashed while peers held `Dead(A, N)` @@ -97,14 +99,6 @@ pub fn register_default_subsystems( detail: format!("node_id is not a valid ID: {e}"), } })?, - // Use the explicit SWIM UDP addr if set; otherwise let the OS - // pick an ephemeral port by binding to port 0 on the listen addr. - swim_addr: config.swim_udp_addr.unwrap_or_else(|| { - let mut a = config.listen_addr; - a.set_port(0); - a - }), - seeds: config.seed_nodes.clone(), incarnation_store: Some(Arc::new(CatalogIncarnationStore::new(catalog))), }; @@ -112,7 +106,8 @@ pub fn register_default_subsystems( swim_cfg, Arc::clone(&ctx.routing), Arc::clone(&ctx.topology), - vec![], + swim.transport, + swim.subscribers, ))); registry.register(Arc::new(ReachabilitySubsystem::new( @@ -187,6 +182,15 @@ pub async fn start_cluster( Ok(cluster_state) } +/// The shared cluster handles every default subsystem runs on. +pub struct SubsystemHandles { + pub topology: Arc>, + pub routing: Arc>, + pub transport: Arc, + /// The `RaftLoop`'s own `MultiRaft` handle. + pub raft_multi_raft: Arc>, +} + /// Spawn the default cluster subsystems sharing `raft_multi_raft` with /// the running [`crate::raft_loop::RaftLoop`]. /// @@ -196,19 +200,30 @@ pub async fn start_cluster( /// subsystems use the same `Arc>` the loop owns — /// no double-ownership, no orphan Arcs blocking shutdown. /// +/// `swim` carries the SWIM socket bound before [`start_cluster`] and the +/// host's subscribers, which join the routing hook before the first probe. +/// /// The returned [`RunningCluster`] keeps subsystem background tasks /// alive; dropping it signals all of them to shut down. The host /// **must** call [`RunningCluster::shutdown_all`] explicitly during /// orderly shutdown so subsystems release their `MultiRaft` Arc /// before the loop exits. +/// +/// `migration_tracker` receives the state of every migration the +/// rebalancer's executor runs. pub async fn start_cluster_subsystems( config: &ClusterConfig, - topology: Arc>, - routing: Arc>, - transport: Arc, - raft_multi_raft: Arc>, + handles: SubsystemHandles, catalog: &Arc, + swim: SwimWiring, + migration_tracker: Arc, ) -> Result { + let SubsystemHandles { + topology, + routing, + transport, + raft_multi_raft, + } = handles; let health = ClusterHealth::new(); let (decommission_signal, _) = tokio::sync::watch::channel(false); let ctx = BootstrapCtx::new( @@ -220,15 +235,18 @@ pub async fn start_cluster_subsystems( decommission_signal, ); - let executor = Arc::new(MigrationExecutor::new( - Arc::clone(&raft_multi_raft), - Arc::clone(&routing), - Arc::clone(&topology), - Arc::clone(&transport), - )); + let executor = Arc::new( + MigrationExecutor::new( + Arc::clone(&raft_multi_raft), + Arc::clone(&routing), + Arc::clone(&topology), + Arc::clone(&transport), + ) + .with_tracker(migration_tracker), + ); let mut registry = SubsystemRegistry::new(); - register_default_subsystems(&mut registry, config, &ctx, executor, catalog)?; + register_default_subsystems(&mut registry, config, &ctx, executor, catalog, swim)?; registry .start_all(&ctx) diff --git a/nodedb-cluster/src/calvin/completion.rs b/nodedb-cluster/src/calvin/completion.rs index 4f27a7532..a6e24fa46 100644 --- a/nodedb-cluster/src/calvin/completion.rs +++ b/nodedb-cluster/src/calvin/completion.rs @@ -1,12 +1,23 @@ // SPDX-License-Identifier: BUSL-1.1 -use std::collections::{BTreeMap, BTreeSet}; +use std::collections::{BTreeMap, VecDeque}; use std::sync::{Arc, Mutex}; +use std::time::{Duration, Instant}; use tokio::sync::{mpsc, oneshot}; +use crate::calvin::completion_entry::{PendingCompletion, outcome_for}; +use crate::calvin::completion_waiter::{CompletionReport, CompletionWaiter}; use crate::calvin::sequencer::AbortReason; +/// The sequencer assignment of one submission: `(epoch, position, +/// participants)`. +pub type Assignment = (u64, u32, usize); + +/// Receives one submission's [`Assignment`]. It reads a closed channel when +/// the sequencer rejected or discarded the submission without sequencing it. +pub type AssignmentReceiver = oneshot::Receiver; + /// Calvin transaction identity in the sequencer-assigned coordinate space. /// /// `(epoch, position)` is the unique key the sequencer Raft state machine @@ -31,15 +42,15 @@ impl TxnId { /// all expected vshards acked but the global cross-shard verdict was ABORT /// (`Aborted`), the executor reported an OLLP prediction mismatch that forces a /// retry (`Mismatch`), or the scheduler rejected the transaction's routing as -/// terminally broken (`Failed`). `Aborted` and `Failed` are NEVER retried, -/// unlike `Mismatch`. +/// terminally broken (`Failed`). `Failed` is never retried. `Aborted` retries +/// only for `AbortReason::PredictionDrift`, which is the same drift as +/// `Mismatch`. #[derive(Clone, Debug, PartialEq, Eq)] pub enum AttemptOutcome { Completed, - /// The global cross-shard verdict was ABORT. `reason` names the cause; it is - /// `None` for a pre-existing durable verdict that recorded no reason. + /// The global cross-shard verdict was ABORT. `reason` names the cause. Aborted { - reason: Option, + reason: AbortReason, }, Mismatch, Failed { @@ -47,22 +58,19 @@ pub enum AttemptOutcome { }, } -/// One participant vShard's durable commit vote. `Abort(None)` is a -/// pre-existing durable `SequencerEntry::Vote { commit: false }`, written before -/// abort reasons existed on the wire. +/// One participant vShard's durable commit vote. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum ParticipantVote { Commit, - Abort(Option), + Abort(AbortReason), } /// A commit/abort decision: the tally aggregated from votes, and the -/// authoritative verdict stored on a `PendingCompletion`. `Abort(None)` carries -/// no reason, either from a legacy vote or a legacy stored verdict. +/// authoritative verdict stored on a `PendingCompletion`. #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum VerdictOutcome { Commit, - Abort(Option), + Abort(AbortReason), } impl VerdictOutcome { @@ -91,88 +99,11 @@ pub struct ParticipantProgress { pub routing_failed: bool, } -pub(crate) struct PendingCompletion { - /// `pub(crate)`: also read/written by the vote/verdict-tally methods in - /// `completion_verdict.rs` (a sibling module in the same crate). - pub(crate) expected_participants: usize, - acked_vshards: BTreeSet, - completion_tx: Option>, - /// Set when an OLLP mismatch is observed before the coordinator registers - /// its waiter, so the outcome is not lost across registration order (mirrors - /// how `acked_vshards` persists ack state regardless of registration order). - mismatched: bool, - /// Set when a terminal routing failure is observed before the coordinator - /// registers its waiter, mirroring `mismatched`. Takes precedence over both - /// `mismatched` and completion: a routing failure is never retried and never - /// falsely reported as success. - routing_failed: Option, - /// Durable per-participant commit votes tallied from `SequencerEntry::Vote` - /// and `SequencerEntry::AbortVote`, keyed by vshard so a re-proposed vote (retry) overwrites deterministically. - /// Once complete, the leader aggregates the tally into the global `verdict` - /// that gates each participant's flush/drop at the cross-shard commit barrier. - pub(crate) votes: BTreeMap, - /// The authoritative commit/abort verdict, applied from a replicated - /// `SequencerEntry::Verdict` or `AbortVerdict`; `None` until applied. This is the durable - /// barrier gate: a participant parked in `AwaitingVerdict` resumes into its - /// flush (commit) or drop (abort) once it is set. It is also the ONLY - /// authority for "already decided" — a post-failover leader re-proposes from - /// `votes` until it is stored (see `drain_unproposed_verdicts`). - pub(crate) verdict: Option, - /// Dedup guard for the LOCAL emit path: set the first time the tally becomes - /// complete so the verdict signal is emitted exactly once across vote - /// re-proposals. Per-node, non-durable, never reset on failover — so it is - /// NOT consulted by `drain_unproposed_verdicts` (which trusts only the - /// durable `verdict`, letting a promoted leader re-propose a still-unstored one). - pub(crate) verdict_proposed: bool, -} - -/// Choose the terminal outcome for a COMPLETED entry (all expected vshards -/// acked) by consulting the durable global verdict. -/// -/// The verdict is applied from a replicated `SequencerEntry::Verdict` that is -/// strictly ordered BEFORE the abort's `CompletionAck` in the sequencer Raft -/// log, so an ABORT verdict is always stored by the time this entry's -/// completion fires. A stored abort becomes `Aborted`, carrying the reason the -/// verdict recorded; `None` (single-shard / no-verdict paths) and a commit -/// verdict are `Completed`. We deliberately do NOT gate on `verdict.is_some()` -/// — that would stall the no-verdict completion paths. -fn outcome_for(entry: &PendingCompletion) -> AttemptOutcome { - match entry.verdict { - Some(VerdictOutcome::Abort(reason)) => AttemptOutcome::Aborted { reason }, - _ => AttemptOutcome::Completed, - } -} - -impl PendingCompletion { - pub(crate) fn new(expected_participants: usize) -> Self { - Self { - expected_participants, - acked_vshards: BTreeSet::new(), - completion_tx: None, - mismatched: false, - routing_failed: None, - votes: BTreeMap::new(), - verdict: None, - verdict_proposed: false, - } - } - - fn is_complete(&self) -> bool { - // Require a KNOWN participant count (>0). The `expected_participants == 0` - // default means "not yet seeded" — completion must not fire until the - // count is known (via `note_assigned` on the leader, or `register_completion` - // from the routed assignment on a remote coordinator). Without this guard a - // replicated ack that races ahead of seeding, or a bare `register_completion`, - // would spuriously report `Completed` with zero acks. - self.expected_participants > 0 && self.acked_vshards.len() >= self.expected_participants - } -} - /// `pub(crate)`: also read by the vote/verdict-tally methods in /// `completion_verdict.rs` (a sibling module in the same crate). #[derive(Default)] pub(crate) struct Inner { - assignments: BTreeMap>, + assignments: BTreeMap>, pub(crate) completions: BTreeMap, /// Per-vShard senders for the verdict push, keyed by vShard id. Each local /// Calvin scheduler registers its receiver's sender here at construction; @@ -181,6 +112,36 @@ pub(crate) struct Inner { /// notification never disagree. pub(crate) verdict_signal_senders: BTreeMap>, + /// Terminal entries with no waiter, oldest first, with the instant each + /// became one (`completion_gc`). + pub(crate) waiterless: VecDeque<(Instant, TxnId)>, + /// How long a terminal entry waits for a waiter before eviction. `None` + /// takes [`super::completion_gc::DEFAULT_WAITERLESS_TTL`]. + pub(crate) waiterless_ttl: Option, +} + +impl Inner { + /// Remove `txn`'s entry and deliver `outcome` to `waiter`. `signal` names + /// the event in the warning logged when the receiver is gone. + pub(crate) fn fire( + &mut self, + txn: TxnId, + waiter: CompletionWaiter, + outcome: AttemptOutcome, + ack_results: Vec>, + signal: &'static str, + ) { + self.completions.remove(&txn); + if !waiter.send(outcome, ack_results) { + tracing::warn!( + epoch = txn.epoch, + position = txn.position, + signal, + "calvin completion receiver dropped before its outcome fired; \ + client likely timed out on completion wait" + ); + } + } } pub struct CalvinCompletionRegistry { @@ -219,7 +180,11 @@ impl CalvinCompletionRegistry { Self::new(verdict_tx) } - pub fn register_submission(&self, inbox_seq: u64) -> oneshot::Receiver<(u64, u32, usize)> { + /// Register interest in the assignment of submission `inbox_seq`. + /// + /// Call it before the submission reaches the inbox channel, so the tick + /// that drains it always finds the sender. `Inbox::submit_with` does so. + pub fn register_submission(&self, inbox_seq: u64) -> AssignmentReceiver { let (tx, rx) = oneshot::channel(); self.inner .lock() @@ -229,6 +194,20 @@ impl CalvinCompletionRegistry { rx } + /// Drop the assignment sender of submission `inbox_seq`, if one is held. + /// + /// The sequencer calls it for every submission it rejects or discards + /// without sequencing. The waiting caller then reads a closed channel at + /// once. A caller that gives up waiting calls it too, so no sender stays + /// behind. + pub fn drop_assignment(&self, inbox_seq: u64) { + self.inner + .lock() + .unwrap_or_else(|p| p.into_inner()) + .assignments + .remove(&inbox_seq); + } + pub fn note_assigned(&self, inbox_seq: u64, txn: TxnId, expected_participants: usize) { let mut inner = self.inner.lock().unwrap_or_else(|p| p.into_inner()); if let Some(tx) = inner.assignments.remove(&inbox_seq) @@ -263,6 +242,25 @@ impl CalvinCompletionRegistry { expected_participants: usize, ) -> oneshot::Receiver { let (tx, rx) = oneshot::channel(); + self.register_waiter(txn, expected_participants, CompletionWaiter::Outcome(tx)); + rx + } + + /// [`Self::register_completion`] for a coordinator that also reads each + /// participant's apply result, as its `CompletionAck` carried it. The + /// acks reach every sequencer replica, so the results arrive whether or + /// not this node hosts a replica of each participant. + pub fn register_completion_report( + &self, + txn: TxnId, + expected_participants: usize, + ) -> oneshot::Receiver { + let (tx, rx) = oneshot::channel(); + self.register_waiter(txn, expected_participants, CompletionWaiter::Report(tx)); + rx + } + + fn register_waiter(&self, txn: TxnId, expected_participants: usize, tx: CompletionWaiter) { let mut inner = self.inner.lock().unwrap_or_else(|p| p.into_inner()); let entry = inner .completions @@ -272,52 +270,54 @@ impl CalvinCompletionRegistry { // Routing failure takes precedence over everything else: it is terminal // and must never be masked by a later ack or mismatch signal. if let Some(detail) = entry.routing_failed.take() { - inner.completions.remove(&txn); - if tx.send(AttemptOutcome::Failed { detail }).is_err() { - tracing::warn!( - epoch = txn.epoch, - position = txn.position, - "calvin completion receiver dropped before routing-failure signal; \ - client likely timed out on completion wait" - ); - } + inner.fire( + txn, + tx, + AttemptOutcome::Failed { detail }, + Vec::new(), + "routing failure", + ); + } else if entry.abandoned { + // The entry keeps its stored verdict for the participants that + // still probe it, and parks as a waiterless one. + super::completion_parts::send_parts_lost(txn, tx); + inner.settle_waiterless(txn); } else if entry.mismatched { - inner.completions.remove(&txn); - if tx.send(AttemptOutcome::Mismatch).is_err() { - tracing::warn!( - epoch = txn.epoch, - position = txn.position, - "calvin completion receiver dropped before OLLP-mismatch signal; \ - client likely timed out on completion wait" - ); - } + inner.fire( + txn, + tx, + AttemptOutcome::Mismatch, + Vec::new(), + "OLLP mismatch", + ); } else if entry.is_complete() { // Acks raced ahead of waiter registration: consult the stored // verdict so an already-complete ABORT surfaces as `Aborted`, not a // false `Completed`. let outcome = outcome_for(entry); - inner.completions.remove(&txn); - if tx.send(outcome).is_err() { - tracing::warn!( - epoch = txn.epoch, - position = txn.position, - "calvin completion receiver dropped before all-acked signal; \ - client likely timed out on completion wait" - ); - } + let results = entry.take_ack_results(); + inner.fire(txn, tx, outcome, results, "all acked"); } else { entry.completion_tx = Some(tx); } - rx } pub fn note_completion_ack(&self, txn: TxnId, vshard_id: u32) { + self.note_completion_ack_with(txn, vshard_id, Vec::new()); + } + + /// Record `vshard_id`'s `CompletionAck` for `txn`, with the apply result + /// it carried. An empty `result` records none. + pub fn note_completion_ack_with(&self, txn: TxnId, vshard_id: u32, result: Vec) { let mut inner = self.inner.lock().unwrap_or_else(|p| p.into_inner()); let entry = inner .completions .entry(txn) .or_insert_with(|| PendingCompletion::new(0)); entry.acked_vshards.insert(vshard_id); + if !result.is_empty() { + entry.ack_results.insert(vshard_id, result); + } if entry.is_complete() { // Consult the stored global verdict: `Verdict` is applied strictly // before this abort's `CompletionAck` in the sequencer Raft log, so @@ -334,18 +334,11 @@ impl CalvinCompletionRegistry { // `mismatched`/`routing_failed` persist across the same race. if let Some(tx) = entry.completion_tx.take() { let outcome = outcome_for(entry); - inner.completions.remove(&txn); - if tx.send(outcome).is_err() { - tracing::warn!( - epoch = txn.epoch, - position = txn.position, - vshard_id, - "calvin completion receiver dropped before final ack; \ - client likely timed out on completion wait" - ); - } + let results = entry.take_ack_results(); + inner.fire(txn, tx, outcome, results, "final ack"); } } + inner.settle_waiterless(txn); } /// Record an OLLP prediction mismatch for `txn`, the second terminal outcome @@ -363,16 +356,15 @@ impl CalvinCompletionRegistry { .or_insert_with(|| PendingCompletion::new(0)); entry.mismatched = true; if let Some(tx) = entry.completion_tx.take() { - inner.completions.remove(&txn); - if tx.send(AttemptOutcome::Mismatch).is_err() { - tracing::warn!( - epoch = txn.epoch, - position = txn.position, - "calvin completion receiver dropped before OLLP-mismatch signal; \ - client likely timed out on completion wait" - ); - } + inner.fire( + txn, + tx, + AttemptOutcome::Mismatch, + Vec::new(), + "OLLP mismatch", + ); } + inner.settle_waiterless(txn); } /// Record a terminal, NON-retryable routing failure for `txn` — the @@ -393,16 +385,24 @@ impl CalvinCompletionRegistry { .or_insert_with(|| PendingCompletion::new(0)); entry.routing_failed = Some(detail.clone()); if let Some(tx) = entry.completion_tx.take() { - inner.completions.remove(&txn); - if tx.send(AttemptOutcome::Failed { detail }).is_err() { - tracing::warn!( - epoch = txn.epoch, - position = txn.position, - "calvin completion receiver dropped before routing-failure signal; \ - client likely timed out on completion wait" - ); - } + inner.fire( + txn, + tx, + AttemptOutcome::Failed { detail }, + Vec::new(), + "routing failure", + ); } + inner.settle_waiterless(txn); + } + + /// Set how long a terminal entry with no waiter stays for one to + /// register. The host sets it longer than its statement deadline. + pub fn set_waiterless_ttl(&self, ttl: Duration) { + self.inner + .lock() + .unwrap_or_else(|p| p.into_inner()) + .waiterless_ttl = Some(ttl); } /// What this registry holds for participant `vshard` of `txn`. @@ -434,12 +434,85 @@ impl CalvinCompletionRegistry { .completions .len() } + + /// Test-only: the number of assignment senders held. + #[cfg(test)] + pub fn pending_assignments_len(&self) -> usize { + self.inner + .lock() + .unwrap_or_else(|p| p.into_inner()) + .assignments + .len() + } } #[cfg(test)] mod tests { use super::*; + /// A replica on which no coordinator waits keeps no completion entry, + /// and no ack result, past the eviction window. An entry that has a + /// waiter stays until its outcome fires. + #[tokio::test] + async fn waiterless_terminal_entries_are_evicted_with_their_results() { + let reg = CalvinCompletionRegistry::new_detached(); + reg.set_waiterless_ttl(Duration::ZERO); + + let acked = TxnId::new(3, 0); + reg.note_assigned(1, acked, 1); + reg.note_completion_ack_with(acked, 5, vec![0xAB; 4096]); + assert_eq!( + reg.pending_completions_len(), + 0, + "a complete entry nobody waits for is evicted with its result" + ); + + let mismatched = TxnId::new(3, 1); + reg.note_ollp_mismatch(mismatched); + assert_eq!(reg.pending_completions_len(), 0); + + let awaited = TxnId::new(3, 2); + let rx = reg.register_completion_report(awaited, 2); + reg.note_completion_ack_with(awaited, 5, b"five".to_vec()); + reg.note_completion_ack_with(TxnId::new(3, 3), 6, b"other".to_vec()); + assert!( + reg.participant_progress(awaited, 5).is_some(), + "an entry with a waiter is never evicted" + ); + reg.note_completion_ack_with(awaited, 6, b"six".to_vec()); + let report = rx.await.expect("completion fires"); + assert_eq!(report.ack_results, vec![b"five".to_vec(), b"six".to_vec()]); + } + + /// The coordinator hosts no replica of either participant: it learns + /// their apply results only from the `CompletionAck`s its sequencer + /// replica applies. The report carries each participant's result, in + /// vShard order, whether the acks land before or after it registers. + #[tokio::test] + async fn a_report_carries_every_participants_ack_result() { + let reg = CalvinCompletionRegistry::new_detached(); + let txn = TxnId::new(11, 4); + reg.note_completion_ack_with(txn, 20, b"twenty".to_vec()); + let rx = reg.register_completion_report(txn, 3); + reg.note_completion_ack_with(txn, 10, b"ten".to_vec()); + reg.note_completion_ack_with(txn, 30, Vec::new()); + let report = rx.await.expect("completion fires"); + assert_eq!(report.outcome, AttemptOutcome::Completed); + assert_eq!( + report.ack_results, + vec![b"ten".to_vec(), b"twenty".to_vec()] + ); + assert_eq!(reg.pending_completions_len(), 0); + + let late = TxnId::new(11, 5); + reg.note_completion_ack_with(late, 7, b"seven".to_vec()); + let report = reg + .register_completion_report(late, 1) + .await + .expect("an already complete txn fires on registration"); + assert_eq!(report.ack_results, vec![b"seven".to_vec()]); + } + #[tokio::test] async fn completion_entry_removed_after_all_acks() { let reg = CalvinCompletionRegistry::new_detached(); @@ -644,7 +717,7 @@ mod tests { reg.note_assigned(1, txn, 2); reg.note_verdict( txn, - VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)), + VerdictOutcome::Abort(AbortReason::SerializationConflict), ); let rx = reg.register_completion(txn, 2); reg.note_completion_ack(txn, 10); @@ -653,7 +726,7 @@ mod tests { assert_eq!( outcome, AttemptOutcome::Aborted { - reason: Some(AbortReason::SerializationConflict) + reason: AbortReason::SerializationConflict }, "a stored ABORT verdict must surface as Aborted, not Completed" ); @@ -668,10 +741,7 @@ mod tests { let reg = CalvinCompletionRegistry::new_detached(); let txn = TxnId::new(32, 1); reg.note_assigned(1, txn, 2); - reg.note_verdict( - txn, - VerdictOutcome::Abort(Some(AbortReason::ParticipantError)), - ); + reg.note_verdict(txn, VerdictOutcome::Abort(AbortReason::ParticipantError)); reg.note_completion_ack(txn, 10); reg.note_completion_ack(txn, 20); let rx = reg.register_completion(txn, 2); @@ -679,7 +749,7 @@ mod tests { assert_eq!( outcome, AttemptOutcome::Aborted { - reason: Some(AbortReason::ParticipantError) + reason: AbortReason::ParticipantError } ); assert_eq!(reg.pending_completions_len(), 0); @@ -700,21 +770,30 @@ mod tests { assert_eq!(reg.pending_completions_len(), 0); } - #[tokio::test] - async fn legacy_abort_verdict_reports_aborted_with_no_reason() { - // A durable `SequencerEntry::Verdict { commit: false }` written before - // abort reasons existed applies as `Abort(None)`. The coordinator must - // still get `Aborted`, with `reason: None` marking the unknown cause. + /// A dropped assignment closes the caller's channel at once and leaves no + /// sender behind. A later `note_assigned` for the seq finds none. + #[test] + fn drop_assignment_closes_the_callers_channel_and_frees_the_sender() { let reg = CalvinCompletionRegistry::new_detached(); - let txn = TxnId::new(35, 0); - reg.note_assigned(1, txn, 2); - reg.note_verdict(txn, VerdictOutcome::Abort(None)); - let rx = reg.register_completion(txn, 2); - reg.note_completion_ack(txn, 10); - reg.note_completion_ack(txn, 20); - let outcome = rx.await.expect("completion fires"); - assert_eq!(outcome, AttemptOutcome::Aborted { reason: None }); - assert_eq!(reg.pending_completions_len(), 0); + let mut rejected = reg.register_submission(4); + let mut kept = reg.register_submission(5); + assert_eq!(reg.pending_assignments_len(), 2); + + reg.drop_assignment(4); + assert_eq!( + rejected.try_recv(), + Err(oneshot::error::TryRecvError::Closed), + "the caller must see a closed channel, not wait out its timeout" + ); + assert_eq!(reg.pending_assignments_len(), 1); + + reg.note_assigned(5, TxnId::new(2, 0), 3); + assert_eq!(kept.try_recv(), Ok((2, 0, 3))); + assert_eq!(reg.pending_assignments_len(), 0); + + // Dropping a seq with no sender is a no-op. + reg.drop_assignment(4); + assert_eq!(reg.pending_assignments_len(), 0); } #[tokio::test] diff --git a/nodedb-cluster/src/calvin/completion_entry.rs b/nodedb-cluster/src/calvin/completion_entry.rs new file mode 100644 index 000000000..60f64a2a6 --- /dev/null +++ b/nodedb-cluster/src/calvin/completion_entry.rs @@ -0,0 +1,117 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One transaction's completion entry in the +//! [`super::completion::CalvinCompletionRegistry`], and the outcome it fires. + +use std::collections::{BTreeMap, BTreeSet}; + +use super::completion::{AttemptOutcome, ParticipantVote, VerdictOutcome}; +use super::completion_waiter::CompletionWaiter; + +pub(crate) struct PendingCompletion { + pub(crate) expected_participants: usize, + pub(crate) acked_vshards: BTreeSet, + /// The apply result each acked participant's `CompletionAck` carried, + /// by vShard. A participant whose ack carried none has no entry. + pub(crate) ack_results: BTreeMap>, + pub(crate) completion_tx: Option, + /// Set when an OLLP mismatch is observed before the coordinator registers + /// its waiter, so the outcome is not lost across registration order (mirrors + /// how `acked_vshards` persists ack state regardless of registration order). + pub(crate) mismatched: bool, + /// Set when a terminal routing failure is observed before the coordinator + /// registers its waiter, mirroring `mismatched`. Takes precedence over both + /// `mismatched` and completion: a routing failure is never retried and never + /// falsely reported as success. + pub(crate) routing_failed: Option, + /// Set when the multi-part transaction lost its parts + /// (`SequencerEntry::TxnPartsAbandoned`). Its outcome is + /// `Aborted { PartsLost }` without waiting for acks: no participant + /// staged the whole transaction. + pub(crate) abandoned: bool, + /// Durable per-participant commit votes tallied from `SequencerEntry::Vote` + /// and `SequencerEntry::AbortVote`, keyed by vshard so a re-proposed vote + /// (retry) overwrites deterministically. Once complete, the leader + /// aggregates the tally into the global `verdict` that gates each + /// participant's flush/drop at the cross-shard commit barrier. + pub(crate) votes: BTreeMap, + /// The authoritative commit/abort verdict, applied from a replicated + /// `SequencerEntry::Verdict` or `AbortVerdict`; `None` until applied. This + /// is the durable barrier gate: a participant parked in `AwaitingVerdict` + /// resumes into its flush (commit) or drop (abort) once it is set. It is + /// also the ONLY authority for "already decided" — a post-failover leader + /// re-proposes from `votes` until it is stored (see + /// `drain_unproposed_verdicts`). + pub(crate) verdict: Option, + /// Dedup guard for the LOCAL emit path: set the first time the tally becomes + /// complete so the verdict signal is emitted exactly once across vote + /// re-proposals. Per-node, non-durable, never reset on failover — so it is + /// NOT consulted by `drain_unproposed_verdicts` (which trusts only the + /// durable `verdict`, letting a promoted leader re-propose a still-unstored one). + pub(crate) verdict_proposed: bool, + /// Queued for eviction as a terminal entry with no waiter + /// (`completion_gc`). + pub(crate) parked: bool, +} + +/// Choose the terminal outcome for a COMPLETED entry (all expected vshards +/// acked) by consulting the durable global verdict. +/// +/// The verdict is applied from a replicated `SequencerEntry::Verdict` that is +/// strictly ordered BEFORE the abort's `CompletionAck` in the sequencer Raft +/// log, so an ABORT verdict is always stored by the time this entry's +/// completion fires. A stored abort becomes `Aborted`, carrying the reason the +/// verdict recorded; `None` (single-shard / no-verdict paths) and a commit +/// verdict are `Completed`. We deliberately do NOT gate on `verdict.is_some()` +/// — that would stall the no-verdict completion paths. +pub(crate) fn outcome_for(entry: &PendingCompletion) -> AttemptOutcome { + match entry.verdict { + Some(VerdictOutcome::Abort(reason)) => AttemptOutcome::Aborted { reason }, + _ => AttemptOutcome::Completed, + } +} + +impl PendingCompletion { + pub(crate) fn new(expected_participants: usize) -> Self { + Self { + expected_participants, + acked_vshards: BTreeSet::new(), + ack_results: BTreeMap::new(), + completion_tx: None, + mismatched: false, + routing_failed: None, + abandoned: false, + votes: BTreeMap::new(), + verdict: None, + verdict_proposed: false, + parked: false, + } + } + + pub(crate) fn has_waiter(&self) -> bool { + self.completion_tx.is_some() + } + + /// Whether the entry's outcome is decided: every expected participant + /// acked, or an OLLP mismatch or a routing failure is recorded. + pub(crate) fn is_terminal(&self) -> bool { + self.is_complete() || self.mismatched || self.routing_failed.is_some() || self.abandoned + } + + /// Every acked participant's apply result, in vShard order. + pub(crate) fn take_ack_results(&mut self) -> Vec> { + std::mem::take(&mut self.ack_results) + .into_values() + .collect() + } + + pub(crate) fn is_complete(&self) -> bool { + // Require a KNOWN participant count (>0). The `expected_participants == 0` + // default means "not yet seeded" — completion must not fire until the + // count is known (via `note_assigned` on the leader, or `register_completion` + // from the routed assignment on a remote coordinator). Without this guard a + // replicated ack that races ahead of seeding, or a bare `register_completion`, + // would spuriously report `Completed` with zero acks. + self.expected_participants > 0 && self.acked_vshards.len() >= self.expected_participants + } +} diff --git a/nodedb-cluster/src/calvin/completion_gc.rs b/nodedb-cluster/src/calvin/completion_gc.rs new file mode 100644 index 000000000..6d2eb9373 --- /dev/null +++ b/nodedb-cluster/src/calvin/completion_gc.rs @@ -0,0 +1,68 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Eviction of completion entries no waiter will collect. +//! +//! Every sequencer replica applies every `CompletionAck`, so every replica +//! builds a completion entry for every transaction, and the entry holds each +//! participant's apply result. Only the replica on the coordinator's node +//! gains a waiter. A terminal entry keeps its outcome so a waiter that +//! registers after the last ack still receives it. A coordinator registers +//! within its statement deadline, so a terminal entry that gained no waiter +//! within the eviction window never will, and is removed with its results. +//! +//! An entry becomes terminal when every expected participant acked, or when +//! an OLLP mismatch or a routing failure is recorded for it. A terminal +//! entry with no waiter joins a queue stamped with the instant it did. Each +//! later registry change sweeps the queue front, so the queue holds at most +//! the entries of one window. + +use std::time::{Duration, Instant}; + +use super::completion::{Inner, TxnId}; + +/// The eviction window when the host sets none. Longer than any statement +/// deadline, so a coordinator's waiter always registers inside it. +pub const DEFAULT_WAITERLESS_TTL: Duration = Duration::from_secs(120); + +impl Inner { + /// Queue `txn`'s entry when it just became a waiterless terminal entry, + /// then evict every queued entry whose window passed. + pub(crate) fn settle_waiterless(&mut self, txn: TxnId) { + // no-determinism: node-local eviction of finished entries; never in the log. + let now = Instant::now(); + self.park_waiterless(txn, now); + self.sweep_waiterless(now); + } + + /// Queue `txn`'s entry for eviction when it is terminal, has no waiter, + /// and is not queued yet. + pub(crate) fn park_waiterless(&mut self, txn: TxnId, now: Instant) { + let Some(entry) = self.completions.get_mut(&txn) else { + return; + }; + if entry.parked || entry.has_waiter() || !entry.is_terminal() { + return; + } + entry.parked = true; + self.waiterless.push_back((now, txn)); + } + + /// Remove every queued entry that stayed waiterless for the whole + /// window. + pub(crate) fn sweep_waiterless(&mut self, now: Instant) { + let ttl = self.waiterless_ttl.unwrap_or(DEFAULT_WAITERLESS_TTL); + while let Some(&(since, txn)) = self.waiterless.front() { + if now.saturating_duration_since(since) < ttl { + break; + } + self.waiterless.pop_front(); + if self + .completions + .get(&txn) + .is_some_and(|entry| !entry.has_waiter()) + { + self.completions.remove(&txn); + } + } + } +} diff --git a/nodedb-cluster/src/calvin/completion_parts.rs b/nodedb-cluster/src/calvin/completion_parts.rs new file mode 100644 index 000000000..c08b128c1 --- /dev/null +++ b/nodedb-cluster/src/calvin/completion_parts.rs @@ -0,0 +1,100 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The completion registry's side of a multi-part transaction that lost its +//! parts. + +use super::completion::{AttemptOutcome, CalvinCompletionRegistry, TxnId, VerdictOutcome}; +use super::completion_entry::PendingCompletion; +use super::completion_verdict::VerdictSignal; +use super::sequencer::AbortReason; + +impl CalvinCompletionRegistry { + /// Record that the multi-part transaction `txn` lost its parts, applied + /// from a `SequencerEntry::TxnPartsAbandoned` on every replica. + /// + /// - The stored verdict becomes `Abort(PartsLost)`, pushed to every local + /// scheduler. A participant that staged the parts that target it drops + /// them at the verdict. + /// - The coordinator's outcome is `Aborted { PartsLost }` at once. It + /// waits for no ack: no participant staged the whole transaction, and + /// no verdict can ever commit it. + /// + /// Idempotent. A later ack of `txn` finds no entry for its outcome and + /// parks as a waiterless one. + pub fn note_parts_abandoned(&self, txn: TxnId) { + let verdict = VerdictOutcome::Abort(AbortReason::PartsLost); + let mut inner = self.inner.lock().unwrap_or_else(|p| p.into_inner()); + let waiter = { + let entry = inner + .completions + .entry(txn) + .or_insert_with(|| PendingCompletion::new(0)); + entry.abandoned = true; + entry.verdict = Some(verdict); + entry.completion_tx.take() + }; + let signal = VerdictSignal { + epoch: txn.epoch, + position: txn.position, + verdict, + }; + for tx in inner.verdict_signal_senders.values() { + let _ = tx.try_send(signal); + } + // The entry stays: a participant that staged every part targeting + // it probes the stored verdict when it parks, after this. + if let Some(waiter) = waiter { + send_parts_lost(txn, waiter); + } + inner.settle_waiterless(txn); + } +} + +/// Answer `waiter` with `Aborted { PartsLost }`, keeping `txn`'s entry and +/// its stored verdict for the participants that still probe it. +pub(crate) fn send_parts_lost(txn: TxnId, waiter: super::completion_waiter::CompletionWaiter) { + let outcome = AttemptOutcome::Aborted { + reason: AbortReason::PartsLost, + }; + if !waiter.send(outcome, Vec::new()) { + tracing::warn!( + epoch = txn.epoch, + position = txn.position, + "calvin completion receiver dropped before its parts-lost outcome fired" + ); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn lost_parts_abort_a_registered_waiter_without_acks() { + let reg = CalvinCompletionRegistry::new_detached(); + let txn = TxnId::new(4, 1); + let rx = reg.register_completion(txn, 3); + reg.note_parts_abandoned(txn); + assert_eq!( + rx.await.expect("outcome"), + AttemptOutcome::Aborted { + reason: AbortReason::PartsLost + } + ); + assert_eq!(reg.verdict(txn), Some(false)); + } + + #[tokio::test] + async fn lost_parts_before_registration_abort_the_later_waiter() { + let reg = CalvinCompletionRegistry::new_detached(); + let txn = TxnId::new(4, 2); + reg.note_parts_abandoned(txn); + let rx = reg.register_completion(txn, 3); + assert_eq!( + rx.await.expect("outcome"), + AttemptOutcome::Aborted { + reason: AbortReason::PartsLost + } + ); + } +} diff --git a/nodedb-cluster/src/calvin/completion_verdict.rs b/nodedb-cluster/src/calvin/completion_verdict.rs index 60cfb3e11..1cf8845b2 100644 --- a/nodedb-cluster/src/calvin/completion_verdict.rs +++ b/nodedb-cluster/src/calvin/completion_verdict.rs @@ -2,20 +2,17 @@ //! Vote/verdict-tally methods for [`super::completion::CalvinCompletionRegistry`]. //! -//! Split out of `completion.rs` (which hit the file-size limit): this module -//! holds every method that participates in the cross-shard commit-barrier -//! vote tally and verdict push, plus their tests. It reaches into -//! `completion.rs`'s otherwise-private `Inner` / `PendingCompletion` internals -//! via `pub(crate)` fields — no visibility is widened beyond this crate. +//! This module holds every method that takes part in the cross-shard +//! commit-barrier vote tally and verdict push, plus their tests. It reads +//! `Inner` and `PendingCompletion` through `pub(crate)` fields. use std::collections::BTreeMap; use tokio::sync::mpsc; use super::TxnId; -use super::completion::{ - CalvinCompletionRegistry, ParticipantVote, PendingCompletion, VerdictOutcome, -}; +use super::completion::{CalvinCompletionRegistry, ParticipantVote, VerdictOutcome}; +use super::completion_entry::PendingCompletion; use super::sequencer::AbortReason; /// Push notification that a staged cross-shard txn's authoritative global @@ -53,6 +50,16 @@ impl CalvinCompletionRegistry { .insert(vshard, tx); } + /// Stop pushing verdicts to `vshard`'s scheduler: this node no longer + /// replicates the vShard. + pub fn unregister_verdict_signal_sender(&self, vshard: u32) { + self.inner + .lock() + .unwrap_or_else(|p| p.into_inner()) + .verdict_signal_senders + .remove(&vshard); + } + /// Seed the expected participant count for `txn` deterministically from the /// replicated `SequencerEntry::EpochBatch` — this runs on every replica (not /// just the epoch's originating leader), so vote-completeness becomes @@ -173,9 +180,6 @@ impl CalvinCompletionRegistry { ); entry.verdict = Some(verdict); } - // Same decision, but a legacy abort recorded no reason: adopt the - // one this apply carries. - Some(VerdictOutcome::Abort(None)) => entry.verdict = Some(verdict), Some(_) => {} None => entry.verdict = Some(verdict), } @@ -223,26 +227,34 @@ impl CalvinCompletionRegistry { /// Aggregate a complete vote tally: commit only when every participant voted /// commit, otherwise abort with the highest-precedence reason. /// -/// Precedence: `ParticipantError` outranks `SerializationConflict` — a peer that -/// never staged makes a stale read-set unverifiable, and the infrastructure -/// failure is the actionable diagnosis. +/// Precedence: `PartsLost` outranks everything, since no participant holds +/// the whole transaction. `ParticipantError` outranks `CollectionSuperseded`, which +/// outranks `PredictionDrift`, which outranks `SerializationConflict`. A peer +/// that never staged makes every other verdict unverifiable, a superseded +/// collection makes a retry against the same read-set pointless, and a +/// drifted prediction retries with a fresh reconnaissance. The tally is +/// order-independent. fn tally_verdict(votes: &BTreeMap) -> VerdictOutcome { - let mut aborted = false; + fn rank(reason: AbortReason) -> u8 { + match reason { + AbortReason::PartsLost => 4, + AbortReason::ParticipantError => 3, + AbortReason::CollectionSuperseded => 2, + AbortReason::PredictionDrift => 1, + AbortReason::SerializationConflict => 0, + } + } let mut reason: Option = None; for vote in votes.values() { - let ParticipantVote::Abort(vote_reason) = vote else { - continue; - }; - aborted = true; - match (reason, vote_reason) { - (Some(AbortReason::ParticipantError), _) | (Some(_), None) => {} - _ => reason = *vote_reason, + if let ParticipantVote::Abort(vote_reason) = *vote + && reason.is_none_or(|held| rank(vote_reason) > rank(held)) + { + reason = Some(vote_reason); } } - if aborted { - VerdictOutcome::Abort(reason) - } else { - VerdictOutcome::Commit + match reason { + Some(reason) => VerdictOutcome::Abort(reason), + None => VerdictOutcome::Commit, } } @@ -258,16 +270,14 @@ mod tests { reg.note_vote( txn, 2, - ParticipantVote::Abort(Some(AbortReason::SerializationConflict)), + ParticipantVote::Abort(AbortReason::SerializationConflict), ); let tally = reg.vote_tally(txn).expect("entry created by note_vote"); assert_eq!(tally.len(), 2); assert_eq!(tally.get(&1), Some(&ParticipantVote::Commit)); assert_eq!( tally.get(&2), - Some(&ParticipantVote::Abort(Some( - AbortReason::SerializationConflict - ))) + Some(&ParticipantVote::Abort(AbortReason::SerializationConflict)) ); } @@ -309,14 +319,14 @@ mod tests { reg.note_vote( txn, 2, - ParticipantVote::Abort(Some(AbortReason::SerializationConflict)), + ParticipantVote::Abort(AbortReason::SerializationConflict), ); // Any abort vote makes the aggregated verdict an abort. assert_eq!( rx.try_recv().expect("verdict emitted"), ( txn, - VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)) + VerdictOutcome::Abort(AbortReason::SerializationConflict) ) ); assert!(rx.try_recv().is_err()); @@ -333,19 +343,16 @@ mod tests { reg.note_vote( txn, 1, - ParticipantVote::Abort(Some(AbortReason::SerializationConflict)), + ParticipantVote::Abort(AbortReason::SerializationConflict), ); reg.note_vote( txn, 2, - ParticipantVote::Abort(Some(AbortReason::ParticipantError)), + ParticipantVote::Abort(AbortReason::ParticipantError), ); assert_eq!( rx.try_recv().expect("verdict emitted"), - ( - txn, - VerdictOutcome::Abort(Some(AbortReason::ParticipantError)) - ), + (txn, VerdictOutcome::Abort(AbortReason::ParticipantError)), "a peer that never staged makes the stale read-set unverifiable" ); } @@ -361,22 +368,46 @@ mod tests { reg.note_vote( txn, 1, - ParticipantVote::Abort(Some(AbortReason::ParticipantError)), + ParticipantVote::Abort(AbortReason::ParticipantError), ); reg.note_vote( txn, 2, - ParticipantVote::Abort(Some(AbortReason::SerializationConflict)), + ParticipantVote::Abort(AbortReason::SerializationConflict), ); assert_eq!( rx.try_recv().expect("verdict emitted"), - ( - txn, - VerdictOutcome::Abort(Some(AbortReason::ParticipantError)) - ) + (txn, VerdictOutcome::Abort(AbortReason::ParticipantError)) ); } + /// A superseded collection aborts a transaction other slices voted to + /// commit, outranks a conflict, and yields to a participant error, in any + /// vote order. + #[test] + fn a_superseded_collection_aborts_the_whole_transaction() { + let superseded = ParticipantVote::Abort(AbortReason::CollectionSuperseded); + let conflict = ParticipantVote::Abort(AbortReason::SerializationConflict); + let errored = ParticipantVote::Abort(AbortReason::ParticipantError); + let tally = |votes: [&ParticipantVote; 2]| { + let votes: BTreeMap = + (1..).zip(votes.into_iter().copied()).collect(); + tally_verdict(&votes) + }; + let aborted = VerdictOutcome::Abort; + + assert_eq!( + tally([&ParticipantVote::Commit, &superseded]), + aborted(AbortReason::CollectionSuperseded) + ); + for votes in [[&superseded, &conflict], [&conflict, &superseded]] { + assert_eq!(tally(votes), aborted(AbortReason::CollectionSuperseded)); + } + for votes in [[&superseded, &errored], [&errored, &superseded]] { + assert_eq!(tally(votes), aborted(AbortReason::ParticipantError)); + } + } + #[tokio::test] async fn seed_expected_is_idempotent_max_and_enables_verdict_without_note_assigned() { // Mirrors complete_all_true_tally_emits_commit_verdict_once, but seeds @@ -476,7 +507,7 @@ mod tests { let txn = TxnId::new(51, 0); reg.note_verdict( txn, - VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)), + VerdictOutcome::Abort(AbortReason::SerializationConflict), ); // Both locally registered vShard schedulers receive the broadcast; each @@ -526,7 +557,7 @@ mod tests { // Must not panic despite the closed channel; the verdict still stores. reg.note_verdict( txn, - VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)), + VerdictOutcome::Abort(AbortReason::SerializationConflict), ); assert_eq!(reg.verdict(txn), Some(false)); } @@ -573,14 +604,14 @@ mod tests { reg.note_vote( txn, 2, - ParticipantVote::Abort(Some(AbortReason::SerializationConflict)), + ParticipantVote::Abort(AbortReason::SerializationConflict), ); assert_eq!(reg.verdict(txn), None); assert_eq!( reg.drain_unproposed_verdicts(), vec![( txn, - VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)) + VerdictOutcome::Abort(AbortReason::SerializationConflict) )], "any abort vote makes the re-proposed verdict an abort" ); diff --git a/nodedb-cluster/src/calvin/completion_waiter.rs b/nodedb-cluster/src/calvin/completion_waiter.rs new file mode 100644 index 000000000..7faf0edda --- /dev/null +++ b/nodedb-cluster/src/calvin/completion_waiter.rs @@ -0,0 +1,48 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The coordinator's side of a Calvin completion: what fires, and what it +//! carries. +//! +//! Every participant vShard proposes a `CompletionAck` to the sequencer log, +//! and every sequencer replica applies it, so a coordinator learns each +//! participant's apply result wherever that participant runs. A coordinator +//! that registers for a [`CompletionReport`] receives those results with the +//! outcome. A coordinator that registers for the outcome alone receives only +//! the [`AttemptOutcome`]. + +use tokio::sync::oneshot; + +use super::completion::AttemptOutcome; + +/// A completion as a coordinator that asked for its apply results receives +/// it. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct CompletionReport { + pub outcome: AttemptOutcome, + /// Each participant vShard's apply result, as its `CompletionAck` + /// carried it, in vShard order. A participant whose ack carried none + /// contributes none. Empty for every outcome but a completion. + pub ack_results: Vec>, +} + +/// The registered receiver of one transaction's terminal outcome. +pub(crate) enum CompletionWaiter { + Outcome(oneshot::Sender), + Report(oneshot::Sender), +} + +impl CompletionWaiter { + /// Deliver `outcome` with `ack_results`. `false` when the receiver is + /// gone. + pub(crate) fn send(self, outcome: AttemptOutcome, ack_results: Vec>) -> bool { + match self { + Self::Outcome(tx) => tx.send(outcome).is_ok(), + Self::Report(tx) => tx + .send(CompletionReport { + outcome, + ack_results, + }) + .is_ok(), + } + } +} diff --git a/nodedb-cluster/src/calvin/mod.rs b/nodedb-cluster/src/calvin/mod.rs index 659690779..7f72bd68d 100644 --- a/nodedb-cluster/src/calvin/mod.rs +++ b/nodedb-cluster/src/calvin/mod.rs @@ -2,21 +2,28 @@ pub mod applied_acks; pub mod completion; +pub mod completion_entry; +pub mod completion_gc; +mod completion_parts; mod completion_verdict; +pub mod completion_waiter; pub mod sequencer; pub mod types; pub use applied_acks::{AppliedAckLog, AppliedCompletionAck}; pub use completion::{ - AttemptOutcome, CalvinCompletionRegistry, ParticipantProgress, ParticipantVote, TxnId, - VerdictOutcome, + Assignment, AssignmentReceiver, AttemptOutcome, CalvinCompletionRegistry, ParticipantProgress, + ParticipantVote, TxnId, VerdictOutcome, }; pub use completion_verdict::VerdictSignal; +pub use completion_waiter::CompletionReport; pub use sequencer::{ - AbortReason, AdmittedTx, ConflictKey, EpochCheck, Inbox, InboxReceiver, RejectedTx, - ReservationInbox, ReservationInboxReceiver, ReservationRequest, SEQUENCER_GROUP_ID, + AbortReason, AdmittedTx, ConflictKey, CutInstantHook, EpochCheck, Inbox, InboxReceiver, + PartsIntake, PartsOffer, PartsOfferStatus, RejectedTx, ReservationInbox, + ReservationInboxReceiver, ReservationRequest, RestorePointHook, SEQUENCER_GROUP_ID, SequencerConfig, SequencerEntry, SequencerError, SequencerHalt, SequencerMetrics, - SequencerReceivers, SequencerService, SequencerStateMachine, UnrecoverableEpochHook, new_inbox, - new_reservation_inbox, validate_batch, + SequencerReceivers, SequencerRestorePoint, SequencerService, SequencerSnapshot, + SequencerStateMachine, UnrecoverableEpochHook, new_inbox, new_reservation_inbox, + validate_batch, }; pub use types::{EngineKeySet, EpochBatch, ReadWriteSet, SequencedTxn, SortedVec, TxClass}; diff --git a/nodedb-cluster/src/calvin/sequencer/config.rs b/nodedb-cluster/src/calvin/sequencer/config.rs index f1bf7dd13..141779ea2 100644 --- a/nodedb-cluster/src/calvin/sequencer/config.rs +++ b/nodedb-cluster/src/calvin/sequencer/config.rs @@ -35,22 +35,43 @@ pub struct SequencerConfig { /// Default: 512. pub vshard_channel_depth: usize, - /// Maximum serialized byte size of the `plans` blob for a single - /// transaction. Transactions exceeding this cap are rejected at the inbox - /// with `TxnTooLarge`. + /// Maximum plan bytes one sequencer entry carries: a single-entry + /// transaction's `plans`, or one part of a multi-part transaction. A + /// coordinator splits a larger transaction into parts. The inbox refuses a + /// single entry or a part above it with `TxnTooLarge`. /// /// Default: 1 MiB (1 << 20). pub max_plans_bytes_per_txn: usize, - /// Maximum number of distinct vShards a single transaction may touch. - /// Transactions exceeding this cap are rejected at the inbox with - /// `FanoutTooWide`. + /// Maximum vShards one sequencer entry's plans target: a single-entry + /// transaction's participants, or one part's targets. A coordinator + /// splits a transaction with more participants into parts. The inbox + /// refuses a part above it with `FanoutTooWide`. A multi-part + /// transaction's participants are bounded only by the vShard count. /// /// Default: 64. pub max_participating_vshards_per_txn: usize, + /// Maximum part bytes the sequencer leader queues across every part + /// stream, streamed in and not yet proposed. A stream that finds the + /// queue full is told to retry, so the leader's memory stays bounded + /// whatever the transaction's size. An empty queue always takes one + /// part. + /// + /// Default: 64 MiB (64 << 20), four epochs of `max_bytes_per_epoch`. + pub max_queued_part_bytes: usize, + + /// How long the leader keeps a part stream that sends nothing. A + /// coordinator that stops streaming leaves its transaction holding + /// locks on every participant, so the leader then abandons it. + /// + /// Default: 30 s. + pub part_stream_stall: Duration, + /// Maximum number of transactions drained from the inbox per epoch. - /// The epoch tick stops draining once this count is reached. + /// The epoch tick stops draining once this count is reached. It must be + /// below [`nodedb_types::MAX_POSITIONS_PER_EPOCH`]: each position owns + /// one nanosecond of its epoch's system-time millisecond. /// /// Default: 1 024. pub max_txns_per_epoch: usize, @@ -102,6 +123,8 @@ impl Default for SequencerConfig { vshard_channel_depth: 512, max_plans_bytes_per_txn: 1 << 20, max_participating_vshards_per_txn: 64, + max_queued_part_bytes: 64 << 20, + part_stream_stall: Duration::from_secs(30), max_txns_per_epoch: 1_024, max_bytes_per_epoch: 16 << 20, tenant_inbox_quota: (inbox_capacity / 8).max(1), @@ -112,9 +135,20 @@ impl Default for SequencerConfig { } impl SequencerConfig { - /// Validate that all caps are `>= 1`. Returns an error describing the - /// first violation found. + /// Validate that all caps are `>= 1` and that an epoch fits its + /// system-time millisecond. Returns an error describing the first + /// violation found. pub fn validate(&self) -> Result<(), ClusterError> { + if self.max_txns_per_epoch >= nodedb_types::MAX_POSITIONS_PER_EPOCH { + return Err(ClusterError::Config { + detail: format!( + "SequencerConfig.max_txns_per_epoch must be < {}, got {}: each position \ + owns one nanosecond of its epoch's millisecond", + nodedb_types::MAX_POSITIONS_PER_EPOCH, + self.max_txns_per_epoch + ), + }); + } let caps: &[(&'static str, usize)] = &[ ("inbox_capacity", self.inbox_capacity), ("vshard_channel_depth", self.vshard_channel_depth), @@ -123,6 +157,7 @@ impl SequencerConfig { "max_participating_vshards_per_txn", self.max_participating_vshards_per_txn, ), + ("max_queued_part_bytes", self.max_queued_part_bytes), ("max_txns_per_epoch", self.max_txns_per_epoch), ("max_bytes_per_epoch", self.max_bytes_per_epoch), ("tenant_inbox_quota", self.tenant_inbox_quota), @@ -190,6 +225,28 @@ mod tests { assert!(cfg.validate().is_err()); } + #[test] + fn validate_rejects_an_epoch_past_its_millisecond() { + let cfg = SequencerConfig { + max_txns_per_epoch: nodedb_types::MAX_POSITIONS_PER_EPOCH, + ..SequencerConfig::default() + }; + let err = cfg + .validate() + .expect_err("an epoch of 1e6 positions is refused"); + let text = err.to_string(); + assert!(text.contains("max_txns_per_epoch"), "{text}"); + assert!(text.contains("1000000"), "{text}"); + + let at_limit = SequencerConfig { + max_txns_per_epoch: nodedb_types::MAX_POSITIONS_PER_EPOCH - 1, + ..SequencerConfig::default() + }; + at_limit + .validate() + .expect("the largest epoch that fits is accepted"); + } + #[test] fn tenant_quota_floor_is_one_for_tiny_capacity() { // (1 / 8).max(1) == 1 — use a variable to avoid constant-folding lint. diff --git a/nodedb-cluster/src/calvin/sequencer/entry.rs b/nodedb-cluster/src/calvin/sequencer/entry.rs index 66b232ff6..63f5dd4c4 100644 --- a/nodedb-cluster/src/calvin/sequencer/entry.rs +++ b/nodedb-cluster/src/calvin/sequencer/entry.rs @@ -23,12 +23,27 @@ use crate::calvin::types::{EpochBatch, LockKeyWire, ReleaseReason, TxnIdWire}; Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack, + rkyv::Archive, + rkyv::Serialize, + rkyv::Deserialize, )] pub enum AbortReason { /// A participant's read-set was stale at validation. SerializationConflict, /// A participant returned an error, so its read-set was never validated. ParticipantError, + /// A collection the transaction names no longer holds the incarnation + /// its coordinator planned against: a purge and a same-name create + /// replaced it. + CollectionSuperseded, + /// A participant found state other than the reconnaissance its + /// coordinator planned against, and wrote nothing. The coordinator reads + /// again and resubmits. + PredictionDrift, + /// A multi-part transaction lost its parts: the sequencer leader that + /// held them changed before it proposed them all. No participant staged + /// the whole transaction, and the coordinator resubmits it. + PartsLost, } /// An entry in the replicated sequencer log. @@ -57,6 +72,16 @@ pub enum SequencerEntry { epoch: u64, position: u32, vshard_id: u32, + /// The participant's apply result for the coordinator, opaque to the + /// sequencer: the host crate encodes and decodes it. Every replica + /// applies the ack, so the coordinator reads it whether or not its + /// node hosts a replica of this vShard. Empty when the apply has no + /// result to report. + result: Vec, + /// The node whose scheduler applied the slice and proposed the ack: + /// the vShard's data-group leader when it proposed. For tracing; the + /// first ack of a vShard in log order answers for it. + from_node: u64, }, /// OLLP predicate-mismatch signal. Proposed by the per-vShard scheduler when /// the active executor returns `OllpRetryRequired`. Applied on ALL sequencer-group @@ -78,30 +103,21 @@ pub enum SequencerEntry { position: u32, detail: String, }, - /// One participant vShard's durable commit vote for a staged cross-shard txn. - /// Generalizes `OllpMismatch` (an implicit vote=abort). Unlike + /// One participant vShard's durable COMMIT vote for a staged cross-shard + /// txn: its read-set is valid. An abort vote is `AbortVote`. Unlike /// `OllpMismatch`/`CompletionAck` which key only on `(epoch, position)`, `Vote` /// carries `vshard` because the verdict aggregator must attribute exactly one - /// vote per participant to know when the tally is complete. `commit: true` - /// is the read-set-valid vote; `commit: false` appears only in log entries - /// written before `AbortVote` existed, and carries no reason. + /// vote per participant to know when the tally is complete. Vote { epoch: u64, position: u32, vshard: u32, - commit: bool, - }, - /// The global commit/abort verdict for a staged cross-shard txn, proposed by - /// the sequencer leader once every participant has voted (see `Vote`). - /// `commit: true` means every participant voted commit. Every replica applies - /// this to store the authoritative decision; participants later flush (commit) - /// or drop (abort) their staged buffer on it. `commit: false` appears only in - /// log entries written before `AbortVerdict` existed. - Verdict { - epoch: u64, - position: u32, - commit: bool, }, + /// The global COMMIT verdict for a staged cross-shard txn, proposed by the + /// sequencer leader once every participant voted commit (see `Vote`). An + /// abort verdict is `AbortVerdict`. Every replica applies it to store the + /// authoritative decision. Participants then flush their staged buffer. + Verdict { epoch: u64, position: u32 }, /// Install a SHARED reservation on `key` for interactive txn `owner` (its /// stable Calvin `(epoch, position)` lock id). Leader-proposed; applied on /// every replica so each installs an identical shared lock. @@ -117,8 +133,6 @@ pub enum SequencerEntry { reason: ReleaseReason, }, /// One participant vShard's durable ABORT vote, carrying why it aborted. - /// `Vote` stays the commit-only vote: adding a field to it would break - /// decode of already-durable log entries, so the reason rides a new variant. AbortVote { epoch: u64, position: u32, @@ -126,8 +140,7 @@ pub enum SequencerEntry { reason: AbortReason, }, /// The global ABORT verdict for a staged cross-shard txn, carrying the - /// winning participant reason. `Verdict` stays the commit-only verdict, for - /// the same wire-compatibility reason as `AbortVote`. + /// winning participant reason. Participants drop their staged buffer. AbortVerdict { epoch: u64, position: u32, @@ -138,7 +151,42 @@ pub enum SequencerEntry { /// log order. A scheduler reports the marker once every transaction /// delivered to it before the marker finished, and gives every /// transaction delivered after it a commit HLC above `hlc`. - CutMarker { hlc: u64 }, + /// + /// `restore_point` names the cluster restore point the cut takes, `0` for + /// a backup's cut. Every replica records the sequencer's place at the + /// point when it applies the marker. + CutMarker { hlc: u64, restore_point: u64 }, + /// The first entry of a sequencer log a cluster restore rebuilt. It sets + /// the next epoch the sequencer proposes to `next_epoch`, the epoch that + /// followed the restore point, so no restored epoch is minted again. It + /// sets the applied epoch instant to `epoch_system_ms`, the highest one + /// applied before the point, so every later epoch instant is above it. + /// `0` when no epoch applied before the point. + EpochFloor { + next_epoch: u64, + epoch_system_ms: i64, + }, + /// Part `index` of the plans of the multi-part transaction sequenced at + /// `(epoch, position)`. The leader proposes every part after the header's + /// epoch batch, in part order. `targets` are the vShards the part's tasks + /// route to, sorted. Only their schedulers receive it. `chunk` is set on + /// a part that holds one byte range of one task. A part of a transaction + /// that is not open (unknown, complete, or abandoned) is ignored. + TxnPart { + epoch: u64, + position: u32, + index: u32, + first_task: u32, + targets: Vec, + plans: Vec, + chunk: Option, + }, + /// The leader's claim that the multi-part transaction at + /// `(epoch, position)` lost parts no leader holds any more. Applied only + /// while the transaction is open: it then aborts with + /// [`AbortReason::PartsLost`]. A transaction whose last part applied + /// before this entry ignores it. + TxnPartsAbandoned { epoch: u64, position: u32 }, } #[cfg(test)] @@ -262,7 +310,6 @@ mod tests { epoch: 5, position: 2, vshard: 9, - commit: true, }; let bytes = zerompk::to_msgpack_vec(&entry).expect("encode"); let decoded: SequencerEntry = zerompk::from_msgpack(&bytes).expect("decode"); @@ -274,7 +321,6 @@ mod tests { let entry = SequencerEntry::Verdict { epoch: 5, position: 2, - commit: false, }; let bytes = zerompk::to_msgpack_vec(&entry).expect("encode"); let decoded: SequencerEntry = zerompk::from_msgpack(&bytes).expect("decode"); @@ -306,58 +352,6 @@ mod tests { assert_eq!(entry, decoded); } - #[test] - fn vote_and_verdict_keep_their_pre_existing_wire_shape() { - // Guards the compat claim behind adding `AbortVote`/`AbortVerdict`: - // already-durable log entries must still decode. Variant payloads are - // positional arrays under a strict length check, so the name tag and the - // field count are what a field addition would have broken. - let vote = zerompk::to_msgpack_vec(&SequencerEntry::Vote { - epoch: 5, - position: 2, - vshard: 9, - commit: true, - }) - .expect("encode"); - assert_eq!(vote[0], 0x92, "enum stays a 2-element [tag, payload] array"); - assert_eq!(&vote[1..6], b"\xA4Vote", "fixstr(4) name tag"); - assert_eq!(vote[6], 0x94, "Vote payload must stay a 4-element array"); - - let verdict = zerompk::to_msgpack_vec(&SequencerEntry::Verdict { - epoch: 5, - position: 2, - commit: false, - }) - .expect("encode"); - assert_eq!(&verdict[1..9], b"\xA7Verdict"); - assert_eq!( - verdict[9], 0x93, - "Verdict payload must stay a 3-element array" - ); - - // And both still round-trip through the widened enum. - let decoded: SequencerEntry = zerompk::from_msgpack(&vote).expect("decode legacy Vote"); - assert_eq!( - decoded, - SequencerEntry::Vote { - epoch: 5, - position: 2, - vshard: 9, - commit: true, - } - ); - let decoded: SequencerEntry = - zerompk::from_msgpack(&verdict).expect("decode legacy Verdict"); - assert_eq!( - decoded, - SequencerEntry::Verdict { - epoch: 5, - position: 2, - commit: false, - } - ); - } - #[test] fn reserve_read_msgpack_roundtrip() { let entry = SequencerEntry::ReserveRead { diff --git a/nodedb-cluster/src/calvin/sequencer/entry_limits.rs b/nodedb-cluster/src/calvin/sequencer/entry_limits.rs new file mode 100644 index 000000000..ffb09d8cf --- /dev/null +++ b/nodedb-cluster/src/calvin/sequencer/entry_limits.rs @@ -0,0 +1,454 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The shape a submitted transaction and its streamed parts must have for +//! the sequencer to carry them. +//! +//! The per-entry caps bound one sequencer log entry, never a transaction: +//! +//! - A single-entry transaction carries its plans in `plans`. They fit one +//! entry: at most `max_plans_bytes` bytes, for at most +//! `max_participating_vshards` participants. +//! - A multi-part transaction's header carries no plans. Its manifest names +//! how many parts target each participant. +//! - Each streamed part fits one entry the same way, counting its targets. +//! [`PartCursor`] checks the parts in order against the manifest. No cap +//! bounds the whole transaction. + +use std::collections::BTreeMap; + +use crate::calvin::sequencer::error::SequencerError; +use crate::calvin::types::{MultiPartPlans, StreamedPart, TxClass}; + +/// The caps one sequencer entry obeys. +#[derive(Debug, Clone, Copy)] +pub(crate) struct EntryLimits { + pub max_plans_bytes: usize, + pub max_participating_vshards: usize, +} + +/// Refuse a transaction the sequencer cannot carry. +pub(crate) fn check_entry_shape( + tx_class: &TxClass, + limits: EntryLimits, +) -> Result<(), SequencerError> { + let Some(manifest) = &tx_class.multi_part else { + if tx_class.plans.len() > limits.max_plans_bytes { + return Err(SequencerError::TxnTooLarge { + bytes: tx_class.plans.len(), + limit: limits.max_plans_bytes, + }); + } + let vshards = tx_class.participating_vshards().len(); + if vshards > limits.max_participating_vshards { + return Err(SequencerError::FanoutTooWide { + vshards, + limit: limits.max_participating_vshards, + }); + } + return Ok(()); + }; + if !tx_class.plans.is_empty() { + return Err(malformed("the header carries plans beside its parts")); + } + check_manifest(tx_class, manifest) +} + +fn check_manifest(tx_class: &TxClass, manifest: &MultiPartPlans) -> Result<(), SequencerError> { + if manifest.part_count == 0 || manifest.total_tasks == 0 || manifest.per_vshard.is_empty() { + return Err(malformed("the manifest names no parts, tasks or targets")); + } + let sorted = manifest + .per_vshard + .windows(2) + .all(|pair| pair[0].vshard < pair[1].vshard); + if !sorted { + return Err(malformed("the manifest's vShards are unsorted")); + } + let participants = tx_class.participating_vshards(); + for entry in &manifest.per_vshard { + if entry.parts == 0 || entry.parts > manifest.part_count { + return Err(malformed(&format!( + "vShard {} is owed {} of {} parts", + entry.vshard, entry.parts, manifest.part_count + ))); + } + if participants + .binary_search_by_key(&entry.vshard, |vshard| vshard.as_u32()) + .is_err() + { + return Err(malformed(&format!( + "the manifest targets vShard {}, which does not participate", + entry.vshard + ))); + } + } + Ok(()) +} + +/// A task split over chunk parts, part way through. +#[derive(Debug, Clone, Copy)] +struct OpenChunk { + task: u32, + next_offset: u64, + total_len: u64, +} + +/// The check of one stream's parts, in index order, against its manifest. +#[derive(Debug)] +pub(crate) struct PartCursor { + part_count: u32, + total_tasks: u32, + next_index: u32, + /// Parts each vShard is still owed. + owed: BTreeMap, + last_first_task: Option, + open_chunk: Option, +} + +impl PartCursor { + pub(crate) fn new(manifest: &MultiPartPlans) -> Self { + Self { + part_count: manifest.part_count, + total_tasks: manifest.total_tasks, + next_index: 0, + owed: manifest + .per_vshard + .iter() + .map(|entry| (entry.vshard, entry.parts)) + .collect(), + last_first_task: None, + open_chunk: None, + } + } + + /// The index of the next part the stream owes. + pub(crate) fn next_index(&self) -> u32 { + self.next_index + } + + /// Whether every part arrived. + pub(crate) fn is_complete(&self) -> bool { + self.next_index == self.part_count + } + + /// Take `streamed`, the stream's next part. A refused part changes + /// nothing. + pub(crate) fn admit( + &mut self, + streamed: &StreamedPart, + limits: EntryLimits, + ) -> Result<(), SequencerError> { + let part = &streamed.part; + let targets = &streamed.targets; + if streamed.index != self.next_index || self.next_index >= self.part_count { + return Err(malformed(&format!( + "part {} arrived where part {} of {} was owed", + streamed.index, self.next_index, self.part_count + ))); + } + if part.plans.len() > limits.max_plans_bytes { + return Err(SequencerError::TxnTooLarge { + bytes: part.plans.len(), + limit: limits.max_plans_bytes, + }); + } + if targets.len() > limits.max_participating_vshards { + return Err(SequencerError::FanoutTooWide { + vshards: targets.len(), + limit: limits.max_participating_vshards, + }); + } + if part.plans.is_empty() || targets.is_empty() { + return Err(malformed("a part carries no plans or no targets")); + } + if targets.windows(2).any(|pair| pair[0] >= pair[1]) { + return Err(malformed("a part's targets are unsorted")); + } + if let Some(stray) = targets + .iter() + .find(|target| self.owed.get(target).is_none_or(|owed| *owed == 0)) + { + return Err(malformed(&format!( + "a part targets vShard {stray}, which is owed no more parts" + ))); + } + let (open_chunk, last_first_task) = self.next_task_state(streamed)?; + let last = streamed.index + 1 == self.part_count; + if last { + let owed_after = + |vshard: &u32, owed: &u32| *owed - u32::from(targets.binary_search(vshard).is_ok()); + if open_chunk.is_some() || self.owed.iter().any(|(v, o)| owed_after(v, o) > 0) { + return Err(malformed( + "the last part leaves a task or a vShard's parts unfinished", + )); + } + } + for target in targets { + if let Some(owed) = self.owed.get_mut(target) { + *owed -= 1; + } + } + self.open_chunk = open_chunk; + self.last_first_task = Some(last_first_task); + self.next_index += 1; + Ok(()) + } + + /// The chunk state and last first task after `streamed`, or why its + /// tasks do not follow the ones before it. + fn next_task_state( + &self, + streamed: &StreamedPart, + ) -> Result<(Option, u32), SequencerError> { + let part = &streamed.part; + let len = part.plans.len() as u64; + if let Some(open) = self.open_chunk { + let continues = part.chunk.is_some_and(|chunk| { + part.first_task == open.task + && chunk.offset == open.next_offset + && chunk.total_len == open.total_len + }); + if !continues { + return Err(malformed(&format!( + "part {} does not continue task {} at byte {}", + streamed.index, open.task, open.next_offset + ))); + } + return Ok((advance(open, len)?, open.task)); + } + let in_order = match self.last_first_task { + None => part.first_task == 0, + Some(last) => part.first_task > last, + }; + if !in_order || part.first_task >= self.total_tasks { + return Err(malformed(&format!( + "part {} starts at task {} of {}, out of order", + streamed.index, part.first_task, self.total_tasks + ))); + } + let open_chunk = match part.chunk { + None => None, + Some(chunk) if chunk.offset == 0 && chunk.total_len > 0 => advance( + OpenChunk { + task: part.first_task, + next_offset: 0, + total_len: chunk.total_len, + }, + len, + )?, + Some(_) => return Err(malformed("a task's first chunk does not start at byte 0")), + }; + Ok((open_chunk, part.first_task)) + } +} + +/// `open` after `len` more bytes: still open, or `None` once whole. +fn advance(open: OpenChunk, len: u64) -> Result, SequencerError> { + let next_offset = open.next_offset.saturating_add(len); + if next_offset > open.total_len { + return Err(malformed(&format!( + "task {} chunks run past its {} bytes", + open.task, open.total_len + ))); + } + Ok((next_offset < open.total_len).then_some(OpenChunk { + next_offset, + ..open + })) +} + +fn malformed(detail: &str) -> SequencerError { + SequencerError::MalformedParts { + detail: detail.to_owned(), + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::TenantId; + use nodedb_types::id::{CollectionKey, DatabaseId}; + + use super::*; + use crate::calvin::types::{ + EngineKeySet, PartStreamId, PlanPart, ReadWriteSet, SortedVec, TaskChunk, VShardParts, + VersionedReadSet, + }; + + const LIMITS: EntryLimits = EntryLimits { + max_plans_bytes: 4, + max_participating_vshards: 1, + }; + + /// A class over two collections on distinct vShards, and those vShards. + fn two_vshard_class() -> (TxClass, u32, u32) { + let mut seen: Option<(String, u32)> = None; + for i in 0u32..512 { + let name = format!("col_{i}"); + let vshard = CollectionKey::from_bare(DatabaseId::DEFAULT, &name) + .vshard() + .as_u32(); + let Some((first, first_vshard)) = seen.clone() else { + seen = Some((name, vshard)); + continue; + }; + if first_vshard == vshard { + continue; + } + let write_set = ReadWriteSet::new(vec![ + EngineKeySet::Document { + collection: first, + surrogates: SortedVec::new(vec![1]), + }, + EngineKeySet::Document { + collection: name, + surrogates: SortedVec::new(vec![2]), + }, + ]); + let class = TxClass::new( + ReadWriteSet::new(vec![]), + write_set, + Vec::new(), + TenantId::new(1), + None, + VersionedReadSet::default(), + ) + .expect("valid class"); + return (class, first_vshard.min(vshard), first_vshard.max(vshard)); + } + panic!("no two distinct-vshard collections in 512 tries"); + } + + fn manifest(part_count: u32, total_tasks: u32, per_vshard: &[(u32, u32)]) -> MultiPartPlans { + MultiPartPlans { + stream: PartStreamId { node: 1, seq: 1 }, + part_count, + total_tasks, + user_write: true, + client_write: true, + per_vshard: per_vshard + .iter() + .map(|&(vshard, parts)| VShardParts { vshard, parts }) + .collect(), + } + } + + fn part(index: u32, first_task: u32, bytes: usize, target: u32) -> StreamedPart { + StreamedPart { + index, + targets: vec![target], + part: PlanPart { + first_task, + plans: vec![7; bytes], + chunk: None, + }, + } + } + + fn chunk( + index: u32, + task: u32, + offset: u64, + bytes: usize, + total: u64, + target: u32, + ) -> StreamedPart { + let mut streamed = part(index, task, bytes, target); + streamed.part.chunk = Some(TaskChunk { + offset, + total_len: total, + }); + streamed + } + + #[test] + fn a_single_entry_over_the_entry_caps_is_refused() { + let (mut class, _, _) = two_vshard_class(); + assert!(matches!( + check_entry_shape(&class, LIMITS), + Err(SequencerError::FanoutTooWide { vshards: 2, .. }) + )); + class.plans = vec![0; 5]; + let wide = EntryLimits { + max_participating_vshards: 8, + ..LIMITS + }; + assert!(matches!( + check_entry_shape(&class, wide), + Err(SequencerError::TxnTooLarge { bytes: 5, .. }) + )); + } + + #[test] + fn a_header_manifest_must_name_participants() { + let (mut class, low, high) = two_vshard_class(); + class.multi_part = Some(manifest(2, 2, &[(low, 1), (high, 1)])); + check_entry_shape(&class, LIMITS).expect("a header over both participants"); + for bad in [ + manifest(0, 2, &[(low, 1)]), + manifest(2, 2, &[(high, 1), (low, 1)]), + manifest(2, 2, &[(low, 3)]), + manifest(2, 2, &[(u32::MAX, 1)]), + ] { + class.multi_part = Some(bad); + assert!(matches!( + check_entry_shape(&class, LIMITS), + Err(SequencerError::MalformedParts { .. }) + )); + } + } + + /// Whole-task parts and a chunked task, each within the entry caps, + /// carry a transaction over them. + #[test] + fn parts_that_each_fit_an_entry_carry_a_larger_transaction() { + let (_, low, high) = two_vshard_class(); + let mut cursor = PartCursor::new(&manifest(4, 3, &[(low, 1), (high, 3)])); + let parts = [ + part(0, 0, 4, low), + chunk(1, 1, 0, 4, 10, high), + chunk(2, 1, 4, 4, 10, high), + chunk(3, 1, 8, 2, 10, high), + ]; + for streamed in &parts { + cursor.admit(streamed, LIMITS).expect("fits"); + } + assert!(cursor.is_complete()); + } + + #[test] + fn a_part_that_breaks_the_stream_is_refused_and_changes_nothing() { + let (_, low, high) = two_vshard_class(); + let m = manifest(2, 2, &[(low, 1), (high, 1)]); + let cases = [ + (part(1, 0, 1, low), "an index gap"), + (part(0, 1, 1, low), "a first part past task 0"), + (part(0, 0, 5, low), "over the byte cap"), + (part(0, 0, 1, u32::MAX), "a stray target"), + (chunk(0, 0, 3, 1, 4, low), "a first chunk past byte 0"), + ]; + for (streamed, what) in &cases { + let mut cursor = PartCursor::new(&m); + assert!(cursor.admit(streamed, LIMITS).is_err(), "{what}"); + assert_eq!(cursor.next_index(), 0, "{what}"); + } + + let mut cursor = PartCursor::new(&m); + cursor.admit(&part(0, 0, 1, low), LIMITS).expect("first"); + assert!( + cursor.admit(&part(1, 1, 1, low), LIMITS).is_err(), + "the last part leaves the other vShard's part owed" + ); + let mut chunked = PartCursor::new(&manifest(2, 1, &[(low, 2)])); + chunked + .admit(&chunk(0, 0, 0, 2, 5, low), LIMITS) + .expect("first chunk"); + assert!( + chunked.admit(&chunk(1, 0, 3, 2, 5, low), LIMITS).is_err(), + "a chunk that skips bytes" + ); + assert!( + chunked.admit(&chunk(1, 0, 2, 2, 5, low), LIMITS).is_err(), + "the last chunk leaves the task unfinished" + ); + } +} diff --git a/nodedb-cluster/src/calvin/sequencer/error.rs b/nodedb-cluster/src/calvin/sequencer/error.rs index 3b74a85d4..34999da05 100644 --- a/nodedb-cluster/src/calvin/sequencer/error.rs +++ b/nodedb-cluster/src/calvin/sequencer/error.rs @@ -52,12 +52,18 @@ pub enum SequencerError { #[error("transaction plans blob is {bytes} bytes, exceeds limit of {limit} bytes")] TxnTooLarge { bytes: usize, limit: usize }, - /// The transaction touches more vShards than the configured fan-out cap. + /// A single-entry transaction or one part targets more vShards than one + /// sequencer entry carries. /// - /// The caller must split the transaction or reduce its write set. - #[error("transaction touches {vshards} vshards, exceeds limit of {limit}")] + /// The coordinator must split the plans into parts that fit. + #[error("transaction entry targets {vshards} vshards, exceeds limit of {limit}")] FanoutTooWide { vshards: usize, limit: usize }, + /// A multi-part transaction's manifest or one of its streamed parts is + /// malformed. + #[error("multi-part transaction is malformed: {detail}")] + MalformedParts { detail: String }, + /// The submitting tenant already has `in_flight >= quota` transactions /// in the inbox. Back off and retry. #[error("tenant {tenant} inbox quota exceeded: {in_flight} in flight, quota {quota}")] diff --git a/nodedb-cluster/src/calvin/sequencer/inbox.rs b/nodedb-cluster/src/calvin/sequencer/inbox.rs index 630518de9..50d9ab910 100644 --- a/nodedb-cluster/src/calvin/sequencer/inbox.rs +++ b/nodedb-cluster/src/calvin/sequencer/inbox.rs @@ -17,9 +17,11 @@ use std::sync::atomic::{AtomicU64, Ordering}; use tokio::sync::mpsc; use tracing::warn; +use crate::calvin::completion::{AssignmentReceiver, CalvinCompletionRegistry}; use crate::calvin::sequencer::config::SequencerConfig; use crate::calvin::sequencer::error::SequencerError; -use crate::calvin::types::TxClass; +use crate::calvin::sequencer::parts_intake::{PartsIntake, PartsOffer}; +use crate::calvin::types::{PartStreamId, StreamedPart, TxClass}; use nodedb_types::TenantId; @@ -69,6 +71,8 @@ pub struct Inbox { tenant_in_flight: Arc>>, max_plans_bytes: usize, max_participating_vshards: usize, + /// The leader's queue of streamed parts, shared with the receiver. + parts: Arc, tenant_quota: usize, max_dependent_read_bytes: usize, max_dependent_read_passives: usize, @@ -90,6 +94,8 @@ pub struct InboxReceiver { /// `depth_counter` are NOT decremented while an item sits here — they are /// decremented when the item is finally moved into the output vector. pending: Option, + /// The leader's queue of streamed parts, shared with every sender. + parts: Arc, } impl Inbox { @@ -98,8 +104,10 @@ impl Inbox { /// Checks are performed in fail-fast order; no state mutation occurs on /// rejection: /// - /// 1. Plans blob too large → `TxnTooLarge`. - /// 2. Too many participating vShards → `FanoutTooWide`. + /// 1. A single entry above the per-entry byte cap → `TxnTooLarge`. A + /// multi-part header with plans or a malformed manifest → + /// `MalformedParts`. + /// 2. A single entry above the per-entry vShard cap → `FanoutTooWide`. /// 3. Dependent-read payload too large → `DependentReadTooLarge`. /// 4. Dependent-read fan-in too wide → `DependentReadFanoutTooWide`. /// 5. Tenant in-flight quota exceeded → `TenantQuotaExceeded`. @@ -109,22 +117,44 @@ impl Inbox { /// /// This call is **non-blocking**: it never waits for the epoch ticker. pub fn submit(&self, tx_class: TxClass) -> Result { - // Check 1: plans byte size. - if tx_class.plans.len() > self.max_plans_bytes { - return Err(SequencerError::TxnTooLarge { - bytes: tx_class.plans.len(), - limit: self.max_plans_bytes, - }); - } + self.admit(tx_class, |_| (), |_| ()).map(|(seq, ())| seq) + } - // Check 2: vshard fan-out. - let vshards = tx_class.participating_vshards().len(); - if vshards > self.max_participating_vshards { - return Err(SequencerError::FanoutTooWide { - vshards, - limit: self.max_participating_vshards, - }); - } + /// [`Self::submit`], registering for the submission's assignment in + /// `registry` before the submission reaches the channel. + /// + /// The epoch tick can drain the submission the moment it is sent. Its + /// `note_assigned` then always finds the sender, so no assignment is lost. + /// A send that fails drops the registration again. + pub fn submit_with( + &self, + tx_class: TxClass, + registry: &CalvinCompletionRegistry, + ) -> Result<(u64, AssignmentReceiver), SequencerError> { + self.admit( + tx_class, + |seq| registry.register_submission(seq), + |seq| registry.drop_assignment(seq), + ) + } + + /// The submit body. `register` runs with the assigned seq before the + /// send. `unregister` runs with it when the send fails. + fn admit( + &self, + tx_class: TxClass, + register: impl FnOnce(u64) -> R, + unregister: impl FnOnce(u64), + ) -> Result<(u64, R), SequencerError> { + // Checks 1 & 2: a single entry fits one entry, and a multi-part + // header is well formed. Its parts are checked as they stream in. + crate::calvin::sequencer::entry_limits::check_entry_shape( + &tx_class, + crate::calvin::sequencer::entry_limits::EntryLimits { + max_plans_bytes: self.max_plans_bytes, + max_participating_vshards: self.max_participating_vshards, + }, + )?; // Checks 3 & 4: dependent-read caps. if let Some(spec) = &tx_class.dependent_reads { @@ -180,12 +210,14 @@ impl Inbox { inbox_seq = seq, ); + let registered = register(seq); match self.tx.try_send(admitted) { Ok(()) => { self.depth_counter.fetch_add(1, Ordering::Relaxed); - Ok(seq) + Ok((seq, registered)) } Err(e) => { + unregister(seq); // Roll back the tenant counter because the message was never // enqueued. { @@ -212,6 +244,12 @@ impl Inbox { pub fn depth(&self) -> usize { self.depth_counter.load(Ordering::Relaxed) as usize } + + /// Offer streamed parts of a multi-part transaction to this node's + /// sequencer leader queue. + pub fn offer_parts(&self, stream: PartStreamId, parts: Vec) -> PartsOffer { + self.parts.offer(stream, parts) + } } impl InboxReceiver { @@ -276,26 +314,28 @@ impl InboxReceiver { /// Drain and discard all items including the `pending` slot. /// - /// Used by the non-leader discard path. Decrements `tenant_in_flight` and - /// `depth_counter` per item. Returns the total count discarded. - pub fn drain_all_discard(&mut self) -> usize { - let mut count = 0; + /// Used by the non-leader and halted discard paths. Decrements + /// `tenant_in_flight` and `depth_counter` per item. Returns the + /// `inbox_seq` of every discarded item, so the caller can drop each + /// one's assignment. + pub fn drain_all_discard(&mut self) -> Vec { + let mut discarded = Vec::new(); // Discard the pending slot. if let Some(pending) = self.pending.take() { self.decrement_tenant(&pending.tx_class.tenant_id); self.depth_counter.fetch_sub(1, Ordering::Relaxed); - count += 1; + discarded.push(pending.inbox_seq); } // Drain the channel. while let Ok(tx) = self.rx.try_recv() { self.decrement_tenant(&tx.tx_class.tenant_id); self.depth_counter.fetch_sub(1, Ordering::Relaxed); - count += 1; + discarded.push(tx.inbox_seq); } - count + discarded } /// The inbox's configured capacity. @@ -303,6 +343,11 @@ impl InboxReceiver { self.capacity } + /// The leader's queue of streamed parts. + pub fn parts_intake(&self) -> Arc { + Arc::clone(&self.parts) + } + /// Approximate current depth (items queued). pub fn depth(&self) -> u64 { self.depth_counter.load(Ordering::Relaxed) @@ -329,6 +374,7 @@ pub fn new_inbox(capacity: usize, config: &SequencerConfig) -> (Inbox, InboxRece let next_seq = Arc::new(AtomicU64::new(0)); let depth_counter = Arc::new(AtomicU64::new(0)); let tenant_in_flight: Arc>> = Arc::new(Mutex::new(BTreeMap::new())); + let parts = Arc::new(PartsIntake::new(config)); let inbox = Inbox { tx, @@ -337,6 +383,7 @@ pub fn new_inbox(capacity: usize, config: &SequencerConfig) -> (Inbox, InboxRece tenant_in_flight: Arc::clone(&tenant_in_flight), max_plans_bytes: config.max_plans_bytes_per_txn, max_participating_vshards: config.max_participating_vshards_per_txn, + parts: Arc::clone(&parts), tenant_quota: config.tenant_inbox_quota, max_dependent_read_bytes: config.max_dependent_read_bytes_per_txn, max_dependent_read_passives: config.max_dependent_read_passives_per_txn, @@ -347,6 +394,7 @@ pub fn new_inbox(capacity: usize, config: &SequencerConfig) -> (Inbox, InboxRece depth_counter, tenant_in_flight, pending: None, + parts, }; (inbox, receiver) } @@ -466,6 +514,63 @@ mod tests { assert_eq!(n2, 0); } + /// The tick can drain a submission before the submitter's next line runs. + /// Registering inside `submit_with` means the assignment still reaches the + /// submitter. + #[test] + fn submit_with_delivers_the_assignment_to_a_drain_right_after_submit() { + let registry = CalvinCompletionRegistry::new_detached(); + let (inbox, mut rx) = new_inbox(10, &default_config()); + let (seq, mut assignment) = inbox + .submit_with(make_tx_class(), ®istry) + .expect("submit"); + + let mut out = Vec::new(); + assert_eq!(rx.drain_into_capped(&mut out, 100, usize::MAX), 1); + registry.note_assigned(out[0].inbox_seq, crate::calvin::TxnId::new(3, 0), 2); + + assert_eq!(out[0].inbox_seq, seq); + assert_eq!(assignment.try_recv(), Ok((3, 0, 2))); + } + + /// A submission the channel refuses holds no assignment sender. + #[test] + fn submit_with_drops_the_registration_when_the_send_fails() { + let registry = CalvinCompletionRegistry::new_detached(); + let config = SequencerConfig { + inbox_capacity: 1, + tenant_inbox_quota: 10, + ..default_config() + }; + let (inbox, _rx) = new_inbox(1, &config); + inbox + .submit_with(make_tx_class(), ®istry) + .expect("first submit fills the channel"); + let err = inbox + .submit_with(make_tx_class(), ®istry) + .expect_err("channel full"); + assert_eq!(err, SequencerError::Overloaded); + assert_eq!(registry.pending_assignments_len(), 1); + } + + #[test] + fn drain_all_discard_returns_every_discarded_seq() { + let (inbox, mut rx) = new_inbox(10, &default_config()); + for _ in 0..3 { + inbox.submit(make_tx_class()).expect("submit"); + } + // Park one item in the pending slot: a zero-byte cap defers it. + let mut tx = make_tx_class(); + tx.plans = vec![0u8; 4]; + inbox.submit(tx).expect("submit"); + let mut out = Vec::new(); + rx.drain_into_capped(&mut out, 100, 0); + assert_eq!(out.len(), 3); + + assert_eq!(rx.drain_all_discard(), vec![3]); + assert_eq!(rx.depth(), 0); + } + #[test] fn capacity_is_reported_correctly() { let config = default_config(); diff --git a/nodedb-cluster/src/calvin/sequencer/mod.rs b/nodedb-cluster/src/calvin/sequencer/mod.rs index cbf566fcc..60367d229 100644 --- a/nodedb-cluster/src/calvin/sequencer/mod.rs +++ b/nodedb-cluster/src/calvin/sequencer/mod.rs @@ -2,10 +2,12 @@ pub mod config; pub mod entry; +pub(crate) mod entry_limits; pub mod epoch_guard; pub mod error; pub mod inbox; pub mod metrics; +pub mod parts_intake; pub mod replay; pub mod reservation_inbox; pub mod service; @@ -18,9 +20,13 @@ pub use epoch_guard::{EpochCheck, SequencerHalt, UnrecoverableEpochHook}; pub use error::SequencerError; pub use inbox::{AdmittedTx, Inbox, InboxReceiver, RejectedTx, new_inbox}; pub use metrics::{ConflictKey, SequencerMetrics}; +pub use parts_intake::{PartsIntake, PartsOffer, PartsOfferStatus}; pub use reservation_inbox::{ ReservationInbox, ReservationInboxReceiver, ReservationRequest, new_reservation_inbox, }; pub use service::{SequencerReceivers, SequencerService}; -pub use state_machine::SequencerStateMachine; +pub use state_machine::{ + CutInstantHook, RestorePointHook, SequencerRestorePoint, SequencerSnapshot, + SequencerStateMachine, +}; pub use validator::validate_batch; diff --git a/nodedb-cluster/src/calvin/sequencer/parts_intake.rs b/nodedb-cluster/src/calvin/sequencer/parts_intake.rs new file mode 100644 index 000000000..db2e60a8c --- /dev/null +++ b/nodedb-cluster/src/calvin/sequencer/parts_intake.rs @@ -0,0 +1,371 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The sequencer leader's bounded queue of streamed parts. +//! +//! A coordinator submits a multi-part transaction's header, then streams its +//! parts here in order. The leader proposes queued parts on its epoch ticks. +//! +//! - A stream opens when the leader proposes its header's epoch batch. An +//! offer to a stream that is not open is answered `Unknown`. +//! - Each part is checked in order against the header's manifest. A +//! malformed part closes the stream and is answered `Rejected`. The leader +//! then abandons the transaction. +//! - The queue holds at most `max_queued_part_bytes` across every stream. +//! An offer that pushes the total past it is answered `Full`, and the coordinator +//! retries. An empty queue always takes one part, so a stream never +//! stalls on the cap. +//! - A part offered twice is taken once. The answer names the next part the +//! stream owes, so a coordinator resumes where the leader stands. +//! - A stream that sends nothing for `part_stream_stall` before its last +//! part is closed, and the leader abandons its transaction. + +use std::collections::{BTreeMap, VecDeque}; +use std::sync::Mutex; +use std::time::{Duration, Instant}; + +use crate::calvin::TxnId; +use crate::calvin::sequencer::config::SequencerConfig; +use crate::calvin::sequencer::entry_limits::{EntryLimits, PartCursor}; +use crate::calvin::types::{MultiPartPlans, PartStreamId, StreamedPart}; + +/// How the leader answered an offer of parts. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +#[repr(u8)] +pub enum PartsOfferStatus { + /// Every offered part the stream owed is queued. + Accepted = 0, + /// The queue is full. Retry from `next_index`. + Full = 1, + /// This leader holds no such stream: the header is not proposed yet, or + /// the stream closed. + Unknown = 2, + /// A part is malformed. The stream is closed and the transaction aborts. + Rejected = 3, +} + +impl PartsOfferStatus { + /// The status a wire code names, or `None` for an unknown code. + pub fn from_wire(code: u8) -> Option { + match code { + 0 => Some(Self::Accepted), + 1 => Some(Self::Full), + 2 => Some(Self::Unknown), + 3 => Some(Self::Rejected), + _ => None, + } + } +} + +/// The answer to an offer of parts. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct PartsOffer { + pub status: PartsOfferStatus, + /// The index of the next part the stream owes. + pub next_index: u32, + /// Why a part was rejected. + pub detail: Option, +} + +/// One open stream. +#[derive(Debug)] +struct Stream { + txn: TxnId, + /// The leader term the header was proposed in. + term: u64, + cursor: PartCursor, + unproposed: VecDeque, + /// no-determinism: stall detection only, never in the log. + last_offer: Instant, +} + +#[derive(Debug, Default)] +struct IntakeState { + streams: BTreeMap, + /// Streams in header order, for fair proposal. + order: VecDeque, + queued_bytes: usize, +} + +impl IntakeState { + fn remove(&mut self, stream: PartStreamId) -> Option { + let removed = self.streams.remove(&stream)?; + self.order.retain(|id| *id != stream); + let bytes: usize = removed.unproposed.iter().map(part_bytes).sum(); + self.queued_bytes = self.queued_bytes.saturating_sub(bytes); + Some(removed) + } +} + +fn part_bytes(part: &StreamedPart) -> usize { + part.part.plans.len() +} + +/// The leader's queue of streamed parts, shared by the inbox that takes +/// offers and the service that proposes them. +#[derive(Debug)] +pub struct PartsIntake { + state: Mutex, + limits: EntryLimits, + max_queued_bytes: usize, + stall: Duration, +} + +impl PartsIntake { + pub fn new(config: &SequencerConfig) -> Self { + Self { + state: Mutex::new(IntakeState::default()), + limits: EntryLimits { + max_plans_bytes: config.max_plans_bytes_per_txn, + max_participating_vshards: config.max_participating_vshards_per_txn, + }, + max_queued_bytes: config.max_queued_part_bytes, + stall: config.part_stream_stall, + } + } + + fn lock(&self) -> std::sync::MutexGuard<'_, IntakeState> { + self.state.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Queue `parts` of `stream`, in index order, as far as the stream and + /// the queue take them. + pub fn offer(&self, stream: PartStreamId, parts: Vec) -> PartsOffer { + let mut guard = self.lock(); + let state = &mut *guard; + let Some(open) = state.streams.get_mut(&stream) else { + return PartsOffer { + status: PartsOfferStatus::Unknown, + next_index: 0, + detail: None, + }; + }; + // no-determinism: stall detection only, never in the log. + open.last_offer = Instant::now(); + let mut status = PartsOfferStatus::Accepted; + for part in parts { + let next = open.cursor.next_index(); + if part.index < next { + continue; + } + if part.index > next { + break; + } + let bytes = part_bytes(&part); + if state.queued_bytes > 0 + && state.queued_bytes.saturating_add(bytes) > self.max_queued_bytes + { + status = PartsOfferStatus::Full; + break; + } + if let Err(error) = open.cursor.admit(&part, self.limits) { + let next_index = open.cursor.next_index(); + state.remove(stream); + return PartsOffer { + status: PartsOfferStatus::Rejected, + next_index, + detail: Some(error.to_string()), + }; + } + open.unproposed.push_back(part); + state.queued_bytes += bytes; + } + PartsOffer { + status, + next_index: open.cursor.next_index(), + detail: None, + } + } + + /// Open `stream` for the header of `txn`, proposed in `term`. + pub(crate) fn open( + &self, + stream: PartStreamId, + txn: TxnId, + term: u64, + manifest: &MultiPartPlans, + ) { + let mut state = self.lock(); + state.remove(stream); + state.order.push_back(stream); + state.streams.insert( + stream, + Stream { + txn, + term, + cursor: PartCursor::new(manifest), + unproposed: VecDeque::new(), + // no-determinism: stall detection only, never in the log. + last_offer: Instant::now(), + }, + ); + } + + /// Close every stream. Returns how many were open. + pub(crate) fn clear(&self) -> usize { + let mut state = self.lock(); + let open = state.streams.len(); + *state = IntakeState::default(); + open + } + + /// Close every stream opened in a term other than `term`, and every + /// stream whose coordinator sent nothing for the stall window before its + /// last part. Returns how many closed. + pub(crate) fn close_stale(&self, term: Option, now: Instant) -> usize { + let mut state = self.lock(); + let stale: Vec = state + .streams + .iter() + .filter(|(_, open)| { + Some(open.term) != term + || (!open.cursor.is_complete() + && now.saturating_duration_since(open.last_offer) > self.stall) + }) + .map(|(id, _)| *id) + .collect(); + for id in &stale { + state.remove(*id); + } + stale.len() + } + + /// Take the next part to propose: the front part of the first stream, + /// in header order, that has one. + pub(crate) fn next_part(&self) -> Option<(PartStreamId, TxnId, StreamedPart)> { + let mut guard = self.lock(); + let state = &mut *guard; + let (id, open) = state.order.iter().find_map(|id| { + let open = state.streams.get(id)?; + (!open.unproposed.is_empty()).then_some((*id, open.txn)) + })?; + let part = state.streams.get_mut(&id)?.unproposed.pop_front()?; + state.queued_bytes = state.queued_bytes.saturating_sub(part_bytes(&part)); + Some((id, open, part)) + } + + /// Put back a part whose proposal failed, at the front of its stream. + pub(crate) fn return_part(&self, stream: PartStreamId, part: StreamedPart) { + let mut guard = self.lock(); + let state = &mut *guard; + if let Some(open) = state.streams.get_mut(&stream) { + state.queued_bytes += part_bytes(&part); + open.unproposed.push_front(part); + } + } + + /// Close every stream whose parts are all proposed and for which `done` + /// holds. + pub(crate) fn close_finished(&self, done: impl Fn(TxnId) -> bool) { + let mut state = self.lock(); + let finished: Vec = state + .streams + .iter() + .filter(|(_, open)| { + open.cursor.is_complete() && open.unproposed.is_empty() && done(open.txn) + }) + .map(|(id, _)| *id) + .collect(); + for id in finished { + state.remove(id); + } + } + + /// Whether an open stream carries the parts of `txn`. + pub(crate) fn holds(&self, txn: TxnId) -> bool { + self.lock().streams.values().any(|open| open.txn == txn) + } + + /// Part bytes queued and not yet proposed. + pub fn queued_bytes(&self) -> usize { + self.lock().queued_bytes + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::calvin::types::{PlanPart, VShardParts}; + + const STREAM: PartStreamId = PartStreamId { node: 2, seq: 9 }; + + fn manifest(parts: u32) -> MultiPartPlans { + MultiPartPlans { + stream: STREAM, + part_count: parts, + total_tasks: parts, + user_write: true, + client_write: true, + per_vshard: vec![VShardParts { vshard: 5, parts }], + } + } + + fn part(index: u32, bytes: usize) -> StreamedPart { + StreamedPart { + index, + targets: vec![5], + part: PlanPart { + first_task: index, + plans: vec![1; bytes], + chunk: None, + }, + } + } + + fn intake(max_queued_part_bytes: usize) -> PartsIntake { + PartsIntake::new(&SequencerConfig { + max_plans_bytes_per_txn: 8, + max_queued_part_bytes, + ..SequencerConfig::default() + }) + } + + #[test] + fn an_offer_before_the_header_is_unknown() { + let intake = intake(16); + let offer = intake.offer(STREAM, vec![part(0, 4)]); + assert_eq!(offer.status, PartsOfferStatus::Unknown); + } + + /// The queue stops at its byte cap and answers `Full`. Proposing drains + /// it, and the stream resumes where it stopped. A resent part is taken + /// once. + #[test] + fn a_full_queue_pushes_back_and_resumes_after_proposals() { + let intake = intake(16); + intake.open(STREAM, TxnId::new(1, 0), 3, &manifest(6)); + let offer = intake.offer(STREAM, (0..6).map(|i| part(i, 8)).collect()); + assert_eq!(offer.status, PartsOfferStatus::Full); + assert_eq!(offer.next_index, 2); + assert_eq!(intake.queued_bytes(), 16); + + let (_, _, first) = intake.next_part().expect("a queued part"); + assert_eq!(first.index, 0); + let offer = intake.offer(STREAM, (1..6).map(|i| part(i, 8)).collect()); + assert_eq!(offer.next_index, 3, "part 1 is taken once, part 2 fits"); + assert!(intake.queued_bytes() <= 16); + } + + #[test] + fn a_malformed_part_closes_the_stream() { + let intake = intake(64); + intake.open(STREAM, TxnId::new(1, 0), 3, &manifest(2)); + let offer = intake.offer(STREAM, vec![part(0, 4), part(1, 9)]); + assert_eq!(offer.status, PartsOfferStatus::Rejected); + assert!(offer.detail.is_some()); + assert!(!intake.holds(TxnId::new(1, 0))); + assert_eq!(intake.queued_bytes(), 0); + } + + #[test] + fn a_term_change_or_a_stall_closes_the_stream() { + let intake = intake(64); + intake.open(STREAM, TxnId::new(1, 0), 3, &manifest(2)); + assert_eq!(intake.close_stale(Some(3), Instant::now()), 0); + assert_eq!(intake.close_stale(Some(4), Instant::now()), 1); + assert!(!intake.holds(TxnId::new(1, 0))); + + intake.open(STREAM, TxnId::new(1, 0), 4, &manifest(2)); + let later = Instant::now() + Duration::from_secs(60); + assert_eq!(intake.close_stale(Some(4), later), 1, "stalled"); + } +} diff --git a/nodedb-cluster/src/calvin/sequencer/replay.rs b/nodedb-cluster/src/calvin/sequencer/replay.rs index 79d40796e..860e67be4 100644 --- a/nodedb-cluster/src/calvin/sequencer/replay.rs +++ b/nodedb-cluster/src/calvin/sequencer/replay.rs @@ -57,6 +57,10 @@ impl SequencerStateMachine { /// * `ReserveRead` targeting `vshard_id` → [`SchedulerInput::Reserve`]. /// * `ReleaseReservation` targeting `vshard_id` → [`SchedulerInput::Release`]. /// * `CutMarker` → [`SchedulerInput::CutMarker`], for every vShard. + /// * `TxnPart` of a transaction in the epoch range, targeting `vshard_id` + /// → [`SchedulerInput::TxnPart`]. + /// * `TxnPartsAbandoned` of a transaction in the epoch range → + /// [`SchedulerInput::PartsAbandoned`], for every vShard. /// * All other variants carry no per-vShard scheduler input. /// /// Entries are emitted in Raft-log order (and, within an epoch batch, in @@ -145,7 +149,7 @@ impl SequencerStateMachine { per_vshard.epoch_system_ms = batch.epoch_system_ms; per_vshard.epoch_vshard_txn_count = vshard_txn_counts.get(&vshard_id).copied().unwrap_or(0); - result.push(SchedulerInput::Txn(per_vshard)); + result.push(SchedulerInput::Txn(Box::new(per_vshard))); } } // Re-fan the read reservation exactly as the live `ReserveRead` @@ -164,9 +168,47 @@ impl SequencerStateMachine { } // A cut marker reaches every vShard, exactly as the live // `CutMarker` arm fans it out. - SequencerEntry::CutMarker { hlc } => { + SequencerEntry::CutMarker { hlc, .. } => { result.push(SchedulerInput::CutMarker { hlc }); } + // A part reaches the vShards it targets, exactly as the live + // `TxnPart` arm fans it out. The scheduler ignores a part of a + // transaction it holds no header for. + SequencerEntry::TxnPart { + epoch, + position, + index, + first_task, + targets, + plans, + chunk, + } if epoch >= from_epoch + && epoch <= to_epoch + && targets.binary_search(&vshard_id).is_ok() => + { + result.push(SchedulerInput::TxnPart { + txn: crate::calvin::types::TxnIdWire { epoch, position }, + index, + first_task, + plans: std::sync::Arc::new(plans), + chunk, + }); + } + SequencerEntry::TxnPart { .. } => {} + // An abandonment reaches every vShard. The scheduler acts only + // on a transaction it holds, which it holds only when it + // participates. + SequencerEntry::TxnPartsAbandoned { epoch, position } + if epoch >= from_epoch && epoch <= to_epoch => + { + result.push(SchedulerInput::PartsAbandoned { + txn: crate::calvin::types::TxnIdWire { epoch, position }, + }); + } + SequencerEntry::TxnPartsAbandoned { .. } => {} + // An epoch floor only moves the next epoch the sequencer + // proposes. No scheduler input comes from it. + SequencerEntry::EpochFloor { .. } => {} // Reservation entries for a different vShard carry nothing for us. SequencerEntry::ReserveRead { .. } => {} SequencerEntry::ReleaseReservation { .. } => {} diff --git a/nodedb-cluster/src/calvin/sequencer/service/core.rs b/nodedb-cluster/src/calvin/sequencer/service/core.rs index 3bd1bc201..f1e7b44d6 100644 --- a/nodedb-cluster/src/calvin/sequencer/service/core.rs +++ b/nodedb-cluster/src/calvin/sequencer/service/core.rs @@ -31,12 +31,10 @@ use tokio::sync::mpsc; use crate::calvin::sequencer::config::SEQUENCER_GROUP_ID; use crate::calvin::sequencer::config::SequencerConfig; use crate::calvin::sequencer::entry::SequencerEntry; -use crate::calvin::sequencer::inbox::{AdmittedTx, InboxReceiver}; +use crate::calvin::sequencer::inbox::InboxReceiver; use crate::calvin::sequencer::reservation_inbox::ReservationInboxReceiver; use crate::calvin::sequencer::service::verdict_entry::verdict_entry; use crate::calvin::sequencer::state_machine::SequencerStateMachine; -use crate::calvin::sequencer::validator::validate_batch_with_assignments; -use crate::calvin::types::EpochBatch; use crate::calvin::{CalvinCompletionRegistry, TxnId, VerdictOutcome}; use crate::error::ClusterError; use crate::multi_raft::MultiRaft; @@ -66,10 +64,10 @@ pub struct SequencerReceivers { /// Drives the epoch ticker. Must be spawned as a Tokio task on the Control /// Plane. `Send + Sync`. pub struct SequencerService { - config: SequencerConfig, - node_id: u64, - multi_raft: Arc>, - inbox_receiver: InboxReceiver, + pub(super) config: SequencerConfig, + pub(super) node_id: u64, + pub(super) multi_raft: Arc>, + pub(super) inbox_receiver: InboxReceiver, /// Carries hot-key read-reservation requests from the Control Plane. Only /// the leader services it (see `process_reservations`); a follower drains /// and discards it so awaiting callers fall back to plain OCC. @@ -89,17 +87,21 @@ pub struct SequencerService { /// sequencer group's log has been replayed into the state machine. On /// leader failover, `inbox_receiver` is simply dropped (in-flight /// submissions are not in the log and will be retried). - current_epoch: Option, + pub(super) current_epoch: Option, + /// The `epoch_system_ms` this service last minted, or `None` before its + /// first mint. A proposed batch applies later, so the next mint reads + /// this as well as the state machine's applied instant. + pub(super) last_minted_ms: Option, /// The state machine committed sequencer entries are applied into on this /// node. Read to derive the epoch seed, and on every tick to see whether it /// has halted. - state_machine: Arc>, + pub(super) state_machine: Arc>, /// Whether the halt has already been reported. The tick runs at epoch /// cadence (milliseconds), so the report is latched to one line rather than /// burying the original cause under a per-tick repeat. - halt_reported: bool, + pub(super) halt_reported: bool, pub metrics: Arc, - completion_registry: Arc, + pub(super) completion_registry: Arc, /// Receives `(txn, commit)` verdict signals emitted by this node's /// completion registry when a staged cross-shard txn's vote tally becomes /// complete. Only the leader turns a signal into a `Verdict` proposal. @@ -107,6 +109,13 @@ pub struct SequencerService { /// local, avoiding a borrow conflict with `self.tick()` in a sibling /// `select!` arm; it is always `Some` after construction. verdict_rx: Option>, + /// The streamed parts of the multi-part transactions this leader + /// sequenced, shared with the inbox that takes them (see + /// [`super::parts`]). + pub(super) parts_intake: Arc, + /// Abandonments proposed in the current term and not yet applied, so + /// each is proposed once per term. + pub(super) abandoning: super::parts::Abandoning, } impl SequencerService { @@ -130,6 +139,7 @@ impl SequencerService { inbox, reservations, } = receivers; + let parts_intake = inbox.parts_intake(); Self { config, node_id, @@ -139,11 +149,14 @@ impl SequencerService { next_reservation_position: RESERVATION_POSITION_BAND, reservation_epoch: 0, current_epoch: None, + last_minted_ms: None, state_machine, halt_reported: false, metrics: SequencerMetrics::new(), completion_registry, verdict_rx: Some(verdict_rx), + parts_intake, + abandoning: super::parts::Abandoning::default(), } } @@ -224,11 +237,14 @@ impl SequencerService { // multi_raft is_leader API directly. if !self.is_leader() { // Drain and discard: clients will retry against the real leader. - let discarded = self.inbox_receiver.drain_all_discard(); + let discarded = self.discard_inbox(); // Discard reservation requests too: dropping each `Reserve`'s `reply` // sender makes the CP awaiter observe a closed channel and fall back // to plain OCC — correct degradation when this node is not leader. let reservations_discarded = self.reservation_receiver.drain_all_discard(); + // Parts held here are gone with the leadership. The next leader + // abandons their transactions. + self.drop_part_streams(); debug!( node_id = self.node_id, "not sequencer leader; discarding {discarded} inbox items \ @@ -285,171 +301,21 @@ impl SequencerService { } return; }; - - // Drain inbox up to per-epoch caps. - let mut candidates: Vec = Vec::new(); - let drained = self.inbox_receiver.drain_into_capped( - &mut candidates, - self.config.max_txns_per_epoch, - self.config.max_bytes_per_epoch, - ); - if drained == 0 { - debug!( - node_id = self.node_id, - epoch, "epoch tick: inbox empty, no proposal" - ); - return; - } - - // Pre-validation. - let (admitted, rejected) = validate_batch_with_assignments(epoch, candidates); - - self.metrics - .admitted_total - .fetch_add(admitted.len() as u64, Ordering::Relaxed); - - // Record per-conflict metrics and increment the aggregate counter. - for r in &rejected { - self.metrics - .rejected_conflict_total - .fetch_add(1, Ordering::Relaxed); - if let Some(ctx) = r.conflict_context.clone() { - self.metrics.record_conflict(ctx); - } - } - - if admitted.is_empty() { - debug!( - epoch, - rejected = rejected.len(), - "epoch tick: all candidates rejected, no proposal" - ); - self.current_epoch = Some(epoch + 1); - return; - } - - // Read wall clock ONCE on the sequencer leader. This is the single - // deterministic timestamp source for every transaction in this epoch. - // All replicas receive this value via Raft replication; engine handlers - // use it instead of reading the wall clock independently. - let epoch_system_ms = std::time::SystemTime::now() // no-determinism: read once on leader; replicated to all replicas via Raft - .duration_since(std::time::UNIX_EPOCH) - .map(|d| d.as_millis() as i64) - .unwrap_or(0); - - // Encode and propose. - let batch = EpochBatch { - epoch, - txns: admitted.iter().map(|(_, txn)| txn.clone()).collect(), - epoch_system_ms, - }; - for (inbox_seq, txn) in &admitted { - self.completion_registry.note_assigned( - *inbox_seq, - crate::calvin::TxnId::new(epoch, txn.position), - txn.tx_class.participating_vshards().len(), - ); - } - let entry = SequencerEntry::EpochBatch { batch }; - let txns_count = entry_txn_count(&entry); - let _replicate_span = - tracing::info_span!("sequencer_replicate", epoch, txns_count,).entered(); - match self.propose_entry(&entry) { - Ok(log_index) => { - debug!( - epoch, - log_index, - admitted = entry_txn_count(&entry), - rejected = rejected.len(), - "sequencer proposed epoch batch" - ); - } - Err(e) => { - warn!(epoch, error = %e, "sequencer propose failed; epoch will be retried on next tick if still leader"); - // Do NOT advance epoch on propose failure — the same epoch - // will be re-attempted on the next tick if the node is still - // the leader. This is safe because the epoch has not been - // committed to the Raft log. - return; - } - } - self.current_epoch = Some(epoch + 1); - } - - /// Derive the epoch seed once, then reuse it for the life of this service. - /// - /// Delegates the (heavily reasoned) safety gate to - /// [`super::epoch_seed::derive_epoch_seed`]; `None` means it is not yet - /// safe to mint an epoch on this node and the caller must skip the tick. - /// - /// Publishes the outcome to `metrics.epoch_seeded` so the readiness probe - /// can tell whether a Calvin submit landing here can be sequenced. - fn ensure_epoch_seeded(&mut self) -> Option { - let seed = self.derive_or_cached_epoch(); - self.metrics - .epoch_seeded - .store(seed.is_some(), Ordering::Relaxed); - seed + self.mint_epoch(epoch); + // Parts follow their headers' batches, this tick's included. + self.propose_parts(); + self.abandon_orphaned_parts(); } - /// The seed itself, without the readiness publication. - fn derive_or_cached_epoch(&mut self) -> Option { - // Checked ahead of the cached seed, not just before deriving one: a halt - // can land long after the seed was taken. A halted state machine refuses - // every epoch batch, so a minted epoch would only manufacture identities - // that nothing on this node will ever apply. - if self.state_machine_halted() { - return None; - } - if let Some(epoch) = self.current_epoch { - return Some(epoch); - } - let epoch = super::epoch_seed::derive_epoch_seed( - self.node_id, - &self.multi_raft, - &self.state_machine, - )?; - self.current_epoch = Some(epoch); - Some(epoch) - } - - /// Whether this node's sequencer state machine has stopped applying epoch - /// batches after an unrecoverable epoch regression. - fn state_machine_halted(&self) -> bool { - self.state_machine - .lock() - .unwrap_or_else(|p| p.into_inner()) - .is_halted() - } - - /// Fail every queued submission fast while the state machine is halted. - /// - /// A halt scopes the fault to sequencing: reads, non-Calvin writes, metadata - /// and every other engine on this node are unaffected, so the node keeps - /// serving. What it must not do is keep accepting Calvin work — nothing will - /// ever sequence it. Dropping each submission's reply channel makes the - /// awaiting Control-Plane caller observe a closed channel immediately and - /// surface an error, instead of every writer hanging to its deadline behind - /// a queue that will never drain. Reservation requests degrade to plain OCC - /// the same way they do on a follower. - fn shed_submissions_after_halt(&mut self) { + /// Drain and discard every queued submission, and drop each one's + /// assignment so its caller reads a closed channel at once. Returns how + /// many were discarded. + pub(super) fn discard_inbox(&mut self) -> usize { let discarded = self.inbox_receiver.drain_all_discard(); - let reservations_discarded = self.reservation_receiver.drain_all_discard(); - if !self.halt_reported { - self.halt_reported = true; - tracing::error!( - node_id = self.node_id, - "sequencer state machine halted on an epoch regression; this node has stopped \ - sequencing and is failing Calvin submissions fast. Every other query path \ - keeps serving — operator intervention is required to resume sequencing." - ); - } - if discarded > 0 || reservations_discarded > 0 { - debug!( - node_id = self.node_id, - discarded, reservations_discarded, "sequencer halted; shed queued submissions" - ); + for inbox_seq in &discarded { + self.completion_registry.drop_assignment(*inbox_seq); } + discarded.len() } /// Re-propose every complete-but-unstored cross-shard verdict. @@ -497,32 +363,18 @@ impl SequencerService { } } -fn entry_txn_count(entry: &SequencerEntry) -> usize { - match entry { - SequencerEntry::EpochBatch { batch } => batch.txns.len(), - SequencerEntry::CompletionAck { .. } => 0, - SequencerEntry::OllpMismatch { .. } => 0, - SequencerEntry::TxnRoutingFailed { .. } => 0, - SequencerEntry::Vote { .. } => 0, - SequencerEntry::Verdict { .. } => 0, - SequencerEntry::AbortVote { .. } => 0, - SequencerEntry::AbortVerdict { .. } => 0, - SequencerEntry::ReserveRead { .. } => 0, - SequencerEntry::ReleaseReservation { .. } => 0, - SequencerEntry::CutMarker { .. } => 0, - } -} - // ── Tests ──────────────────────────────────────────────────────────────────── #[cfg(test)] -mod tests { +pub(super) mod tests { use std::collections::HashMap; use std::time::{Duration, Instant}; + use tokio::sync::oneshot; + use super::*; use crate::calvin::sequencer::config::SequencerConfig; - use crate::calvin::sequencer::inbox::{Inbox, new_inbox}; + use crate::calvin::sequencer::inbox::{AdmittedTx, Inbox, new_inbox}; use crate::calvin::sequencer::reservation_inbox::{ReservationInbox, new_reservation_inbox}; use crate::calvin::sequencer::validator::validate_batch; use crate::calvin::types::{ @@ -553,7 +405,10 @@ mod tests { panic!("could not find two distinct-vshard collections in 512 tries"); } - fn make_tx_class(surr_a: u32, surr_b: u32) -> TxClass { + pub(in crate::calvin::sequencer::service) fn make_tx_class( + surr_a: u32, + surr_b: u32, + ) -> TxClass { let (col_a, col_b) = find_two_distinct_collections(); let write_set = ReadWriteSet::new(vec![ EngineKeySet::Document { @@ -716,11 +571,11 @@ mod tests { /// Live parts of a service under test. The inboxes are kept alive because /// dropping them would close the receivers the service holds. - struct Harness { - service: SequencerService, - state_machine: Arc>, - multi_raft: Arc>, - _inbox: Inbox, + pub(in crate::calvin::sequencer::service) struct Harness { + pub(in crate::calvin::sequencer::service) service: SequencerService, + pub(in crate::calvin::sequencer::service) state_machine: Arc>, + pub(in crate::calvin::sequencer::service) multi_raft: Arc>, + pub(in crate::calvin::sequencer::service) inbox: Inbox, /// Kept alive so the service's receiver stays open, and used directly by /// the tests that submit reservation requests to a leader tick. reservations: ReservationInbox, @@ -728,7 +583,7 @@ mod tests { _dir: tempfile::TempDir, } - fn make_harness() -> Harness { + pub(in crate::calvin::sequencer::service) fn make_harness() -> Harness { let dir = tempfile::tempdir().expect("tempdir"); let routing = RoutingTable::uniform(1, &[1], 1); let mut mr = MultiRaft::new(1, routing, dir.path().to_path_buf()); @@ -762,7 +617,7 @@ mod tests { service, state_machine, multi_raft, - _inbox: inbox, + inbox, reservations, _verdict_tx: verdict_tx, _dir: dir, @@ -787,7 +642,7 @@ mod tests { /// Drive the single-voter sequencer group to leadership so proposals append /// to its log. - fn elect(multi_raft: &Arc>) { + pub(in crate::calvin::sequencer::service) fn elect(multi_raft: &Arc>) { let mut mr = multi_raft.lock().unwrap_or_else(|p| p.into_inner()); if let Some(node) = mr.groups_mut().get_mut(&SEQUENCER_GROUP_ID) { // no-determinism: test-only forced election deadline so the single @@ -840,6 +695,56 @@ mod tests { assert_eq!(harness.service.ensure_epoch_seeded(), Some(9)); } + /// Epoch instants order Calvin versions, so a leader mints each one + /// strictly above every instant it minted and every one its state + /// machine applied. A wall clock that steps back, a restart that replays + /// the log, and a new leader with a slower clock all mint above history. + #[test] + fn epoch_instants_stay_monotonic_across_clock_steps_and_leader_changes() { + let mut harness = make_harness(); + let first = harness.service.next_epoch_system_ms(2_000_000_000_000); + assert_eq!(first, 2_000_000_000_000); + let stepped_back = harness.service.next_epoch_system_ms(1_999_999_999_000); + assert_eq!(stepped_back, first + 1, "a clock step back mints above"); + + // A node that replayed a batch minted at 1_700_000_000_000 by an + // earlier leader, now leading with a clock behind that instant. + let mut successor = make_harness(); + successor + .state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .apply(1, &epoch_batch_bytes(0)); + let minted = successor.service.next_epoch_system_ms(1_600_000_000_000); + assert_eq!( + minted, 1_700_000_000_001, + "a new leader mints above history" + ); + let next = successor.service.next_epoch_system_ms(1_600_000_000_000); + assert_eq!(next, minted + 1, "an unapplied mint still counts"); + + // A log a cluster restore rebuilt opens with the point's epoch + // floor. Its leader mints above the point's instant. + let mut restored = make_harness(); + restored + .state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .apply( + 1, + &zerompk::to_msgpack_vec(&SequencerEntry::EpochFloor { + next_epoch: 4, + epoch_system_ms: 1_900_000_000_000, + }) + .expect("encode"), + ); + assert_eq!( + restored.service.next_epoch_system_ms(1_600_000_000_000), + 1_900_000_000_001, + "a restored leader mints above the restore point" + ); + } + /// A brand-new node has an empty log and an empty state machine. Nothing /// has been applied because nothing was ever proposed, and nothing can be /// proposed until an epoch is minted — so a gate that waited for an applied @@ -982,6 +887,129 @@ mod tests { ); } + // ── Unsequenced submissions ────────────────────────────────────────────── + + /// A single-vshard class that reads `read_col` and writes `write_col`. + fn make_rw_class(read_col: &str, write_col: &str) -> TxClass { + TxClass::new_single_vshard( + ReadWriteSet::new(vec![EngineKeySet::Document { + collection: read_col.to_owned(), + surrogates: SortedVec::new(vec![1]), + }]), + ReadWriteSet::new(vec![EngineKeySet::Document { + collection: write_col.to_owned(), + surrogates: SortedVec::new(vec![1]), + }]), + vec![], + TenantId::new(1), + None, + crate::calvin::types::VersionedReadSet::default(), + ) + .expect("valid TxClass") + } + + fn closed(rx: &mut crate::calvin::AssignmentReceiver) -> bool { + rx.try_recv() == Err(oneshot::error::TryRecvError::Closed) + } + + /// A follower discards its inbox. Each discarded caller must read a closed + /// channel at once instead of waiting out its timeout. + #[test] + fn non_leader_discard_fails_each_submission_at_once() { + let mut harness = make_harness(); + let registry = Arc::clone(&harness.service.completion_registry); + let (_, mut rx) = harness + .inbox + .submit_with(make_tx_class(1, 2), ®istry) + .expect("submit"); + + harness.service.tick(); + + assert!(closed(&mut rx)); + assert_eq!(registry.pending_assignments_len(), 0); + } + + /// The validator rejects the later txn of a read/write cycle. Its caller + /// reads a closed channel. The admitted txn's caller gets its assignment. + #[test] + fn validator_rejection_fails_the_rejected_submission_at_once() { + let mut harness = make_harness(); + elect(&harness.multi_raft); + let registry = Arc::clone(&harness.service.completion_registry); + let (col_a, col_b) = find_two_distinct_collections(); + let (_, mut winner) = harness + .inbox + .submit_with(make_rw_class(&col_a, &col_b), ®istry) + .expect("submit"); + let (_, mut loser) = harness + .inbox + .submit_with(make_rw_class(&col_b, &col_a), ®istry) + .expect("submit"); + + harness.service.mint_epoch(4); + + assert!(closed(&mut loser)); + // Two participants: the vShard it writes and the vShard it reads. + assert_eq!(winner.try_recv(), Ok((4, 0, 2))); + assert_eq!(registry.pending_assignments_len(), 0); + assert_eq!(harness.service.current_epoch, Some(5)); + } + + /// A batch whose proposal failed is not in the log. Its callers must fail + /// at once, not hold an `(epoch, position)` the next tick hands to another + /// transaction. + #[test] + fn failed_proposal_fails_every_submission_of_the_batch() { + let mut harness = make_harness(); + let registry = Arc::clone(&harness.service.completion_registry); + let (_, mut first) = harness + .inbox + .submit_with(make_tx_class(1, 2), ®istry) + .expect("submit"); + let (_, mut second) = harness + .inbox + .submit_with(make_tx_class(3, 4), ®istry) + .expect("submit"); + + // Not elected: the proposal fails. + harness.service.mint_epoch(0); + + assert!(closed(&mut first)); + assert!(closed(&mut second)); + assert_eq!(registry.pending_assignments_len(), 0); + assert_eq!( + harness.service.current_epoch, None, + "a failed proposal must not consume the epoch" + ); + } + + /// A halted state machine sheds the inbox. Each shed caller must read a + /// closed channel at once. + #[test] + fn halted_shed_fails_each_submission_at_once() { + let mut harness = make_harness(); + elect(&harness.multi_raft); + { + let mut sm = harness + .state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()); + sm.apply(1, &epoch_batch_bytes(0)); + sm.apply(2, &epoch_batch_bytes(0)); + assert!(sm.is_halted()); + } + let registry = Arc::clone(&harness.service.completion_registry); + let (_, mut rx) = harness + .inbox + .submit_with(make_tx_class(1, 2), ®istry) + .expect("submit"); + + harness.service.tick(); + + assert!(closed(&mut rx)); + assert_eq!(registry.pending_assignments_len(), 0); + } + /// The seed must not be taken while the sequencer group is still replaying: /// that is exactly the startup window in which the state machine's counter /// still reads 0 no matter how much history the log holds. diff --git a/nodedb-cluster/src/calvin/sequencer/service/epoch_mint.rs b/nodedb-cluster/src/calvin/sequencer/service/epoch_mint.rs new file mode 100644 index 000000000..a630c9ad8 --- /dev/null +++ b/nodedb-cluster/src/calvin/sequencer/service/epoch_mint.rs @@ -0,0 +1,191 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The minting half of a leader tick: drain the inbox into one epoch, +//! validate it, and propose the admitted batch. + +use std::sync::atomic::Ordering; + +use tracing::{debug, warn}; + +use crate::calvin::TxnId; +use crate::calvin::sequencer::entry::SequencerEntry; +use crate::calvin::sequencer::validator::validate_batch_with_assignments; +use crate::calvin::types::EpochBatch; + +use super::core::SequencerService; + +impl SequencerService { + /// The epoch instant the next minted epoch carries, given the wall clock + /// reads `wall_ms`. Records it as this service's last mint. + /// + /// The instant is strictly above every instant this service minted and + /// every one the state machine applied. The state machine rebuilds its + /// instant from the replayed log, so the order holds across a wall clock + /// that steps back, a leader change and a restart. + pub(super) fn next_epoch_system_ms(&mut self, wall_ms: i64) -> i64 { + let applied = self + .state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .last_epoch_system_ms(); + let minted = monotonic_epoch_ms(wall_ms, self.last_minted_ms.max(applied)); + self.last_minted_ms = Some(minted); + minted + } + + /// Sequence the queued submissions into epoch `epoch`. + /// + /// Every drained submission leaves with an answer. An admitted one gets + /// its assignment once the batch is in the Raft log. A rejected one, and + /// every one of a batch whose proposal failed, has its assignment + /// dropped, so its caller reads a closed channel at once. None of those + /// is in the log, so a retry never applies a transaction twice. + /// + /// The epoch advances when the batch is proposed, and when every + /// candidate was rejected. A failed proposal leaves it unchanged, so the + /// next tick re-attempts the same epoch. + pub(super) fn mint_epoch(&mut self, epoch: u64) { + let mut candidates = Vec::new(); + let drained = self.inbox_receiver.drain_into_capped( + &mut candidates, + self.config.max_txns_per_epoch, + self.config.max_bytes_per_epoch, + ); + if drained == 0 { + debug!( + node_id = self.node_id, + epoch, "epoch tick: inbox empty, no proposal" + ); + return; + } + + let (admitted, rejected) = validate_batch_with_assignments(epoch, candidates); + + self.metrics + .admitted_total + .fetch_add(admitted.len() as u64, Ordering::Relaxed); + + // Record per-conflict metrics and fail each rejected submission. + let rejected_count = rejected.len(); + for rejection in rejected { + self.completion_registry + .drop_assignment(rejection.admitted.inbox_seq); + self.metrics + .rejected_conflict_total + .fetch_add(1, Ordering::Relaxed); + if let Some(ctx) = rejection.conflict_context { + self.metrics.record_conflict(ctx); + } + } + + if admitted.is_empty() { + debug!( + epoch, + rejected = rejected_count, + "epoch tick: all candidates rejected, no proposal" + ); + self.current_epoch = Some(epoch + 1); + return; + } + + // Read wall clock ONCE on the sequencer leader. This is the single + // deterministic timestamp source for every transaction in this epoch. + // All replicas receive this value via Raft replication; engine handlers + // use it instead of reading the wall clock independently. + let wall_ms = std::time::SystemTime::now() // no-determinism: read once on leader; replicated to all replicas via Raft + .duration_since(std::time::UNIX_EPOCH) + .map(|d| i64::try_from(d.as_millis()).unwrap_or(i64::MAX)) + .unwrap_or(0); + let epoch_system_ms = self.next_epoch_system_ms(wall_ms); + + // A multi-part transaction enters the batch as its header. Its + // coordinator streams the parts once the batch is proposed, and each + // is proposed as its own entry. + let streams: Vec<(TxnId, crate::calvin::types::MultiPartPlans)> = admitted + .iter() + .filter_map(|(_, txn)| { + let manifest = txn.tx_class.multi_part.clone()?; + Some((TxnId::new(epoch, txn.position), manifest)) + }) + .collect(); + + // `(inbox_seq, txn, participants)` of each admitted submission. + let assignments: Vec<(u64, TxnId, usize)> = admitted + .iter() + .map(|(inbox_seq, txn)| { + ( + *inbox_seq, + TxnId::new(epoch, txn.position), + txn.tx_class.participating_vshards().len(), + ) + }) + .collect(); + let batch = EpochBatch { + epoch, + txns: admitted.into_iter().map(|(_, txn)| txn).collect(), + epoch_system_ms, + }; + let txns_count = batch.txns.len(); + let entry = SequencerEntry::EpochBatch { batch }; + let _replicate_span = + tracing::info_span!("sequencer_replicate", epoch, txns_count).entered(); + match self.propose_entry(&entry) { + Ok(log_index) => { + // Open each stream before its submitter learns the + // assignment, so its first offer finds the stream. + for (txn, manifest) in streams { + self.open_part_stream(txn, &manifest); + } + for (inbox_seq, txn, participants) in assignments { + self.completion_registry + .note_assigned(inbox_seq, txn, participants); + } + debug!( + epoch, + log_index, + admitted = txns_count, + rejected = rejected_count, + "sequencer proposed epoch batch" + ); + self.current_epoch = Some(epoch + 1); + } + Err(e) => { + warn!( + epoch, + error = %e, + "sequencer propose failed; the batch's submissions are failed and \ + the epoch is retried on the next tick if still leader" + ); + for (inbox_seq, _, _) in assignments { + self.completion_registry.drop_assignment(inbox_seq); + } + } + } + } +} + +/// The epoch instant a mint carries: the wall clock, raised strictly above +/// `floor`, the highest instant minted or applied before it. +fn monotonic_epoch_ms(wall_ms: i64, floor: Option) -> i64 { + match floor { + Some(floor) => wall_ms.max(floor.saturating_add(1)), + None => wall_ms, + } +} + +#[cfg(test)] +mod tests { + use super::monotonic_epoch_ms; + + #[test] + fn the_first_mint_takes_the_wall_clock() { + assert_eq!(monotonic_epoch_ms(1_000, None), 1_000); + } + + #[test] + fn a_wall_clock_that_steps_back_still_mints_above_the_floor() { + assert_eq!(monotonic_epoch_ms(900, Some(1_000)), 1_001); + assert_eq!(monotonic_epoch_ms(1_000, Some(1_000)), 1_001); + assert_eq!(monotonic_epoch_ms(5_000, Some(1_000)), 5_000); + } +} diff --git a/nodedb-cluster/src/calvin/sequencer/service/epoch_seed.rs b/nodedb-cluster/src/calvin/sequencer/service/epoch_seed.rs index 086628ff0..ee9f84b8f 100644 --- a/nodedb-cluster/src/calvin/sequencer/service/epoch_seed.rs +++ b/nodedb-cluster/src/calvin/sequencer/service/epoch_seed.rs @@ -1,11 +1,13 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Derivation of the sequencer leader's starting epoch. +//! Derivation of the sequencer leader's starting epoch, and the service's +//! response to a halted state machine. //! -//! Split out of the service so the reasoning that guards it stays next to the -//! one function that implements it. +//! The reasoning that guards the seed stays next to the one function that +//! implements it. use std::sync::Mutex; +use std::sync::atomic::Ordering; use tracing::{debug, info, warn}; @@ -13,6 +15,82 @@ use crate::calvin::sequencer::config::SEQUENCER_GROUP_ID; use crate::calvin::sequencer::state_machine::SequencerStateMachine; use crate::multi_raft::MultiRaft; +use super::core::SequencerService; + +impl SequencerService { + /// Derive the epoch seed once, then reuse it for the life of this service. + /// + /// Delegates the safety gate to [`derive_epoch_seed`]. `None` means it is + /// not yet safe to mint an epoch on this node and the caller must skip + /// minting for this tick. + /// + /// Publishes the outcome to `metrics.epoch_seeded` so the readiness probe + /// can tell whether a Calvin submit landing here can be sequenced. + pub(super) fn ensure_epoch_seeded(&mut self) -> Option { + let seed = self.derive_or_cached_epoch(); + self.metrics + .epoch_seeded + .store(seed.is_some(), Ordering::Relaxed); + seed + } + + /// The seed itself, without the readiness publication. + fn derive_or_cached_epoch(&mut self) -> Option { + // Checked ahead of the cached seed, not just before deriving one: a halt + // can land long after the seed was taken. A halted state machine refuses + // every epoch batch, so a minted epoch would only manufacture identities + // that nothing on this node will ever apply. + if self.state_machine_halted() { + return None; + } + if let Some(epoch) = self.current_epoch { + return Some(epoch); + } + let epoch = derive_epoch_seed(self.node_id, &self.multi_raft, &self.state_machine)?; + self.current_epoch = Some(epoch); + Some(epoch) + } + + /// Whether this node's sequencer state machine has stopped applying epoch + /// batches after an unrecoverable epoch regression. + pub(super) fn state_machine_halted(&self) -> bool { + self.state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .is_halted() + } + + /// Fail every queued submission fast while the state machine is halted. + /// + /// A halt scopes the fault to sequencing: reads, non-Calvin writes, metadata + /// and every other engine on this node are unaffected, so the node keeps + /// serving. What it must not do is keep accepting Calvin work — nothing will + /// ever sequence it. Dropping each submission's assignment makes the + /// awaiting Control-Plane caller observe a closed channel immediately and + /// surface an error, instead of every writer hanging to its deadline behind + /// a queue that will never drain. Reservation requests degrade to plain OCC + /// the same way they do on a follower. + pub(super) fn shed_submissions_after_halt(&mut self) { + let discarded = self.discard_inbox(); + let reservations_discarded = self.reservation_receiver.drain_all_discard(); + if !self.halt_reported { + self.halt_reported = true; + tracing::error!( + node_id = self.node_id, + "sequencer state machine halted on an epoch regression; this node has stopped \ + sequencing and is failing Calvin submissions fast. Every other query path \ + keeps serving — operator intervention is required to resume sequencing." + ); + } + if discarded > 0 || reservations_discarded > 0 { + debug!( + node_id = self.node_id, + discarded, reservations_discarded, "sequencer halted; shed queued submissions" + ); + } + } +} + /// Derive — once — the first epoch this node may propose, returning `None` /// while it is not yet safe to derive one. /// diff --git a/nodedb-cluster/src/calvin/sequencer/service/mod.rs b/nodedb-cluster/src/calvin/sequencer/service/mod.rs index 403115343..ec0e527a0 100644 --- a/nodedb-cluster/src/calvin/sequencer/service/mod.rs +++ b/nodedb-cluster/src/calvin/sequencer/service/mod.rs @@ -1,12 +1,14 @@ // SPDX-License-Identifier: BUSL-1.1 pub mod core; +pub mod epoch_mint; pub mod epoch_seed; +pub mod parts; pub mod reservations; pub mod verdict_entry; // `self::` is required: a bare `core` in a `use` path resolves to the `core` // crate, not this module's sibling. pub use self::core::{RESERVATION_POSITION_BAND, SequencerReceivers, SequencerService}; -// Re-exported so existing call sites (`service::SequencerMetrics`) don't break. +// The sequencer's metrics types, reachable at `service::SequencerMetrics`. pub use crate::calvin::sequencer::metrics::{ConflictKey, SequencerMetrics}; diff --git a/nodedb-cluster/src/calvin/sequencer/service/parts.rs b/nodedb-cluster/src/calvin/sequencer/service/parts.rs new file mode 100644 index 000000000..e9d0038a7 --- /dev/null +++ b/nodedb-cluster/src/calvin/sequencer/service/parts.rs @@ -0,0 +1,418 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The leader's side of multi-part transactions: opening their part +//! streams, proposing the parts they stream in, and abandoning the ones no +//! stream will finish. +//! +//! The leader opens a transaction's stream once its header's epoch batch is +//! proposed. The coordinator streams the parts into the shared +//! [`PartsIntake`], whose byte cap bounds the leader's memory. The leader +//! proposes queued parts in order on each tick. +//! +//! - Fairness: one tick proposes parts up to `max_bytes_per_epoch` bytes, at +//! least one part. Epoch batches keep their cadence between them. A part +//! carries a position fixed by its header, so it never reorders anything on +//! the vShards it shares with later transactions: their flushes wait for it +//! by position. +//! - A failed proposal puts the part back for the next tick. +//! - A node that stops leading closes every stream. A leader whose term +//! changed closes the streams of the old term: Raft can drop an earlier +//! term's proposals. A stream that stalls is closed too. +//! - A leader abandons every open transaction whose stream it does not +//! hold: one another leader sequenced, one this node sequenced before a +//! restart or a term change, or one whose stream closed. The abandonment +//! applies only while the transaction is open, so a last part that commits +//! first wins. +//! +//! A stream stays open until every part is proposed, its header applied, +//! and the transaction closed in the state machine, so the leader never +//! abandons a transaction it is still carrying. +//! +//! [`PartsIntake`]: crate::calvin::sequencer::parts_intake::PartsIntake + +use std::time::Instant; + +use tracing::{debug, warn}; + +use crate::calvin::TxnId; +use crate::calvin::sequencer::config::SEQUENCER_GROUP_ID; +use crate::calvin::sequencer::entry::SequencerEntry; +use crate::calvin::types::MultiPartPlans; + +use super::core::SequencerService; + +/// The abandonments this leader proposed and has not seen applied, and the +/// term it proposed them in. +/// +/// Raft can drop a proposal of an earlier term. So a term change forgets +/// them all, and the leader proposes each again, as it closes the streams +/// of an earlier term. A leader regained with no non-leader tick between is +/// covered too: the term still changed. +#[derive(Debug, Default)] +pub(crate) struct Abandoning { + term: Option, + txns: std::collections::BTreeSet, +} + +impl Abandoning { + /// Forget every abandonment unless it was proposed in `term`. + fn keep_term(&mut self, term: Option) { + if self.term != term { + self.txns.clear(); + self.term = term; + } + } + + /// Forget every abandonment. + fn clear(&mut self) { + self.txns.clear(); + self.term = None; + } +} + +impl SequencerService { + /// Open the part stream of `txn`, whose header's batch was + /// proposed. + pub(super) fn open_part_stream(&mut self, txn: TxnId, manifest: &MultiPartPlans) { + let Some(term) = self.sequencer_term() else { + warn!( + epoch = txn.epoch, + position = txn.position, + "sequencer: leadership lost right after proposing a multi-part header; \ + the next leader abandons it" + ); + return; + }; + self.parts_intake.open(manifest.stream, txn, term, manifest); + } + + /// This node's term in the sequencer group while it leads, else `None`. + fn sequencer_term(&self) -> Option { + let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + mr.leader_term(SEQUENCER_GROUP_ID) + } + + /// Close every part stream: this node no longer leads. + pub(super) fn drop_part_streams(&mut self) { + let dropped = self.parts_intake.clear(); + self.abandoning.clear(); + if dropped > 0 { + warn!( + node_id = self.node_id, + dropped, + "sequencer: leadership lost with multi-part transactions streaming; \ + the next leader abandons them" + ); + } + } + + /// Close stale streams, then propose queued parts in order, up to one + /// epoch's byte budget and at least one part. Stops at the first failed + /// proposal. + pub(super) fn propose_parts(&mut self) { + let closed = self.parts_intake.close_stale( + self.sequencer_term(), + // no-determinism: stall detection only, never in the log. + Instant::now(), + ); + if closed > 0 { + warn!( + node_id = self.node_id, + closed, + "sequencer: closed part streams of an earlier term or with a stalled \ + coordinator; abandoning their transactions" + ); + } + let budget = self.config.max_bytes_per_epoch; + let mut spent = 0usize; + while spent < budget { + let Some((stream, txn, streamed)) = self.parts_intake.next_part() else { + return; + }; + let entry = SequencerEntry::TxnPart { + epoch: txn.epoch, + position: txn.position, + index: streamed.index, + first_task: streamed.part.first_task, + targets: streamed.targets.clone(), + plans: streamed.part.plans.clone(), + chunk: streamed.part.chunk, + }; + match self.propose_entry(&entry) { + Ok(_) => spent = spent.saturating_add(streamed.part.plans.len().max(1)), + Err(error) => { + warn!( + epoch = txn.epoch, + position = txn.position, + part = streamed.index, + %error, + "sequencer: part proposal failed; retrying next tick" + ); + self.parts_intake.return_part(stream, streamed); + return; + } + } + } + } + + /// Abandon every open multi-part transaction whose stream this leader + /// does not hold, and close the streams that finished. + /// + /// Runs only once the epoch seed exists: the group then replayed its log, + /// so the state machine knows every header committed before this term. + pub(super) fn abandon_orphaned_parts(&mut self) { + let (open, applied_epoch) = { + let sm = self.state_machine.lock().unwrap_or_else(|p| p.into_inner()); + (sm.open_multi_part_txns(), sm.last_applied_epoch()) + }; + self.parts_intake.close_finished(|txn| { + applied_epoch.is_some_and(|applied| applied >= txn.epoch) + && open.binary_search(&txn).is_err() + }); + let term = self.sequencer_term(); + self.abandoning.keep_term(term); + self.abandoning + .txns + .retain(|txn| open.binary_search(txn).is_ok()); + for txn in open { + if self.parts_intake.holds(txn) || self.abandoning.txns.contains(&txn) { + continue; + } + match self.propose_entry(&SequencerEntry::TxnPartsAbandoned { + epoch: txn.epoch, + position: txn.position, + }) { + Ok(_) => { + debug!( + epoch = txn.epoch, + position = txn.position, + "sequencer: abandoning a multi-part transaction no stream carries" + ); + self.abandoning.txns.insert(txn); + } + Err(error) => warn!( + epoch = txn.epoch, + position = txn.position, + %error, + "sequencer: abandonment proposal failed; retrying next tick" + ), + } + } + } +} + +#[cfg(test)] +mod tests { + use crate::calvin::sequencer::parts_intake::PartsOfferStatus; + use crate::calvin::types::{ + EpochBatch, PartStreamId, PlanPart, SequencedTxn, StreamedPart, VShardParts, + }; + + use super::super::core::tests::{Harness, elect, make_harness, make_tx_class}; + use super::*; + + const TXN: TxnId = TxnId { + epoch: 0, + position: 0, + }; + const STREAM: PartStreamId = PartStreamId { node: 1, seq: 42 }; + + fn part(index: u32, bytes: usize, target: u32) -> StreamedPart { + StreamedPart { + index, + targets: vec![target], + part: PlanPart { + first_task: index, + plans: vec![0x90; bytes], + chunk: None, + }, + } + } + + /// Apply, as a log replay does, the epoch-0 batch whose one txn is the + /// header of `parts` parts, every one targeting the first participant. + /// Returns the manifest and that participant. + fn apply_header(harness: &Harness, parts: u32) -> (MultiPartPlans, u32) { + let mut tx_class = make_tx_class(1, 2); + tx_class.plans = Vec::new(); + let target = tx_class.participating_vshards()[0].as_u32(); + let manifest = MultiPartPlans { + stream: STREAM, + part_count: parts, + total_tasks: parts, + user_write: true, + client_write: true, + per_vshard: vec![VShardParts { + vshard: target, + parts, + }], + }; + tx_class.multi_part = Some(manifest.clone()); + let batch = EpochBatch { + epoch: 0, + txns: vec![SequencedTxn { + epoch: 0, + position: 0, + tx_class, + epoch_system_ms: 1_700_000_000_000, + epoch_vshard_txn_count: 1, + lock_owner: None, + }], + epoch_system_ms: 1_700_000_000_000, + }; + let bytes = zerompk::to_msgpack_vec(&SequencerEntry::EpochBatch { batch }).expect("encode"); + let mut sm = harness + .state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()); + sm.apply(1, &bytes); + assert_eq!(sm.open_multi_part_txns(), [TXN]); + (manifest, target) + } + + fn log_tip(harness: &Harness) -> u64 { + harness + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .last_log_index(SEQUENCER_GROUP_ID) + .expect("group is mounted") + } + + /// A leader that holds no stream for an open txn, after a restart or a + /// leader change, abandons it once. + #[test] + fn a_leader_without_the_stream_abandons_the_open_txn_once() { + let mut harness = make_harness(); + elect(&harness.multi_raft); + apply_header(&harness, 2); + + let before = log_tip(&harness); + harness.service.abandon_orphaned_parts(); + assert_eq!(log_tip(&harness), before + 1, "one abandonment"); + harness.service.abandon_orphaned_parts(); + assert_eq!(log_tip(&harness), before + 1, "proposed once"); + } + + /// An abandonment proposed in an earlier term is proposed again: Raft + /// can drop that term's proposals. A leader regained with no + /// non-leader tick between still sees the term change. + #[test] + fn an_abandonment_of_an_earlier_term_is_proposed_again() { + let mut harness = make_harness(); + elect(&harness.multi_raft); + apply_header(&harness, 2); + let term = harness + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .leader_term(SEQUENCER_GROUP_ID) + .expect("leader"); + harness.service.abandoning.term = Some(term.saturating_sub(1)); + harness.service.abandoning.txns.insert(TXN); + + let before = log_tip(&harness); + harness.service.abandon_orphaned_parts(); + assert_eq!(log_tip(&harness), before + 1, "proposed again this term"); + harness.service.abandon_orphaned_parts(); + assert_eq!(log_tip(&harness), before + 1, "once per term"); + } + + /// The leader that sequenced a txn takes its streamed parts, proposes + /// them, and never abandons it while it is open. + #[test] + fn the_leader_proposes_streamed_parts_and_keeps_the_txn() { + let mut harness = make_harness(); + elect(&harness.multi_raft); + let (manifest, target) = apply_header(&harness, 2); + harness.service.open_part_stream(TXN, &manifest); + + let offer = harness + .inbox + .offer_parts(STREAM, vec![part(0, 4, target), part(1, 4, target)]); + assert_eq!(offer.status, PartsOfferStatus::Accepted); + assert_eq!(offer.next_index, 2); + + let before = log_tip(&harness); + harness.service.propose_parts(); + assert_eq!(log_tip(&harness), before + 2, "both parts"); + harness.service.abandon_orphaned_parts(); + assert_eq!(log_tip(&harness), before + 2, "no abandonment"); + assert!(harness.service.parts_intake.holds(TXN)); + } + + /// A stream opened in an earlier term is closed mid-stream: Raft can + /// drop that term's proposals. Its coordinator's next offer is answered + /// `Unknown`, and the open txn is abandoned. + #[test] + fn a_term_change_mid_stream_closes_it_and_abandons_the_txn() { + let mut harness = make_harness(); + elect(&harness.multi_raft); + let (manifest, target) = apply_header(&harness, 3); + let term = harness + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .leader_term(SEQUENCER_GROUP_ID) + .expect("leader"); + harness + .service + .parts_intake + .open(STREAM, TXN, term.saturating_sub(1), &manifest); + let offer = harness.inbox.offer_parts(STREAM, vec![part(0, 4, target)]); + assert_eq!(offer.status, PartsOfferStatus::Accepted); + + let before = log_tip(&harness); + harness.service.propose_parts(); + assert_eq!(log_tip(&harness), before, "no part of the old term"); + let offer = harness.inbox.offer_parts(STREAM, vec![part(1, 4, target)]); + assert_eq!(offer.status, PartsOfferStatus::Unknown); + harness.service.abandon_orphaned_parts(); + assert_eq!(log_tip(&harness), before + 1, "one abandonment"); + } + + /// A txn of 72 one-MiB parts, over the 64 MiB RPC limit, streams in + /// batches. The leader's queue never holds more than its cap, tells the + /// stream to wait when full, and proposes every part in order. + #[test] + fn a_stream_over_64_mib_is_bounded_and_fully_proposed() { + const PARTS: u32 = 72; + const PART_BYTES: usize = 1 << 20; + let mut harness = make_harness(); + elect(&harness.multi_raft); + let (manifest, target) = apply_header(&harness, PARTS); + harness.service.open_part_stream(TXN, &manifest); + let cap = harness.service.config.max_queued_part_bytes; + let before = log_tip(&harness); + + let mut next = 0u32; + let mut pushed_back = false; + let mut ticks = 0; + while next < PARTS { + let batch: Vec = (next..PARTS.min(next + 8)) + .map(|index| part(index, PART_BYTES, target)) + .collect(); + let offer = harness.inbox.offer_parts(STREAM, batch); + assert!(harness.service.parts_intake.queued_bytes() <= cap); + next = offer.next_index; + if offer.status == PartsOfferStatus::Full { + pushed_back = true; + harness.service.propose_parts(); + ticks += 1; + } + assert!(ticks < 1_000, "the stream makes progress"); + } + while harness.service.parts_intake.queued_bytes() > 0 { + harness.service.propose_parts(); + } + + assert!(pushed_back, "the queue pushed back before 72 MiB queued"); + assert_eq!(log_tip(&harness), before + u64::from(PARTS)); + harness.service.abandon_orphaned_parts(); + assert_eq!( + log_tip(&harness), + before + u64::from(PARTS), + "a fully streamed txn is never abandoned" + ); + } +} diff --git a/nodedb-cluster/src/calvin/sequencer/service/verdict_entry.rs b/nodedb-cluster/src/calvin/sequencer/service/verdict_entry.rs index f3c9cfb02..ebfe7e006 100644 --- a/nodedb-cluster/src/calvin/sequencer/service/verdict_entry.rs +++ b/nodedb-cluster/src/calvin/sequencer/service/verdict_entry.rs @@ -6,27 +6,20 @@ use crate::calvin::sequencer::entry::SequencerEntry; use crate::calvin::{TxnId, VerdictOutcome}; -/// Encode a decision as the entry the leader proposes. An abort takes the -/// reason-carrying `AbortVerdict`, so the coordinator reports the actual cause. -/// A reasonless abort — a pre-existing vote re-tallied after failover — keeps -/// the original `Verdict` shape. +/// Encode a decision as the entry the leader proposes. A commit is `Verdict`. +/// An abort is `AbortVerdict`, which carries the reason the coordinator +/// reports. pub(crate) fn verdict_entry(txn: TxnId, outcome: VerdictOutcome) -> SequencerEntry { match outcome { VerdictOutcome::Commit => SequencerEntry::Verdict { epoch: txn.epoch, position: txn.position, - commit: true, }, - VerdictOutcome::Abort(Some(reason)) => SequencerEntry::AbortVerdict { + VerdictOutcome::Abort(reason) => SequencerEntry::AbortVerdict { epoch: txn.epoch, position: txn.position, reason, }, - VerdictOutcome::Abort(None) => SequencerEntry::Verdict { - epoch: txn.epoch, - position: txn.position, - commit: false, - }, } } @@ -36,10 +29,10 @@ mod tests { use crate::calvin::AbortReason; #[test] - fn abort_with_a_reason_proposes_abort_verdict() { + fn abort_proposes_abort_verdict_with_its_reason() { let entry = verdict_entry( TxnId::new(4, 1), - VerdictOutcome::Abort(Some(AbortReason::ParticipantError)), + VerdictOutcome::Abort(AbortReason::ParticipantError), ); assert_eq!( entry, @@ -52,21 +45,12 @@ mod tests { } #[test] - fn commit_and_reasonless_abort_keep_the_original_verdict_shape() { + fn commit_proposes_verdict() { assert_eq!( verdict_entry(TxnId::new(4, 2), VerdictOutcome::Commit), SequencerEntry::Verdict { epoch: 4, position: 2, - commit: true, - } - ); - assert_eq!( - verdict_entry(TxnId::new(4, 3), VerdictOutcome::Abort(None)), - SequencerEntry::Verdict { - epoch: 4, - position: 3, - commit: false, } ); } diff --git a/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs b/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs index 4ff93a135..94d5d3dae 100644 --- a/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs +++ b/nodedb-cluster/src/calvin/sequencer/state_machine/apply.rs @@ -12,14 +12,14 @@ use std::sync::atomic::Ordering; -use tokio::sync::mpsc; use tracing::{error, warn}; use crate::calvin::sequencer::entry::SequencerEntry; use crate::calvin::sequencer::epoch_guard::{EpochCheck, SequencerHalt, classify}; -use crate::calvin::types::SchedulerInput; +use crate::calvin::types::{EpochBatch, SchedulerInput, TxnIdWire}; +use crate::calvin::{ParticipantVote, TxnId, VerdictOutcome}; -use super::core::SequencerStateMachine; +use super::core::{Delivery, SequencerStateMachine}; impl SequencerStateMachine { /// Apply a committed Raft log entry. @@ -85,228 +85,25 @@ impl SequencerStateMachine { }; match entry { - SequencerEntry::EpochBatch { mut batch } => { - // Re-derive the participating_vshards field which is skipped - // during serialization (it is computed from write_set collection names). - // A class whose participants cannot be derived makes the entry - // as unusable as one that fails to decode, so it is skipped the - // same way. - for txn in &mut batch.txns { - if let Err(err) = txn.tx_class.restore_derived() { - error!( - epoch = batch.epoch, - raft_index = index, - error = %err, - "sequencer state machine: epoch batch carries a transaction \ - with underivable participants; skipping entry" - ); - crate::diag::sequencer_participants_underivable( - batch.epoch, - index, - &err.to_string(), - ); - return; - } - } - - // A halted state machine has already diverged from the log; - // resuming fan-out mid-divergence is how a detected fault turns - // into corrupted lock-table and completion state. - if self.halted { - error!( - epoch = batch.epoch, - raft_index = index, - "sequencer state machine is halted on an epoch regression; \ - refusing to apply further epoch batches" - ); - return; - } - - let expected = self.next_epoch(); - let check = classify(expected, batch.epoch); - match check { - EpochCheck::InOrder => {} - // Entries are missing on THIS replica, but the batch in - // hand is intact and self-describing. Dropping it would - // add fresh data loss on top of the entries already - // missed, so it is fanned out and the hole is reported — - // the scheduler recovers the missed range by replaying the - // sequencer Raft log. - EpochCheck::Ahead => { - error!( - epoch = batch.epoch, - expected, - raft_index = index, - "sequencer state machine: epoch gap detected; this node missed \ - entries. Fanning out the batch in hand; the skipped epochs must \ - be recovered by log replay." - ); - self.metrics - .epochs_skipped_gap - .fetch_add(1, Ordering::Relaxed); - crate::diag::sequencer_epoch_gap( - expected, - batch.epoch, - check.direction(), - batch.txns.len(), - index, - ); - } - // A NEW log entry (an index never applied here — the - // re-delivery guard at the top of `apply` already returned - // for the ones that were) carrying an already-consumed - // epoch. Every `(epoch, position)` in this batch aliases one - // that has already run here, so fanning it out would collide - // with live lock-table and completion entries — and dropping - // it would silently discard committed writes. Neither is - // acceptable: halt and escalate. - EpochCheck::Behind => { - error!( - epoch = batch.epoch, - expected, - raft_index = index, - txns = batch.txns.len(), - "sequencer state machine: epoch regression; a committed epoch was \ - proposed a second time. Halting the sequencer state machine \ - rather than aliasing committed transaction identities." - ); - self.metrics - .epochs_refused_regression - .fetch_add(1, Ordering::Relaxed); - crate::diag::sequencer_epoch_gap( - expected, - batch.epoch, - check.direction(), - batch.txns.len(), - index, - ); - self.halted = true; - if let Some(hook) = self.unrecoverable_hook.as_ref() { - hook(SequencerHalt { - expected_epoch: expected, - found_epoch: batch.epoch, - txns_in_batch: batch.txns.len(), - raft_index: index, - }); - } - return; - } - } - - let mut fanned_out = 0u64; - let mut dropped = 0u64; - // Collected for a single end-of-call diagnostics report, - // never emitted per-txn — a sustained backpressure storm can - // drop many positions in one apply() call and per-txn - // emission would report-storm. - let mut drop_pairs: Vec<(u32, &'static str)> = Vec::new(); - - // Per-vShard count of how many of this epoch's positions target - // each vShard. Delivered to each scheduler so it knows how many - // positions of the epoch it must apply before the epoch is fully - // applied on its vShard — the input to its per-`(epoch, position)` - // applied gate and fully-applied watermark. Every position of an - // epoch targeting a given vShard is stamped with the same count. - // Shared with the replay path via `compute_vshard_txn_counts` so - // the two paths can never drift. - let vshard_txn_counts = - crate::calvin::sequencer::replay::compute_vshard_txn_counts(&batch); - for txn in &batch.txns { - // Seed the expected vote-participant count deterministically on - // EVERY replica (not just the epoch's originating leader), so a - // post-failover sequencer leader can still detect vote - // completeness and aggregate the verdict. - self.completion_registry.seed_expected( - crate::calvin::TxnId::new(batch.epoch, txn.position), - txn.tx_class.participating_vshards().len(), - ); - } - - for txn in &batch.txns { - // Build a per-shard copy with epoch_system_ms stamped from - // the batch. This is the deterministic time anchor that engine - // handlers use instead of reading the wall clock themselves. - let mut txn_with_ts = txn.clone(); - txn_with_ts.epoch_system_ms = batch.epoch_system_ms; - - // Fan out only to vshards that participate in this txn. - let vshards = txn.tx_class.participating_vshards(); - for vshard_id in vshards { - let vshard = vshard_id.as_u32(); - if let Some(sender) = self.vshard_senders.get(&vshard) { - // Stamp the per-vShard position count for the vShard - // this copy is delivered to. - let mut per_vshard = txn_with_ts.clone(); - per_vshard.epoch_vshard_txn_count = - vshard_txn_counts.get(&vshard).copied().unwrap_or(0); - match sender.try_send(SchedulerInput::Txn(per_vshard)) { - Ok(()) => { - fanned_out += 1; - } - Err(mpsc::error::TrySendError::Full(_)) => { - warn!( - epoch = batch.epoch, - position = txn.position, - vshard, - "sequencer apply: vshard channel full (backpressure); \ - dropping txn. Scheduler will catch up via log replay." - ); - self.record_catch_up(vshard, index); - dropped += 1; - drop_pairs.push((vshard, "full")); - } - Err(mpsc::error::TrySendError::Closed(_)) => { - warn!( - vshard, - epoch = batch.epoch, - "sequencer apply: vshard sender gone; \ - scheduler may have exited" - ); - self.record_catch_up(vshard, index); - dropped += 1; - drop_pairs.push((vshard, "closed")); - } - } - } - // If no sender registered for this vshard, silently skip — - // this node may not host that vshard. - } - } - - if dropped > 0 { - crate::diag::sequencer_backpressure_drop(batch.epoch, dropped, &drop_pairs); - } - - self.metrics - .txns_fanned_out - .fetch_add(fanned_out, Ordering::Relaxed); - self.metrics - .txns_dropped_backpressure - .fetch_add(dropped, Ordering::Relaxed); - self.metrics.epochs_applied.fetch_add(1, Ordering::Relaxed); - self.last_applied_epoch = batch.epoch; - } + SequencerEntry::EpochBatch { batch } => self.apply_epoch_batch(index, batch), SequencerEntry::CompletionAck { epoch, position, vshard_id, - } => { - let txn = crate::calvin::TxnId::new(epoch, position); - self.completion_registry.note_completion_ack(txn, vshard_id); - self.completion_registry - .applied_acks - .record(crate::calvin::AppliedCompletionAck { - index, - txn, - vshard_id, - }); - } + result, + from_node, + } => self.apply_completion_ack( + index, + TxnId::new(epoch, position), + vshard_id, + from_node, + result, + ), // Broadcast the OLLP predicate-mismatch signal to ALL replicas so the // coordinator's registry fires wherever it lives (including remote nodes). - SequencerEntry::OllpMismatch { epoch, position } => { - self.completion_registry - .note_ollp_mismatch(crate::calvin::TxnId::new(epoch, position)); - } + SequencerEntry::OllpMismatch { epoch, position } => self + .completion_registry + .note_ollp_mismatch(TxnId::new(epoch, position)), // Broadcast the terminal routing-failure signal to ALL replicas so // the coordinator's registry fires wherever it lives (including // remote nodes), mirroring `OllpMismatch`. @@ -314,169 +111,389 @@ impl SequencerStateMachine { epoch, position, detail, - } => { - self.completion_registry - .note_routing_failed(crate::calvin::TxnId::new(epoch, position), detail); - } - // Durable per-participant commit vote for a staged cross-shard txn. - // The registry tallies votes per vshard; once every participant has - // voted the leader aggregates them into the global verdict that gates - // the cross-shard commit barrier (flush on commit, drop on abort). + } => self + .completion_registry + .note_routing_failed(TxnId::new(epoch, position), detail), + // Durable per-participant votes for a staged cross-shard txn. The + // registry tallies them per vshard. Once every participant voted, + // the leader aggregates them into the global verdict that gates the + // cross-shard commit barrier (flush on commit, drop on abort). SequencerEntry::Vote { epoch, position, vshard, - commit, - } => { - self.completion_registry.note_vote( - crate::calvin::TxnId::new(epoch, position), - vshard, - if commit { - crate::calvin::ParticipantVote::Commit - } else { - // A pre-existing abort vote records no reason. - crate::calvin::ParticipantVote::Abort(None) - }, - ); - } - // Durable per-participant ABORT vote carrying its cause. Split from - // `Vote` so the reason reaches the coordinator without changing - // `Vote`'s wire shape. + } => self.completion_registry.note_vote( + TxnId::new(epoch, position), + vshard, + ParticipantVote::Commit, + ), SequencerEntry::AbortVote { epoch, position, vshard, reason, - } => { - self.completion_registry.note_vote( - crate::calvin::TxnId::new(epoch, position), - vshard, - crate::calvin::ParticipantVote::Abort(Some(reason)), - ); - } - // Authoritative commit/abort verdict for a staged cross-shard txn, - // proposed by the leader once every participant voted. Applied on - // ALL replicas to store the durable decision, which releases every - // participant parked at the cross-shard commit barrier into its - // flush (commit) or drop (abort). - SequencerEntry::Verdict { + } => self.completion_registry.note_vote( + TxnId::new(epoch, position), + vshard, + ParticipantVote::Abort(reason), + ), + // Authoritative verdict for a staged cross-shard txn, proposed by + // the leader once every participant voted. Applied on ALL replicas + // to store the durable decision, which releases every participant + // parked at the cross-shard commit barrier into its flush (commit) + // or drop (abort). + SequencerEntry::Verdict { epoch, position } => self + .completion_registry + .note_verdict(TxnId::new(epoch, position), VerdictOutcome::Commit), + SequencerEntry::AbortVerdict { epoch, position, - commit, - } => { - // A pre-existing abort verdict records no reason; `Abort(None)` - // is exactly that unknown cause. - let outcome = if commit { - crate::calvin::VerdictOutcome::Commit - } else { - crate::calvin::VerdictOutcome::Abort(None) - }; - self.completion_registry - .note_verdict(crate::calvin::TxnId::new(epoch, position), outcome); + reason, + } => self + .completion_registry + .note_verdict(TxnId::new(epoch, position), VerdictOutcome::Abort(reason)), + // The owning vShard's scheduler installs the SHARED lock. + SequencerEntry::ReserveRead { owner, vshard, key } => self.forward_reservation( + index, + vshard, + owner, + SchedulerInput::Reserve { owner, key }, + "read reservation", + ), + // The owning vShard's scheduler releases every shared lock of `owner`. + SequencerEntry::ReleaseReservation { + owner, + vshard, + reason, + } => self.forward_reservation( + index, + vshard, + owner, + SchedulerInput::Release { owner, reason }, + "reservation release", + ), + SequencerEntry::EpochFloor { + next_epoch, + epoch_system_ms, + } => self.apply_epoch_floor(next_epoch, epoch_system_ms), + SequencerEntry::CutMarker { hlc, restore_point } => { + self.apply_cut_marker(index, hlc, restore_point) } - // Authoritative ABORT verdict carrying the winning participant - // reason. Split from `Verdict` for the same wire-shape reason as - // `AbortVote`. - SequencerEntry::AbortVerdict { + SequencerEntry::TxnPart { epoch, position, - reason, - } => { - self.completion_registry.note_verdict( - crate::calvin::TxnId::new(epoch, position), - crate::calvin::VerdictOutcome::Abort(Some(reason)), + index: part, + first_task, + targets, + plans, + chunk, + } => self.apply_txn_part( + index, + TxnId::new(epoch, position), + super::parts::PartEntry { + index: part, + first_task, + targets, + plans, + chunk, + }, + ), + SequencerEntry::TxnPartsAbandoned { epoch, position } => { + self.apply_parts_abandoned(index, TxnId::new(epoch, position)) + } + } + } + + /// Apply a committed epoch batch: re-derive each txn's participants, + /// check the epoch order, then fan the batch out. + fn apply_epoch_batch(&mut self, index: u64, mut batch: EpochBatch) { + // Re-derive the participating_vshards field which is skipped + // during serialization (it is computed from write_set collection names). + // A class whose participants cannot be derived makes the entry + // as unusable as one that fails to decode, so it is skipped the + // same way. + for txn in &mut batch.txns { + if let Err(err) = txn.tx_class.restore_derived() { + error!( + epoch = batch.epoch, + raft_index = index, + error = %err, + "sequencer state machine: epoch batch carries a transaction \ + with underivable participants; skipping entry" ); + crate::diag::sequencer_participants_underivable( + batch.epoch, + index, + &err.to_string(), + ); + return; } - // Fan a hot-key read reservation out to its owning vShard's scheduler, - // which installs the SHARED lock. Same `try_send` backpressure - // discipline as the epoch-batch fan-out: a full/closed channel logs - // and drops (this node may not host the vShard, in which case there is - // simply no sender registered). - SequencerEntry::ReserveRead { owner, vshard, key } => { - if let Some(sender) = self.vshard_senders.get(&vshard) { - match sender.try_send(SchedulerInput::Reserve { owner, key }) { - Ok(()) => {} - Err(mpsc::error::TrySendError::Full(_)) => { - warn!( - vshard, - owner_epoch = owner.epoch, - owner_position = owner.position, - "sequencer apply: vshard channel full (backpressure); \ - dropping read reservation" - ); - self.record_catch_up(vshard, index); - } - Err(mpsc::error::TrySendError::Closed(_)) => { - warn!( - vshard, - "sequencer apply: vshard sender gone; \ - scheduler may have exited (reservation)" - ); - self.record_catch_up(vshard, index); - } - } - } + } + if self.epoch_in_order(index, &batch) { + self.last_epoch_system_ms = Some( + self.last_epoch_system_ms + .map_or(batch.epoch_system_ms, |seen| { + seen.max(batch.epoch_system_ms) + }), + ); + self.open_multi_parts(index, &batch); + self.fan_out_epoch_batch(index, batch); + } + } + + /// Whether `batch` may be fanned out. `false` when the state machine is + /// halted or the batch re-mints a consumed epoch. A forward gap is + /// reported and still returns `true`. + fn epoch_in_order(&mut self, index: u64, batch: &EpochBatch) -> bool { + // A halted state machine has already diverged from the log; + // resuming fan-out mid-divergence is how a detected fault turns + // into corrupted lock-table and completion state. + if self.halted { + error!( + epoch = batch.epoch, + raft_index = index, + "sequencer state machine is halted on an epoch regression; \ + refusing to apply further epoch batches" + ); + return false; + } + + let expected = self.next_epoch(); + let check = classify(expected, batch.epoch); + match check { + EpochCheck::InOrder => {} + // Entries are missing on THIS replica, but the batch in + // hand is intact and self-describing. Dropping it would + // add fresh data loss on top of the entries already + // missed, so it is fanned out and the hole is reported — + // the scheduler recovers the missed range by replaying the + // sequencer Raft log. + EpochCheck::Ahead => { + error!( + epoch = batch.epoch, + expected, + raft_index = index, + "sequencer state machine: epoch gap detected; this node missed \ + entries. Fanning out the batch in hand; the skipped epochs must \ + be recovered by log replay." + ); + self.metrics + .epochs_skipped_gap + .fetch_add(1, Ordering::Relaxed); + crate::diag::sequencer_epoch_gap( + expected, + batch.epoch, + check.direction(), + batch.txns.len(), + index, + ); } - // Fan a backup's cut marker out to every vShard scheduler this - // node hosts. Same `try_send` discipline as `ReserveRead`: a - // dropped marker is recovered by the scheduler's catch-up drain, - // which replays it in log order. - SequencerEntry::CutMarker { hlc } => { - for (&vshard, sender) in &self.vshard_senders { - match sender.try_send(SchedulerInput::CutMarker { hlc }) { - Ok(()) => {} - Err(mpsc::error::TrySendError::Full(_)) => { - warn!( - vshard, - hlc, - "sequencer apply: vshard channel full (backpressure); \ - dropping cut marker" - ); - self.record_catch_up(vshard, index); - } - Err(mpsc::error::TrySendError::Closed(_)) => { - warn!( - vshard, - "sequencer apply: vshard sender gone; \ - scheduler may have exited (cut marker)" - ); - self.record_catch_up(vshard, index); - } - } + // A NEW log entry (an index never applied here — the + // re-delivery guard at the top of `apply` already returned + // for the ones that were) carrying an already-consumed + // epoch. Every `(epoch, position)` in this batch aliases one + // that has already run here, so fanning it out would collide + // with live lock-table and completion entries — and dropping + // it would silently discard committed writes. Neither is + // acceptable: halt and escalate. + EpochCheck::Behind => { + error!( + epoch = batch.epoch, + expected, + raft_index = index, + txns = batch.txns.len(), + "sequencer state machine: epoch regression; a committed epoch was \ + proposed a second time. Halting the sequencer state machine \ + rather than aliasing committed transaction identities." + ); + self.metrics + .epochs_refused_regression + .fetch_add(1, Ordering::Relaxed); + crate::diag::sequencer_epoch_gap( + expected, + batch.epoch, + check.direction(), + batch.txns.len(), + index, + ); + self.halted = true; + if let Some(hook) = self.unrecoverable_hook.as_ref() { + hook(SequencerHalt { + expected_epoch: expected, + found_epoch: batch.epoch, + txns_in_batch: batch.txns.len(), + raft_index: index, + }); } + return false; } - // Fan a reservation release out to its owning vShard's scheduler. - // Same `try_send` discipline as `ReserveRead`. - SequencerEntry::ReleaseReservation { - owner, - vshard, - reason, - } => { - if let Some(sender) = self.vshard_senders.get(&vshard) { - match sender.try_send(SchedulerInput::Release { owner, reason }) { - Ok(()) => {} - Err(mpsc::error::TrySendError::Full(_)) => { - warn!( - vshard, - owner_epoch = owner.epoch, - owner_position = owner.position, - "sequencer apply: vshard channel full (backpressure); \ - dropping reservation release" - ); - self.record_catch_up(vshard, index); - } - Err(mpsc::error::TrySendError::Closed(_)) => { - warn!( - vshard, - "sequencer apply: vshard sender gone; \ - scheduler may have exited (reservation release)" - ); - self.record_catch_up(vshard, index); - } + } + true + } + + /// Fan each txn of `batch` out to the scheduler of every participating + /// vShard this node hosts, then advance the applied epoch. + fn fan_out_epoch_batch(&mut self, index: u64, mut batch: EpochBatch) { + let mut fanned_out = 0u64; + let mut dropped = 0u64; + // Collected for a single end-of-call diagnostics report, + // never emitted per-txn — a sustained backpressure storm can + // drop many positions in one apply() call and per-txn + // emission would report-storm. + let mut drop_pairs: Vec<(u32, &'static str)> = Vec::new(); + + // Per-vShard count of how many of this epoch's positions target + // each vShard. Delivered to each scheduler so it knows how many + // positions of the epoch it must apply before the epoch is fully + // applied on its vShard — the input to its per-`(epoch, position)` + // applied gate and fully-applied watermark. Every position of an + // epoch targeting a given vShard is stamped with the same count. + // Shared with the replay path via `compute_vshard_txn_counts` so + // the two paths can never drift. + let vshard_txn_counts = crate::calvin::sequencer::replay::compute_vshard_txn_counts(&batch); + for txn in &batch.txns { + // Seed the expected vote-participant count deterministically on + // EVERY replica (not just the epoch's originating leader), so a + // post-failover sequencer leader can still detect vote + // completeness and aggregate the verdict. + self.completion_registry.seed_expected( + crate::calvin::TxnId::new(batch.epoch, txn.position), + txn.tx_class.participating_vshards().len(), + ); + } + + let epoch_system_ms = batch.epoch_system_ms; + for txn in &mut batch.txns { + // Stamp epoch_system_ms from the batch. This is the + // deterministic time anchor that engine handlers use + // instead of reading the wall clock themselves. + txn.epoch_system_ms = epoch_system_ms; + + // Fan out only to vshards that participate in this txn. + let vshards = txn.tx_class.participating_vshards(); + for vshard_id in vshards { + let vshard = vshard_id.as_u32(); + // This node may not host the vShard: then nothing is sent. + if !self.vshard_senders.contains_key(&vshard) { + continue; + } + // Stamp the per-vShard position count for the vShard this + // copy is delivered to. + let mut per_vshard = txn.clone(); + per_vshard.epoch_vshard_txn_count = + vshard_txn_counts.get(&vshard).copied().unwrap_or(0); + match self.deliver(index, vshard, SchedulerInput::Txn(Box::new(per_vshard))) { + Delivery::NotHosted => {} + Delivery::Sent => { + fanned_out += 1; + self.undurable + .note(index, vshard, batch.epoch, txn.position); + } + // The armed catch-up replays it in log order. + Delivery::Deferred => { + dropped += 1; + drop_pairs.push((vshard, "catch_up")); + } + Delivery::DroppedFull => { + warn!( + epoch = batch.epoch, + position = txn.position, + vshard, + "sequencer apply: vshard channel full (backpressure); \ + txn left to the catch-up replay" + ); + dropped += 1; + drop_pairs.push((vshard, "full")); + } + Delivery::DroppedClosed => { + warn!( + vshard, + epoch = batch.epoch, + "sequencer apply: vshard sender gone; scheduler may have exited" + ); + dropped += 1; + drop_pairs.push((vshard, "closed")); } } } } + + if dropped > 0 { + crate::diag::sequencer_backpressure_drop(batch.epoch, dropped, &drop_pairs); + } + + self.metrics + .txns_fanned_out + .fetch_add(fanned_out, Ordering::Relaxed); + self.metrics + .txns_dropped_backpressure + .fetch_add(dropped, Ordering::Relaxed); + self.metrics.epochs_applied.fetch_add(1, Ordering::Relaxed); + self.last_applied_epoch = batch.epoch; + } + + /// Record `vshard_id`'s completion ack for `txn`, with its apply result, + /// and log the ack at its Raft `index`. + /// + /// The first ack of a vShard in log order answers for it, on every node + /// alike: the result depends on the log alone. Only the vShard's + /// data-group leader proposes an ack, so a node that left the group adds + /// none. `from_node` names the proposer for tracing. + fn apply_completion_ack( + &self, + index: u64, + txn: TxnId, + vshard_id: u32, + from_node: u64, + result: Vec, + ) { + tracing::trace!( + epoch = txn.epoch, + position = txn.position, + vshard_id, + from_node, + "sequencer apply: completion ack" + ); + self.completion_registry + .note_completion_ack_with(txn, vshard_id, result); + self.completion_registry + .applied_acks + .record(crate::calvin::AppliedCompletionAck { + index, + txn, + vshard_id, + }); + } + + /// Send a reservation `input` for `owner` to `vshard`'s scheduler. + /// + /// Same delivery as the epoch-batch fan-out: an armed vShard, or a full + /// or closed channel, leaves the input to the catch-up replay. This node + /// may not host the vShard. Then nothing is sent. `what` names the input + /// in the warnings. + fn forward_reservation( + &self, + index: u64, + vshard: u32, + owner: TxnIdWire, + input: SchedulerInput, + what: &'static str, + ) { + match self.deliver(index, vshard, input) { + Delivery::NotHosted | Delivery::Sent | Delivery::Deferred => {} + Delivery::DroppedFull => warn!( + vshard, + owner_epoch = owner.epoch, + owner_position = owner.position, + what, + "sequencer apply: vshard channel full (backpressure); \ + reservation input left to the catch-up replay" + ), + Delivery::DroppedClosed => warn!( + vshard, + what, "sequencer apply: vshard sender gone; scheduler may have exited" + ), + } } } @@ -485,6 +502,8 @@ mod tests { use std::collections::HashMap; use std::sync::Arc; + use tokio::sync::mpsc; + use super::*; use crate::calvin::CalvinCompletionRegistry; use crate::calvin::types::{ @@ -883,7 +902,7 @@ mod tests { let (tx_b, _rx_b) = mpsc::channel(1); // Pre-fill channel A so it is full. let pre_fill: SequencedTxn = batch.txns[0].clone(); - let _ = tx_a.try_send(SchedulerInput::Txn(pre_fill)); + let _ = tx_a.try_send(SchedulerInput::Txn(Box::new(pre_fill))); let mut senders = HashMap::new(); senders.insert(va, tx_a); senders.insert(vb, tx_b); @@ -956,7 +975,6 @@ mod tests { let data = encode_entry(&SequencerEntry::Verdict { epoch: 9, position: 4, - commit: true, }); sm.apply(1, &data); @@ -988,7 +1006,7 @@ mod tests { assert_eq!( rx.await.expect("completion fires"), crate::calvin::AttemptOutcome::Aborted { - reason: Some(crate::calvin::AbortReason::ParticipantError) + reason: crate::calvin::AbortReason::ParticipantError } ); assert_eq!(sm.last_applied_epoch(), None); @@ -1011,16 +1029,14 @@ mod tests { assert_eq!( registry.vote_tally(txn).and_then(|t| t.get(&3).copied()), - Some(crate::calvin::ParticipantVote::Abort(Some( + Some(crate::calvin::ParticipantVote::Abort( crate::calvin::AbortReason::ParticipantError - ))) + )) ); } #[tokio::test] - async fn apply_legacy_abort_vote_tallies_without_a_reason() { - // A `Vote { commit: false }` durable before abort reasons existed must - // still decode and tally — with an unknown cause, never an invented one. + async fn apply_vote_tallies_a_commit_vote() { let registry = CalvinCompletionRegistry::new_detached(); let mut sm = SequencerStateMachine::new(HashMap::new(), Arc::clone(®istry)); let txn = crate::calvin::TxnId::new(9, 7); @@ -1029,16 +1045,69 @@ mod tests { epoch: 9, position: 7, vshard: 3, - commit: false, }); sm.apply(1, &data); assert_eq!( registry.vote_tally(txn).and_then(|t| t.get(&3).copied()), - Some(crate::calvin::ParticipantVote::Abort(None)) + Some(crate::calvin::ParticipantVote::Commit) ); } + /// Every txn a replica fans out carries the batch's `epoch_system_ms`, + /// whatever value the proposer left on the txn itself. + #[test] + fn fanned_out_txns_carry_the_batch_epoch_system_ms() { + let (mut batch, va, vb) = make_batch_with_two_vshards(); + batch.epoch_system_ms = 1_800_000_000_000; + for txn in &mut batch.txns { + txn.epoch_system_ms = 0; + } + let (tx_a, mut rx_a) = mpsc::channel(64); + let (tx_b, mut rx_b) = mpsc::channel(64); + let mut senders = HashMap::new(); + senders.insert(va, tx_a); + senders.insert(vb, tx_b); + let mut sm = SequencerStateMachine::new(senders, CalvinCompletionRegistry::new_detached()); + + sm.apply(1, &encode_entry(&SequencerEntry::EpochBatch { batch })); + + for rx in [&mut rx_a, &mut rx_b] { + match rx.try_recv() { + Ok(SchedulerInput::Txn(txn)) => { + assert_eq!(txn.epoch_system_ms, 1_800_000_000_000); + assert_eq!(txn.epoch_vshard_txn_count, 1); + } + _ => panic!("each participating vShard receives the txn"), + } + } + } + + /// The state machine keeps the highest epoch instant it applied, the + /// floor a leader seeded from it mints above. + #[test] + fn applied_epoch_instant_is_the_highest_applied() { + let mut sm = + SequencerStateMachine::new(HashMap::new(), CalvinCompletionRegistry::new_detached()); + assert_eq!(sm.last_epoch_system_ms(), None); + for (index, (epoch, ms)) in [(0u64, 5_000i64), (1, 4_000), (2, 6_000)] + .into_iter() + .enumerate() + { + let (mut batch, _, _) = make_batch_with_two_vshards(); + batch.epoch = epoch; + for txn in &mut batch.txns { + txn.epoch = epoch; + } + batch.epoch_system_ms = ms; + sm.apply( + index as u64 + 1, + &encode_entry(&SequencerEntry::EpochBatch { batch }), + ); + } + assert_eq!(sm.last_epoch_system_ms(), Some(6_000)); + } + #[test] fn catch_up_from_records_dropped_index_and_min_collapses() { let (batch, va, vb) = make_batch_with_two_vshards(); @@ -1046,7 +1115,7 @@ mod tests { let (tx_a, _rx_a) = mpsc::channel(1); // vshard B has room and a live receiver → never drops. let (tx_b, _rx_b) = mpsc::channel(64); - let _ = tx_a.try_send(SchedulerInput::Txn(batch.txns[0].clone())); + let _ = tx_a.try_send(SchedulerInput::Txn(Box::new(batch.txns[0].clone()))); let mut senders = HashMap::new(); senders.insert(va, tx_a); senders.insert(vb, tx_b); @@ -1079,6 +1148,45 @@ mod tests { assert_eq!(sm.take_catch_up_from(va), None); } + /// An armed vShard takes no live input, even with room on its channel: + /// a later input overtaking an earlier dropped one would reach the + /// scheduler out of log order. The replay delivers both. + #[test] + fn an_armed_vshard_defers_every_later_input_to_the_replay() { + let (batch, va, _vb) = make_batch_with_two_vshards(); + let (tx_a, mut rx_a) = mpsc::channel(1); + let _ = tx_a.try_send(SchedulerInput::Txn(Box::new(batch.txns[0].clone()))); + let mut senders = HashMap::new(); + senders.insert(va, tx_a); + let mut sm = SequencerStateMachine::new(senders, CalvinCompletionRegistry::new_detached()); + + sm.apply( + 4, + &encode_entry(&SequencerEntry::EpochBatch { + batch: batch.clone(), + }), + ); + // The scheduler reads the pre-filled input: the channel has room. + assert!(rx_a.try_recv().is_ok()); + let mut later = batch; + later.epoch = 1; + for txn in &mut later.txns { + txn.epoch = 1; + } + sm.apply( + 7, + &encode_entry(&SequencerEntry::EpochBatch { batch: later }), + ); + + assert!(rx_a.try_recv().is_err(), "the later input is not sent live"); + assert_eq!(sm.peek_catch_up_from(va), Some(4)); + // A replay through 5 leaves index 7 owed: still armed, from 6. + sm.clear_catch_up_up_to(va, 5); + assert_eq!(sm.peek_catch_up_from(va), Some(6)); + sm.clear_catch_up_up_to(va, 7); + assert_eq!(sm.peek_catch_up_from(va), None); + } + /// PEEK must not consume: the scheduler drain reads the armed index, and /// only clears it after a confirmed replay. A take-then-early-return (the /// old shape) silently lost the miss when the replay could not complete. @@ -1086,7 +1194,7 @@ mod tests { fn peek_catch_up_from_does_not_consume() { let (batch, va, _vb) = make_batch_with_two_vshards(); let (tx_a, _rx_a) = mpsc::channel(1); - let _ = tx_a.try_send(SchedulerInput::Txn(batch.txns[0].clone())); + let _ = tx_a.try_send(SchedulerInput::Txn(Box::new(batch.txns[0].clone()))); let mut senders = HashMap::new(); senders.insert(va, tx_a); let mut sm = SequencerStateMachine::new(senders, CalvinCompletionRegistry::new_detached()); @@ -1121,7 +1229,13 @@ mod tests { /// across all vShards, so no replica's replay range is compacted away. #[test] fn min_catch_up_from_is_lowest_armed_index_across_vshards() { - let senders = HashMap::new(); + let mut senders = HashMap::new(); + let mut receivers = Vec::new(); + for vshard in 1..=3 { + let (tx, rx) = mpsc::channel(4); + senders.insert(vshard, tx); + receivers.push(rx); + } let sm = SequencerStateMachine::new(senders, CalvinCompletionRegistry::new_detached()); assert_eq!(sm.min_catch_up_from(), None); @@ -1139,6 +1253,53 @@ mod tests { assert_eq!(sm.min_catch_up_from(), None); } + /// A vShard that leaves this node takes its catch-up with it. Its armed + /// index never holds sequencer compaction down afterward. + #[test] + fn a_retired_vshard_releases_the_compaction_floor() { + let (tx, _rx) = mpsc::channel(4); + let mut sm = SequencerStateMachine::new( + HashMap::from([(7, tx)]), + CalvinCompletionRegistry::new_detached(), + ); + sm.arm_catch_up_from(7, 1); + assert_eq!(sm.min_catch_up_from(), Some(1)); + + sm.remove_vshard_sender(7); + assert_eq!(sm.min_catch_up_from(), None); + assert_eq!(sm.peek_catch_up_from(7), None); + } + + /// An exiting scheduler can arm after its sender is gone. That arm never + /// counts toward the floor while no sender is registered. + #[test] + fn an_arm_without_a_sender_never_holds_the_floor() { + let mut sm = + SequencerStateMachine::new(HashMap::new(), CalvinCompletionRegistry::new_detached()); + sm.arm_catch_up_from(9, 40); + assert_eq!(sm.min_catch_up_from(), None); + + let (tx, _rx) = mpsc::channel(4); + sm.set_vshard_sender(9, tx); + assert_eq!(sm.min_catch_up_from(), Some(40)); + } + + /// Replacing a vShard's sender keeps its pending catch-up. Only a vShard + /// that leaves this node drops it. + #[test] + fn a_replaced_sender_keeps_the_pending_catch_up() { + let (tx, _rx) = mpsc::channel(4); + let mut sm = SequencerStateMachine::new( + HashMap::from([(5, tx)]), + CalvinCompletionRegistry::new_detached(), + ); + sm.arm_catch_up_from(5, 12); + let (tx2, _rx2) = mpsc::channel(4); + sm.set_vshard_sender(5, tx2); + assert_eq!(sm.peek_catch_up_from(5), Some(12)); + assert_eq!(sm.min_catch_up_from(), Some(12)); + } + #[test] fn catch_up_from_records_dropped_index_on_closed_channel() { let (batch, va, vb) = make_batch_with_two_vshards(); @@ -1169,7 +1330,6 @@ mod tests { let data = encode_entry(&SequencerEntry::Verdict { epoch: 1, position: 0, - commit: true, }); sm.apply(42, &data); assert_eq!(sm.current_committed_index(), Some(42)); diff --git a/nodedb-cluster/src/calvin/sequencer/state_machine/core.rs b/nodedb-cluster/src/calvin/sequencer/state_machine/core.rs index a7cfde5ba..64a252010 100644 --- a/nodedb-cluster/src/calvin/sequencer/state_machine/core.rs +++ b/nodedb-cluster/src/calvin/sequencer/state_machine/core.rs @@ -17,6 +17,33 @@ use crate::calvin::types::SchedulerInput; use super::counters::StateMachineMetrics; +/// One cluster restore point, as this replica applies the sequencer's cut +/// marker for it. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct SequencerRestorePoint { + pub id: u64, + /// The point's watermark HLC. + pub hlc: u64, + /// The marker's sequencer log index. + pub index: u64, + /// The first epoch the sequencer proposes after the marker. + pub next_epoch: u64, + /// The highest epoch instant (ms) applied before the marker, or `None` + /// when no epoch applied. + pub epoch_system_ms: Option, +} + +/// Records a restore point's sequencer place. Runs on the Raft tick thread, +/// so it must not block. +pub type RestorePointHook = Arc; + +/// Records the epoch instant at a cut marker: the marker's watermark HLC, its +/// sequencer log index, and the highest epoch instant (ms) applied before +/// it, `None` when no epoch applied. Every epoch before the marker has an +/// instant at or below it, and every epoch after it has a higher one. Runs +/// on the Raft tick thread, so it must not block. +pub type CutInstantHook = Arc) + Send + Sync>; + /// The Calvin sequencer Raft state machine. /// /// One instance per replica (including leader). Applied on every `CommitApplier` @@ -27,6 +54,11 @@ pub struct SequencerStateMachine { /// has been applied yet (using `u64::MAX` avoids a separate `Option` and /// makes the "nothing applied" state explicit). pub(super) last_applied_epoch: u64, + /// The highest `epoch_system_ms` of an epoch batch applied here, or + /// `None` before the first one. Log replay rebuilds it, so a leader that + /// seeds from it after a restart or a leader change never mints an epoch + /// instant at or below a committed one. + pub(super) last_epoch_system_ms: Option, /// Raft log index of the last committed entry applied on this replica. /// `NOT_YET_APPLIED` means nothing has been applied yet. Advanced for EVERY /// applied entry (not just `EpochBatch`), so it is a safe upper bound for the @@ -34,15 +66,19 @@ pub struct SequencerStateMachine { pub(super) last_committed_index: u64, /// Per-vshard output channels. The scheduler subscribes on the other end. pub(super) vshard_senders: HashMap>, - /// Per-vShard "catch up from this Raft index" bookkeeping. - /// - /// When a fan-out `try_send` to a vShard's scheduler channel fails (Full or - /// Closed), the input for that vShard was dropped. The current entry's Raft - /// index is recorded here with MIN-COLLAPSE (the smallest dropped index per - /// vShard wins), so the scheduler-side drain replays the sequencer Raft log - /// from the earliest miss forward. Bounded by the number of hosted vShards — - /// a vShard contributes at most one entry until its catch-up is drained. - pub(super) catch_up_from: Mutex>, + /// Per-vShard armed catch-up: the Raft indexes whose inputs the vShard's + /// scheduler still owes a replay of. + /// + /// - A fan-out `try_send` that fails (Full or Closed) arms the vShard at + /// the entry's index. + /// - While a vShard is armed, every later input for it is deferred to the + /// replay, never sent live. The scheduler therefore receives its inputs + /// in exact log order: the channel holds only inputs from before the + /// arm, and the replay delivers the rest. + /// - The scheduler-side drain replays the range and clears it. + /// + /// Bounded by the number of hosted vShards. + pub(super) catch_up_from: Mutex>, /// Set once an already-consumed epoch was proposed again. While set, no /// further `EpochBatch` is applied: this replica's epoch sequence and the /// proposing leader's have diverged, and fanning out under a colliding @@ -52,12 +88,49 @@ pub struct SequencerStateMachine { /// Host escalation for the halt above. `None` in tests and in embedded /// callers with no fail-stop path; production wires it to node shutdown. pub(super) unrecoverable_hook: Option, + /// Host hook that records the sequencer's place at a cluster restore + /// point. `None` in tests and embedded callers. + pub(super) restore_point_hook: Option, + /// Host hook that records the epoch instant at every cut marker. `None` + /// in tests and embedded callers. + pub(super) cut_instant_hook: Option, pub metrics: Arc, pub(super) completion_registry: Arc, + /// Every multi-part transaction whose header applied here and whose + /// parts have not all applied, nor been abandoned (see [`super::parts`]). + pub(super) open_parts: super::parts::OpenParts, + /// `Txn` inputs this node's schedulers received and may not have made + /// durable (see [`super::undurable`]). Local to this replica. + pub(super) undurable: super::undurable::UndurableInputs, } pub(super) const NOT_YET_APPLIED: u64 = u64::MAX; +/// The Raft indexes a vShard's armed catch-up covers. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct CatchUpRange { + /// The lowest index whose input the scheduler has not received. + pub from: u64, + /// The highest index whose input was dropped or deferred to the replay. + pub through: u64, +} + +/// What became of one input for a vShard's scheduler. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum Delivery { + /// This node hosts no scheduler for the vShard. + NotHosted, + /// Sent on the scheduler's channel. + Sent, + /// The vShard's catch-up is armed: the replay delivers it, in log order. + Deferred, + /// The channel was full: the catch-up is armed from this index. + DroppedFull, + /// The scheduler's channel is closed: the catch-up is armed from this + /// index. + DroppedClosed, +} + impl SequencerStateMachine { /// Construct a fresh state machine with no applied epochs. pub fn new( @@ -66,13 +139,18 @@ impl SequencerStateMachine { ) -> Self { Self { last_applied_epoch: NOT_YET_APPLIED, + last_epoch_system_ms: None, last_committed_index: NOT_YET_APPLIED, vshard_senders, catch_up_from: Mutex::new(HashMap::new()), halted: false, unrecoverable_hook: None, + restore_point_hook: None, + cut_instant_hook: None, metrics: StateMachineMetrics::new(), completion_registry, + open_parts: super::parts::OpenParts::default(), + undurable: super::undurable::UndurableInputs::default(), } } @@ -89,6 +167,22 @@ impl SequencerStateMachine { self } + /// Install the host's hook that records the sequencer's place at a + /// cluster restore point. + #[must_use] + pub fn with_restore_point_hook(mut self, hook: RestorePointHook) -> Self { + self.restore_point_hook = Some(hook); + self + } + + /// Install the host's hook that records the epoch instant at every cut + /// marker. + #[must_use] + pub fn with_cut_instant_hook(mut self, hook: CutInstantHook) -> Self { + self.cut_instant_hook = Some(hook); + self + } + /// Whether this state machine has halted on an unrecoverable epoch /// regression and is refusing to apply further epoch batches. pub fn is_halted(&self) -> bool { @@ -105,6 +199,15 @@ impl SequencerStateMachine { } } + /// The highest epoch instant (`epoch_system_ms`) of an epoch batch + /// applied here, or `None` before the first one. + /// + /// Like [`Self::next_epoch`], it answers for the whole log only once + /// every committed entry has been applied here. + pub fn last_epoch_system_ms(&self) -> Option { + self.last_epoch_system_ms + } + /// The epoch number that the next proposal should use. /// /// INVARIANT: the first epoch a restarted leader proposes must be strictly @@ -133,8 +236,20 @@ impl SequencerStateMachine { /// Remove the output sender for a vshard (e.g. when a vshard is migrated /// away from this node). + /// + /// Its catch-up goes with it. No scheduler here will replay it, so an arm + /// left behind would hold sequencer compaction down forever. pub fn remove_vshard_sender(&mut self, vshard: u32) { self.vshard_senders.remove(&vshard); + self.drop_catch_up(vshard); + } + + /// Forget `vshard`'s armed catch-up. + fn drop_catch_up(&self, vshard: u32) { + self.catch_up_from + .lock() + .unwrap_or_else(|p| p.into_inner()) + .remove(&vshard); } /// The highest epoch number that has been committed and applied on this @@ -161,15 +276,44 @@ impl SequencerStateMachine { } } - /// Record that a fan-out to `vshard` was dropped at Raft index `index`. + /// Record that the input for `vshard` at Raft index `index` goes to the + /// replay. /// - /// Min-collapse: the smallest dropped index per vShard is retained, so the - /// scheduler-side drain replays from the earliest miss forward. O(1), no I/O. + /// The range widens to cover `index`: the smallest index stays the + /// replay's start, and the largest one tells the drain whether its clear + /// leaves inputs owed. O(1), no I/O. pub(super) fn record_catch_up(&self, vshard: u32, index: u64) { let mut map = self.catch_up_from.lock().unwrap_or_else(|p| p.into_inner()); - map.entry(vshard) - .and_modify(|i| *i = (*i).min(index)) - .or_insert(index); + widen(&mut map, vshard, index); + } + + /// Deliver `input`, the input for `vshard` at Raft index `index`, to the + /// vShard's scheduler. + /// + /// An armed vShard takes no live input: the input is deferred to the + /// replay, so the scheduler never receives a later input before an + /// earlier one. Otherwise a `try_send` sends it, and a full or closed + /// channel arms the catch-up from `index`. Never blocks. + pub(super) fn deliver(&self, index: u64, vshard: u32, input: SchedulerInput) -> Delivery { + let Some(sender) = self.vshard_senders.get(&vshard) else { + return Delivery::NotHosted; + }; + let mut map = self.catch_up_from.lock().unwrap_or_else(|p| p.into_inner()); + if map.contains_key(&vshard) { + widen(&mut map, vshard, index); + return Delivery::Deferred; + } + match sender.try_send(input) { + Ok(()) => Delivery::Sent, + Err(mpsc::error::TrySendError::Full(_)) => { + widen(&mut map, vshard, index); + Delivery::DroppedFull + } + Err(mpsc::error::TrySendError::Closed(_)) => { + widen(&mut map, vshard, index); + Delivery::DroppedClosed + } + } } /// Take (remove and return) the catch-up-from Raft index for `vshard`. @@ -179,7 +323,7 @@ impl SequencerStateMachine { /// drop is pending for the vShard. The next drop re-records a fresh index. pub fn take_catch_up_from(&self, vshard: u32) -> Option { let mut map = self.catch_up_from.lock().unwrap_or_else(|p| p.into_inner()); - map.remove(&vshard) + map.remove(&vshard).map(|range| range.from) } /// Arm a catch-up for `vshard` from `index` (min-collapse), so the @@ -205,36 +349,74 @@ impl SequencerStateMachine { /// dropping it — the loss the old take-then-early-return had. pub fn peek_catch_up_from(&self, vshard: u32) -> Option { let map = self.catch_up_from.lock().unwrap_or_else(|p| p.into_inner()); - map.get(&vshard).copied() + map.get(&vshard).map(|range| range.from) } - /// Clear `vshard`'s catch-up entry only if its recorded index is `<= up_to`. + /// Mark `vshard`'s catch-up replayed through `up_to`. /// - /// Called after a successful replay of `lo ..= up_to`: the recorded miss is - /// now covered, so clear it — unless a concurrent drop has already lowered - /// the entry to an index the just-finished replay did not cover (only - /// possible for an index `<= up_to` given min-collapse, hence the guard is a - /// belt-and-braces no-op in that case). A newer drop recorded at an index - /// `> up_to` is preserved for the next drain. + /// Called after a replay of `from ..= up_to`. An input deferred or + /// dropped past `up_to` while the replay ran keeps the vShard armed from + /// `up_to + 1`, so the next drain delivers it and no live input overtakes + /// it. With none, the vShard is disarmed and live delivery resumes. An + /// armed range that starts past `up_to` is left as it is. pub fn clear_catch_up_up_to(&self, vshard: u32, up_to: u64) { let mut map = self.catch_up_from.lock().unwrap_or_else(|p| p.into_inner()); - if let Some(&idx) = map.get(&vshard) - && idx <= up_to - { + let Some(range) = map.get_mut(&vshard) else { + return; + }; + if range.from > up_to { + return; + } + if range.through > up_to { + range.from = up_to.saturating_add(1); + } else { map.remove(&vshard); } } - /// The smallest armed catch-up index across ALL vShards, or `None` when no - /// catch-up is pending. + /// Arm `vshard`'s catch-up past the last applied entry, so every later + /// input for it goes to the replay. Returns the armed start: the first + /// index the replay owes, or the start already armed when that is lower. + /// + /// The scheduler calls it when it stops reading its channel: the channel + /// then holds a finite run of inputs, and the log holds the rest. + pub fn arm_catch_up_past_applied(&self, vshard: u32) -> u64 { + let next = self + .current_committed_index() + .map_or(0, |i| i.saturating_add(1)); + let mut map = self.catch_up_from.lock().unwrap_or_else(|p| p.into_inner()); + widen(&mut map, vshard, next); + map.get(&vshard).map_or(next, |range| range.from) + } + + /// The smallest armed catch-up index across the hosted vShards, or `None` + /// when no catch-up is pending. /// /// The sequencer-group log compactor floors its compaction index at this /// value so a dropped/undelivered fan-out is always replayable from the /// retained log — the hold-down the scheduler-side drain's `LogCompacted` - /// arm depends on. Only hosted vShards ever arm a catch-up, so this never - /// pins compaction on a vShard this node does not serve. + /// arm depends on. Only a vShard with a registered sender counts. A + /// scheduler that is exiting can arm after its sender is gone, and that + /// arm must never pin compaction on a vShard this node does not serve. pub fn min_catch_up_from(&self) -> Option { let map = self.catch_up_from.lock().unwrap_or_else(|p| p.into_inner()); - map.values().copied().min() + map.iter() + .filter(|(vshard, _)| self.vshard_senders.contains_key(vshard)) + .map(|(_, range)| range.from) + .min() } } + +/// Widen `vshard`'s armed range in `map` to cover `index`, arming it at +/// `index` when unarmed. +fn widen(map: &mut HashMap, vshard: u32, index: u64) { + map.entry(vshard) + .and_modify(|range| { + range.from = range.from.min(index); + range.through = range.through.max(index); + }) + .or_insert(CatchUpRange { + from: index, + through: index, + }); +} diff --git a/nodedb-cluster/src/calvin/sequencer/state_machine/mod.rs b/nodedb-cluster/src/calvin/sequencer/state_machine/mod.rs index 95c914411..d6cf02875 100644 --- a/nodedb-cluster/src/calvin/sequencer/state_machine/mod.rs +++ b/nodedb-cluster/src/calvin/sequencer/state_machine/mod.rs @@ -3,8 +3,15 @@ pub mod apply; pub mod core; pub mod counters; +pub mod parts; +pub mod restore; +pub mod snapshot; +pub mod undurable; // `self::` is required: a bare `core` in a `use` path resolves to the `core` // crate, not this module's sibling. -pub use self::core::SequencerStateMachine; +pub use self::core::{ + CutInstantHook, RestorePointHook, SequencerRestorePoint, SequencerStateMachine, +}; pub use counters::StateMachineMetrics; +pub use snapshot::SequencerSnapshot; diff --git a/nodedb-cluster/src/calvin/sequencer/state_machine/parts.rs b/nodedb-cluster/src/calvin/sequencer/state_machine/parts.rs new file mode 100644 index 000000000..5e5e8bc9d --- /dev/null +++ b/nodedb-cluster/src/calvin/sequencer/state_machine/parts.rs @@ -0,0 +1,615 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The state machine's side of multi-part transactions. +//! +//! A multi-part transaction is open from its header's epoch batch until its +//! last part applies, or until a `TxnPartsAbandoned` for it applies first. +//! Every replica decides both from the log alone, so every replica opens and +//! closes the same transactions at the same entries: +//! +//! - A part of an open transaction fans out to the schedulers of its targets +//! this node hosts, as a [`SchedulerInput::TxnPart`]. A part of a +//! transaction that is not open is ignored. +//! - An abandonment of an open transaction aborts it with +//! `AbortReason::PartsLost` in the completion registry, and fans a +//! [`SchedulerInput::PartsAbandoned`] out to every participant this node +//! hosts. An abandonment of a transaction that is not open is ignored. +//! +//! The log compactor keeps every open transaction's header (see +//! [`SequencerStateMachine::min_open_parts_index`]), so a replica that +//! replays the log meets the header before its parts. + +use std::collections::{BTreeMap, BTreeSet}; +use std::sync::Arc; + +use tracing::{debug, error, warn}; + +use crate::calvin::TxnId; +use crate::calvin::types::{EpochBatch, SchedulerInput, TxnIdWire}; + +use super::core::{Delivery, SequencerStateMachine}; +use super::snapshot::OpenTxnImage; + +/// One open multi-part transaction. +#[derive(Debug)] +struct OpenTxn { + /// Raft index of the header's epoch batch. + header_index: u64, + /// Every participating vShard, for the abandonment fan-out. + participants: Vec, + /// How many parts the header announced. + count: u32, + /// The parts applied so far. + seen: BTreeSet, +} + +/// The open multi-part transactions of a state machine. +#[derive(Debug, Default)] +pub struct OpenParts { + txns: BTreeMap, +} + +impl OpenParts { + /// Every open transaction, as a snapshot carries it. + pub(super) fn images(&self) -> Vec { + self.txns + .iter() + .map(|(txn, open)| OpenTxnImage { + epoch: txn.epoch, + position: txn.position, + header_index: open.header_index, + participants: open.participants.clone(), + count: open.count, + seen: open.seen.iter().copied().collect(), + }) + .collect() + } + + /// The open transactions a snapshot carries. + pub(super) fn from_images(images: Vec) -> Self { + let txns = images + .into_iter() + .map(|image| { + ( + image.txn(), + OpenTxn { + header_index: image.header_index, + participants: image.participants, + count: image.count, + seen: image.seen.into_iter().collect(), + }, + ) + }) + .collect(); + Self { txns } + } +} + +impl SequencerStateMachine { + /// Open every multi-part transaction of `batch`, the epoch batch at Raft + /// index `index`. + pub(super) fn open_multi_parts(&mut self, index: u64, batch: &EpochBatch) { + for txn in &batch.txns { + let Some(manifest) = &txn.tx_class.multi_part else { + continue; + }; + self.open_parts.txns.insert( + TxnId::new(batch.epoch, txn.position), + OpenTxn { + header_index: index, + participants: txn + .tx_class + .participating_vshards() + .iter() + .map(|vshard| vshard.as_u32()) + .collect(), + count: manifest.part_count, + seen: BTreeSet::new(), + }, + ); + } + } + + /// Apply part `part` of the transaction `txn`, at Raft index `index`. + pub(super) fn apply_txn_part(&mut self, index: u64, txn: TxnId, part: PartEntry) { + let Some(open) = self.open_parts.txns.get_mut(&txn) else { + debug!( + epoch = txn.epoch, + position = txn.position, + part = part.index, + "sequencer apply: part of a transaction that is not open; ignored" + ); + return; + }; + if part.index >= open.count { + error!( + epoch = txn.epoch, + position = txn.position, + part = part.index, + count = open.count, + raft_index = index, + "sequencer apply: part index past the header's part count; ignored" + ); + return; + } + if !open.seen.insert(part.index) { + return; + } + let complete = open.seen.len() == open.count as usize; + if complete { + self.open_parts.txns.remove(&txn); + } + let plans = Arc::new(part.plans); + for vshard in part.targets { + self.send_part_input( + index, + vshard, + SchedulerInput::TxnPart { + txn: TxnIdWire { + epoch: txn.epoch, + position: txn.position, + }, + index: part.index, + first_task: part.first_task, + plans: Arc::clone(&plans), + chunk: part.chunk, + }, + ); + } + } + + /// Apply the abandonment of the transaction `txn`, at Raft index `index`. + pub(super) fn apply_parts_abandoned(&mut self, index: u64, txn: TxnId) { + let Some(open) = self.open_parts.txns.remove(&txn) else { + debug!( + epoch = txn.epoch, + position = txn.position, + "sequencer apply: abandonment of a transaction that is not open; ignored" + ); + return; + }; + warn!( + epoch = txn.epoch, + position = txn.position, + parts_seen = open.seen.len(), + parts = open.count, + "sequencer apply: multi-part transaction lost its parts; aborting it" + ); + self.completion_registry.note_parts_abandoned(txn); + for vshard in open.participants { + self.send_part_input( + index, + vshard, + SchedulerInput::PartsAbandoned { + txn: TxnIdWire { + epoch: txn.epoch, + position: txn.position, + }, + }, + ); + } + } + + /// Deliver `input` to `vshard`'s scheduler when this node hosts it. An + /// armed vShard, or a full or closed channel, leaves it to the replay, + /// as for every other input. + fn send_part_input(&self, index: u64, vshard: u32, input: SchedulerInput) { + match self.deliver(index, vshard, input) { + Delivery::NotHosted | Delivery::Sent | Delivery::Deferred => {} + Delivery::DroppedFull => warn!( + vshard, + raft_index = index, + "sequencer apply: vshard channel full (backpressure); part input left to replay" + ), + Delivery::DroppedClosed => warn!( + vshard, + "sequencer apply: vshard sender gone; scheduler may have exited (part input)" + ), + } + } + + /// Every open multi-part transaction, in sequence order. The sequencer + /// leader abandons the ones whose parts it does not hold. + pub fn open_multi_part_txns(&self) -> Vec { + self.open_parts.txns.keys().copied().collect() + } + + /// The lowest Raft index of an open transaction's header, or `None` + /// when none is open. The log compactor keeps it. + pub fn min_open_parts_index(&self) -> Option { + self.open_parts + .txns + .values() + .map(|open| open.header_index) + .min() + } +} + +/// The fields of a `SequencerEntry::TxnPart` the apply path reads. +#[derive(Debug)] +pub(super) struct PartEntry { + pub index: u32, + pub first_task: u32, + pub targets: Vec, + pub plans: Vec, + pub chunk: Option, +} + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + + use nodedb_types::TenantId; + use nodedb_types::id::{CollectionKey, DatabaseId}; + use tokio::sync::mpsc; + + use super::*; + use crate::calvin::CalvinCompletionRegistry; + use crate::calvin::sequencer::entry::SequencerEntry; + use crate::calvin::types::{ + EngineKeySet, MultiPartPlans, PartStreamId, ReadWriteSet, SequencedTxn, SortedVec, + TaskChunk, TxClass, VShardParts, VersionedReadSet, + }; + + /// Two collections homed on distinct vShards, with those vShards. + fn two_homes() -> ((String, u32), (String, u32)) { + let home = |name: &str| { + CollectionKey::from_bare(DatabaseId::DEFAULT, name) + .vshard() + .as_u32() + }; + let first = ("col_0".to_owned(), home("col_0")); + let second = (1u32..512) + .map(|i| format!("col_{i}")) + .find(|name| home(name) != first.1) + .map(|name| { + let vshard = home(&name); + (name, vshard) + }) + .expect("two distinct homes in 512 names"); + (first, second) + } + + /// The epoch-0 batch whose one txn is a two-part header: part 0 targets + /// `va`, part 1 targets `vb`. + fn header_batch() -> (EpochBatch, u32, u32) { + header_batch_of(2) + } + + /// The epoch-0 batch whose one txn is the header of `parts` parts. + fn header_batch_of(parts: u32) -> (EpochBatch, u32, u32) { + let ((col_a, va), (col_b, vb)) = two_homes(); + let write_set = ReadWriteSet::new(vec![ + EngineKeySet::Document { + collection: col_a, + surrogates: SortedVec::new(vec![1]), + }, + EngineKeySet::Document { + collection: col_b, + surrogates: SortedVec::new(vec![2]), + }, + ]); + let mut tx_class = TxClass::new( + ReadWriteSet::new(vec![]), + write_set, + vec![], + TenantId::new(1), + None, + VersionedReadSet::default(), + ) + .expect("valid TxClass"); + let mut per_vshard = vec![ + VShardParts { + vshard: va, + parts: parts - 1, + }, + VShardParts { + vshard: vb, + parts: 1, + }, + ]; + per_vshard.sort_by_key(|entry| entry.vshard); + tx_class.multi_part = Some(MultiPartPlans { + stream: PartStreamId { node: 1, seq: 1 }, + part_count: parts, + total_tasks: parts, + user_write: true, + client_write: true, + per_vshard, + }); + let batch = EpochBatch { + epoch: 0, + txns: vec![SequencedTxn { + epoch: 0, + position: 0, + tx_class, + epoch_system_ms: 1_700_000_000_000, + epoch_vshard_txn_count: 1, + lock_owner: None, + }], + epoch_system_ms: 1_700_000_000_000, + }; + (batch, va, vb) + } + + fn encode(entry: &SequencerEntry) -> Vec { + zerompk::to_msgpack_vec(entry).expect("encode") + } + + fn part_entry(index: u32, target: u32) -> SequencerEntry { + sized_part_entry(index, target, 1) + } + + fn sized_part_entry(index: u32, target: u32, bytes: usize) -> SequencerEntry { + SequencerEntry::TxnPart { + epoch: 0, + position: 0, + index, + first_task: index, + targets: vec![target], + plans: vec![0x90; bytes], + chunk: None, + } + } + + fn abandon_entry() -> SequencerEntry { + SequencerEntry::TxnPartsAbandoned { + epoch: 0, + position: 0, + } + } + + struct Fixture { + sm: SequencerStateMachine, + registry: Arc, + rx_a: mpsc::Receiver, + rx_b: mpsc::Receiver, + va: u32, + vb: u32, + } + + /// A state machine that applied the header at Raft index 1, with the + /// header's `Txn` input already taken from both channels. + fn fixture() -> Fixture { + fixture_of(2) + } + + /// [`fixture`] for a header of `parts` parts. + fn fixture_of(parts: u32) -> Fixture { + let (batch, va, vb) = header_batch_of(parts); + let depth = usize::try_from(parts).unwrap_or(usize::MAX) + 16; + let (tx_a, mut rx_a) = mpsc::channel(depth); + let (tx_b, mut rx_b) = mpsc::channel(depth); + let registry = CalvinCompletionRegistry::new_detached(); + let mut senders = HashMap::new(); + senders.insert(va, tx_a); + senders.insert(vb, tx_b); + let mut sm = SequencerStateMachine::new(senders, registry.clone()); + sm.apply(1, &encode(&SequencerEntry::EpochBatch { batch })); + assert!(matches!(rx_a.try_recv(), Ok(SchedulerInput::Txn(_)))); + assert!(matches!(rx_b.try_recv(), Ok(SchedulerInput::Txn(_)))); + Fixture { + sm, + registry, + rx_a, + rx_b, + va, + vb, + } + } + + /// A part reaches only its targets. An abandonment while a part is + /// missing aborts the txn, reaches every participant, and ignores the + /// parts that follow it. + #[test] + fn an_abandonment_while_a_part_is_missing_aborts_the_whole_txn() { + let mut f = fixture(); + let txn = TxnId::new(0, 0); + assert_eq!(f.sm.open_multi_part_txns(), [txn]); + assert_eq!(f.sm.min_open_parts_index(), Some(1)); + + f.sm.apply(2, &encode(&part_entry(0, f.va))); + assert!(matches!( + f.rx_a.try_recv(), + Ok(SchedulerInput::TxnPart { index: 0, .. }) + )); + assert!(f.rx_b.try_recv().is_err(), "part 0 does not target vb"); + + f.sm.apply(3, &encode(&abandon_entry())); + assert_eq!(f.registry.verdict(txn), Some(false)); + assert!(matches!( + f.rx_a.try_recv(), + Ok(SchedulerInput::PartsAbandoned { .. }) + )); + assert!(matches!( + f.rx_b.try_recv(), + Ok(SchedulerInput::PartsAbandoned { .. }) + )); + assert!(f.sm.open_multi_part_txns().is_empty()); + assert_eq!(f.sm.min_open_parts_index(), None); + + f.sm.apply(4, &encode(&part_entry(1, f.vb))); + assert!(f.rx_b.try_recv().is_err(), "a part after the abandonment"); + } + + /// The last part closes the txn. An abandonment after it is ignored, and + /// a part delivered twice fans out once. + #[test] + fn an_abandonment_after_the_last_part_is_ignored() { + let mut f = fixture(); + let txn = TxnId::new(0, 0); + f.sm.apply(2, &encode(&part_entry(0, f.va))); + f.sm.apply(3, &encode(&part_entry(0, f.va))); + f.sm.apply(4, &encode(&part_entry(1, f.vb))); + assert!(f.sm.open_multi_part_txns().is_empty()); + assert!(matches!( + f.rx_a.try_recv(), + Ok(SchedulerInput::TxnPart { index: 0, .. }) + )); + assert!(f.rx_a.try_recv().is_err(), "the duplicate part is dropped"); + assert!(matches!( + f.rx_b.try_recv(), + Ok(SchedulerInput::TxnPart { index: 1, .. }) + )); + + f.sm.apply(5, &encode(&abandon_entry())); + assert_eq!(f.registry.verdict(txn), None); + assert!(f.rx_a.try_recv().is_err()); + assert!(f.rx_b.try_recv().is_err()); + } + + /// A txn of 72 one-MiB parts, over the 64 MiB RPC limit, stays open and + /// uncommitted until its last part applies. Every part reaches its + /// target whole, a chunk part keeps its place in its task, and nothing + /// aborts it. + #[test] + fn a_txn_over_64_mib_of_parts_applies_whole() { + const PARTS: u32 = 72; + let mut f = fixture_of(PARTS); + let txn = TxnId::new(0, 0); + for index in 0..PARTS - 1 { + let mut entry = sized_part_entry(index, f.va, 1 << 20); + if index == 5 + && let SequencerEntry::TxnPart { chunk, .. } = &mut entry + { + *chunk = Some(TaskChunk { + offset: 0, + total_len: 1 << 20, + }); + } + f.sm.apply(u64::from(index) + 2, &encode(&entry)); + assert_eq!( + f.sm.open_multi_part_txns(), + [txn], + "open before the last part" + ); + } + f.sm.apply(u64::from(PARTS) + 1, &encode(&part_entry(PARTS - 1, f.vb))); + assert!( + f.sm.open_multi_part_txns().is_empty(), + "the last part closes it" + ); + assert_eq!(f.registry.verdict(txn), None, "nothing aborted it"); + + let mut bytes = 0usize; + let mut chunks = 0; + while let Ok(input) = f.rx_a.try_recv() { + let SchedulerInput::TxnPart { plans, chunk, .. } = input else { + panic!("only parts after the header"); + }; + bytes += plans.len(); + chunks += usize::from(chunk.is_some()); + } + assert_eq!(bytes, 71 << 20); + assert_eq!(chunks, 1); + assert!(matches!( + f.rx_b.try_recv(), + Ok(SchedulerInput::TxnPart { index: 71, .. }) + )); + } + + /// A leader change mid-stream: the next leader's abandonment lands after + /// 40 of 72 parts. The whole txn aborts, every participant is told, and + /// the parts the old leader proposed after it change nothing. + #[test] + fn a_leader_change_mid_stream_aborts_the_whole_txn() { + const PARTS: u32 = 72; + let mut f = fixture_of(PARTS); + let txn = TxnId::new(0, 0); + for index in 0..40 { + f.sm.apply( + u64::from(index) + 2, + &encode(&sized_part_entry(index, f.va, 1 << 10)), + ); + } + f.sm.apply(42, &encode(&abandon_entry())); + assert_eq!(f.registry.verdict(txn), Some(false)); + assert!(f.sm.open_multi_part_txns().is_empty()); + for index in 40..PARTS - 1 { + f.sm.apply(u64::from(index) + 3, &encode(&part_entry(index, f.va))); + } + f.sm.apply(99, &encode(&part_entry(PARTS - 1, f.vb))); + + let mut parts = 0; + let mut abandoned = 0; + while let Ok(input) = f.rx_a.try_recv() { + match input { + SchedulerInput::TxnPart { .. } => parts += 1, + SchedulerInput::PartsAbandoned { .. } => abandoned += 1, + other => panic!("unexpected input {other:?}"), + } + } + assert_eq!((parts, abandoned), (40, 1)); + assert!(matches!( + f.rx_b.try_recv(), + Ok(SchedulerInput::PartsAbandoned { .. }) + )); + assert!(f.rx_b.try_recv().is_err(), "no part after the abandonment"); + } + + /// A part dropped at a full channel arms the catch-up. The next part is + /// deferred to the replay too, though the channel has room again, so the + /// scheduler never receives part `I + 1` before part `I`. + #[test] + fn a_part_after_a_dropped_part_is_never_sent_live() { + let (batch, va, _vb) = header_batch_of(3); + let (tx_a, mut rx_a) = mpsc::channel(1); + let mut senders = HashMap::new(); + senders.insert(va, tx_a); + let mut sm = SequencerStateMachine::new(senders, CalvinCompletionRegistry::new_detached()); + // The header fills the one-slot channel. + sm.apply(1, &encode(&SequencerEntry::EpochBatch { batch })); + sm.apply(2, &encode(&part_entry(0, va))); + assert_eq!(sm.peek_catch_up_from(va), Some(2), "part 0 dropped"); + + assert!(matches!(rx_a.try_recv(), Ok(SchedulerInput::Txn(_)))); + sm.apply(3, &encode(&part_entry(1, va))); + assert!(rx_a.try_recv().is_err(), "part 1 waits for the replay"); + sm.clear_catch_up_up_to(va, 2); + assert_eq!( + sm.peek_catch_up_from(va), + Some(3), + "the replay still owes part 1" + ); + } + + /// Catch-up replay hands a vShard the header, the parts that target it, + /// and every abandonment, in log order. + #[test] + fn replay_hands_a_vshard_only_the_parts_that_target_it() { + let (batch, va, vb) = header_batch(); + let sm = + SequencerStateMachine::new(HashMap::new(), CalvinCompletionRegistry::new_detached()); + let log: Vec = [ + SequencerEntry::EpochBatch { batch }, + part_entry(0, va), + abandon_entry(), + ] + .iter() + .zip(1u64..) + .map(|(entry, index)| nodedb_raft::LogEntry { + term: 1, + index, + data: encode(entry), + }) + .collect(); + + let for_a = sm.replay_epochs_for_vshard(&log, va, 0, 0); + assert!(matches!( + for_a.as_slice(), + [ + SchedulerInput::Txn(_), + SchedulerInput::TxnPart { index: 0, .. }, + SchedulerInput::PartsAbandoned { .. }, + ] + )); + let for_b = sm.replay_epochs_for_vshard(&log, vb, 0, 0); + assert!(matches!( + for_b.as_slice(), + [ + SchedulerInput::Txn(_), + SchedulerInput::PartsAbandoned { .. } + ] + )); + } +} diff --git a/nodedb-cluster/src/calvin/sequencer/state_machine/restore.rs b/nodedb-cluster/src/calvin/sequencer/state_machine/restore.rs new file mode 100644 index 000000000..f66828787 --- /dev/null +++ b/nodedb-cluster/src/calvin/sequencer/state_machine/restore.rs @@ -0,0 +1,185 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The sequencer entries a cut and a cluster restore write: the cut marker a +//! backup or restore point takes, and the epoch floor a restored log opens +//! with. + +use tracing::warn; + +use crate::calvin::types::SchedulerInput; + +use super::core::{Delivery, NOT_YET_APPLIED, SequencerRestorePoint, SequencerStateMachine}; + +impl SequencerStateMachine { + /// A restored sequencer log opens with the epoch that followed its + /// restore point. No epoch below it is minted again, and no epoch + /// instant at or below `epoch_system_ms` (`0` for none). + pub(super) fn apply_epoch_floor(&mut self, next_epoch: u64, epoch_system_ms: i64) { + if let Some(floor) = next_epoch.checked_sub(1) + && (self.last_applied_epoch == NOT_YET_APPLIED || self.last_applied_epoch < floor) + { + self.last_applied_epoch = floor; + } + if epoch_system_ms > 0 { + self.last_epoch_system_ms = self.last_epoch_system_ms.max(Some(epoch_system_ms)); + } + } + + /// Record a restore point's place in this log, then fan the cut marker + /// out to every vShard scheduler this node hosts. Same delivery as + /// `ReserveRead`: a marker for an armed vShard, or one dropped at a full + /// channel, is replayed by the scheduler's catch-up drain in log order. + pub(super) fn apply_cut_marker(&mut self, index: u64, hlc: u64, restore_point: u64) { + if restore_point != 0 + && let Some(hook) = &self.restore_point_hook + { + hook(SequencerRestorePoint { + id: restore_point, + hlc, + index, + next_epoch: self.next_epoch(), + epoch_system_ms: self.last_epoch_system_ms, + }); + } + // Recorded before the marker reaches any scheduler, so a scheduler + // that passed the marker implies its instant is recorded. + if let Some(hook) = &self.cut_instant_hook { + hook(hlc, index, self.last_epoch_system_ms); + } + let vshards: Vec = self.vshard_senders.keys().copied().collect(); + for vshard in vshards { + match self.deliver(index, vshard, SchedulerInput::CutMarker { hlc }) { + Delivery::NotHosted | Delivery::Sent | Delivery::Deferred => {} + Delivery::DroppedFull => warn!( + vshard, + hlc, + "sequencer apply: vshard channel full (backpressure); \ + cut marker left to the catch-up replay" + ), + Delivery::DroppedClosed => warn!( + vshard, + "sequencer apply: vshard sender gone; \ + scheduler may have exited (cut marker)" + ), + } + } + } +} + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + use std::sync::{Arc, Mutex}; + + use tokio::sync::mpsc; + + use super::*; + use crate::calvin::CalvinCompletionRegistry; + use crate::calvin::sequencer::entry::SequencerEntry; + + fn encode(entry: &SequencerEntry) -> Vec { + zerompk::to_msgpack_vec(entry).unwrap() + } + + fn floor(next_epoch: u64, epoch_system_ms: i64) -> Vec { + encode(&SequencerEntry::EpochFloor { + next_epoch, + epoch_system_ms, + }) + } + + #[test] + fn a_restored_log_resumes_at_its_epoch_floor() { + let mut sm = + SequencerStateMachine::new(HashMap::new(), CalvinCompletionRegistry::new_detached()); + sm.apply(11, &floor(40, 1_900_000_000_000)); + assert_eq!(sm.last_applied_epoch(), Some(39)); + assert_eq!(sm.next_epoch(), 40); + assert_eq!(sm.last_epoch_system_ms(), Some(1_900_000_000_000)); + + // A floor never lowers an epoch or an epoch instant already applied. + sm.apply(12, &floor(5, 1_000)); + assert_eq!(sm.next_epoch(), 40); + assert_eq!(sm.last_epoch_system_ms(), Some(1_900_000_000_000)); + + // A floor of epoch 0 and no instant leaves a fresh machine at epoch 0. + let mut fresh = + SequencerStateMachine::new(HashMap::new(), CalvinCompletionRegistry::new_detached()); + fresh.apply(1, &floor(0, 0)); + assert_eq!(fresh.next_epoch(), 0); + assert_eq!(fresh.last_epoch_system_ms(), None); + } + + #[test] + fn a_restore_point_cut_marker_reports_the_sequencer_place() { + let seen = Arc::new(Mutex::new(Vec::new())); + let sink = Arc::clone(&seen); + let (tx, mut rx) = mpsc::channel(4); + let mut senders = HashMap::new(); + senders.insert(3, tx); + let mut sm = SequencerStateMachine::new(senders, CalvinCompletionRegistry::new_detached()) + .with_restore_point_hook(Arc::new(move |point: SequencerRestorePoint| { + sink.lock().unwrap().push(point); + })); + sm.apply(20, &floor(8, 1_800_000_000_000)); + + sm.apply( + 21, + &encode(&SequencerEntry::CutMarker { + hlc: 900, + restore_point: 0, + }), + ); + sm.apply( + 22, + &encode(&SequencerEntry::CutMarker { + hlc: 1_000, + restore_point: 77, + }), + ); + + assert_eq!( + *seen.lock().unwrap(), + [SequencerRestorePoint { + id: 77, + hlc: 1_000, + index: 22, + next_epoch: 8, + epoch_system_ms: Some(1_800_000_000_000), + }], + "only a restore point's marker is reported" + ); + for hlc in [900, 1_000] { + assert!(matches!( + rx.try_recv(), + Ok(SchedulerInput::CutMarker { hlc: got }) if got == hlc + )); + } + } + + /// Every cut marker reports its log index and the epoch instant applied + /// before it. + #[test] + fn every_cut_marker_reports_its_epoch_instant() { + let seen = Arc::new(Mutex::new(Vec::new())); + let sink = Arc::clone(&seen); + let mut sm = + SequencerStateMachine::new(HashMap::new(), CalvinCompletionRegistry::new_detached()) + .with_cut_instant_hook(Arc::new(move |hlc, index, instant| { + sink.lock().unwrap().push((hlc, index, instant)); + })); + let marker = |hlc| { + encode(&SequencerEntry::CutMarker { + hlc, + restore_point: 0, + }) + }; + sm.apply(1, &marker(10)); + sm.apply(2, &floor(4, 1_800_000_000_000)); + sm.apply(3, &marker(20)); + assert_eq!( + *seen.lock().unwrap(), + [(10, 1, None), (20, 3, Some(1_800_000_000_000))] + ); + } +} diff --git a/nodedb-cluster/src/calvin/sequencer/state_machine/snapshot.rs b/nodedb-cluster/src/calvin/sequencer/state_machine/snapshot.rs new file mode 100644 index 000000000..8b63c1fd3 --- /dev/null +++ b/nodedb-cluster/src/calvin/sequencer/state_machine/snapshot.rs @@ -0,0 +1,225 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The sequencer state machine's Raft snapshot. +//! +//! A replica rebuilds the state machine by applying the sequencer log. A +//! follower that installs a snapshot never applies the entries the snapshot +//! covers, so the snapshot carries the state they built: +//! +//! - the last applied epoch and its epoch instant, so a later leader never +//! mints an epoch or an instant at or below a committed one; +//! - every open multi-part transaction, with the parts already applied, so +//! the follower closes it on the same entry as every other replica, and +//! never abandons one the others committed. +//! +//! The leader captures the snapshot at its applied index, on the Raft tick +//! thread between apply batches, so it holds exactly the entries through +//! that index. The halt flag is local to a replica and is not carried. + +use crate::calvin::TxnId; +use crate::error::ClusterError; + +use super::core::{NOT_YET_APPLIED, SequencerStateMachine}; + +/// The state the sequencer log built through `applied_index`. +#[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +#[msgpack(map)] +pub struct SequencerSnapshot { + applied_index: u64, + last_applied_epoch: Option, + last_epoch_system_ms: Option, + open_txns: Vec, +} + +/// One open multi-part transaction in a [`SequencerSnapshot`]. +#[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +#[msgpack(map)] +pub(super) struct OpenTxnImage { + pub epoch: u64, + pub position: u32, + pub header_index: u64, + pub participants: Vec, + pub count: u32, + pub seen: Vec, +} + +impl OpenTxnImage { + pub(super) fn txn(&self) -> TxnId { + TxnId::new(self.epoch, self.position) + } +} + +impl SequencerSnapshot { + /// The sequencer log index the snapshot holds the state through. + pub fn applied_index(&self) -> u64 { + self.applied_index + } + + /// Encode the snapshot as a Raft snapshot payload. + pub fn encode(&self) -> Result, ClusterError> { + zerompk::to_msgpack_vec(self).map_err(|e| ClusterError::Codec { + detail: format!("sequencer snapshot encode: {e}"), + }) + } + + /// Decode a Raft snapshot payload. + pub fn decode(bytes: &[u8]) -> Result { + zerompk::from_msgpack(bytes).map_err(|e| ClusterError::Codec { + detail: format!("sequencer snapshot decode: {e}"), + }) + } +} + +impl SequencerStateMachine { + /// Capture the state this replica built through `applied_index`, the + /// group's applied index. + /// + /// The caller holds the state machine's lock on the Raft tick thread, + /// between apply batches. An entry with no payload changes no state, so + /// the state machine's own applied index can trail `applied_index`. + pub fn capture_snapshot(&self, applied_index: u64) -> SequencerSnapshot { + SequencerSnapshot { + applied_index, + last_applied_epoch: self.last_applied_epoch(), + last_epoch_system_ms: self.last_epoch_system_ms, + open_txns: self.open_parts.images(), + } + } + + /// Replace the state the log built with `snapshot`, installed as the + /// group's log boundary. Entries after the snapshot apply on top. + pub fn restore_snapshot(&mut self, snapshot: SequencerSnapshot) { + self.last_applied_epoch = snapshot.last_applied_epoch.unwrap_or(NOT_YET_APPLIED); + self.last_epoch_system_ms = snapshot.last_epoch_system_ms; + self.last_committed_index = snapshot.applied_index; + self.open_parts = super::parts::OpenParts::from_images(snapshot.open_txns); + } +} + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + + use tokio::sync::mpsc; + + use super::*; + use crate::calvin::CalvinCompletionRegistry; + use crate::calvin::sequencer::entry::SequencerEntry; + use crate::calvin::types::{ + EngineKeySet, EpochBatch, MultiPartPlans, PartStreamId, ReadWriteSet, SchedulerInput, + SequencedTxn, SortedVec, TxClass, VShardParts, VersionedReadSet, + }; + use nodedb_types::TenantId; + use nodedb_types::id::{CollectionKey, DatabaseId}; + + fn encode(entry: &SequencerEntry) -> Vec { + zerompk::to_msgpack_vec(entry).expect("encode entry") + } + + /// The epoch-0 batch whose one txn is a two-part header writing `col_0` + /// and a collection on another vShard, and `col_0`'s vShard, which both + /// parts target. A multi-vShard write set makes the class valid. + fn header_batch() -> (EpochBatch, u32) { + let home = |name: &str| { + CollectionKey::from_bare(DatabaseId::DEFAULT, name) + .vshard() + .as_u32() + }; + let vshard = home("col_0"); + let other = (1u32..512) + .map(|i| format!("col_{i}")) + .find(|name| home(name) != vshard) + .expect("a second home in 512 names"); + let write_set = ReadWriteSet::new(vec![ + EngineKeySet::Document { + collection: "col_0".to_owned(), + surrogates: SortedVec::new(vec![1]), + }, + EngineKeySet::Document { + collection: other, + surrogates: SortedVec::new(vec![2]), + }, + ]); + let mut tx_class = TxClass::new( + ReadWriteSet::new(Vec::new()), + write_set, + Vec::new(), + TenantId::new(1), + None, + VersionedReadSet::default(), + ) + .expect("tx class"); + tx_class.multi_part = Some(MultiPartPlans { + stream: PartStreamId { node: 1, seq: 1 }, + part_count: 2, + total_tasks: 2, + user_write: true, + client_write: true, + per_vshard: vec![VShardParts { vshard, parts: 2 }], + }); + let batch = EpochBatch { + epoch: 0, + txns: vec![SequencedTxn { + epoch: 0, + position: 0, + tx_class, + epoch_system_ms: 1_700_000_000_000, + epoch_vshard_txn_count: 1, + lock_owner: None, + }], + epoch_system_ms: 1_700_000_000_000, + }; + (batch, vshard) + } + + fn part(index: u32, target: u32) -> SequencerEntry { + SequencerEntry::TxnPart { + epoch: 0, + position: 0, + index, + first_task: index, + targets: vec![target], + plans: vec![0x90], + chunk: None, + } + } + + /// A follower that installs a snapshot taken between a txn's parts holds + /// the txn open with its applied parts. The last part closes it on the + /// follower as on every other replica, and reaches the follower's + /// scheduler. A snapshot round-trips through its payload. + #[test] + fn a_snapshot_carries_an_open_txn_across_an_install() { + let (batch, vshard) = header_batch(); + let mut leader = + SequencerStateMachine::new(HashMap::new(), CalvinCompletionRegistry::new_detached()); + leader.apply(1, &encode(&SequencerEntry::EpochBatch { batch })); + leader.apply(2, &encode(&part(0, vshard))); + let snapshot = leader.capture_snapshot(3); + let bytes = snapshot.encode().expect("encode"); + let decoded = SequencerSnapshot::decode(&bytes).expect("decode"); + assert_eq!(decoded, snapshot); + assert_eq!(decoded.applied_index(), 3); + + let (tx, mut rx) = mpsc::channel(8); + let mut senders = HashMap::new(); + senders.insert(vshard, tx); + let mut follower = + SequencerStateMachine::new(senders, CalvinCompletionRegistry::new_detached()); + follower.restore_snapshot(decoded); + assert_eq!(follower.open_multi_part_txns(), [TxnId::new(0, 0)]); + assert_eq!(follower.min_open_parts_index(), Some(1)); + assert_eq!(follower.last_applied_epoch(), Some(0)); + assert_eq!(follower.current_committed_index(), Some(3)); + + follower.apply(4, &encode(&part(1, vshard))); + assert!( + follower.open_multi_part_txns().is_empty(), + "the last part closes it" + ); + assert!(matches!( + rx.try_recv(), + Ok(SchedulerInput::TxnPart { index: 1, .. }) + )); + } +} diff --git a/nodedb-cluster/src/calvin/sequencer/state_machine/undurable.rs b/nodedb-cluster/src/calvin/sequencer/state_machine/undurable.rs new file mode 100644 index 000000000..139008326 --- /dev/null +++ b/nodedb-cluster/src/calvin/sequencer/state_machine/undurable.rs @@ -0,0 +1,88 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Transaction inputs this node's schedulers received and may not have made +//! durable yet. +//! +//! A scheduler that restarts catches up from the first index the sequencer +//! log still holds, and skips every position its applied state names. An +//! input it received live, and had not installed when the node stopped, is +//! in neither place once the log compacted past it: the transaction is lost +//! on this replica. The sequencer log therefore keeps every such input. The +//! state machine records each `Txn` input it hands a scheduler, live or by +//! the catch-up replay, and the log compactor drops the ones the scheduler +//! made durable before it reads the lowest index left. +//! +//! The record holds one entry per input not yet seen durable, so it is +//! bounded by the inputs in flight on this node's schedulers. + +use std::collections::BTreeMap; + +use super::core::SequencerStateMachine; + +/// One received input: its vShard, epoch and position. +type Input = (u32, u64, u32); + +/// The inputs not yet seen durable, by sequencer log index. +#[derive(Debug, Default)] +pub(super) struct UndurableInputs { + by_index: BTreeMap>, +} + +impl UndurableInputs { + /// Record the input of `vshard` at `index`. + pub(super) fn note(&mut self, index: u64, vshard: u32, epoch: u64, position: u32) { + let inputs = self.by_index.entry(index).or_default(); + if !inputs.contains(&(vshard, epoch, position)) { + inputs.push((vshard, epoch, position)); + } + } + + /// Drop every input `durable` answers true for, then return the lowest + /// index that still holds one. + fn prune(&mut self, durable: impl Fn(u32, u64, u32) -> bool) -> Option { + self.by_index.retain(|_, inputs| { + inputs.retain(|&(vshard, epoch, position)| !durable(vshard, epoch, position)); + !inputs.is_empty() + }); + self.by_index.keys().next().copied() + } +} + +impl SequencerStateMachine { + /// Record that the catch-up replay handed the `Txn` input of `vshard` at + /// `index` to its scheduler. + pub fn note_replayed_txn(&mut self, index: u64, vshard: u32, epoch: u64, position: u32) { + self.undurable.note(index, vshard, epoch, position); + } + + /// The lowest sequencer index whose `Txn` input a scheduler here received + /// and has not made durable, after dropping every input `durable` answers + /// true for. `durable(vshard, epoch, position)` must also answer true for + /// a vShard this node no longer serves: its replica holds no state. + pub fn undurable_floor(&mut self, durable: impl Fn(u32, u64, u32) -> bool) -> Option { + self.undurable.prune(durable) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn the_floor_is_the_lowest_input_not_yet_durable() { + let mut inputs = UndurableInputs::default(); + inputs.note(10, 1, 4, 0); + inputs.note(10, 2, 4, 0); + inputs.note(12, 1, 5, 0); + inputs.note(12, 1, 5, 0); + assert_eq!(inputs.prune(|_, _, _| false), Some(10)); + + // vShard 1 installed epoch 4; vShard 2 has not. + assert_eq!( + inputs.prune(|vshard, epoch, _| vshard == 1 && epoch == 4), + Some(10) + ); + assert_eq!(inputs.prune(|vshard, _, _| vshard == 2), Some(12)); + assert_eq!(inputs.prune(|_, _, _| true), None); + } +} diff --git a/nodedb-cluster/src/calvin/sequencer/validator.rs b/nodedb-cluster/src/calvin/sequencer/validator.rs index 188fe6fc2..b1c9166c7 100644 --- a/nodedb-cluster/src/calvin/sequencer/validator.rs +++ b/nodedb-cluster/src/calvin/sequencer/validator.rs @@ -139,6 +139,20 @@ fn flatten_set(set: &ReadWriteSet) -> Vec { }); } } + // The whole array on each vShard it names. + EngineKeySet::Array { + collection, + vshards, + } => { + for &vshard in vshards.as_slice() { + out.push(FlatKey { + discriminant: 4, + engine_name: "array", + collection: collection.clone(), + key_bytes: vshard.to_le_bytes().to_vec(), + }); + } + } } } out diff --git a/nodedb-cluster/src/calvin/types/mod.rs b/nodedb-cluster/src/calvin/types/mod.rs index ebb0442fa..0828bd18d 100644 --- a/nodedb-cluster/src/calvin/types/mod.rs +++ b/nodedb-cluster/src/calvin/types/mod.rs @@ -1,16 +1,22 @@ // SPDX-License-Identifier: BUSL-1.1 pub mod lock_wire; +pub mod multi_part; pub mod primitives; +pub mod read_write_set; pub mod scheduler_input; pub mod sequencer; pub mod transaction; pub use lock_wire::{LockKeyWire, ReleaseReason, TxnIdWire}; +pub use multi_part::{ + MultiPartPlans, PartStreamId, PlanPart, StreamedPart, TaskChunk, VShardParts, +}; pub use primitives::{ DependentReadSpec, EngineKeySet, EngineTag, PassiveReadKey, ReadKeyIdent, SortedVec, VersionedReadEntry, VersionedReadSet, }; +pub use read_write_set::ReadWriteSet; pub use scheduler_input::SchedulerInput; pub use sequencer::{EpochBatch, SequencedTxn}; -pub use transaction::{ReadWriteSet, TxClass}; +pub use transaction::{CalvinIncarnation, TxClass}; diff --git a/nodedb-cluster/src/calvin/types/multi_part.rs b/nodedb-cluster/src/calvin/types/multi_part.rs new file mode 100644 index 000000000..4524821c8 --- /dev/null +++ b/nodedb-cluster/src/calvin/types/multi_part.rs @@ -0,0 +1,237 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Plans too large for one sequencer entry, carried as ordered parts. +//! +//! A multi-part transaction is sequenced once, by a header in an epoch batch. +//! The header is the transaction's [`TxClass`] with its full read and write +//! sets and no plans, so every participant takes all its locks at the +//! header's position. Its plans travel as [`PlanPart`]s, each in its own +//! `SequencerEntry::TxnPart` after the header. A participant stages the +//! transaction once every part that targets it has arrived. +//! +//! - A part holds consecutive whole tasks, encoded as the plan batch a +//! single-entry transaction carries in `plans`. `first_task` is the index +//! of its first task in the whole transaction. +//! - A task too large for one part travels as a run of chunk parts. Each +//! holds one byte range of the task's own encoding and nothing else. +//! - A part targets the vShards its tasks route to. Only those receive it. +//! - The header's manifest names how many parts target each participant, so +//! a participant knows when it holds all of its parts. +//! +//! The coordinator streams the parts to the sequencer leader after it +//! submits the header. The manifest names the stream. + +use serde::{Deserialize, Serialize}; + +use super::transaction::TxClass; + +/// The name of one multi-part transaction's part stream: the coordinator +/// node and a sequence that node never reuses. +#[derive( + Debug, + Clone, + Copy, + Default, + PartialEq, + Eq, + PartialOrd, + Ord, + Hash, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct PartStreamId { + pub node: u64, + pub seq: u64, +} + +/// Where a chunk part's bytes sit in the encoding of its one task. +#[derive( + Debug, + Clone, + Copy, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct TaskChunk { + /// Offset of the part's bytes in the task's encoding. + pub offset: u64, + /// Length of the task's whole encoding. + pub total_len: u64, +} + +/// One part of a multi-part transaction's plans. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct PlanPart { + /// Index of the part's first task in the whole transaction. For a chunk + /// part, the index of its one task. + pub first_task: u32, + /// The part's tasks as a plan batch, or for a chunk part, one byte range + /// of its task's encoding. + pub plans: Vec, + /// Set on a chunk part. + #[serde(default)] + #[msgpack(default)] + pub chunk: Option, +} + +/// A part with its index and the vShards it targets, as the coordinator +/// streams it and the sequencer leader queues it. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct StreamedPart { + pub index: u32, + /// The vShards the part's tasks route to, sorted. + pub targets: Vec, + pub part: PlanPart, +} + +/// How many parts target one participating vShard. +#[derive( + Debug, + Clone, + Copy, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct VShardParts { + pub vshard: u32, + pub parts: u32, +} + +/// The plan manifest of a multi-part transaction. Its size grows with the +/// participant count, never with the plan bytes. +#[derive( + Debug, + Clone, + Default, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct MultiPartPlans { + /// The stream that carries the parts to the sequencer leader. + pub stream: PartStreamId, + /// How many parts the transaction carries. + pub part_count: u32, + /// How many tasks the whole transaction carries. + pub total_tasks: u32, + /// Whether any task of the whole transaction is a user write, as opposed + /// to a derived side effect. Opaque to the sequencer: the host decides it. + pub user_write: bool, + /// Whether any task of the whole transaction lies outside `body_plans`: + /// the client wrote, beside any trigger body. Opaque to the sequencer. + pub client_write: bool, + /// How many parts target each vShard, sorted by vShard. A vShard absent + /// here receives no part. + pub per_vshard: Vec, +} + +impl MultiPartPlans { + /// How many parts target `vshard`. + pub fn parts_for(&self, vshard: u32) -> u32 { + self.per_vshard + .binary_search_by_key(&vshard, |entry| entry.vshard) + .map_or(0, |at| self.per_vshard[at].parts) + } +} + +impl TxClass { + /// Whether the transaction carries its plans as parts. + pub fn is_multi_part(&self) -> bool { + self.multi_part.is_some() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn manifest() -> MultiPartPlans { + MultiPartPlans { + stream: PartStreamId { node: 3, seq: 11 }, + part_count: 3, + total_tasks: 7, + user_write: true, + client_write: true, + per_vshard: vec![ + VShardParts { + vshard: 1, + parts: 1, + }, + VShardParts { + vshard: 4, + parts: 2, + }, + ], + } + } + + #[test] + fn a_vshard_waits_only_for_the_parts_that_target_it() { + let manifest = manifest(); + assert_eq!(manifest.parts_for(4), 2); + assert_eq!(manifest.parts_for(1), 1); + assert_eq!(manifest.parts_for(7), 0); + } + + #[test] + fn a_manifest_and_a_chunk_part_round_trip() { + let manifest = manifest(); + let bytes = zerompk::to_msgpack_vec(&manifest).expect("encode"); + let decoded: MultiPartPlans = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(decoded, manifest); + + let part = StreamedPart { + index: 2, + targets: vec![4], + part: PlanPart { + first_task: 5, + plans: vec![7; 16], + chunk: Some(TaskChunk { + offset: 16, + total_len: 40, + }), + }, + }; + let bytes = zerompk::to_msgpack_vec(&part).expect("encode"); + let decoded: StreamedPart = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(decoded, part); + } +} diff --git a/nodedb-cluster/src/calvin/types/read_write_set.rs b/nodedb-cluster/src/calvin/types/read_write_set.rs new file mode 100644 index 000000000..ea51c97a1 --- /dev/null +++ b/nodedb-cluster/src/calvin/types/read_write_set.rs @@ -0,0 +1,99 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! [`ReadWriteSet`]: the read set or write set of a Calvin transaction. + +use nodedb_types::id::{CollectionKey, DatabaseId, VShardId}; +use serde::{Deserialize, Serialize}; + +use crate::error::CalvinError; + +use super::primitives::EngineKeySet; + +/// A set of keys spanning one or more engines, forming either the read set +/// or the write set of a Calvin transaction. +/// +/// Cross-engine atomic transactions — e.g. a Document+Vector insert that must +/// land atomically — require all affected engines to appear in a single +/// `ReadWriteSet`. Decomposing by engine would break atomicity. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct ReadWriteSet(pub Vec); + +impl ReadWriteSet { + pub fn new(sets: Vec) -> Self { + Self(sets) + } + + pub fn is_empty(&self) -> bool { + self.0.iter().all(|s| s.is_empty()) + } + + /// Derive the set of vShards participating in this read/write set. + /// + /// For Document/Vector/KV entries the vshard is derived from the + /// collection name (collection-level routing, consistent with the + /// per-vshard Raft groups that own each collection). KV collections + /// are also assigned a single vshard at creation time. + /// + /// For Edge entries the participating vShards are the edge's + /// `home_vshards` (the `from_key(src)` / `from_key(dst)` key-hashed + /// homes), NOT the collection name: a graph edge is dual-homed across + /// its two endpoint vShards so it can be written atomically to both. + /// + /// This derivation is re-run on decode rather than serialized, so the + /// serialized bytes remain deterministic regardless of how `VShardId` + /// is computed. + pub fn participating_vshards(&self) -> Result, CalvinError> { + self.participating_vshards_in_database(DatabaseId::DEFAULT) + } + + /// Derive participants using database-scoped collection homes. + /// + /// Key-set collection names are database-qualified, because the Data + /// Plane reads storage by them. Each one is de-qualified into a + /// [`CollectionKey`] before hashing, so the participant set matches the + /// vShard every other path homes the collection to. + pub fn participating_vshards_in_database( + &self, + database_id: DatabaseId, + ) -> Result, CalvinError> { + let mut seen = std::collections::HashSet::new(); + let mut result = Vec::new(); + for engine_set in &self.0 { + match engine_set { + EngineKeySet::Edge { + home_vshards: homes, + .. + } + | EngineKeySet::Array { vshards: homes, .. } => { + for &home in homes.as_slice() { + let vshard = VShardId::new(home); + if seen.insert(vshard.as_u32()) { + result.push(vshard); + } + } + } + EngineKeySet::Document { .. } + | EngineKeySet::Vector { .. } + | EngineKeySet::Kv { .. } => { + let vshard = + CollectionKey::from_qualified_str(database_id, engine_set.collection())? + .vshard(); + if seen.insert(vshard.as_u32()) { + result.push(vshard); + } + } + } + } + result.sort_by_key(|v| v.as_u32()); + Ok(result) + } +} diff --git a/nodedb-cluster/src/calvin/types/scheduler_input.rs b/nodedb-cluster/src/calvin/types/scheduler_input.rs index 4e0e7be50..23724d495 100644 --- a/nodedb-cluster/src/calvin/types/scheduler_input.rs +++ b/nodedb-cluster/src/calvin/types/scheduler_input.rs @@ -11,7 +11,10 @@ //! //! [`SequencerEntry`]: super::super::sequencer::entry::SequencerEntry +use std::sync::Arc; + use super::lock_wire::{LockKeyWire, ReleaseReason, TxnIdWire}; +use super::multi_part::TaskChunk; use super::sequencer::SequencedTxn; /// One item in a per-vShard scheduler input stream. @@ -20,8 +23,9 @@ use super::sequencer::SequencedTxn; /// form is the replicated `SequencerEntry`, decoded and fanned out into these. #[derive(Debug)] pub enum SchedulerInput { - /// A sequenced transaction to process (lock-acquire + dispatch). - Txn(SequencedTxn), + /// A sequenced transaction to process (lock-acquire + dispatch). Boxed + /// so the other, small variants do not carry its size. + Txn(Box), /// Install a SHARED reservation on `key` for interactive txn `owner`. Reserve { owner: TxnIdWire, key: LockKeyWire }, /// Release ALL of `owner`'s shared reservations on this vShard. @@ -33,4 +37,17 @@ pub enum SchedulerInput { /// transaction delivered before it must finish before the scheduler /// reports it; every transaction delivered after it commits above `hlc`. CutMarker { hlc: u64 }, + /// Part `index` of the plans of the multi-part transaction `txn`, whose + /// first task is the transaction's task `first_task`. The bytes are + /// shared by every scheduler of this node the part targets. `chunk` is + /// set on a part that holds one byte range of one task. + TxnPart { + txn: TxnIdWire, + index: u32, + first_task: u32, + plans: Arc>, + chunk: Option, + }, + /// The multi-part transaction `txn` lost its parts and aborts. + PartsAbandoned { txn: TxnIdWire }, } diff --git a/nodedb-cluster/src/calvin/types/sequencer.rs b/nodedb-cluster/src/calvin/types/sequencer.rs index 238a3106a..e8ef728af 100644 --- a/nodedb-cluster/src/calvin/types/sequencer.rs +++ b/nodedb-cluster/src/calvin/types/sequencer.rs @@ -110,7 +110,7 @@ mod tests { use nodedb_types::id::{CollectionKey, DatabaseId}; use super::super::primitives::{EngineKeySet, SortedVec, VersionedReadSet}; - use super::super::transaction::ReadWriteSet; + use super::super::read_write_set::ReadWriteSet; use super::*; fn doc_set(collection: &str, surrogates: Vec) -> EngineKeySet { diff --git a/nodedb-cluster/src/calvin/types/transaction.rs b/nodedb-cluster/src/calvin/types/transaction.rs index bd681251e..b53080eb5 100644 --- a/nodedb-cluster/src/calvin/types/transaction.rs +++ b/nodedb-cluster/src/calvin/types/transaction.rs @@ -2,104 +2,18 @@ //! Calvin transaction class types. //! -//! Provides [`ReadWriteSet`] and [`TxClass`] — the core transaction -//! representation submitted to the sequencer. +//! Provides [`TxClass`] — the core transaction representation submitted to +//! the sequencer — over the [`ReadWriteSet`] key sets. use nodedb_types::TenantId; -use nodedb_types::id::{CollectionKey, DatabaseId, VShardId}; +use nodedb_types::id::{DatabaseId, VShardId}; use serde::{Deserialize, Serialize}; use crate::error::CalvinError; use super::lock_wire::TxnIdWire; -use super::primitives::{DependentReadSpec, EngineKeySet, VersionedReadSet}; - -// ── ReadWriteSet ────────────────────────────────────────────────────────────── - -/// A set of keys spanning one or more engines, forming either the read set -/// or the write set of a Calvin transaction. -/// -/// Cross-engine atomic transactions — e.g. a Document+Vector insert that must -/// land atomically — require all affected engines to appear in a single -/// `ReadWriteSet`. Decomposing by engine would break atomicity. -#[derive( - Debug, - Clone, - PartialEq, - Eq, - Serialize, - Deserialize, - zerompk::ToMessagePack, - zerompk::FromMessagePack, -)] -pub struct ReadWriteSet(pub Vec); - -impl ReadWriteSet { - pub fn new(sets: Vec) -> Self { - Self(sets) - } - - pub fn is_empty(&self) -> bool { - self.0.iter().all(|s| s.is_empty()) - } - - /// Derive the set of vShards participating in this read/write set. - /// - /// For Document/Vector/KV entries the vshard is derived from the - /// collection name (collection-level routing, consistent with the - /// per-vshard Raft groups that own each collection). KV collections - /// are also assigned a single vshard at creation time. - /// - /// For Edge entries the participating vShards are the edge's - /// `home_vshards` (the `from_key(src)` / `from_key(dst)` key-hashed - /// homes), NOT the collection name: a graph edge is dual-homed across - /// its two endpoint vShards so it can be written atomically to both. - /// - /// This derivation is re-run on decode rather than serialized, so the - /// serialized bytes remain deterministic regardless of how `VShardId` - /// is computed. - pub fn participating_vshards(&self) -> Result, CalvinError> { - self.participating_vshards_in_database(DatabaseId::DEFAULT) - } - - /// Derive participants using database-scoped collection homes. - /// - /// Key-set collection names are database-qualified, because the Data - /// Plane reads storage by them. Each one is de-qualified into a - /// [`CollectionKey`] before hashing, so the participant set matches the - /// vShard every other path homes the collection to. - pub fn participating_vshards_in_database( - &self, - database_id: DatabaseId, - ) -> Result, CalvinError> { - let mut seen = std::collections::HashSet::new(); - let mut result = Vec::new(); - for engine_set in &self.0 { - match engine_set { - EngineKeySet::Edge { home_vshards, .. } => { - for &home in home_vshards.as_slice() { - let vshard = VShardId::new(home); - if seen.insert(vshard.as_u32()) { - result.push(vshard); - } - } - } - EngineKeySet::Document { .. } - | EngineKeySet::Vector { .. } - | EngineKeySet::Kv { .. } => { - let vshard = - CollectionKey::from_qualified_str(database_id, engine_set.collection())? - .vshard(); - if seen.insert(vshard.as_u32()) { - result.push(vshard); - } - } - } - } - result.sort_by_key(|v| v.as_u32()); - Ok(result) - } -} +use super::primitives::{DependentReadSpec, VersionedReadSet}; +use super::read_write_set::ReadWriteSet; // ── TxClass ─────────────────────────────────────────────────────────────────── @@ -174,6 +88,64 @@ pub struct TxClass { #[serde(default)] #[msgpack(default)] pub lock_owner: Option, + /// The metadata-group index the coordinator had applied when it planned + /// the transaction. Every replica's scheduler runs the transaction only + /// once its own metadata apply reached this index, so it never writes a + /// collection this node has not registered yet. `0` holds nothing. + #[serde(default)] + #[msgpack(default)] + pub metadata_floor: u64, + /// WAL event-source code every participant stamps on the transaction's + /// writes, so a server-run body's writes fire no triggers. `0` is the + /// client default. + #[serde(default)] + #[msgpack(default)] + pub event_source: u8, + /// Indexes into `plans` of the plans a trigger body buffered. Every + /// participant commits their rows under the trigger source, so they fire + /// no trigger. Empty for a transaction no body joined. + #[serde(default)] + #[msgpack(default)] + pub body_plans: Vec, + /// Opaque msgpack-encoded `PUBLISH TO` messages the transaction's trigger + /// bodies sent, decoded by the `nodedb` crate. The participant named by + /// [`TxClass::publish_vshard`] commits them in its redo record. Empty for + /// a transaction that published nothing. + #[serde(default)] + #[msgpack(default)] + pub publishes: Vec, + /// Opaque msgpack-encoded dedup key of the cross-shard trigger request + /// the transaction applies, decoded by the `nodedb` crate. The participant + /// named by [`TxClass::applied_key_home`] writes it into its redo record, + /// so the key is durable exactly when the request's writes are. Empty + /// for every other transaction. + #[serde(default)] + #[msgpack(default)] + pub applied_key: Vec, + /// The vShard the cross-shard request addresses. + #[serde(default)] + #[msgpack(default)] + pub applied_key_vshard: u32, + /// Every user collection the plans name, with the incarnation the + /// coordinator's catalog held when it planned the transaction. A + /// participant whose catalog no longer holds one votes abort, so the + /// transaction never lands in a same-name recreate. + #[serde(default)] + #[msgpack(default)] + pub incarnations: Vec, + /// The plan manifest when the plans travel as parts (see + /// [`super::multi_part`]). `plans` is empty then. `None` for a + /// transaction whose plans ride its own sequencer entry. + #[serde(default)] + #[msgpack(default)] + pub multi_part: Option, + /// The RESTORE that re-issues the transaction's writes. Every participant + /// raises the tenant's restore mark under it instead of the user write + /// mark, so a retry of the same RESTORE finds only its own marks. `0` for + /// every other transaction. + #[serde(default)] + #[msgpack(default)] + pub restore_id: u64, /// Cached participating-vshard set. Re-derived on decode; not serialized. #[serde(skip)] #[msgpack(ignore)] @@ -227,7 +199,7 @@ impl TxClass { dependent_reads: Option, versioned_reads: VersionedReadSet, ) -> Result { - Self::new_checked( + Self::unchecked( read_set, write_set, plans, @@ -235,8 +207,8 @@ impl TxClass { database_id, dependent_reads, versioned_reads, - false, ) + .checked(false) } /// Construct a validated transaction class that is permitted to resolve to a @@ -284,7 +256,7 @@ impl TxClass { dependent_reads: Option, versioned_reads: VersionedReadSet, ) -> Result { - Self::new_checked( + Self::unchecked( read_set, write_set, plans, @@ -292,15 +264,12 @@ impl TxClass { database_id, dependent_reads, versioned_reads, - true, ) + .checked(true) } - /// Shared construction body. `allow_single_vshard` relaxes the participant - /// floor from 2 (multi-vshard) to 1 (single-vshard opt-in); an empty write - /// set and a zero-participant write set are rejected on both paths. - #[allow(clippy::too_many_arguments)] // shared validation for both constructor modes - fn new_checked( + /// The class the constructors validate, with no participants derived yet. + fn unchecked( read_set: ReadWriteSet, write_set: ReadWriteSet, plans: Vec, @@ -308,65 +277,74 @@ impl TxClass { database_id: DatabaseId, dependent_reads: Option, versioned_reads: VersionedReadSet, - allow_single_vshard: bool, - ) -> Result { - if write_set.is_empty() { + ) -> Self { + Self { + read_set, + write_set, + plans, + tenant_id, + database_id, + dependent_reads, + versioned_reads, + lock_owner: None, + metadata_floor: 0, + event_source: 0, + body_plans: Vec::new(), + publishes: Vec::new(), + applied_key: Vec::new(), + applied_key_vshard: 0, + incarnations: Vec::new(), + multi_part: None, + restore_id: 0, + participating_vshards: Vec::new(), + } + } + + /// Shared construction body. `allow_single_vshard` relaxes the participant + /// floor from 2 (multi-vshard) to 1 (single-vshard opt-in); an empty write + /// set and a zero-participant write set are rejected on both paths. + fn checked(mut self, allow_single_vshard: bool) -> Result { + if self.write_set.is_empty() { return Err(CalvinError::EmptyWriteSet); } - let mut participating_vshards = write_set.participating_vshards_in_database(database_id)?; + let write_vshards = self + .write_set + .participating_vshards_in_database(self.database_id)?; let min_participants = if allow_single_vshard { 1 } else { 2 }; // The participant FLOOR is computed from the WRITE set ONLY, and BEFORE - // the read-set union below: a txn that writes a single shard but reads N + // the read-set union: a txn that writes a single shard but reads N // additional shards is a legitimate single-write-shard txn and must not // trip the `>= 2` floor. - if participating_vshards.len() < min_participants { - let vshard = participating_vshards - .first() - .map(|v| v.as_u32()) - .unwrap_or(0); + if write_vshards.len() < min_participants { + let vshard = write_vshards.first().map(|v| v.as_u32()).unwrap_or(0); return Err(CalvinError::SingleVshardTxn { vshard }); } - // Union the read set's participating vShards: a shard that is only READ - // (never written) still participates so it can validate the read at the - // commit serialization point. This union MUST be applied identically in - // `new_checked` and `restore_derived` — `participating_vshards` is - // `#[serde(skip)]` and re-derived on decode, so an encoded and a decoded - // `TxClass` would disagree on their participant set if the two diverged. - for v in read_set.participating_vshards_in_database(database_id)? { - if !participating_vshards - .iter() - .any(|e| e.as_u32() == v.as_u32()) - { - participating_vshards.push(v); - } - } - // Extend participating_vshards with passive vshards from dependent_reads. - if let Some(ref spec) = dependent_reads { - for &passive_vshard in spec.passive_reads.keys() { - let v = VShardId::new(passive_vshard); - if !participating_vshards - .iter() - .any(|e| e.as_u32() == passive_vshard) - { - participating_vshards.push(v); - } - } + self.participating_vshards = self.union_participants(write_vshards)?; + Ok(self) + } + + /// `write_vshards` joined with the read set's vShards and the passive + /// dependent-read vShards, sorted by id and deduplicated. + /// + /// A shard that is only READ still participates, so it can validate the + /// read at the commit serialization point. `participating_vshards` is + /// `#[serde(skip)]` and re-derived on decode, so construction and + /// [`Self::restore_derived`] both derive it here. No caller reads the + /// unsorted order. The sorted order is stable across encode and decode. + fn union_participants( + &self, + mut participants: Vec, + ) -> Result, CalvinError> { + participants.extend( + self.read_set + .participating_vshards_in_database(self.database_id)?, + ); + if let Some(spec) = &self.dependent_reads { + participants.extend(spec.passive_reads.keys().map(|&v| VShardId::new(v))); } - // Stable ordering across encode/decode (participants ride the Raft log): - // one final sort after ALL unions (write floor + read + passive), kept in - // lockstep with `restore_derived`. - participating_vshards.sort_by_key(|v| v.as_u32()); - Ok(Self { - read_set, - write_set, - plans, - tenant_id, - database_id, - dependent_reads, - versioned_reads, - lock_owner: None, - participating_vshards, - }) + participants.sort_unstable_by_key(|v| v.as_u32()); + participants.dedup_by_key(|v| v.as_u32()); + Ok(participants) } /// Ergonomic constructor for dependent-read Calvin transactions. @@ -407,30 +385,10 @@ impl TxClass { /// class, so a failure here means the decoded bytes are not a class any /// constructor built. pub fn restore_derived(&mut self) -> Result<(), CalvinError> { - let mut vshards = self + let write_vshards = self .write_set .participating_vshards_in_database(self.database_id)?; - // Union the read set's participating vShards — MUST match `new_checked`'s - // union exactly so a decoded `TxClass` derives the identical participant - // set the encoder computed (participants are not serialized). - for v in self - .read_set - .participating_vshards_in_database(self.database_id)? - { - if !vshards.iter().any(|e| e.as_u32() == v.as_u32()) { - vshards.push(v); - } - } - if let Some(ref spec) = self.dependent_reads { - for &passive_vshard in spec.passive_reads.keys() { - if !vshards.iter().any(|e| e.as_u32() == passive_vshard) { - vshards.push(VShardId::new(passive_vshard)); - } - } - } - // Final stable sort after all unions — lockstep with `new_checked`. - vshards.sort_by_key(|v| v.as_u32()); - self.participating_vshards = vshards; + self.participating_vshards = self.union_participants(write_vshards)?; Ok(()) } @@ -438,14 +396,104 @@ impl TxClass { pub fn set_lock_owner(&mut self, owner: Option) { self.lock_owner = owner; } + + /// Set the WAL event-source code the transaction's writes carry. + pub fn set_event_source(&mut self, code: u8) { + self.event_source = code; + } + + /// Mark the transaction as the re-issue of RESTORE `restore_id`. + pub fn set_restore_id(&mut self, restore_id: u64) { + self.restore_id = restore_id; + } + + /// Set the indexes into `plans` of the plans a trigger body buffered. + pub fn set_body_plans(&mut self, body_plans: Vec) { + self.body_plans = body_plans; + } + + /// Set the encoded messages the transaction's trigger bodies published. + pub fn set_publishes(&mut self, publishes: Vec) { + self.publishes = publishes; + } + + /// The participant whose redo record carries the transaction's messages: + /// the lowest vShard the transaction writes. Every write participant + /// resolves a redo record, so this one always commits them. `None` when + /// the write set names no vShard. + /// + /// Fails when a write-set collection name lacks the qualifier of the + /// class's database. Construction rejects such a class. + pub fn publish_vshard(&self) -> Result, CalvinError> { + Ok(self.write_vshards()?.into_iter().min()) + } + + /// Set the encoded dedup key of the cross-shard request the transaction + /// applies, and the vShard the request addresses. + pub fn set_applied_key(&mut self, applied_key: Vec, vshard: u32) { + self.applied_key = applied_key; + self.applied_key_vshard = vshard; + } + + /// The participant whose redo record carries the applied key: the vShard + /// the request addresses when the transaction writes it, else the lowest + /// vShard it writes. `None` for a transaction with no applied key. + /// + /// Fails as [`Self::publish_vshard`] does. + pub fn applied_key_home(&self) -> Result, CalvinError> { + if self.applied_key.is_empty() { + return Ok(None); + } + let writes = self.write_vshards()?; + if writes.contains(&self.applied_key_vshard) { + return Ok(Some(self.applied_key_vshard)); + } + Ok(writes.into_iter().min()) + } + + /// Every vShard of the class's database the write set names. + fn write_vshards(&self) -> Result, CalvinError> { + Ok(self + .write_set + .participating_vshards_in_database(self.database_id)? + .into_iter() + .map(|vshard| vshard.as_u32()) + .collect()) + } + + /// Set the collections the plans name, with their planned incarnations. + pub fn set_incarnations(&mut self, incarnations: Vec) { + self.incarnations = incarnations; + } +} + +/// A user collection a Calvin transaction names, and the incarnation its +/// coordinator planned against. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct CalvinIncarnation { + /// The collection as the plans name it: database-qualified outside the + /// default database. + pub collection: String, + pub incarnation: nodedb_types::Hlc, } #[cfg(test)] mod tests { use super::super::primitives::{ - EngineTag, PassiveReadKey, ReadKeyIdent, SortedVec, VersionedReadEntry, VersionedReadSet, + EngineKeySet, EngineTag, PassiveReadKey, ReadKeyIdent, SortedVec, VersionedReadEntry, + VersionedReadSet, }; use super::*; + use nodedb_types::id::CollectionKey; use nodedb_types::{KeyRepr, Lsn}; fn doc_set(collection: &str, surrogates: Vec) -> EngineKeySet { @@ -717,12 +765,16 @@ mod tests { collection: "kv_col".to_owned(), key: ReadKeyIdent::Point(KeyRepr::KvKey(Box::from(&b"k1"[..]))), read_lsn: Lsn::new(7), + home_vshard: None, + served_by: 0, }, VersionedReadEntry { engine: EngineTag::Document, collection: "doc_col".to_owned(), key: ReadKeyIdent::Predicate, read_lsn: Lsn::new(11), + home_vshard: None, + served_by: 0, }, ]) } @@ -825,6 +877,131 @@ mod tests { assert_eq!(decoded.database_id, DatabaseId::DEFAULT); assert_eq!(decoded.plans, vec![0x01, 0x02]); assert_eq!(decoded.participating_vshards().len(), 2); + assert_eq!(decoded.metadata_floor, 0, "a legacy class holds nothing"); + assert_eq!(decoded.event_source, 0, "a legacy class is a client's"); + assert!(decoded.body_plans.is_empty()); + assert!(decoded.publishes.is_empty()); + assert_eq!(decoded.restore_id, 0, "a legacy class re-issues no RESTORE"); + } + + /// The event-source code survives the codec, so every participant stamps + /// the same source on the transaction's writes. + #[test] + fn event_source_roundtrips_through_the_codec() { + let mut tx = TxClass::new( + ReadWriteSet::new(vec![]), + two_home_write_set(), + vec![0x01], + TenantId::new(3), + None, + VersionedReadSet::default(), + ) + .expect("valid TxClass"); + assert_eq!(tx.event_source, 0); + tx.set_event_source(2); + tx.set_body_plans(vec![1]); + tx.set_restore_id(77); + let bytes = zerompk::to_msgpack_vec(&tx).expect("encode"); + let decoded: TxClass = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(decoded.event_source, 2); + assert_eq!(decoded.body_plans, vec![1]); + assert_eq!( + decoded.restore_id, 77, + "every participant raises the restore's mark" + ); + } + + /// The published messages survive the codec, and one participant, the + /// lowest write vShard, carries them. + #[test] + fn publishes_roundtrip_and_name_the_lowest_write_vshard() { + let mut tx = TxClass::new( + ReadWriteSet::new(vec![]), + two_home_write_set(), + vec![0x01], + TenantId::new(3), + None, + VersionedReadSet::default(), + ) + .expect("valid TxClass"); + assert!(tx.publishes.is_empty()); + tx.set_publishes(vec![0x91, 0xa1, 0x61]); + let bytes = zerompk::to_msgpack_vec(&tx).expect("encode"); + let mut decoded: TxClass = zerompk::from_msgpack(&bytes).expect("decode"); + decoded.restore_derived().expect("restore derived"); + assert_eq!(decoded.publishes, vec![0x91, 0xa1, 0x61]); + let lowest = decoded + .participating_vshards() + .iter() + .map(|vshard| vshard.as_u32()) + .min(); + assert!(lowest.is_some()); + assert_eq!(decoded.publish_vshard(), Ok(lowest)); + } + + /// A class whose write set lacks its database's qualifier cannot name a + /// write vShard. Both homes report the error instead of `None`. + #[test] + fn redo_homes_report_an_underivable_write_set() { + let mut tx = make_tx_class(multi_vshard_write_set()); + tx.set_applied_key(vec![0x80], 0); + // The bare document collection names carry no `9/` qualifier. + tx.database_id = DatabaseId::new(9); + assert!(tx.publish_vshard().is_err()); + assert!(tx.applied_key_home().is_err()); + } + + /// The applied key rides the vShard its request addresses when the + /// transaction writes it, else the lowest write vShard. + #[test] + fn applied_key_home_prefers_the_addressed_vshard() { + let mut tx = TxClass::new( + ReadWriteSet::new(vec![]), + two_home_write_set(), + vec![0x01], + TenantId::new(3), + None, + VersionedReadSet::default(), + ) + .expect("valid TxClass"); + assert_eq!(tx.applied_key_home(), Ok(None), "no key, no home"); + let mut homes: Vec = tx + .participating_vshards() + .iter() + .map(|vshard| vshard.as_u32()) + .collect(); + homes.sort_unstable(); + let (lowest, highest) = (homes[0], homes[homes.len() - 1]); + tx.set_applied_key(vec![0x80], highest); + let bytes = zerompk::to_msgpack_vec(&tx).expect("encode"); + let mut decoded: TxClass = zerompk::from_msgpack(&bytes).expect("decode"); + decoded.restore_derived().expect("restore derived"); + assert_eq!(decoded.applied_key, vec![0x80]); + assert_eq!(decoded.applied_key_home(), Ok(Some(highest))); + let unwritten = (0..u32::MAX) + .find(|vshard| !homes.contains(vshard)) + .unwrap_or(u32::MAX); + decoded.set_applied_key(vec![0x80], unwritten); + assert_eq!(decoded.applied_key_home(), Ok(Some(lowest))); + } + + /// The metadata floor survives the codec, so every replica waits for the + /// catalog the coordinator planned against. + #[test] + fn metadata_floor_roundtrips_through_the_codec() { + let mut tx = TxClass::new( + ReadWriteSet::new(vec![]), + two_home_write_set(), + vec![0x01], + TenantId::new(3), + None, + VersionedReadSet::default(), + ) + .expect("valid TxClass"); + tx.metadata_floor = 42; + let bytes = zerompk::to_msgpack_vec(&tx).expect("encode"); + let decoded: TxClass = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(decoded.metadata_floor, 42); } /// Find two distinct string keys whose `from_key` vShards differ. @@ -1010,7 +1187,7 @@ mod tests { assert_eq!( tx.participating_vshards(), decoded.participating_vshards(), - "restore_derived must reproduce new_checked's read∪write participants" + "restore_derived must reproduce the constructor's read∪write participants" ); } diff --git a/nodedb-cluster/src/catalog/boot_epoch.rs b/nodedb-cluster/src/catalog/boot_epoch.rs new file mode 100644 index 000000000..f31b28e35 --- /dev/null +++ b/nodedb-cluster/src/catalog/boot_epoch.rs @@ -0,0 +1,69 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The node's boot epoch: a counter raised and made durable at every boot. +//! +//! A peer keeps its replay window for this node across this node's restart. +//! So every boot sends its frames in a sequence range above every range an +//! earlier boot used (see [`crate::rpc_codec::PeerSeqSender::enter_boot_epoch`]). +//! The counter lives in the catalog, not the clock: a clock stepped back, a +//! bad RTC or a restored VM snapshot never lowers it. + +use redb::ReadableTable; + +use crate::error::Result; + +use super::core::ClusterCatalog; +use super::schema::{KEY_BOOT_EPOCH, METADATA_TABLE, catalog_err}; + +impl ClusterCatalog { + /// Raise the boot epoch by one, commit it durably, and return it. The + /// first boot returns 1. Blocks on disk: call it off the async threads, + /// once per boot, before any transport sends. + pub fn advance_boot_epoch(&self) -> Result { + let txn = self.db.begin_write().map_err(catalog_err)?; + let epoch = { + let mut table = txn.open_table(METADATA_TABLE).map_err(catalog_err)?; + let stored = match table.get(KEY_BOOT_EPOCH).map_err(catalog_err)? { + Some(guard) => { + let bytes: [u8; 8] = guard.value().try_into().map_err(|_| { + catalog_err(format!( + "metadata key {KEY_BOOT_EPOCH} has unexpected length {} (expected 8)", + guard.value().len() + )) + })?; + u64::from_le_bytes(bytes) + } + None => 0, + }; + let epoch = stored.checked_add(1).ok_or_else(|| { + catalog_err(format!( + "metadata key {KEY_BOOT_EPOCH} is exhausted at {stored}" + )) + })?; + table + .insert(KEY_BOOT_EPOCH, epoch.to_le_bytes().as_slice()) + .map_err(catalog_err)?; + epoch + }; + txn.commit().map_err(catalog_err)?; + Ok(epoch) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn every_boot_gets_a_higher_epoch_that_survives_a_reopen() { + let dir = tempfile::tempdir().expect("tempdir"); + let path = dir.path().join("cluster.redb"); + { + let catalog = ClusterCatalog::open(&path).expect("open"); + assert_eq!(catalog.advance_boot_epoch().expect("advance"), 1); + assert_eq!(catalog.advance_boot_epoch().expect("advance"), 2); + } + let catalog = ClusterCatalog::open(&path).expect("reopen"); + assert_eq!(catalog.advance_boot_epoch().expect("advance"), 3); + } +} diff --git a/nodedb-cluster/src/catalog/core.rs b/nodedb-cluster/src/catalog/core.rs index 4468b0d4c..7c6275254 100644 --- a/nodedb-cluster/src/catalog/core.rs +++ b/nodedb-cluster/src/catalog/core.rs @@ -24,6 +24,9 @@ use super::schema::{ /// Persistent cluster catalog backed by redb. pub struct ClusterCatalog { pub(super) db: redb::Database, + /// Test hook: the next `save_cluster_epoch` fails once. + #[cfg(test)] + fail_next_epoch_write: std::sync::atomic::AtomicBool, } impl ClusterCatalog { @@ -66,7 +69,23 @@ impl ClusterCatalog { "cluster catalog opened" ); - Ok(Self { db }) + Ok(Self { + db, + #[cfg(test)] + fail_next_epoch_write: std::sync::atomic::AtomicBool::new(false), + }) + } + + /// Make the next `save_cluster_epoch` return an error. + #[cfg(test)] + pub(crate) fn fail_next_epoch_write_for_test(&self) { + self.fail_next_epoch_write + .store(true, std::sync::atomic::Ordering::SeqCst); + } + + /// The redb database that holds this catalog. + pub fn database(&self) -> &redb::Database { + &self.db } // ── Metadata ──────────────────────────────────────────────────── @@ -114,6 +133,15 @@ impl ClusterCatalog { /// Persist the cluster epoch (the leader-bumped monotonic fence /// token stamped on every Raft RPC). Overwrites any prior value. pub fn save_cluster_epoch(&self, epoch: u64) -> Result<()> { + #[cfg(test)] + if self + .fail_next_epoch_write + .swap(false, std::sync::atomic::Ordering::SeqCst) + { + return Err(crate::error::ClusterError::Transport { + detail: "injected cluster epoch write failure".into(), + }); + } let bytes = epoch.to_le_bytes(); let txn = self.db.begin_write().map_err(catalog_err)?; { diff --git a/nodedb-cluster/src/catalog/mod.rs b/nodedb-cluster/src/catalog/mod.rs index d61fcb2b5..12398263a 100644 --- a/nodedb-cluster/src/catalog/mod.rs +++ b/nodedb-cluster/src/catalog/mod.rs @@ -21,6 +21,7 @@ //! and is the single place where future breaking schema //! changes land as explicit `v{N} → v{N+1}` arms. +pub mod boot_epoch; pub mod cluster_settings; pub mod core; pub mod ghosts; diff --git a/nodedb-cluster/src/catalog/schema.rs b/nodedb-cluster/src/catalog/schema.rs index 1ba196bd1..abaf6f809 100644 --- a/nodedb-cluster/src/catalog/schema.rs +++ b/nodedb-cluster/src/catalog/schema.rs @@ -44,6 +44,11 @@ pub(super) const KEY_CLUSTER_EPOCH: &str = "cluster_epoch"; /// and dominate any lingering `Dead(stored)` rumour. See /// `crate::swim::incarnation_store`. pub(super) const KEY_SWIM_INCARNATION: &str = "swim_incarnation"; +/// Boot epoch (u64 LE): how many times this node has booted. Raised and made +/// durable at every boot, before any transport sends. It opens each boot's +/// own range of outbound frame sequence numbers. See +/// `ClusterCatalog::advance_boot_epoch`. +pub(super) const KEY_BOOT_EPOCH: &str = "boot_epoch"; pub(super) const KEY_CA_CERT: &str = "ca_cert"; /// Metadata key holding the catalog format version (u32 LE). /// Stored under this key in the metadata table by `ClusterCatalog::open`. diff --git a/nodedb-cluster/src/circuit_breaker.rs b/nodedb-cluster/src/circuit_breaker.rs index a38d24208..c74ccfe39 100644 --- a/nodedb-cluster/src/circuit_breaker.rs +++ b/nodedb-cluster/src/circuit_breaker.rs @@ -51,10 +51,37 @@ pub enum CircuitState { HalfOpen, } +/// How [`CircuitBreaker::check`] let a request through. +/// +/// The caller hands it back with the request's outcome. Only the probe's +/// own failure reopens a half-open circuit. +#[must_use] +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Admission { + /// A request on a closed circuit. + Normal, + /// The one probe a half-open circuit lets through. + Probe { + /// Tells this probe apart from an earlier probe that expired. + id: u64, + }, +} + +/// The probe a half-open circuit has out. +#[derive(Debug, Clone, Copy)] +struct OutstandingProbe { + id: u64, + since: Instant, +} + struct PeerBreaker { state: CircuitState, consecutive_failures: u32, last_state_change: Instant, + /// The half-open probe. `None` when none is out. + probe: Option, + /// Id for the next probe. Ids never repeat for one peer. + next_probe_id: u64, } impl PeerBreaker { @@ -63,6 +90,24 @@ impl PeerBreaker { state: CircuitState::Closed, consecutive_failures: 0, last_state_change: Instant::now(), + probe: None, + next_probe_id: 0, + } + } + + /// Send out a new probe. It replaces any probe still out. + fn issue_probe(&mut self, now: Instant) -> Admission { + let id = self.next_probe_id; + self.next_probe_id = self.next_probe_id.wrapping_add(1); + self.probe = Some(OutstandingProbe { id, since: now }); + Admission::Probe { id } + } + + /// Whether `admission` is the probe this half-open circuit has out. + fn is_current_probe(&self, admission: Admission) -> bool { + match (admission, self.probe) { + (Admission::Probe { id }, Some(probe)) => probe.id == id, + _ => false, } } } @@ -77,64 +122,109 @@ impl CircuitBreaker { /// Check if an RPC to this peer is allowed. /// - /// Returns `Ok(())` if the circuit is closed or half-open (probe allowed). - /// Returns `Err(CircuitOpen)` if the circuit is open and cooldown hasn't expired. - pub fn check(&self, peer: u64) -> Result<()> { + /// Returns the [`Admission`] the caller hands back with the outcome. + /// Returns `Err(CircuitOpen)` while the circuit is open and its cooldown + /// runs, and while a half-open circuit has its probe out. + pub fn check(&self, peer: u64) -> Result { let mut peers = self.peers.write().unwrap_or_else(|p| p.into_inner()); let breaker = peers.entry(peer).or_insert_with(PeerBreaker::new); + let now = Instant::now(); + let refused = ClusterError::CircuitOpen { + node_id: peer, + failures: breaker.consecutive_failures, + }; + // An open circuit refuses until its cooldown passes, then turns + // half-open and lets exactly one probe through. Other requests are + // refused while it is out. A probe that never reports, because its + // caller was dropped, stops counting after one cooldown. The next + // request then goes out as the probe, so the circuit can always + // recover. match breaker.state { - CircuitState::Closed => Ok(()), - CircuitState::HalfOpen => Ok(()), // Allow probe. + CircuitState::Closed => Ok(Admission::Normal), + CircuitState::HalfOpen => { + let probe_out = breaker.probe.is_some_and(|probe| { + now.saturating_duration_since(probe.since) < self.config.cooldown + }); + if probe_out { + return Err(refused); + } + Ok(breaker.issue_probe(now)) + } CircuitState::Open => { - // Check if cooldown has expired → transition to HalfOpen. - if breaker.last_state_change.elapsed() >= self.config.cooldown { - breaker.state = CircuitState::HalfOpen; - breaker.last_state_change = Instant::now(); - Ok(()) - } else { - Err(ClusterError::CircuitOpen { - node_id: peer, - failures: breaker.consecutive_failures, - }) + if now.saturating_duration_since(breaker.last_state_change) < self.config.cooldown { + return Err(refused); } + breaker.state = CircuitState::HalfOpen; + breaker.last_state_change = now; + Ok(breaker.issue_probe(now)) } } } - /// Record a successful RPC to a peer. Resets the circuit to Closed. - pub fn record_success(&self, peer: u64) { + /// Admit a recovery probe, which the circuit never refuses. + /// + /// A closed circuit admits it as a normal request. An open or half-open + /// circuit turns half-open and sends it out as its probe. Its outcome + /// then decides the circuit: a success closes it, a failure reopens it. + pub fn admit_probe(&self, peer: u64) -> Admission { + let mut peers = self.peers.write().unwrap_or_else(|p| p.into_inner()); + let breaker = peers.entry(peer).or_insert_with(PeerBreaker::new); + let now = Instant::now(); + match breaker.state { + CircuitState::Closed => Admission::Normal, + CircuitState::HalfOpen => breaker.issue_probe(now), + CircuitState::Open => { + breaker.state = CircuitState::HalfOpen; + breaker.last_state_change = now; + breaker.issue_probe(now) + } + } + } + + /// Record that the peer answered a request. Resets the circuit to Closed. + /// + /// Any answer proves the link is up, so a late answer closes the circuit + /// as a probe's answer does. + pub fn record_success(&self, peer: u64, _admission: Admission) { let mut peers = self.peers.write().unwrap_or_else(|p| p.into_inner()); let breaker = peers.entry(peer).or_insert_with(PeerBreaker::new); breaker.consecutive_failures = 0; + breaker.probe = None; if breaker.state != CircuitState::Closed { breaker.state = CircuitState::Closed; breaker.last_state_change = Instant::now(); } } - /// Record a failed RPC to a peer. May open the circuit. - pub fn record_failure(&self, peer: u64) { + /// Record a failed request to a peer. Enough failures open the circuit. + pub fn record_failure(&self, peer: u64, admission: Admission) { let mut peers = self.peers.write().unwrap_or_else(|p| p.into_inner()); let breaker = peers.entry(peer).or_insert_with(PeerBreaker::new); - breaker.consecutive_failures += 1; match breaker.state { CircuitState::Closed => { + breaker.consecutive_failures = breaker.consecutive_failures.saturating_add(1); if breaker.consecutive_failures >= self.config.failure_threshold { breaker.state = CircuitState::Open; breaker.last_state_change = Instant::now(); } } + // Only the probe's own failure reopens the circuit. A request + // admitted before the circuit opened can fail late. Its failure + // says nothing the opening failures did not already say. CircuitState::HalfOpen => { - // Probe failed → back to Open. - breaker.state = CircuitState::Open; - breaker.last_state_change = Instant::now(); - } - CircuitState::Open => { - // Already open — refresh the timestamp to extend cooldown. - breaker.last_state_change = Instant::now(); + if breaker.is_current_probe(admission) { + breaker.consecutive_failures = breaker.consecutive_failures.saturating_add(1); + breaker.state = CircuitState::Open; + breaker.last_state_change = Instant::now(); + breaker.probe = None; + } } + // A late failure neither grows the count nor restarts the + // cooldown. Old requests that fail one by one cannot keep the + // circuit open. + CircuitState::Open => {} } } @@ -293,6 +383,7 @@ impl RetryPolicy { | ClusterError::SpatialGather(_) | ClusterError::Bm25Gather(_) | ClusterError::TsGather(_) + | ClusterError::ShufflePush(_) | ClusterError::RemoteUntyped { .. } | ClusterError::ShardExecution { .. } => false, } @@ -303,11 +394,13 @@ impl RetryPolicy { mod tests { use super::*; + use Admission::Normal; + #[test] fn circuit_starts_closed() { let cb = CircuitBreaker::new(CircuitBreakerConfig::default()); assert_eq!(cb.state(42), CircuitState::Closed); - cb.check(42).unwrap(); // Should succeed. + assert_eq!(cb.check(42).expect("closed admits"), Normal); } #[test] @@ -317,12 +410,12 @@ mod tests { cooldown: Duration::from_secs(60), }); - cb.check(1).unwrap(); - cb.record_failure(1); - cb.record_failure(1); + let admission = cb.check(1).expect("closed admits"); + cb.record_failure(1, admission); + cb.record_failure(1, Normal); assert_eq!(cb.state(1), CircuitState::Closed); - cb.record_failure(1); // 3rd failure → opens. + cb.record_failure(1, Normal); // 3rd failure → opens. assert_eq!(cb.state(1), CircuitState::Open); assert_eq!(cb.failure_count(1), 3); @@ -338,17 +431,69 @@ mod tests { cooldown: Duration::from_millis(10), }); - cb.record_failure(1); // Opens immediately (threshold=1). + cb.record_failure(1, Normal); // Opens immediately (threshold=1). assert_eq!(cb.state(1), CircuitState::Open); // Wait for cooldown. std::thread::sleep(Duration::from_millis(15)); // Check should transition to HalfOpen and allow the probe. - cb.check(1).unwrap(); + let admission = cb.check(1).expect("the probe goes out"); + assert!(matches!(admission, Admission::Probe { .. })); + assert_eq!(cb.state(1), CircuitState::HalfOpen); + } + + #[test] + fn half_open_lets_one_probe_through_until_it_reports_or_expires() { + let cb = CircuitBreaker::new(CircuitBreakerConfig { + failure_threshold: 1, + cooldown: Duration::from_millis(500), + }); + cb.record_failure(1, Normal); + std::thread::sleep(Duration::from_millis(550)); + let lost = cb.check(1).expect("the probe goes out"); + assert!(cb.check(1).is_err(), "a second request waits for the probe"); + // The probe never reports: after one cooldown the next request probes. + std::thread::sleep(Duration::from_millis(550)); + let replacement = cb.check(1).expect("a lost probe is replaced"); + assert_ne!(lost, replacement, "the replacement is a new probe"); + cb.record_success(1, replacement); + assert_eq!(cb.state(1), CircuitState::Closed); + } + + #[test] + fn late_failures_do_not_extend_an_open_circuit() { + let cb = CircuitBreaker::new(CircuitBreakerConfig { + failure_threshold: 1, + cooldown: Duration::from_millis(400), + }); + cb.record_failure(1, Normal); + std::thread::sleep(Duration::from_millis(300)); + // A request sent before the circuit opened fails now. + cb.record_failure(1, Normal); + std::thread::sleep(Duration::from_millis(150)); + let _probe = cb + .check(1) + .expect("the cooldown ran from the opening failure"); assert_eq!(cb.state(1), CircuitState::HalfOpen); } + #[test] + fn failures_while_open_do_not_grow_the_count() { + let cb = CircuitBreaker::new(CircuitBreakerConfig { + failure_threshold: 2, + cooldown: Duration::from_secs(60), + }); + cb.record_failure(1, Normal); + cb.record_failure(1, Normal); + assert_eq!(cb.state(1), CircuitState::Open); + for _ in 0..10 { + cb.record_failure(1, Normal); + } + assert_eq!(cb.failure_count(1), 2, "late failures are not counted"); + assert_eq!(cb.state(1), CircuitState::Open); + } + #[test] fn half_open_success_closes_circuit() { let cb = CircuitBreaker::new(CircuitBreakerConfig { @@ -356,11 +501,11 @@ mod tests { cooldown: Duration::from_millis(5), }); - cb.record_failure(1); + cb.record_failure(1, Normal); std::thread::sleep(Duration::from_millis(10)); - cb.check(1).unwrap(); // → HalfOpen + let probe = cb.check(1).expect("the probe goes out"); // → HalfOpen - cb.record_success(1); + cb.record_success(1, probe); assert_eq!(cb.state(1), CircuitState::Closed); assert_eq!(cb.failure_count(1), 0); } @@ -372,12 +517,84 @@ mod tests { cooldown: Duration::from_millis(5), }); - cb.record_failure(1); + cb.record_failure(1, Normal); + std::thread::sleep(Duration::from_millis(10)); + let probe = cb.check(1).expect("the probe goes out"); // → HalfOpen + + cb.record_failure(1, probe); // Probe failed → back to Open. + assert_eq!(cb.state(1), CircuitState::Open); + } + + #[test] + fn a_normal_failure_does_not_reopen_a_half_open_circuit() { + let cb = CircuitBreaker::new(CircuitBreakerConfig { + failure_threshold: 1, + cooldown: Duration::from_millis(5), + }); + // A request admitted while the circuit was closed is still in flight. + let in_flight = cb.check(1).expect("closed admits"); + cb.record_failure(1, Normal); std::thread::sleep(Duration::from_millis(10)); - cb.check(1).unwrap(); // → HalfOpen + let probe = cb.check(1).expect("the probe goes out"); + + cb.record_failure(1, in_flight); + assert_eq!( + cb.state(1), + CircuitState::HalfOpen, + "only the probe's own failure reopens the circuit" + ); + assert!(cb.check(1).is_err(), "the probe is still out"); + + cb.record_success(1, probe); + assert_eq!(cb.state(1), CircuitState::Closed); + } - cb.record_failure(1); // Probe failed → back to Open. + #[test] + fn an_expired_probe_failure_does_not_reopen_the_circuit() { + let cb = CircuitBreaker::new(CircuitBreakerConfig { + failure_threshold: 1, + cooldown: Duration::from_millis(50), + }); + cb.record_failure(1, Normal); + std::thread::sleep(Duration::from_millis(60)); + let expired = cb.check(1).expect("the first probe goes out"); + std::thread::sleep(Duration::from_millis(60)); + let current = cb.check(1).expect("the expired probe is replaced"); + + cb.record_failure(1, expired); + assert_eq!(cb.state(1), CircuitState::HalfOpen); + cb.record_failure(1, current); + assert_eq!(cb.state(1), CircuitState::Open); + } + + #[test] + fn admit_probe_bypasses_an_open_circuit() { + let cb = CircuitBreaker::new(CircuitBreakerConfig { + failure_threshold: 1, + cooldown: Duration::from_secs(60), + }); + assert_eq!(cb.admit_probe(1), Normal, "a closed circuit needs no probe"); + cb.record_failure(1, Normal); + assert!(cb.check(1).is_err()); + + let probe = cb.admit_probe(1); + assert!(matches!(probe, Admission::Probe { .. })); + assert_eq!(cb.state(1), CircuitState::HalfOpen); + cb.record_success(1, probe); + assert_eq!(cb.state(1), CircuitState::Closed); + } + + #[test] + fn a_failed_forced_probe_reopens_the_circuit() { + let cb = CircuitBreaker::new(CircuitBreakerConfig { + failure_threshold: 1, + cooldown: Duration::from_secs(60), + }); + cb.record_failure(1, Normal); + let probe = cb.admit_probe(1); + cb.record_failure(1, probe); assert_eq!(cb.state(1), CircuitState::Open); + assert!(cb.check(1).is_err()); } #[test] @@ -387,11 +604,11 @@ mod tests { cooldown: Duration::from_secs(60), }); - cb.record_failure(1); - cb.record_failure(1); + cb.record_failure(1, Normal); + cb.record_failure(1, Normal); assert_eq!(cb.failure_count(1), 2); - cb.record_success(1); + cb.record_success(1, Normal); assert_eq!(cb.failure_count(1), 0); assert_eq!(cb.state(1), CircuitState::Closed); } diff --git a/nodedb-cluster/src/distributed_array/scatter.rs b/nodedb-cluster/src/distributed_array/scatter.rs index fbde79528..a697059f0 100644 --- a/nodedb-cluster/src/distributed_array/scatter.rs +++ b/nodedb-cluster/src/distributed_array/scatter.rs @@ -127,6 +127,7 @@ fn counts_against_breaker(err: &ClusterError) -> bool { | ClusterError::SpatialGather(_) | ClusterError::Bm25Gather(_) | ClusterError::TsGather(_) + | ClusterError::ShufflePush(_) | ClusterError::RemoteUntyped { .. } => true, } } @@ -176,7 +177,7 @@ pub async fn fan_out( for &shard_id in ¶ms.shard_ids { // Circuit-breaker gate: treat shard_id as the peer identifier. - circuit_breaker.check(shard_id as u64)?; + let admission = circuit_breaker.check(shard_id as u64)?; let env = VShardEnvelope::new( msg_type_from_opcode(opcode)?, @@ -191,8 +192,8 @@ pub async fn fan_out( futs.push(async move { match call_with_wrong_owner_retry(&dispatch, env, timeout_ms).await { - Ok(resp) => Ok((cb_shard, resp.payload)), - Err(e) => Err((cb_shard, e)), + Ok(resp) => Ok((cb_shard, admission, resp.payload)), + Err(e) => Err((cb_shard, admission, e)), } }); } @@ -200,13 +201,13 @@ pub async fn fan_out( let mut results = Vec::with_capacity(params.shard_ids.len()); while let Some(outcome) = futs.next().await { match outcome { - Ok((shard_id, payload)) => { - circuit_breaker.record_success(shard_id as u64); + Ok((shard_id, admission, payload)) => { + circuit_breaker.record_success(shard_id as u64, admission); results.push((shard_id, payload)); } - Err((shard_id, e)) => { + Err((shard_id, admission, e)) => { if counts_against_breaker(&e) { - circuit_breaker.record_failure(shard_id as u64); + circuit_breaker.record_failure(shard_id as u64, admission); } return Err(e); } @@ -234,7 +235,7 @@ pub async fn fan_out_partitioned( let mut futs = futures::stream::FuturesUnordered::new(); for (shard_id, payload) in per_shard { - circuit_breaker.check(*shard_id as u64)?; + let admission = circuit_breaker.check(*shard_id as u64)?; let env = VShardEnvelope::new( msg_type_from_opcode(opcode)?, @@ -249,8 +250,8 @@ pub async fn fan_out_partitioned( futs.push(async move { match call_with_wrong_owner_retry(&dispatch, env, timeout_ms).await { - Ok(resp) => Ok((cb_shard, resp.payload)), - Err(e) => Err((cb_shard, e)), + Ok(resp) => Ok((cb_shard, admission, resp.payload)), + Err(e) => Err((cb_shard, admission, e)), } }); } @@ -258,13 +259,13 @@ pub async fn fan_out_partitioned( let mut results = Vec::with_capacity(per_shard.len()); while let Some(outcome) = futs.next().await { match outcome { - Ok((shard_id, payload)) => { - circuit_breaker.record_success(shard_id as u64); + Ok((shard_id, admission, payload)) => { + circuit_breaker.record_success(shard_id as u64, admission); results.push((shard_id, payload)); } - Err((shard_id, e)) => { + Err((shard_id, admission, e)) => { if counts_against_breaker(&e) { - circuit_breaker.record_failure(shard_id as u64); + circuit_breaker.record_failure(shard_id as u64, admission); } return Err(e); } @@ -468,7 +469,7 @@ mod tests { cooldown: Duration::from_secs(60), }); // Trip the breaker for shard 0. - cb.record_failure(0); + cb.record_failure(0, crate::circuit_breaker::Admission::Normal); let params = FanOutParams { shard_ids: vec![0], diff --git a/nodedb-cluster/src/distributed_graph/barrier.rs b/nodedb-cluster/src/distributed_graph/barrier.rs index 9f8230376..46d5ba964 100644 --- a/nodedb-cluster/src/distributed_graph/barrier.rs +++ b/nodedb-cluster/src/distributed_graph/barrier.rs @@ -7,9 +7,14 @@ use thiserror::Error; -/// A superstep aggregate was read while shards were still missing. +/// A superstep could not complete as one step of the whole graph. #[derive(Debug, Error, PartialEq, Eq)] pub enum BspBarrierError { + /// A routed contribution names a vertex its receiving shard does not + /// own. Applying nothing would drop its mass from the graph. + #[error("'{algorithm}' contribution to vertex '{vertex}' reached a shard that does not own it")] + UnownedContribution { algorithm: String, vertex: String }, + /// A superstep aggregate was read while shards were still missing. #[error( "superstep barrier incomplete for '{algorithm}' at iteration {iteration}: \ {acked} of {expected} shards ACKed" diff --git a/nodedb-cluster/src/distributed_graph/mod.rs b/nodedb-cluster/src/distributed_graph/mod.rs index e6b736271..1bcda8ef2 100644 --- a/nodedb-cluster/src/distributed_graph/mod.rs +++ b/nodedb-cluster/src/distributed_graph/mod.rs @@ -9,7 +9,7 @@ pub mod wcc; pub use barrier::{BspBarrierError, SuperstepTotals}; pub use coordinator::BspCoordinator; -pub use pagerank::ShardPageRankState; +pub use pagerank::{PageRankUpdate, ShardPageRankState}; pub use pattern_match::{ DistributedMatchCoordinator, PatternContinuation, ResolvedContinuationArgs, ShardMatchResult, }; diff --git a/nodedb-cluster/src/distributed_graph/pagerank.rs b/nodedb-cluster/src/distributed_graph/pagerank.rs index 92adc0137..e85dd035b 100644 --- a/nodedb-cluster/src/distributed_graph/pagerank.rs +++ b/nodedb-cluster/src/distributed_graph/pagerank.rs @@ -1,9 +1,25 @@ // SPDX-License-Identifier: BUSL-1.1 //! Per-shard PageRank execution state for distributed BSP. +//! +//! One BSP iteration is one power iteration of single-node PageRank: every +//! new rank is computed from one rank vector. A shard's cross-shard +//! contributions therefore cross the barrier before they are applied: +//! +//! 1. [`ShardPageRankState::scatter`] computes this shard's dangling mass and +//! its contributions to other shards' vertices from the current rank. +//! 2. The coordinator routes them to the owning shards. +//! 3. [`ShardPageRankState::update`] computes the next rank from the same +//! current rank: the redistributed base, the local contributions, and the +//! routed contributions. +//! +//! No rank mass is in flight when a run halts, so the ranks always sum to +//! 1.0. use std::collections::HashMap; +use super::barrier::BspBarrierError; + /// Per-superstep outbound cross-shard contributions: `target_shard_id -> /// [(destination_vertex_name, contribution)]`. The coordinator routes each entry /// to the shard that owns `target_shard_id` for the next superstep. @@ -21,6 +37,25 @@ pub struct ShardPageRankState { pub incoming_contributions: HashMap, } +/// The inputs of one [`ShardPageRankState::update`]. +pub struct PageRankUpdate<'a> { + pub damping: f64, + /// The vertex count of the whole graph. + pub global_n: usize, + /// The dangling mass of the current rank over the whole graph: the sum + /// of every shard's `scatter` dangling mass. + pub global_dangling_sum: f64, + /// `None` for uniform PageRank. `Some(p)` for Personalized PageRank: + /// this shard's globally normalized seed share per owned vertex, + /// aligned with `rank`. Both the teleport mass and the dangling mass + /// spread by `p`, as single-node `build_personalization` spreads them. + pub personalization: Option<&'a [f64]>, + /// Owned vertex index to its owned destination indices. + pub local_edge_iter: &'a dyn Fn(u32) -> Vec, + /// Vertex name to its owned vertex index. + pub node_id_to_local: &'a dyn Fn(&str) -> Option, +} + impl ShardPageRankState { /// Initialize from local CSR partition. pub fn init( @@ -65,89 +100,60 @@ impl ShardPageRankState { } } - /// Execute one superstep. Returns `(local_delta, local_dangling_sum, - /// outbound_contributions)`. - /// - /// # Global dangling mass - /// - /// `global_dangling_sum` is the dangling-node rank mass aggregated by the - /// coordinator from ALL shards' previous-superstep local sums (supplied via - /// the `local_dangling_sum` field of each shard's return value). The base - /// teleport term therefore uses the GLOBAL dangling mass rather than this - /// shard's local dangling mass, so dangling-node rank redistributes across - /// the WHOLE graph, not just the shard that owns the dangling node. - /// - /// On superstep 0 the coordinator passes `0.0` (no previous local sums exist - /// yet) and the base collapses to the plain teleport `(1−d)/n` — identical - /// to before, correct for a single-shard graph with no dangling nodes. The - /// 1-superstep lag converges to the correct fixed point, exactly like the - /// existing cross-shard contribution round-trip. - /// - /// The returned `local_dangling_sum` is this shard's dangling-node mass - /// computed from `self.rank` BEFORE the rank swap; the coordinator sums these - /// across shards into the next superstep's `global_dangling_sum`. - /// - /// # Incoming contributions - /// - /// Cross-shard contributions accumulated via [`Self::add_remote_contribution`] - /// are folded into `next_rank` **after** the local scatter and **before** the - /// rank swap. This ordering is mandatory for a stateless per-superstep round - /// trip: applying them after the swap silently lands them in the *previous* - /// iteration's rank vector. + /// This shard's dangling mass and its cross-shard contributions, both + /// from the current rank. Returns `(local_dangling_sum, outbound)`. /// - /// `node_id_to_local` maps an incoming contribution's destination vertex name - /// to its local dense index; contributions whose target is not owned by this - /// shard are dropped. - /// - /// # Personalization (Personalized PageRank) - /// - /// `personalization` is `None` for standard (uniform) PageRank and - /// `Some(p)` for Personalized PageRank, where `p` is this shard's per-owned-node - /// GLOBALLY-normalized seed share, positionally aligned with `self.rank` / - /// `self.next_rank` (so `Σ_global p_i == 1.0` across ALL shards). When present, - /// both the teleport mass and the dangling mass redistribute by `p` instead of - /// uniformly — identical to single-node `build_personalization` semantics. The - /// scalar `redistributed` total mass is the same in both branches; only HOW it - /// is spread differs (uniform `/ n` vs. per-node `* p_i`). - pub fn superstep( - &mut self, - damping: f64, - global_n: usize, - global_dangling_sum: f64, - personalization: Option<&[f64]>, - local_edge_iter: &dyn Fn(u32) -> Vec, - node_id_to_local: &dyn Fn(&str) -> Option, - ) -> (f64, f64, OutboundContributions) { - let n = global_n as f64; - - // Compute THIS shard's local dangling mass from the PRE-SWAP rank vector - // and return it so the coordinator can aggregate it for the NEXT superstep. + /// The coordinator sums every shard's dangling mass into the next + /// update's `global_dangling_sum`, and routes each contribution to the + /// shard that owns its destination. + pub fn scatter(&self, damping: f64) -> (f64, OutboundContributions) { let local_dangling_sum: f64 = self .rank .iter() - .enumerate() - .filter(|(i, _)| self.is_dangling[*i]) - .map(|(_, r)| r) + .zip(&self.is_dangling) + .filter(|(_, dangling)| **dangling) + .map(|(rank, _)| rank) .sum(); - // Total mass to redistribute this superstep (SCALAR, matches single-node): - // the (1 - damping) teleport budget plus the damped GLOBAL dangling mass - // (aggregated by the coordinator from all shards' previous-superstep local - // sums). When there are no dangling nodes anywhere, `global_dangling_sum` - // is 0.0 and `redistributed == 1 - damping`. - let redistributed = (1.0 - damping) + damping * global_dangling_sum; + let mut outbound: OutboundContributions = HashMap::new(); + for (u, boundary) in &self.boundary_edges { + let u = *u as usize; + let deg = self.out_degrees[u]; + if deg == 0 { + continue; + } + let contrib = damping * self.rank[u] / deg as f64; + for (dst_name, target_shard) in boundary { + outbound + .entry(*target_shard) + .or_default() + .push((dst_name.clone(), contrib)); + } + } + (local_dangling_sum, outbound) + } + + /// Replace the current rank with the next one and return the L1 delta + /// between them. + /// + /// The next rank is the redistributed base, plus every local + /// contribution from the current rank, plus every contribution added by + /// [`Self::add_remote_contribution`]. Those came from the other shards' + /// `scatter` of the same iteration's rank, so the step is one power + /// iteration of the whole graph. + /// + /// A contribution to a vertex this shard does not own is an error: its + /// mass would leave the graph. + pub fn update(&mut self, step: PageRankUpdate<'_>) -> Result { + let redistributed = (1.0 - step.damping) + step.damping * step.global_dangling_sum; - match personalization { - // Uniform: every node gets an equal share of the redistributed mass. + match step.personalization { None => { - let base = redistributed / n; + let base = redistributed / step.global_n as f64; for r in self.next_rank.iter_mut() { *r = base; } } - // PPR: each owned node's base is its GLOBAL seed share of the - // redistributed mass. Summed over all shards this injects exactly - // `redistributed * Σ_global p_i == redistributed` (mass conserved). Some(p) => { for (slot, &pi) in self.next_rank.iter_mut().zip(p) { *slot = redistributed * pi; @@ -155,36 +161,25 @@ impl ShardPageRankState { } } - let mut outbound: HashMap> = HashMap::new(); for u in 0..self.vertex_count { let deg = self.out_degrees[u]; if deg == 0 { continue; } - let contrib = damping * self.rank[u] / deg as f64; - - // Scatter to local edges. - for dst in local_edge_iter(u as u32) { + let contrib = step.damping * self.rank[u] / deg as f64; + for dst in (step.local_edge_iter)(u as u32) { self.next_rank[dst as usize] += contrib; } - - // Scatter to boundary edges (cross-shard). - if let Some(boundary) = self.boundary_edges.get(&(u as u32)) { - for (dst_name, target_shard) in boundary { - outbound - .entry(*target_shard) - .or_default() - .push((dst_name.clone(), contrib)); - } - } } - // Fold incoming cross-shard contributions into next_rank BEFORE the - // swap so they land in the iteration being computed, not the prior one. - for (vertex_name, contrib) in &self.incoming_contributions { - if let Some(local_id) = node_id_to_local(vertex_name) { - self.next_rank[local_id as usize] += contrib; - } + for (vertex, contrib) in self.incoming_contributions.drain() { + let Some(local_id) = (step.node_id_to_local)(&vertex) else { + return Err(BspBarrierError::UnownedContribution { + algorithm: "pagerank".to_owned(), + vertex, + }); + }; + self.next_rank[local_id as usize] += contrib; } let delta: f64 = self @@ -195,9 +190,7 @@ impl ShardPageRankState { .sum(); std::mem::swap(&mut self.rank, &mut self.next_rank); - self.incoming_contributions.clear(); - - (delta, local_dangling_sum, outbound) + Ok(delta) } pub fn add_remote_contribution(&mut self, vertex_name: String, value: f64) { @@ -212,6 +205,19 @@ impl ShardPageRankState { mod tests { use super::*; + fn ring3(node: u32) -> Vec { + match node { + 0 => vec![1], + 1 => vec![2], + 2 => vec![0], + _ => Vec::new(), + } + } + + fn no_remote(_name: &str) -> Option { + None + } + #[test] fn shard_state_init() { let state = ShardPageRankState::init(3, vec![2, 1, 0], |_| None, &|_node| Vec::new()); @@ -239,127 +245,69 @@ mod tests { } #[test] - fn shard_superstep_local_only() { + fn shard_update_local_only() { let mut state = ShardPageRankState::init(3, vec![1, 1, 1], |_| None, &|_| Vec::new()); - // No dangling nodes in this ring, so global_dangling_sum = 0.0 — behaviour - // is unchanged: base == teleport only, mass is conserved. - let (delta, local_dangling_sum, outbound) = state.superstep( - 0.85, - 3, - 0.0, // no dangling nodes anywhere - None, // uniform PageRank - &|node| match node { - 0 => vec![1], - 1 => vec![2], - 2 => vec![0], - _ => Vec::new(), - }, - &|_name| None, - ); + let (local_dangling_sum, outbound) = state.scatter(0.85); assert!(outbound.is_empty()); - assert!(delta >= 0.0); - // No dangling nodes → local_dangling_sum must be 0.0. assert_eq!(local_dangling_sum, 0.0); + let delta = state + .update(PageRankUpdate { + damping: 0.85, + global_n: 3, + global_dangling_sum: 0.0, + personalization: None, + local_edge_iter: &ring3, + node_id_to_local: &no_remote, + }) + .expect("update"); + assert!(delta >= 0.0); let sum: f64 = state.rank.iter().sum(); assert!((sum - 1.0).abs() < 1e-6); } #[test] fn global_dangling_sum_used_in_base() { - // 2 vertices: vertex 0 has out-degree 1 (non-dangling), vertex 1 has - // out-degree 0 (dangling). Simulate a cross-shard scenario by passing a - // global_dangling_sum LARGER than this shard's own dangling mass — as if - // additional dangling mass exists on other shards. Assert each node's - // base reflects the GLOBAL value, not just the local shard's dangling sum. - let n: usize = 10; // fictional global node count — larger than local 2 + // Vertex 0 has out-degree 1, vertex 1 is dangling. The global dangling + // mass is larger than this shard's own, as if other shards hold more. + let n: usize = 10; let damping = 0.85_f64; - // Local dangling mass = rank of vertex 1 = 1/2 (uniform init). - // External dangling mass = 0.3 (simulated from other shards). - let local_dangling_init = 1.0 / 2.0_f64; - let extra_dangling = 0.3_f64; - let global_dangling = local_dangling_init + extra_dangling; + let global_dangling = 0.5 + 0.3; - let mut state = ShardPageRankState::init( - 2, - vec![1, 0], // vertex 1 is dangling - |_| None, - &|_| Vec::new(), // no boundary edges in this focused test - ); - // Edge 0 → 1 (local, no boundary edges needed for this assertion). - let (_delta, _local_dangling_sum, _outbound) = state.superstep( - damping, - n, - global_dangling, - None, // uniform PageRank - &|node| if node == 0 { vec![1] } else { Vec::new() }, - &|_name| None, - ); + let mut state = ShardPageRankState::init(2, vec![1, 0], |_| None, &|_| Vec::new()); + state + .update(PageRankUpdate { + damping, + global_n: n, + global_dangling_sum: global_dangling, + personalization: None, + local_edge_iter: &|node| if node == 0 { vec![1] } else { Vec::new() }, + node_id_to_local: &no_remote, + }) + .expect("update"); - // The base each node receives must be ((1-d) + d*global_dangling) / n. + // Vertex 0 has no in-edge, so its rank is the base alone. let expected_base = ((1.0 - damping) + damping * global_dangling) / n as f64; - // Vertex 0: base + contribution from no in-edge in this pass - // (vertex 1 is dangling; no out-edges scatter to 0 locally). - // Vertex 1: base + damping * rank[0] / 1 (edge 0→1). - // - // We assert the minimum: both nodes start with at least `expected_base` - // in next_rank (the base is applied before any scatter). Since next_rank - // after the swap is state.rank, and vertex 1 also got vertex 0's scatter, - // we check vertex 0's rank == expected_base (it received no in-edge - // scatter this superstep). - let rank_v0 = state.rank[0]; - assert!( - (rank_v0 - expected_base).abs() < 1e-12, - "vertex 0 rank should equal base={expected_base:.15} (no in-edge scatter), got {rank_v0:.15}" - ); + assert!((state.rank[0] - expected_base).abs() < 1e-12); } #[test] - fn personalized_superstep_biases_base_toward_seed() { - // 3-node ring, no dangling nodes. A seed concentrated on vertex 0 - // (p = [1, 0, 0]) must give vertex 0 a strictly-larger teleport base than - // its peers — the distributed analogue of single-node PPR biasing. + fn personalized_update_biases_base_toward_seed() { let damping = 0.85_f64; - let n: usize = 3; - // Globally-normalized seed share: all mass on vertex 0. let p = [1.0_f64, 0.0, 0.0]; - let mut state = ShardPageRankState::init(3, vec![1, 1, 1], |_| None, &|_| Vec::new()); - let (_delta, local_dangling_sum, outbound) = state.superstep( - damping, - n, - 0.0, // no dangling nodes - Some(&p), - &|node| match node { - 0 => vec![1], - 1 => vec![2], - 2 => vec![0], - _ => Vec::new(), - }, - &|_name| None, - ); - assert!(outbound.is_empty()); - assert_eq!(local_dangling_sum, 0.0); - - // redistributed = (1 - d) since dangling sum is 0. The seeded node's base - // is `redistributed * 1.0`, peers' base is `redistributed * 0.0 == 0.0`, - // before any scatter. After scatter, vertex 0 (seeded base + incoming from - // vertex 2 which has base 0 → 0 contribution) must strictly outrank the - // others. + state + .update(PageRankUpdate { + damping, + global_n: 3, + global_dangling_sum: 0.0, + personalization: Some(&p), + local_edge_iter: &ring3, + node_id_to_local: &no_remote, + }) + .expect("update"); let redistributed = 1.0 - damping; - // Vertex 0 receives base = redistributed*1.0 plus a scatter from vertex 2, - // whose pre-swap rank was the uniform init 1/3 (init unchanged here). - // Regardless, vertex 0's rank must exceed both peers'. - assert!( - state.rank[0] > state.rank[1] && state.rank[0] > state.rank[2], - "seeded vertex 0 base must dominate: {:?}", - state.rank - ); - // The seeded node's base alone (before scatter) is `redistributed`. - assert!( - state.rank[0] >= redistributed - 1e-12, - "seeded base should be at least redistributed={redistributed}, got {}", - state.rank[0] - ); + assert!(state.rank[0] > state.rank[1] && state.rank[0] > state.rank[2]); + assert!(state.rank[0] >= redistributed - 1e-12); } #[test] @@ -371,4 +319,92 @@ mod tests { assert!((state.incoming_contributions["n0"] - 0.3).abs() < 1e-10); assert!((state.incoming_contributions["n1"] - 0.3).abs() < 1e-10); } + + /// A contribution to a vertex the shard does not own is refused. + #[test] + fn an_unowned_contribution_is_an_error() { + let mut state = ShardPageRankState::init(1, vec![0], |_| None, &|_| Vec::new()); + state.add_remote_contribution("elsewhere".into(), 0.1); + let result = state.update(PageRankUpdate { + damping: 0.85, + global_n: 2, + global_dangling_sum: 0.0, + personalization: None, + local_edge_iter: &|_| Vec::new(), + node_id_to_local: &no_remote, + }); + assert!(matches!( + result, + Err(BspBarrierError::UnownedContribution { .. }) + )); + } + + /// A 4-ring split over two shards, every edge crossing shards, keeps a + /// total rank of 1.0 after every iteration and stays uniform: no mass is + /// in flight between iterations. + #[test] + fn a_cross_shard_ring_conserves_mass_every_iteration() { + // Shard A owns r0, r2. Shard B owns r1, r3. Ring r0→r1→r2→r3→r0. + let names = [["r0", "r2"], ["r1", "r3"]]; + let succ = |name: &str| -> &'static str { + match name { + "r0" => "r1", + "r1" => "r2", + "r2" => "r3", + _ => "r0", + } + }; + let mut shards: Vec = names + .iter() + .enumerate() + .map(|(shard, owned)| { + let edges = |i: u32| { + vec![( + succ(owned[i as usize]).to_string(), + true, + (1 - shard) as u16, + )] + }; + let mut state = ShardPageRankState::init(2, vec![1, 1], |_| None, &edges); + state.rank = vec![0.25, 0.25]; + state + }) + .collect(); + for _ in 0..30 { + let mut dangling = 0.0; + let mut routed: Vec<(usize, String, f64)> = Vec::new(); + for state in &shards { + let (d, outbound) = state.scatter(0.85); + dangling += d; + for (target, contribs) in outbound { + for (name, c) in contribs { + routed.push((target as usize, name, c)); + } + } + } + for (target, name, c) in routed { + shards[target].add_remote_contribution(name, c); + } + for (shard, state) in shards.iter_mut().enumerate() { + let owned = names[shard]; + state + .update(PageRankUpdate { + damping: 0.85, + global_n: 4, + global_dangling_sum: dangling, + personalization: None, + local_edge_iter: &|_| Vec::new(), + node_id_to_local: &|name| { + owned.iter().position(|n| *n == name).map(|i| i as u32) + }, + }) + .expect("every contribution is owned"); + } + let total: f64 = shards.iter().flat_map(|s| s.rank.iter()).sum(); + assert!((total - 1.0).abs() < 1e-12, "mass conserved, got {total}"); + for rank in shards.iter().flat_map(|s| s.rank.iter()) { + assert!((rank - 0.25).abs() < 1e-12, "ring stays uniform"); + } + } + } } diff --git a/nodedb-cluster/src/error.rs b/nodedb-cluster/src/error.rs index 2aa8b43b7..5ee76b7b1 100644 --- a/nodedb-cluster/src/error.rs +++ b/nodedb-cluster/src/error.rs @@ -229,6 +229,9 @@ pub enum ClusterError { #[error("timeseries gather error: {0}")] TsGather(#[from] crate::distributed_timeseries::TsGatherError), + #[error("shuffle push error: {0}")] + ShufflePush(#[from] crate::transport::ShufflePushError), + /// A remote node answered with an error whose type has no wire mirror. /// `detail` is that error's message. #[error("remote error: {detail}")] @@ -245,3 +248,47 @@ pub enum ClusterError { detail: String, }, } + +impl ClusterError { + /// Whether the error means the link to the peer failed. + /// + /// A link failure counts against the peer's circuit breaker and ends a + /// batch of sends to that peer. Every other error is an answer: the peer + /// is up, and the next request to it can still succeed. + pub fn is_link_failure(&self) -> bool { + matches!( + self, + Self::Transport { .. } | Self::CircuitOpen { .. } | Self::NodeUnreachable { .. } + ) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn only_link_errors_are_link_failures() { + assert!( + ClusterError::Transport { + detail: "reset".into() + } + .is_link_failure() + ); + assert!( + ClusterError::CircuitOpen { + node_id: 1, + failures: 5 + } + .is_link_failure() + ); + assert!(ClusterError::NodeUnreachable { node_id: 1 }.is_link_failure()); + assert!(!ClusterError::GroupNotFound { group_id: 4 }.is_link_failure()); + assert!( + !ClusterError::RemoteUntyped { + detail: "refused".into() + } + .is_link_failure() + ); + } +} diff --git a/nodedb-cluster/src/group_disk/mod.rs b/nodedb-cluster/src/group_disk/mod.rs new file mode 100644 index 000000000..5cad1caeb --- /dev/null +++ b/nodedb-cluster/src/group_disk/mod.rs @@ -0,0 +1,13 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Raft group disks: per-group log storage whose writes are made durable by +//! a writer thread of their own, off the `MultiRaft` lock and off the async +//! threads. + +pub mod staged; +pub mod ticket; +pub mod writer; + +pub use staged::StagedLogStorage; +pub use ticket::DurabilityTicket; +pub use writer::{DiskProgress, GroupDisk}; diff --git a/nodedb-cluster/src/group_disk/staged.rs b/nodedb-cluster/src/group_disk/staged.rs new file mode 100644 index 000000000..435203d40 --- /dev/null +++ b/nodedb-cluster/src/group_disk/staged.rs @@ -0,0 +1,209 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The Raft log storage a mounted group runs on: every write is staged on +//! the group's disk and returns at once. +//! +//! The Raft node keeps its whole log in memory, so it reads nothing back +//! after its restore. A write returns before it is durable, so the node +//! learns how far the disk has come from [`LogStorage::stable_through`], and +//! every caller that sends a reply or a vote request depending on a write +//! first awaits a [`super::DurabilityTicket`]. + +use std::path::Path; +use std::sync::Arc; +use std::thread::JoinHandle; + +use nodedb_raft::message::LogEntry; +use nodedb_raft::state::HardState; +use nodedb_raft::storage::LogStorage; +use tracing::error; + +use super::ticket::DurabilityTicket; +use super::writer::{GroupDisk, StorageOp}; + +/// A group's staged log storage and the owner of its writer thread. +pub struct StagedLogStorage { + disk: Arc, + /// The snapshot boundary as this node's log holds it. + snapshot: (u64, u64), + writer: Option>, +} + +impl StagedLogStorage { + /// Open `group_id`'s log at `path` and start its writer thread. Blocks + /// on disk: call it off the async threads. + pub fn open(group_id: u64, path: &Path) -> crate::Result { + let disk = Arc::new(GroupDisk::open(group_id, path)?); + let snapshot = disk.storage().snapshot_metadata(); + disk.set_restored_stable(snapshot); + let writer = { + let disk = Arc::clone(&disk); + std::thread::Builder::new() + .name(format!("raft-disk-{group_id}")) + .spawn(move || disk.run_writer()) + .map_err(|e| crate::ClusterError::Storage { + detail: format!("start the disk writer of raft group {group_id}: {e}"), + })? + }; + Ok(Self { + disk, + snapshot, + writer: Some(writer), + }) + } + + /// The group's disk, shared with its tickets. + pub fn disk(&self) -> &Arc { + &self.disk + } + + /// A ticket for every write staged so far, or `None` when all of them + /// are durable. + pub fn ticket(&self) -> Option { + DurabilityTicket::for_staged(&self.disk) + } + + /// The sequence number of the last staged write. A caller reads it + /// before a Raft call and passes it to [`Self::reply_ticket`]. + pub fn staged_through(&self) -> u64 { + self.disk.staged_through() + } + + /// A ticket for the writes a reply built after `mark` depends on, or + /// `None` when they are durable. See [`DurabilityTicket::for_reply`]. + pub fn reply_ticket(&self, mark: u64) -> Option { + DurabilityTicket::for_reply(&self.disk, mark) + } +} + +impl Drop for StagedLogStorage { + /// Close the disk and wait for its writer to make the staged writes + /// durable. The log file is closed when this returns, so the group can be + /// opened again. Callers drop a replica off the async threads. + fn drop(&mut self) { + self.disk.close(); + if let Some(writer) = self.writer.take() + && writer.join().is_err() + { + error!( + group_id = self.disk.group_id(), + "raft disk: the writer thread panicked" + ); + } + } +} + +impl LogStorage for StagedLogStorage { + fn append(&mut self, entries: &[LogEntry]) -> nodedb_raft::error::Result<()> { + if !entries.is_empty() { + self.disk.stage(StorageOp::Append(entries.to_vec())); + } + Ok(()) + } + + fn truncate(&mut self, index: u64) -> nodedb_raft::error::Result<()> { + self.disk.stage(StorageOp::Truncate(index)); + Ok(()) + } + + fn load_entries_after(&self, snapshot_index: u64) -> nodedb_raft::error::Result> { + let entries = self.disk.storage().load_entries_after(snapshot_index)?; + if let Some(last) = entries.last() { + self.disk.set_restored_stable((last.index, last.term)); + } + Ok(entries) + } + + fn compact(&mut self, index: u64, term: u64) -> nodedb_raft::error::Result<()> { + if index > self.snapshot.0 { + self.snapshot = (index, term); + } + self.disk.stage(StorageOp::Compact { index, term }); + Ok(()) + } + + fn snapshot_metadata(&self) -> (u64, u64) { + self.snapshot + } + + fn save_hard_state(&mut self, state: &HardState) -> nodedb_raft::error::Result<()> { + self.disk.stage(StorageOp::HardState(state.clone())); + Ok(()) + } + + fn load_hard_state(&self) -> nodedb_raft::error::Result { + self.disk.storage().load_hard_state() + } + + fn save_applied_index(&mut self, index: u64) -> nodedb_raft::error::Result<()> { + self.disk.stage(StorageOp::AppliedIndex(index)); + Ok(()) + } + + fn load_applied_index(&self) -> nodedb_raft::error::Result { + self.disk.storage().load_applied_index() + } + + fn stable_through(&self) -> Option<(u64, u64)> { + Some(self.disk.progress().stable) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn entry(index: u64, term: u64) -> LogEntry { + LogEntry { + term, + index, + data: vec![index as u8], + } + } + + /// Staged writes become durable in order, and a reopen reads them back. + #[test] + fn staged_writes_are_durable_after_their_ticket_and_survive_a_reopen() { + let dir = tempfile::tempdir().expect("tempdir"); + let path = dir.path().join("group-7.redb"); + { + let mut storage = StagedLogStorage::open(7, &path).expect("open"); + storage + .append(&[entry(1, 1), entry(2, 1), entry(3, 1)]) + .expect("stage"); + storage.truncate(3).expect("stage"); + storage.append(&[entry(3, 2)]).expect("stage"); + storage + .save_hard_state(&HardState { + current_term: 2, + voted_for: 5, + }) + .expect("stage"); + let ticket = storage.ticket().expect("writes are staged"); + assert!(ticket.wait_blocking()); + assert_eq!(storage.stable_through(), Some((3, 2))); + assert!(storage.ticket().is_none(), "nothing is left to write"); + } + let storage = StagedLogStorage::open(7, &path).expect("reopen"); + let entries = storage.load_entries_after(0).expect("load"); + let terms: Vec<(u64, u64)> = entries.iter().map(|e| (e.index, e.term)).collect(); + assert_eq!(terms, vec![(1, 1), (2, 1), (3, 2)]); + assert_eq!(storage.load_hard_state().expect("load").voted_for, 5); + assert_eq!(storage.stable_through(), Some((3, 2))); + } + + /// Dropping the storage makes every staged write durable first. + #[test] + fn a_drop_flushes_the_staged_writes() { + let dir = tempfile::tempdir().expect("tempdir"); + let path = dir.path().join("group-8.redb"); + { + let mut storage = StagedLogStorage::open(8, &path).expect("open"); + storage.append(&[entry(1, 1)]).expect("stage"); + storage.save_applied_index(1).expect("stage"); + } + let storage = StagedLogStorage::open(8, &path).expect("reopen"); + assert_eq!(storage.load_entries_after(0).expect("load").len(), 1); + assert_eq!(storage.load_applied_index().expect("load"), 1); + } +} diff --git a/nodedb-cluster/src/group_disk/ticket.rs b/nodedb-cluster/src/group_disk/ticket.rs new file mode 100644 index 000000000..e6def5034 --- /dev/null +++ b/nodedb-cluster/src/group_disk/ticket.rs @@ -0,0 +1,152 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A claim on the durability of a group's staged writes. +//! +//! A caller takes a ticket under the `MultiRaft` lock, right after the Raft +//! call that staged the writes its reply depends on. It releases the lock, +//! awaits the ticket, and only then sends the reply or the vote request. No +//! other group, and no other caller of this group, waits on the disk +//! meanwhile. + +use std::sync::Arc; + +use super::writer::GroupDisk; + +/// The durability of every write a group staged up to one sequence number. +#[derive(Clone)] +pub struct DurabilityTicket { + disk: Arc, + seq: u64, +} + +impl std::fmt::Debug for DurabilityTicket { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("DurabilityTicket") + .field("group_id", &self.disk.group_id()) + .field("seq", &self.seq) + .finish() + } +} + +impl DurabilityTicket { + /// A ticket for every write `disk` staged so far, or `None` when all of + /// them are durable. + pub(super) fn for_staged(disk: &Arc) -> Option { + let seq = disk.staged_through(); + (disk.progress().durable_seq < seq).then(|| Self { + disk: Arc::clone(disk), + seq, + }) + } + + /// A ticket for a reply staged after `mark`, or `None` when the writes it + /// depends on are durable. `mark` is [`GroupDisk::staged_through`] read + /// before the Raft call that built the reply. + /// + /// A reply depends on the latest hard state, whoever staged it: its term + /// and vote must survive a restart. When the call staged writes, the reply + /// depends on them too. Writes are durable in staging order, so the + /// ticket then covers every write staged so far. An `AppliedIndex` write + /// staged by the apply loop is no dependency of a reply that staged + /// nothing. + pub(super) fn for_reply(disk: &Arc, mark: u64) -> Option { + let (staged, hard_state) = disk.staged_marks(); + let seq = if staged > mark { staged } else { hard_state }; + (disk.progress().durable_seq < seq).then(|| Self { + disk: Arc::clone(disk), + seq, + }) + } + + /// The group the ticket belongs to. + pub fn group_id(&self) -> u64 { + self.disk.group_id() + } + + /// Wait until the writes are durable. Fails when the group's disk closed + /// first: the group was unmounted, and the writes will never be durable. + pub async fn durable(self) -> crate::Result<()> { + let mut rx = self.disk.subscribe(); + loop { + let progress = *rx.borrow_and_update(); + if progress.durable_seq >= self.seq { + return Ok(()); + } + if progress.closed || rx.changed().await.is_err() { + return Err(crate::ClusterError::Storage { + detail: format!( + "raft group {}: the disk closed before write {} was durable", + self.disk.group_id(), + self.seq + ), + }); + } + } + } + + /// Block until the writes are durable. Returns whether they are. For + /// callers off the async threads. + pub fn wait_blocking(&self) -> bool { + self.disk.wait_blocking(self.seq) + } +} + +#[cfg(test)] +mod tests { + use nodedb_raft::message::LogEntry; + use nodedb_raft::state::HardState; + + use super::super::writer::StorageOp; + use super::*; + + /// A disk with no writer thread: nothing staged becomes durable. + fn idle_disk() -> (tempfile::TempDir, Arc) { + let dir = tempfile::tempdir().expect("tempdir"); + let path = dir.path().join("group-3.redb"); + let disk = Arc::new(GroupDisk::open(3, &path).expect("open")); + (dir, disk) + } + + #[test] + fn a_reply_that_staged_nothing_waits_on_no_applied_index() { + let (_dir, disk) = idle_disk(); + disk.stage(StorageOp::AppliedIndex(4)); + let mark = disk.staged_through(); + + assert!(DurabilityTicket::for_reply(&disk, mark).is_none()); + assert!( + DurabilityTicket::for_staged(&disk).is_some(), + "the applied index is not durable" + ); + } + + #[test] + fn a_reply_waits_on_the_latest_hard_state_whoever_staged_it() { + let (_dir, disk) = idle_disk(); + disk.stage(StorageOp::HardState(HardState { + current_term: 2, + voted_for: 5, + })); + disk.stage(StorageOp::AppliedIndex(4)); + let mark = disk.staged_through(); + + let ticket = + DurabilityTicket::for_reply(&disk, mark).expect("the hard state is not durable"); + assert_eq!(ticket.seq, 1, "the later applied index is no dependency"); + } + + #[test] + fn a_reply_that_staged_writes_waits_on_every_staged_write() { + let (_dir, disk) = idle_disk(); + disk.stage(StorageOp::AppliedIndex(4)); + let mark = disk.staged_through(); + disk.stage(StorageOp::Append(vec![LogEntry { + term: 1, + index: 5, + data: Vec::new(), + }])); + + let ticket = DurabilityTicket::for_reply(&disk, mark).expect("the append is not durable"); + assert_eq!(ticket.seq, 2); + } +} diff --git a/nodedb-cluster/src/group_disk/writer.rs b/nodedb-cluster/src/group_disk/writer.rs new file mode 100644 index 000000000..7dfefa5ad --- /dev/null +++ b/nodedb-cluster/src/group_disk/writer.rs @@ -0,0 +1,320 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One Raft group's disk: its redb log storage and the writer thread that +//! makes staged writes durable, in the order they were staged. +//! +//! A group's Raft node stages every storage write here and returns at once, +//! so no disk write runs under the `MultiRaft` lock or on an async thread. +//! The writer thread drains the queue and applies each write to redb, one +//! fsync per batch of contiguous appends. It then publishes: +//! - the highest staged sequence number that is durable, for callers that +//! hold a [`super::DurabilityTicket`]; +//! - the last log entry the disk holds durably, which the node's leader +//! counts toward its own commit acknowledgement. +//! +//! A write that fails is retried after [`RETRY_DELAY`] and never skipped: +//! every later write depends on it. Once the disk is closed, a failing write +//! is abandoned after one attempt, and the writer ends. + +use std::collections::VecDeque; +use std::path::Path; +use std::sync::{Condvar, Mutex}; +use std::time::Duration; + +use nodedb_raft::message::LogEntry; +use nodedb_raft::state::HardState; +use nodedb_raft::storage::LogStorage; +use tokio::sync::watch; +use tracing::{error, warn}; + +use crate::raft_storage::RedbLogStorage; + +/// The wait before a failed write is tried again. +const RETRY_DELAY: Duration = Duration::from_millis(50); + +/// One storage write, staged by the group's Raft node. +#[derive(Debug, Clone)] +pub(super) enum StorageOp { + Append(Vec), + Truncate(u64), + Compact { index: u64, term: u64 }, + HardState(HardState), + AppliedIndex(u64), +} + +/// How far the disk has come. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +pub struct DiskProgress { + /// Every write staged at or below this sequence number is durable. + pub durable_seq: u64, + /// The last log entry the disk holds durably, as `(index, term)`. A term + /// of 0 above index 0 means the term is not known here. + pub stable: (u64, u64), + /// The disk is closed. A write not yet durable never becomes durable. + pub closed: bool, +} + +/// The staged writes and the next sequence number. +#[derive(Debug, Default)] +struct Queue { + ops: VecDeque<(u64, StorageOp)>, + /// The sequence number of the last staged write. + staged_through: u64, + /// The sequence number of the last staged hard state, or 0. + hard_state_seq: u64, + closed: bool, +} + +/// One group's disk, shared by its staged storage, its writer thread and the +/// holders of its durability tickets. +pub struct GroupDisk { + group_id: u64, + storage: Mutex, + queue: Mutex, + work: Condvar, + progress: Mutex, + progressed: Condvar, + progress_tx: watch::Sender, +} + +impl GroupDisk { + /// Open `group_id`'s redb log at `path`. Blocks on disk: call it off the + /// async threads. + pub(super) fn open(group_id: u64, path: &Path) -> crate::Result { + let storage = RedbLogStorage::open(path)?; + Ok(Self { + group_id, + storage: Mutex::new(storage), + queue: Mutex::new(Queue::default()), + work: Condvar::new(), + progress: Mutex::new(DiskProgress::default()), + progressed: Condvar::new(), + progress_tx: watch::Sender::new(DiskProgress::default()), + }) + } + + pub fn group_id(&self) -> u64 { + self.group_id + } + + /// The storage itself, for the reads a restore runs before the writer + /// takes any write. + pub(super) fn storage(&self) -> std::sync::MutexGuard<'_, RedbLogStorage> { + self.storage.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Record the last entry the disk holds, read at restore. + pub(super) fn set_restored_stable(&self, stable: (u64, u64)) { + let mut progress = self.progress.lock().unwrap_or_else(|p| p.into_inner()); + progress.stable = stable; + self.progress_tx.send_replace(*progress); + } + + /// Queue `op` behind every earlier write. Never waits on disk. + pub(super) fn stage(&self, op: StorageOp) { + let mut queue = self.queue.lock().unwrap_or_else(|p| p.into_inner()); + queue.staged_through += 1; + let seq = queue.staged_through; + if matches!(op, StorageOp::HardState(_)) { + queue.hard_state_seq = seq; + } + queue.ops.push_back((seq, op)); + self.work.notify_one(); + } + + /// The sequence number of the last staged write. + pub fn staged_through(&self) -> u64 { + self.queue + .lock() + .unwrap_or_else(|p| p.into_inner()) + .staged_through + } + + /// The sequence numbers of the last staged write and of the last staged + /// hard state, read together. + pub(super) fn staged_marks(&self) -> (u64, u64) { + let queue = self.queue.lock().unwrap_or_else(|p| p.into_inner()); + (queue.staged_through, queue.hard_state_seq) + } + + /// How far the disk has come. + pub fn progress(&self) -> DiskProgress { + *self.progress.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// A receiver of every progress update, for async waiters. + pub(super) fn subscribe(&self) -> watch::Receiver { + self.progress_tx.subscribe() + } + + /// Block until `seq` is durable or the disk closed. Returns whether it is + /// durable. For callers off the async threads. + pub(super) fn wait_blocking(&self, seq: u64) -> bool { + let mut progress = self.progress.lock().unwrap_or_else(|p| p.into_inner()); + loop { + if progress.durable_seq >= seq { + return true; + } + if progress.closed { + return false; + } + progress = self + .progressed + .wait(progress) + .unwrap_or_else(|p| p.into_inner()); + } + } + + /// Stop taking writes. The writer makes every write staged so far durable, + /// then ends. + pub(super) fn close(&self) { + let mut queue = self.queue.lock().unwrap_or_else(|p| p.into_inner()); + queue.closed = true; + self.work.notify_one(); + } + + /// The writer thread's body: drain and apply the queue until it is + /// closed and empty. + pub(super) fn run_writer(&self) { + loop { + let (batch, closed) = { + let mut queue = self.queue.lock().unwrap_or_else(|p| p.into_inner()); + while queue.ops.is_empty() && !queue.closed { + queue = self.work.wait(queue).unwrap_or_else(|p| p.into_inner()); + } + (queue.ops.drain(..).collect::>(), queue.closed) + }; + if batch.is_empty() { + break; + } + self.apply_batch(batch, closed); + } + let mut progress = self.progress.lock().unwrap_or_else(|p| p.into_inner()); + progress.closed = true; + self.progress_tx.send_replace(*progress); + self.progressed.notify_all(); + } + + /// Apply `batch` in order, then publish the progress it made. + fn apply_batch(&self, batch: Vec<(u64, StorageOp)>, closed: bool) { + let Some(through) = batch.last().map(|(seq, _)| *seq) else { + return; + }; + let mut stable = self.progress().stable; + for op in merge_appends(batch) { + stable = next_stable(stable, &op); + let mut attempt = 0u32; + while let Err(e) = self.apply(&op) { + attempt += 1; + if closed { + error!( + group_id = self.group_id, + error = %e, + "raft disk: a write failed while the disk closes; it is abandoned" + ); + break; + } + warn!( + group_id = self.group_id, + attempt, + error = %e, + "raft disk: a write failed; it is retried" + ); + std::thread::sleep(RETRY_DELAY); + } + } + let mut progress = self.progress.lock().unwrap_or_else(|p| p.into_inner()); + progress.durable_seq = progress.durable_seq.max(through); + progress.stable = stable; + self.progress_tx.send_replace(*progress); + self.progressed.notify_all(); + } + + fn apply(&self, op: &StorageOp) -> nodedb_raft::error::Result<()> { + let mut storage = self.storage(); + match op { + StorageOp::Append(entries) => storage.append(entries), + StorageOp::Truncate(index) => storage.truncate(*index), + StorageOp::Compact { index, term } => storage.compact(*index, *term), + StorageOp::HardState(state) => storage.save_hard_state(state), + StorageOp::AppliedIndex(index) => storage.save_applied_index(*index), + } + } +} + +/// `batch`'s writes in order, with each run of consecutive appends joined +/// into one append: one redb commit, one fsync. +fn merge_appends(batch: Vec<(u64, StorageOp)>) -> Vec { + let mut merged: Vec = Vec::with_capacity(batch.len()); + for (_, op) in batch { + match (merged.last_mut(), op) { + (Some(StorageOp::Append(run)), StorageOp::Append(more)) => run.extend(more), + (_, op) => merged.push(op), + } + } + merged +} + +/// The last durable entry once `op` is durable, from `stable` before it. +fn next_stable(stable: (u64, u64), op: &StorageOp) -> (u64, u64) { + match op { + StorageOp::Append(entries) => entries + .last() + .map_or(stable, |entry| (entry.index, entry.term)), + // The entry before the cut is still on disk, at a term not known + // here. It is recorded at term 0: no Raft entry above index 0 has + // term 0, so it never matches the log, and index 0 has term 0. + StorageOp::Truncate(index) if stable.0 >= *index => (index.saturating_sub(1), 0), + StorageOp::Compact { index, term } if *index > stable.0 => (*index, *term), + _ => stable, + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn entry(index: u64, term: u64) -> LogEntry { + LogEntry { + term, + index, + data: Vec::new(), + } + } + + #[test] + fn consecutive_appends_join_and_other_writes_keep_their_place() { + let merged = merge_appends(vec![ + (1, StorageOp::Append(vec![entry(1, 1)])), + (2, StorageOp::Append(vec![entry(2, 1)])), + (3, StorageOp::HardState(HardState::new())), + (4, StorageOp::Append(vec![entry(3, 1)])), + ]); + assert_eq!(merged.len(), 3); + assert!(matches!(&merged[0], StorageOp::Append(run) if run.len() == 2)); + assert!(matches!(merged[1], StorageOp::HardState(_))); + assert!(matches!(&merged[2], StorageOp::Append(run) if run.len() == 1)); + } + + #[test] + fn the_stable_entry_follows_appends_truncations_and_compactions() { + let stable = next_stable((0, 0), &StorageOp::Append(vec![entry(1, 1), entry(2, 1)])); + assert_eq!(stable, (2, 1)); + // A cut at 2 leaves entry 1, at a term not known here. + let stable = next_stable(stable, &StorageOp::Truncate(2)); + assert_eq!(stable, (1, 0)); + let stable = next_stable(stable, &StorageOp::Append(vec![entry(2, 2)])); + assert_eq!(stable, (2, 2)); + // A cut above the stable entry leaves it. + assert_eq!(next_stable(stable, &StorageOp::Truncate(5)), (2, 2)); + // A snapshot past the log moves the boundary; one below leaves it. + assert_eq!( + next_stable(stable, &StorageOp::Compact { index: 9, term: 3 }), + (9, 3) + ); + assert_eq!( + next_stable(stable, &StorageOp::Compact { index: 1, term: 1 }), + (2, 2) + ); + } +} diff --git a/nodedb-cluster/src/health.rs b/nodedb-cluster/src/health.rs index 486e79113..21535d112 100644 --- a/nodedb-cluster/src/health.rs +++ b/nodedb-cluster/src/health.rs @@ -16,11 +16,10 @@ use std::time::{Duration, Instant}; use tracing::{debug, info, warn}; use crate::catalog::ClusterCatalog; +use crate::error::ClusterError; use crate::loop_metrics::LoopMetrics; -use crate::rpc_codec::{ - JoinNodeInfo, PingRequest, PongResponse, RaftRpc, TopologyAck, TopologyUpdate, -}; -use crate::topology::{ClusterTopology, NodeState}; +use crate::rpc_codec::{PingRequest, PongResponse, RaftRpc, TopologyAck, TopologyUpdate}; +use crate::topology::{ClusterTopology, NodeInfo, NodeState}; use crate::transport::NexarTransport; /// Default ping interval. @@ -143,8 +142,10 @@ impl HealthMonitor { sender_id: self.node_id, topology_version: topo_version, }); + // The ping is the peer's recovery probe. An open circuit still + // lets it through, and its answer closes the circuit. handles.push(tokio::spawn(async move { - let result = transport.send_rpc(peer_id, ping).await; + let result = transport.send_probe_rpc(peer_id, ping).await; (peer_id, addr, result) })); } @@ -156,17 +157,14 @@ impl HealthMonitor { Err(_) => continue, // JoinError — task panicked, skip. }; - match result { - Ok(RaftRpc::Pong(pong)) => { + match classify_ping(result) { + PingOutcome::Pong(pong) => { topology_changed |= self.handle_pong(peer_id, &pong); } - Ok(_) => { - // Unexpected response type — count as failure. - topology_changed |= self.record_ping_failure(peer_id); - } - Err(_) => { + PingOutcome::Failed => { topology_changed |= self.record_ping_failure(peer_id); } + PingOutcome::NotSent => {} } } @@ -267,6 +265,26 @@ impl HealthMonitor { } } +/// What one ping says about the peer. +#[derive(Debug)] +enum PingOutcome { + /// The peer answered. + Pong(PongResponse), + /// The ping failed, or the peer answered with something else. + Failed, + /// This node's circuit breaker refused the ping before it left. That + /// says nothing about the peer, so it is not a ping failure. + NotSent, +} + +fn classify_ping(result: crate::error::Result) -> PingOutcome { + match result { + Ok(RaftRpc::Pong(pong)) => PingOutcome::Pong(pong), + Err(ClusterError::CircuitOpen { .. }) => PingOutcome::NotSent, + Ok(_) | Err(_) => PingOutcome::Failed, + } +} + /// Broadcast the current topology to every active peer (fire-and-forget). /// /// Shared by [`HealthMonitor`] and the cluster-join path @@ -282,18 +300,7 @@ pub fn broadcast_topology( let topo = topology.read().unwrap_or_else(|p| p.into_inner()); let update = RaftRpc::TopologyUpdate(TopologyUpdate { version: topo.version(), - nodes: topo - .all_nodes() - .map(|n| JoinNodeInfo { - node_id: n.node_id, - addr: n.addr.clone(), - state: n.state.as_u8(), - raft_groups: n.raft_groups.clone(), - wire_version: n.wire_version, - spiffe_id: n.spiffe_id.clone(), - spki_pin: n.spki_pin.map(|arr| arr.to_vec()), - }) - .collect(), + nodes: topo.all_nodes().map(NodeInfo::to_wire).collect(), }); let peers: Vec = topo .active_nodes() @@ -326,18 +333,7 @@ async fn broadcast_topology_to_peer( let topo = topology.read().unwrap_or_else(|p| p.into_inner()); RaftRpc::TopologyUpdate(TopologyUpdate { version: topo.version(), - nodes: topo - .all_nodes() - .map(|n| JoinNodeInfo { - node_id: n.node_id, - addr: n.addr.clone(), - state: n.state.as_u8(), - raft_groups: n.raft_groups.clone(), - wire_version: n.wire_version, - spiffe_id: n.spiffe_id.clone(), - spki_pin: n.spki_pin.map(|arr| arr.to_vec()), - }) - .collect(), + nodes: topo.all_nodes().map(NodeInfo::to_wire).collect(), }) }; if let Err(e) = transport.send_rpc(peer_id, update).await { @@ -364,34 +360,22 @@ pub fn handle_topology_update( let mut topo = topology.write().unwrap_or_else(|p| p.into_inner()); let updated = if update.version > topo.version() { - // Adopt the newer topology. + // This node is the authority on its own SWIM address. A pushed + // topology that carries a stale one for this node is corrected below. + let own_swim = topo.get_node(node_id).and_then(|n| n.swim_addr.clone()); let mut new_topo = ClusterTopology::new(); for node in &update.nodes { - let state = crate::topology::NodeState::from_u8(node.state) - .unwrap_or(crate::topology::NodeState::Active); - let spki_pin: Option<[u8; 32]> = node.spki_pin.as_deref().and_then(|b| { - if b.len() == 32 { - let mut arr = [0u8; 32]; - arr.copy_from_slice(b); - Some(arr) - } else { - None - } - }); - let mut info = crate::topology::NodeInfo::new( - node.node_id, - node.addr.parse().unwrap_or_else(|_| { - "0.0.0.0:0" - .parse() - .expect("invariant: \"0.0.0.0:0\" is a valid SocketAddr literal") - }), - state, - ) - .with_wire_version(node.wire_version) - .with_spiffe_id(node.spiffe_id.clone()) - .with_spki_pin(spki_pin); - info.raft_groups = node.raft_groups.clone(); - new_topo.add_node(info); + new_topo.add_node(NodeInfo::from_wire(node)); + } + new_topo.adopt_version(update.version); + if own_swim.is_some() + && let Some(mut own) = new_topo.get_node(node_id).cloned() + && own.swim_addr != own_swim + { + own.swim_addr = own_swim; + // Bumps the version past the sender's, so the correction spreads + // through the next health round. + new_topo.add_node(own); } *topo = new_topo; true @@ -410,7 +394,7 @@ pub fn handle_topology_update( #[cfg(test)] mod tests { use super::*; - use crate::topology::NodeInfo; + use crate::rpc_codec::JoinNodeInfo; #[test] fn handle_ping_returns_pong() { @@ -443,6 +427,7 @@ mod tests { wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, spiffe_id: None, spki_pin: None, + swim_addr: None, }, JoinNodeInfo { node_id: 2, @@ -452,6 +437,7 @@ mod tests { wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, spiffe_id: None, spki_pin: None, + swim_addr: None, }, ], }; @@ -461,6 +447,11 @@ mod tests { let t = topo.read().unwrap(); assert_eq!(t.node_count(), 2); + assert_eq!( + t.version(), + 3, + "the adopted topology keeps the sender's version" + ); match ack { RaftRpc::TopologyAck(a) => assert_eq!(a.accepted_version, t.version()), @@ -583,4 +574,57 @@ mod tests { let t = topo.read().unwrap(); assert_eq!(t.get_node(2).unwrap().state, NodeState::Active); } + + #[test] + fn an_open_circuit_is_not_a_ping_failure() { + let refused = Err(ClusterError::CircuitOpen { + node_id: 2, + failures: 5, + }); + assert!(matches!(classify_ping(refused), PingOutcome::NotSent)); + } + + #[test] + fn a_link_failure_or_a_wrong_reply_is_a_ping_failure() { + let lost = Err(ClusterError::Transport { + detail: "connection lost".into(), + }); + assert!(matches!(classify_ping(lost), PingOutcome::Failed)); + let wrong = Ok(RaftRpc::TopologyAck(TopologyAck { + responder_id: 2, + accepted_version: 1, + })); + assert!(matches!(classify_ping(wrong), PingOutcome::Failed)); + let pong = Ok(RaftRpc::Pong(PongResponse { + responder_id: 2, + topology_version: 1, + })); + assert!(matches!(classify_ping(pong), PingOutcome::Pong(_))); + } + + /// A pushed topology carrying a stale SWIM address for this node keeps + /// this node's own address and ends up newer than the push. + #[test] + fn topology_update_keeps_this_nodes_swim_address() { + let topo = RwLock::new(ClusterTopology::new()); + topo.write().unwrap().add_node( + NodeInfo::new(1, "10.0.0.1:9400".parse().unwrap(), NodeState::Active) + .with_swim_addr("10.0.0.1:9501".parse().ok()), + ); + let stale = NodeInfo::new(1, "10.0.0.1:9400".parse().unwrap(), NodeState::Active) + .with_swim_addr("10.0.0.1:9401".parse().ok()); + let update = TopologyUpdate { + version: 5, + nodes: vec![stale.to_wire()], + }; + + let (updated, _) = handle_topology_update(1, &topo, &update); + assert!(updated); + let t = topo.read().unwrap(); + assert_eq!( + t.get_node(1).and_then(NodeInfo::swim_socket_addr), + "10.0.0.1:9501".parse().ok() + ); + assert!(t.version() > 5); + } } diff --git a/nodedb-cluster/src/install_snapshot/finalize.rs b/nodedb-cluster/src/install_snapshot/finalize.rs index e516bbb64..dc477c747 100644 --- a/nodedb-cluster/src/install_snapshot/finalize.rs +++ b/nodedb-cluster/src/install_snapshot/finalize.rs @@ -1,80 +1,88 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Final snapshot commit: CRC validation → atomic rename → Raft log boundary advance. +//! Final snapshot commit: CRC validation → stage → host apply → Raft log +//! boundary advance → finish. //! //! Called only when the last chunk (`done == true`) has been written to the -//! `.partial` file. Performs three operations in sequence: +//! `.partial` file. Runs these steps in order: //! //! 1. **CRC validation** — re-reads the assembled file and recomputes the -//! CRC32C. If it disagrees with the running CRC accumulated during chunk -//! writes, the partial file is left in place and `SnapshotCrcMismatch` is -//! returned. The partial file is intentionally *not* deleted on CRC failure -//! so the operator can inspect it. -//! -//! 2. **Atomic rename** — the `.partial` file is renamed to `.snap`. -//! The rename is atomic on POSIX filesystems (same directory, same inode -//! table). If the process crashes between steps 1 and 2, the partial file -//! survives; the GC sweeper will remove it after `orphan_partial_max_age_secs`. -//! -//! 3. **Raft log boundary advance** — calls -//! `MultiRaft::handle_install_snapshot` to advance the Raft log pointer to -//! `last_included_index` / `last_included_term`. This is the same call the -//! existing stub in `handle_rpc.rs` made; we now call it only here, after -//! CRC validation, to prevent advancing Raft state on corrupt data. +//! CRC32C. On a mismatch the partial file is left in place for inspection +//! and `SnapshotCrcMismatch` is returned. +//! 2. **Need check** — a snapshot from a stale term, or one at or below the +//! group's applied index, is not applied: installing it would put the state +//! machine behind the log it keeps applying from. The partial is removed +//! and only the term bookkeeping runs. +//! 3. **Stage** — the partial is renamed to its staged name +//! ([`super::staged`]). From here until step 6 the staged file marks the +//! install as in progress, and boot recovery completes it after a crash. +//! 4. **Host apply** — the [`SnapshotApplier`] installs the snapshot on every +//! Data-Plane core and returns only once the install is durable without +//! the WAL. An error returns [`ClusterError::SnapshotApplyFailed`] with the +//! Raft boundary and durable floor unmoved and the staged file kept. The +//! leader's next send re-installs over it. +//! 5. **Raft advance** — the group adopts the snapshot as its log boundary +//! and durable applied floor, and persists any term bump. +//! 6. **Finish** — the staged file is removed. No copy is kept: the host +//! state is durable, and the leader builds every snapshot it sends from +//! live engine state. -use std::path::PathBuf; use std::sync::{Arc, Mutex}; use nodedb_raft::{InstallSnapshotRequest, InstallSnapshotResponse}; use crate::error::ClusterError; +use crate::install_snapshot::staged::{discard, finish, stage}; use crate::install_snapshot::state::PartialSnapshotState; use crate::multi_raft::MultiRaft; use crate::raft_loop::SnapshotApplier; -/// Validate, rename, and advance Raft state after the last chunk. -/// -/// Returns the `InstallSnapshotResponse` produced by -/// `MultiRaft::handle_install_snapshot` so callers can propagate the -/// Raft term back to the leader. +/// The outcome of [`commit`]. +#[derive(Debug)] +pub struct CommitResult { + /// The reply from `MultiRaft::handle_install_snapshot`, carrying the Raft + /// term back to the leader. + pub response: InstallSnapshotResponse, + /// Whether this node's state machine holds the state through the + /// snapshot index: the host applied the snapshot, or the node had already + /// applied past it. False when the boundary moved on an empty stub with no + /// host state to restore. + pub state_installed: bool, +} + +/// Validate, stage, apply, advance Raft, and finish after the last chunk. pub async fn commit( state: PartialSnapshotState, multi_raft: &Arc>, snapshot_applier: Option<&Arc>, -) -> Result { +) -> Result { let group_id = state.group_id; let partial_path = state.partial_path.clone(); let expected_crc = state.running_crc; + let index = state.last_included_index; + let snapshot_term = state.last_included_term; // Flush and close the partial file before reading it back. // `state.partial_file` may be `None` if the snapshot had zero bytes // (bootstrap stub). In that case skip the I/O validation. if let Some(file) = state.partial_file { - tokio::task::spawn_blocking(move || -> std::io::Result<()> { file.sync_all() }) - .await - .map_err(|e| ClusterError::PartialSnapshotCorrupt { - group_id, - detail: format!("spawn_blocking join error on sync: {e}"), - })? - .map_err(|e| ClusterError::Storage { + blocking(group_id, move || { + file.sync_all().map_err(|e| ClusterError::Storage { detail: format!("sync partial file for group {group_id}: {e}"), - })?; + }) + }) + .await?; } - // CRC validation: re-read the file and compare against running CRC. - // If the file is empty (bootstrap stub), skip. - let file_bytes = tokio::task::spawn_blocking({ + let file_bytes = blocking(group_id, { let path = partial_path.clone(); - move || std::fs::read(&path) + move || { + std::fs::read(&path).map_err(|e| ClusterError::Storage { + detail: format!("read partial file for group {group_id}: {e}"), + }) + } }) - .await - .map_err(|e| ClusterError::PartialSnapshotCorrupt { - group_id, - detail: format!("spawn_blocking join error on read: {e}"), - })? - .map_err(|e| ClusterError::Storage { - detail: format!("read partial file for group {group_id}: {e}"), - })?; + .await?; if !file_bytes.is_empty() { let computed = crc32c::crc32c(&file_bytes); @@ -87,28 +95,86 @@ pub async fn commit( } } - // Atomic rename: .partial → .snap - let snap_path = snap_path_for(&partial_path); - tokio::task::spawn_blocking({ - let from = partial_path.clone(); - let to = snap_path.clone(); - move || std::fs::rename(&from, &to) - }) - .await - .map_err(|e| ClusterError::PartialSnapshotCorrupt { + // Term and leader bookkeeping only. The boundary moves through + // `adopt_snapshot_boundary`, after the host apply. + let bookkeeping = InstallSnapshotRequest { + term: state.term, + leader_id: state.leader_id, + last_included_index: index, + last_included_term: snapshot_term, + offset: 0, + data: Vec::new(), + done: false, group_id, - detail: format!("spawn_blocking join error on rename: {e}"), - })? - .map_err(|e| ClusterError::Storage { - detail: format!("rename partial to snap for group {group_id}: {e}"), - })?; - - // Apply the snapshot to the local Data-Plane state machine BEFORE advancing - // Raft, so the data is visible on this node before the Raft log boundary - // moves. An apply failure is fatal — we return WITHOUT advancing Raft so the - // follower retries the install (no silent partial success). The empty - // bootstrap stub (no engine data) is skipped: there is nothing to apply, and - // group 0 (metadata) is a no-op the applier handles internally. + total_size: 0, + voters: Vec::new(), + learners: Vec::new(), + }; + + // Held from the need check through the boundary advance. No committed + // entry of the group reaches the state machine in between, so none the + // snapshot covers can land on top of the restore. + let apply_gates = multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .apply_gates(); + let mut install_permit = apply_gates.install(group_id).await; + + let needed = multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .snapshot_install_needed(group_id, state.term, index)?; + if !needed { + blocking(group_id, { + let path = partial_path.clone(); + move || discard(&path) + }) + .await?; + let mut mr = multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let response = mr.handle_install_snapshot(&bookkeeping)?; + mr.persist_group_hard_state(group_id)?; + return Ok(CommitResult { + response, + state_installed: true, + }); + } + + // An empty payload carries no metadata state. Adopting it would move + // group 0's boundary past entries this node never applied. + if group_id == crate::metadata_group::METADATA_GROUP_ID + && file_bytes.is_empty() + && snapshot_applier.is_some() + { + blocking(group_id, { + let path = partial_path.clone(); + move || discard(&path) + }) + .await?; + return Err(ClusterError::SnapshotApplyFailed { + group_id, + detail: format!( + "empty metadata snapshot at index {index} carries no state; raft boundary unmoved" + ), + }); + } + + let recv_dir = partial_path + .parent() + .map(std::path::Path::to_path_buf) + .ok_or_else(|| ClusterError::PartialSnapshotCorrupt { + group_id, + detail: format!("partial path {} has no directory", partial_path.display()), + })?; + let staged = blocking(group_id, { + let partial = partial_path.clone(); + let recv_dir = recv_dir.clone(); + move || stage(&partial, &recv_dir, group_id, index, snapshot_term) + }) + .await?; + + // The empty bootstrap stub carries no engine data, so there is nothing + // to apply. + let state_installed = !file_bytes.is_empty() && snapshot_applier.is_some(); if !file_bytes.is_empty() && let Some(applier) = snapshot_applier { @@ -117,54 +183,80 @@ pub async fn commit( .await .map_err(|e| ClusterError::SnapshotApplyFailed { group_id, - detail: e.to_string(), + detail: format!( + "install of snapshot index {index} did not settle on every core, \ + raft boundary unmoved, staged install kept for re-install: {e}" + ), })?; } - // Advance Raft log boundary. Build a minimal InstallSnapshotRequest - // that satisfies `handle_install_snapshot` — `data` is the assembled - // bytes (may be empty for the bootstrap stub), `done` is always `true`. - let req = InstallSnapshotRequest { - term: state.term, - leader_id: state.leader_id, - last_included_index: state.last_included_index, - last_included_term: state.last_included_term, - offset: 0, - data: file_bytes, - done: true, - group_id, - total_size: 0, + let response = { + let mut mr = multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let resp = mr.handle_install_snapshot(&bookkeeping)?; + // A metadata install wrote the routing table its skipped conf + // changes produced. The group's own membership follows it. + if group_id == crate::metadata_group::METADATA_GROUP_ID && state_installed { + mr.sync_group_membership_from_routing(group_id)?; + } + mr.adopt_snapshot_boundary(group_id, index, snapshot_term)?; + install_permit.adopted(index); + // The host recorded the Calvin state the snapshot brought, so a + // replica that refused entries for want of it takes them again. A + // sequencer snapshot decides every data group again: the inputs it + // skipped are gone from this node's Calvin state. + if state_installed { + mr.refresh_snapshot_requirement(group_id); + } + // Persist any term bump (become_follower) durably before replying. + mr.persist_group_hard_state(group_id)?; + resp }; + if let Some(applier) = snapshot_applier { + applier.snapshot_adopted(group_id, index); + } + drop(install_permit); - let mut mr = multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - let resp = mr.handle_install_snapshot(&req)?; - // Persist any term bump (become_follower) durably before replying. - mr.persist_group_hard_state(group_id)?; - Ok(resp) + blocking(group_id, move || finish(&staged, &recv_dir)).await?; + Ok(CommitResult { + response, + state_installed, + }) } -/// Derive the `.snap` path from the `.partial` path (same directory, stem only). -fn snap_path_for(partial: &std::path::Path) -> PathBuf { - let parent = partial - .parent() - .unwrap_or_else(|| std::path::Path::new(".")); - let stem = partial - .file_stem() - .unwrap_or_else(|| std::ffi::OsStr::new("unknown")); - parent.join(format!("{}.snap", stem.to_string_lossy())) +/// Run blocking filesystem work off the async runtime. +async fn blocking(group_id: u64, work: F) -> Result +where + T: Send + 'static, + F: FnOnce() -> Result + Send + 'static, +{ + tokio::task::spawn_blocking(work) + .await + .map_err(|e| ClusterError::PartialSnapshotCorrupt { + group_id, + detail: format!("spawn_blocking join error: {e}"), + })? } #[cfg(test)] mod tests { use super::*; + use crate::install_snapshot::staged::list_staged; use crate::routing::RoutingTable; + /// Every file name left in `dir`. + fn files_in(dir: &std::path::Path) -> Vec { + std::fs::read_dir(dir) + .unwrap() + .map(|e| e.unwrap().file_name().to_string_lossy().into_owned()) + .collect() + } + /// Recording applier: proves the state machine saw the snapshot bytes /// (i.e. the DATA is applied), and can be told to fail. #[derive(Default)] struct RecordingApplier { applied: std::sync::Mutex)>>, - fail: bool, + fail: std::sync::atomic::AtomicBool, } #[async_trait::async_trait] @@ -174,7 +266,7 @@ mod tests { group_id: u64, snapshot_bytes: &[u8], ) -> std::result::Result<(), Box> { - if self.fail { + if self.fail.load(std::sync::atomic::Ordering::SeqCst) { return Err("injected applier failure".into()); } self.applied @@ -213,9 +305,8 @@ mod tests { } /// The data must be applied to the state machine BEFORE the Raft log - /// boundary advances. A snapshot whose data is applied must - /// advance the group's snapshot boundary; one whose apply FAILS must - /// leave Raft untouched (follower retries; no silent divergence). + /// boundary advances. A snapshot whose data is applied must advance the + /// group's snapshot boundary and leave no file behind. #[tokio::test] async fn snapshot_data_applied_before_raft_advances() { let dir = tempfile::tempdir().unwrap(); @@ -224,7 +315,7 @@ mod tests { let (mr, _keep) = multi_raft(); let state = partial_state(dir.path(), b"snapshot-payload", 42); - let resp = commit(state, &mr, Some(&applier)).await.unwrap(); + let resp = commit(state, &mr, Some(&applier)).await.unwrap().response; // Data reached the state machine exactly once, with the payload. let applied = inner.applied.lock().unwrap(); @@ -237,38 +328,83 @@ mod tests { let node = mr.groups_mut().get(&7).unwrap(); assert_eq!(node.log_snapshot_index(), 42); assert_eq!(node.commit_index(), 42); + assert_eq!(node.durable_applied_index(), 42); assert_eq!(resp.term, 1); + + assert!( + files_in(dir.path()).is_empty(), + "a finished install keeps no file" + ); } + /// A failed apply leaves Raft untouched and keeps the staged install. + /// The leader's next send re-installs over it and completes. #[tokio::test] - async fn snapshot_apply_failure_does_not_advance_raft() { + async fn snapshot_apply_failure_keeps_staged_install_for_retry() { let dir = tempfile::tempdir().unwrap(); - let applier: Arc = Arc::new(RecordingApplier { - fail: true, - ..Default::default() - }); + let inner = Arc::new(RecordingApplier::default()); + inner.fail.store(true, std::sync::atomic::Ordering::SeqCst); + let applier: Arc = inner.clone(); let (mr, _keep) = multi_raft(); - let state = partial_state(dir.path(), b"corrupt-in-applier", 42); - let res = commit(state, &mr, Some(&applier)).await; + let state = partial_state(dir.path(), b"first-attempt", 42); + let err = commit(state, &mr, Some(&applier)) + .await + .expect_err("apply failure must surface as an error"); assert!( - res.is_err(), - "apply failure must surface as an error, not silent partial success" + matches!(err, ClusterError::SnapshotApplyFailed { group_id: 7, .. }), + "unexpected error: {err}" ); - // Raft must not advance: the follower retries the install. Advancing - // here without the data is the divergence this order prevents. - let mut mr = mr.lock().unwrap(); - let node = mr.groups_mut().get(&7).unwrap(); + { + let mut guard = mr.lock().unwrap(); + let node = guard.groups_mut().get(&7).unwrap(); + assert_eq!(node.log_snapshot_index(), 0, "boundary must not move"); + assert_eq!(node.commit_index(), 0, "commit index must not move"); + assert_eq!(node.durable_applied_index(), 0, "floor must not move"); + } + let staged = list_staged(dir.path()).unwrap(); assert_eq!( - node.log_snapshot_index(), - 0, - "raft must not advance on apply failure" + staged.len(), + 1, + "the staged install marks the failed install" ); + assert_eq!(staged[0].last_included_index, 42); + + inner.fail.store(false, std::sync::atomic::Ordering::SeqCst); + let state = partial_state(dir.path(), b"second-attempt", 42); + commit(state, &mr, Some(&applier)).await.unwrap(); + + let mut guard = mr.lock().unwrap(); + let node = guard.groups_mut().get(&7).unwrap(); + assert_eq!(node.log_snapshot_index(), 42); assert_eq!( - node.commit_index(), - 0, - "commit index must not advance on apply failure" + inner.applied.lock().unwrap().as_slice(), + &[(7, b"second-attempt".to_vec())] ); + assert!(files_in(dir.path()).is_empty()); + } + + /// A snapshot at or below the applied index is never applied: it would + /// move the state machine behind the log. + #[tokio::test] + async fn snapshot_behind_applied_index_is_not_applied() { + let dir = tempfile::tempdir().unwrap(); + let inner = Arc::new(RecordingApplier::default()); + let applier: Arc = inner.clone(); + let (mr, _keep) = multi_raft(); + mr.lock() + .unwrap() + .adopt_snapshot_boundary(7, 50, 1) + .unwrap(); + + let state = partial_state(dir.path(), b"old-snapshot", 42); + commit(state, &mr, Some(&applier)).await.unwrap(); + + assert!(inner.applied.lock().unwrap().is_empty()); + assert!(!dir.path().join("7.partial").exists()); + assert!(list_staged(dir.path()).unwrap().is_empty()); + let mut guard = mr.lock().unwrap(); + assert_eq!(guard.groups_mut().get(&7).unwrap().log_snapshot_index(), 50); } } diff --git a/nodedb-cluster/src/install_snapshot/mod.rs b/nodedb-cluster/src/install_snapshot/mod.rs index aadfe0f0e..3b55f10fd 100644 --- a/nodedb-cluster/src/install_snapshot/mod.rs +++ b/nodedb-cluster/src/install_snapshot/mod.rs @@ -9,18 +9,24 @@ //! - [`receiver`] — follower `PartialSnapshotState` accumulator; writes chunk //! bytes to `/recv_snapshots/.partial` and validates //! the running CRC. -//! - [`finalize`] — atomic rename + CRC-full validation + Raft log boundary -//! advance. +//! - [`finalize`] — CRC-full validation, staging, host apply, Raft log +//! boundary advance, removal of the staged file. +//! - [`staged`] — the staged-install file that marks an install in progress. +//! - [`recover`] — boot completion of staged installs. //! - [`gc`] — orphan `.partial` file sweeper; removes stale partials that //! predate `orphan_partial_max_age_secs`. pub mod finalize; pub mod gc; pub mod receiver; +pub mod recover; pub mod sender; +pub mod staged; pub mod state; pub use gc::sweep_orphans; pub use receiver::{ChunkOutcome, PartialSnapshotMap, handle_chunk}; +pub use recover::recover_staged_installs; pub use sender::{SendChunkedParams, send_chunked}; +pub use staged::StagedInstall; pub use state::PartialSnapshotState; diff --git a/nodedb-cluster/src/install_snapshot/receiver.rs b/nodedb-cluster/src/install_snapshot/receiver.rs index f011b1c93..b0c173704 100644 --- a/nodedb-cluster/src/install_snapshot/receiver.rs +++ b/nodedb-cluster/src/install_snapshot/receiver.rs @@ -49,8 +49,7 @@ pub enum ChunkOutcome { /// More chunks are expected. Pending, /// The final chunk was received, CRC validated, and the snapshot committed. - /// Contains the `InstallSnapshotResponse` from `MultiRaft::handle_install_snapshot`. - Committed(nodedb_raft::InstallSnapshotResponse), + Committed(super::finalize::CommitResult), } /// Process a single incoming `InstallSnapshotRequest` chunk. diff --git a/nodedb-cluster/src/install_snapshot/recover.rs b/nodedb-cluster/src/install_snapshot/recover.rs new file mode 100644 index 000000000..742af5dbc --- /dev/null +++ b/nodedb-cluster/src/install_snapshot/recover.rs @@ -0,0 +1,225 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Boot recovery of staged snapshot installs. +//! +//! A staged file under `/recv_snapshots/` is an install whose host +//! apply may have started but that did not finish. Recovery completes each +//! one, in one of two ways: +//! +//! - Raft boundary at or past the staged index: the host apply acknowledged +//! and was durable before the boundary moved, so only the file remains. +//! It is removed. +//! - Raft boundary below the staged index: the host state of the group is +//! undetermined, so the snapshot is applied again through the +//! [`SnapshotApplier`], the boundary adopted, and the file removed. +//! +//! The re-apply runs after the Data Plane's WAL replay: a core takes a +//! request only once it has replayed. That order is correct here. The install +//! appended its WAL barrier before any core changed, so replay applied no +//! pre-install record to the group's collections. And the boundary never +//! moved, so no entry after the snapshot applied either: the WAL holds no +//! record for these collections that the re-apply could erase. +//! +//! Recovery also removes every `.snap` file. Nothing reads one: the +//! installed state is durable on its own, and the leader builds every +//! snapshot it sends from live engine state. + +use std::path::Path; + +use tracing::{info, warn}; + +use crate::error::ClusterError; +use crate::multi_raft::MultiRaft; +use crate::raft_loop::SnapshotApplier; + +use super::staged::{discard, finish, list_staged, remove_snap_files}; + +/// Complete every staged install under `/recv_snapshots/`. +/// +/// Returns the number of staged installs completed. A staged install of a +/// group this node no longer mounts is removed: the node holds no Raft state +/// for it. `applier` is `None` only in cluster-only tests with no host state +/// machine. +pub async fn recover_staged_installs( + data_dir: &Path, + multi_raft: &mut MultiRaft, + applier: Option<&dyn SnapshotApplier>, +) -> Result { + let recv_dir = data_dir.join("recv_snapshots"); + let removed = remove_snap_files(&recv_dir)?; + if removed > 0 { + info!(removed, "removed unused snapshot files"); + } + + let mut completed = 0usize; + for staged in list_staged(&recv_dir)? { + let group_id = staged.group_id; + let index = staged.last_included_index; + if !multi_raft.contains_group(group_id) { + warn!( + group_id, + snapshot_index = index, + "staged snapshot install of an unmounted group, removing" + ); + discard(&staged.path)?; + continue; + } + let (_, boundary, _) = multi_raft.snapshot_metadata(group_id)?; + if boundary < index { + let bytes = std::fs::read(&staged.path).map_err(|e| ClusterError::Storage { + detail: format!("read staged install {}: {e}", staged.path.display()), + })?; + if !bytes.is_empty() + && let Some(applier) = applier + { + applier + .apply_snapshot(group_id, &bytes) + .await + .map_err(|e| ClusterError::SnapshotApplyFailed { + group_id, + detail: format!( + "boot re-apply of the staged install at index {index}: {e}" + ), + })?; + if group_id == crate::metadata_group::METADATA_GROUP_ID { + multi_raft.sync_group_membership_from_routing(group_id)?; + } + } + multi_raft.adopt_snapshot_boundary(group_id, index, staged.last_included_term)?; + if !bytes.is_empty() && applier.is_some() { + multi_raft.refresh_snapshot_requirement(group_id); + } + } + finish(&staged, &recv_dir)?; + info!( + group_id, + snapshot_index = index, + "completed staged snapshot install" + ); + completed += 1; + } + Ok(completed) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::install_snapshot::staged::staged_path; + use crate::routing::RoutingTable; + + #[derive(Default)] + struct RecordingApplier { + applied: std::sync::Mutex)>>, + } + + #[async_trait::async_trait] + impl SnapshotApplier for RecordingApplier { + async fn apply_snapshot( + &self, + group_id: u64, + snapshot_bytes: &[u8], + ) -> std::result::Result<(), Box> { + self.applied + .lock() + .unwrap() + .push((group_id, snapshot_bytes.to_vec())); + Ok(()) + } + } + + fn multi_raft(dir: &Path) -> MultiRaft { + let rt = RoutingTable::uniform(1, &[1], 1); + let mut mr = MultiRaft::new(1, rt, dir.to_path_buf()); + mr.add_group(7, vec![]).unwrap(); + mr + } + + fn recv_dir(dir: &Path) -> std::path::PathBuf { + let recv = dir.join("recv_snapshots"); + std::fs::create_dir_all(&recv).unwrap(); + recv + } + + /// An install interrupted before the boundary moved is applied again, + /// then the boundary moves and the staged file goes. + #[tokio::test] + async fn interrupted_install_is_reapplied_then_adopted() { + let dir = tempfile::tempdir().unwrap(); + let recv = recv_dir(dir.path()); + std::fs::write(staged_path(&recv, 7, 42, 2), b"payload").unwrap(); + let mut mr = multi_raft(dir.path()); + let applier = RecordingApplier::default(); + + let completed = + recover_staged_installs(dir.path(), &mut mr, Some(&applier as &dyn SnapshotApplier)) + .await + .unwrap(); + assert_eq!(completed, 1); + + assert_eq!( + *applier.applied.lock().unwrap(), + vec![(7, b"payload".to_vec())] + ); + let node = mr.groups_mut().get(&7).unwrap(); + assert_eq!(node.log_snapshot_index(), 42); + assert_eq!(node.durable_applied_index(), 42); + assert!(list_staged(&recv).unwrap().is_empty()); + } + + /// An install whose boundary moved before the crash was durable already: + /// it is not applied again, only its file removed. + #[tokio::test] + async fn install_past_the_boundary_is_not_reapplied() { + let dir = tempfile::tempdir().unwrap(); + let recv = recv_dir(dir.path()); + let mut mr = multi_raft(dir.path()); + mr.adopt_snapshot_boundary(7, 50, 2).unwrap(); + std::fs::write(staged_path(&recv, 7, 42, 2), b"payload").unwrap(); + let applier = RecordingApplier::default(); + + let completed = + recover_staged_installs(dir.path(), &mut mr, Some(&applier as &dyn SnapshotApplier)) + .await + .unwrap(); + assert_eq!(completed, 1); + + assert!(applier.applied.lock().unwrap().is_empty()); + let node = mr.groups_mut().get(&7).unwrap(); + assert_eq!(node.log_snapshot_index(), 50); + assert!(list_staged(&recv).unwrap().is_empty()); + } + + #[tokio::test] + async fn staged_install_of_unmounted_group_is_removed() { + let dir = tempfile::tempdir().unwrap(); + let recv = recv_dir(dir.path()); + std::fs::write(staged_path(&recv, 9, 42, 2), b"payload").unwrap(); + let mut mr = multi_raft(dir.path()); + let applier = RecordingApplier::default(); + + let completed = + recover_staged_installs(dir.path(), &mut mr, Some(&applier as &dyn SnapshotApplier)) + .await + .unwrap(); + assert_eq!(completed, 0); + assert!(applier.applied.lock().unwrap().is_empty()); + assert!(!staged_path(&recv, 9, 42, 2).exists()); + } + + /// A finished install's `.snap` file is never applied at boot: it is + /// removed. + #[tokio::test] + async fn snap_files_are_removed_not_applied() { + let dir = tempfile::tempdir().unwrap(); + let recv = recv_dir(dir.path()); + std::fs::write(recv.join("7.snap"), b"old-install").unwrap(); + let mut mr = multi_raft(dir.path()); + let applier = RecordingApplier::default(); + + recover_staged_installs(dir.path(), &mut mr, Some(&applier as &dyn SnapshotApplier)) + .await + .unwrap(); + assert!(applier.applied.lock().unwrap().is_empty()); + assert!(!recv.join("7.snap").exists()); + } +} diff --git a/nodedb-cluster/src/install_snapshot/sender.rs b/nodedb-cluster/src/install_snapshot/sender.rs index 0571b2798..c68b00a9e 100644 --- a/nodedb-cluster/src/install_snapshot/sender.rs +++ b/nodedb-cluster/src/install_snapshot/sender.rs @@ -30,6 +30,10 @@ pub struct SendChunkedParams<'a> { pub last_included_term: u64, pub snapshot_bytes: &'a [u8], pub chunk_bytes: u64, + /// The group's voters, the leader included, sent on the final chunk. + pub voters: &'a [u64], + /// The group's learners, sent on the final chunk. + pub learners: &'a [u64], } /// Leader-side chunked send for a single peer. @@ -54,6 +58,8 @@ pub async fn send_chunked( last_included_term, snapshot_bytes, chunk_bytes, + voters, + learners, } = params; // For an empty snapshot we send exactly one stub chunk with done=true. if snapshot_bytes.is_empty() { @@ -67,6 +73,8 @@ pub async fn send_chunked( done: true, group_id, total_size: 0, + voters: voters.to_vec(), + learners: learners.to_vec(), }; let resp = transport @@ -108,6 +116,8 @@ pub async fn send_chunked( done, group_id, total_size: total, + voters: if done { voters.to_vec() } else { Vec::new() }, + learners: if done { learners.to_vec() } else { Vec::new() }, }; let resp = transport diff --git a/nodedb-cluster/src/install_snapshot/staged.rs b/nodedb-cluster/src/install_snapshot/staged.rs new file mode 100644 index 000000000..2820a0ffc --- /dev/null +++ b/nodedb-cluster/src/install_snapshot/staged.rs @@ -0,0 +1,235 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Staged snapshot installs: the on-disk marker of an install in progress. +//! +//! A received snapshot whose CRC passed is renamed from `.partial` to +//! `...staged` before the state machine sees it. The +//! staged file exists from the moment the host apply can start until the Raft +//! boundary has moved, and is then removed: the host install is durable on +//! its own, and the leader builds every snapshot it sends from live engine +//! state, so no copy of a finished install is kept. A staged file found at +//! boot is an install that did not finish, and [`super::recover`] completes +//! it. +//! +//! These functions do blocking filesystem I/O. Async callers run them on +//! `spawn_blocking`. + +use std::path::{Path, PathBuf}; + +use crate::error::ClusterError; + +const STAGED_EXTENSION: &str = "staged"; + +/// Extension of per-group snapshot files. Nothing reads one. Boot recovery +/// removes any it finds. +const SNAP_EXTENSION: &str = "snap"; + +/// One staged install found on disk. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct StagedInstall { + pub group_id: u64, + pub last_included_index: u64, + pub last_included_term: u64, + pub path: PathBuf, +} + +/// Path of the staged file for `(group_id, index, term)` under `recv_dir`. +pub fn staged_path(recv_dir: &Path, group_id: u64, index: u64, term: u64) -> PathBuf { + recv_dir.join(format!("{group_id}.{index}.{term}.{STAGED_EXTENSION}")) +} + +/// Parse `...staged` into `(group, index, term)`. +fn parse_staged_name(name: &str) -> Option<(u64, u64, u64)> { + let stem = name.strip_suffix(STAGED_EXTENSION)?.strip_suffix('.')?; + let mut parts = stem.split('.'); + let group_id = parts.next()?.parse().ok()?; + let index = parts.next()?.parse().ok()?; + let term = parts.next()?.parse().ok()?; + if parts.next().is_some() { + return None; + } + Some((group_id, index, term)) +} + +fn storage_error(action: &str, path: &Path, e: std::io::Error) -> ClusterError { + ClusterError::Storage { + detail: format!("{action} {}: {e}", path.display()), + } +} + +/// Fsync `dir` so a rename or unlink inside it survives a crash. +fn sync_dir(dir: &Path) -> Result<(), ClusterError> { + std::fs::File::open(dir) + .and_then(|d| d.sync_all()) + .map_err(|e| storage_error("fsync directory", dir, e)) +} + +/// Every staged install under `recv_dir`, one per group: the highest index +/// wins, and older staged files of the same group are removed. +pub fn list_staged(recv_dir: &Path) -> Result, ClusterError> { + if !recv_dir.exists() { + return Ok(Vec::new()); + } + let entries = + std::fs::read_dir(recv_dir).map_err(|e| storage_error("read_dir", recv_dir, e))?; + let mut found: Vec = Vec::new(); + for entry in entries { + let path = entry + .map_err(|e| storage_error("iterate", recv_dir, e))? + .path(); + let Some((group_id, last_included_index, last_included_term)) = path + .file_name() + .and_then(|n| n.to_str()) + .and_then(parse_staged_name) + else { + continue; + }; + found.push(StagedInstall { + group_id, + last_included_index, + last_included_term, + path, + }); + } + found.sort_by_key(|s| (s.group_id, std::cmp::Reverse(s.last_included_index))); + + let mut newest: Vec = Vec::with_capacity(found.len()); + for staged in found { + if newest.last().is_some_and(|n| n.group_id == staged.group_id) { + discard(&staged.path)?; + } else { + newest.push(staged); + } + } + Ok(newest) +} + +/// Rename a CRC-checked `partial` file to its staged name and make the +/// rename durable. Removes any older staged file of the same group first, so +/// one group never has two staged installs. +pub fn stage( + partial: &Path, + recv_dir: &Path, + group_id: u64, + index: u64, + term: u64, +) -> Result { + for older in list_staged(recv_dir)? + .into_iter() + .filter(|s| s.group_id == group_id) + { + discard(&older.path)?; + } + let path = staged_path(recv_dir, group_id, index, term); + std::fs::rename(partial, &path).map_err(|e| storage_error("stage", partial, e))?; + sync_dir(recv_dir)?; + Ok(StagedInstall { + group_id, + last_included_index: index, + last_included_term: term, + path, + }) +} + +/// Remove a finished install's staged file and make the removal durable. +pub fn finish(staged: &StagedInstall, recv_dir: &Path) -> Result<(), ClusterError> { + discard(&staged.path)?; + sync_dir(recv_dir) +} + +/// Remove every `.snap` file under `recv_dir`. Returns how many. +pub fn remove_snap_files(recv_dir: &Path) -> Result { + if !recv_dir.exists() { + return Ok(0); + } + let entries = + std::fs::read_dir(recv_dir).map_err(|e| storage_error("read_dir", recv_dir, e))?; + let mut removed = 0usize; + for entry in entries { + let path = entry + .map_err(|e| storage_error("iterate", recv_dir, e))? + .path(); + if path.extension().and_then(|e| e.to_str()) == Some(SNAP_EXTENSION) { + discard(&path)?; + removed += 1; + } + } + if removed > 0 { + sync_dir(recv_dir)?; + } + Ok(removed) +} + +/// Remove a staged or partial file. A missing file is not an error. +pub fn discard(path: &Path) -> Result<(), ClusterError> { + match std::fs::remove_file(path) { + Ok(()) => Ok(()), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(()), + Err(e) => Err(storage_error("remove", path, e)), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn staged_name_round_trips() { + let dir = Path::new("/r"); + let path = staged_path(dir, 7, 42, 3); + let name = path.file_name().unwrap().to_str().unwrap(); + assert_eq!(parse_staged_name(name), Some((7, 42, 3))); + assert_eq!(parse_staged_name("7.snap"), None); + assert_eq!(parse_staged_name("7.partial"), None); + assert_eq!(parse_staged_name("7.42.staged"), None); + assert_eq!(parse_staged_name("7.42.3.1.staged"), None); + } + + #[test] + fn stage_replaces_an_older_staged_install_of_the_group() { + let dir = tempfile::tempdir().unwrap(); + let recv = dir.path(); + let first = recv.join("7.partial"); + std::fs::write(&first, b"a").unwrap(); + stage(&first, recv, 7, 10, 1).unwrap(); + let second = recv.join("7.partial"); + std::fs::write(&second, b"b").unwrap(); + let staged = stage(&second, recv, 7, 20, 1).unwrap(); + + let listed = list_staged(recv).unwrap(); + assert_eq!(listed, vec![staged.clone()]); + assert!(!staged_path(recv, 7, 10, 1).exists()); + + finish(&staged, recv).unwrap(); + assert!(list_staged(recv).unwrap().is_empty()); + } + + #[test] + fn list_keeps_the_newest_staged_install_per_group() { + let dir = tempfile::tempdir().unwrap(); + let recv = dir.path(); + for (group, index) in [(7, 10), (7, 30), (8, 5)] { + std::fs::write(staged_path(recv, group, index, 1), b"x").unwrap(); + } + let listed = list_staged(recv).unwrap(); + let keys: Vec<(u64, u64)> = listed + .iter() + .map(|s| (s.group_id, s.last_included_index)) + .collect(); + assert_eq!(keys, vec![(7, 30), (8, 5)]); + assert!(!staged_path(recv, 7, 10, 1).exists()); + } + + #[test] + fn snap_files_are_removed_and_staged_files_kept() { + let dir = tempfile::tempdir().unwrap(); + let recv = dir.path(); + std::fs::write(recv.join("7.snap"), b"x").unwrap(); + std::fs::write(recv.join("8.snap"), b"y").unwrap(); + std::fs::write(staged_path(recv, 9, 5, 1), b"z").unwrap(); + + assert_eq!(remove_snap_files(recv).unwrap(), 2); + assert!(!recv.join("7.snap").exists()); + assert_eq!(list_staged(recv).unwrap().len(), 1); + } +} diff --git a/nodedb-cluster/src/lease_liveness/dead_holders.rs b/nodedb-cluster/src/lease_liveness/dead_holders.rs new file mode 100644 index 000000000..77b1c6735 --- /dev/null +++ b/nodedb-cluster/src/lease_liveness/dead_holders.rs @@ -0,0 +1,374 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! SWIM Dead records and the lease-liveness rule built on them. + +use std::collections::HashMap; +use std::sync::Mutex; +use std::time::{Duration, Instant}; + +use nodedb_types::{MAX_CLOCK_SKEW_NS, NodeId}; + +use crate::multi_raft::PeerAckSample; +use crate::swim::MemberState; +use crate::swim::subscriber::MembershipSubscriber; + +use super::raft_contact::RaftContactClock; + +/// Longest a holder uses a lease without metadata-leader contact. It is twice +/// the default 5 s maximum election timeout, so a healthy failover never trips it. +pub const LEASE_SELF_FENCE_WINDOW: Duration = Duration::from_secs(10); + +/// Largest clock offset tolerated between the node stamping a lease expiry and +/// the node reading it. +pub const LEASE_CLOCK_SKEW: Duration = Duration::from_nanos(MAX_CLOCK_SKEW_NS); + +/// Wait after SWIM marks a holder Dead before its leases can count as expired: +/// the holder's self-fence window plus the clock-skew margin. +pub const DEAD_HOLDER_LEASE_GRACE: Duration = + LEASE_SELF_FENCE_WINDOW.saturating_add(LEASE_CLOCK_SKEW); + +/// Raft silence the metadata leader must see from a Dead holder before its +/// leases count as expired. Past it, the holder has either lost leader contact +/// for its self-fence window or still hears the leader and applies the release. +pub const DEAD_HOLDER_RAFT_SILENCE: Duration = + LEASE_SELF_FENCE_WINDOW.saturating_add(LEASE_CLOCK_SKEW); + +/// Cap on tracked Dead holders. A holder past the cap keeps its full lease +/// expiry, which is the safe fallback. +const MAX_TRACKED_DEAD_HOLDERS: usize = 4096; + +/// The instant a lease-liveness check is evaluated at. +#[derive(Debug, Clone, Copy)] +pub struct LeaseNow { + /// Local wall time, the frame `expires_at` is stamped in. + pub wall_ns: u64, + /// Monotonic time, the frame the dead grace is measured in. + pub instant: Instant, + /// This node's metadata-group term while it leads the group, else `None`. + /// Only the leader can release a Dead holder's lease early. + pub metadata_leader_term: Option, +} + +/// When each lease holder went SWIM-Dead, plus the metadata leader's view of +/// each holder's Raft contact. +#[derive(Debug, Default)] +pub struct LeaseHolderLiveness { + dead_since: Mutex>, + raft_contact: RaftContactClock, +} + +impl LeaseHolderLiveness { + pub fn new() -> Self { + Self::default() + } + + /// Record `node_id` as Dead since `at`. An earlier record is kept. + pub fn record_dead_at(&self, node_id: u64, at: Instant) { + let mut map = self.dead_since.lock().unwrap_or_else(|p| p.into_inner()); + if let Some(existing) = map.get_mut(&node_id) { + if at < *existing { + *existing = at; + } + return; + } + if map.len() >= MAX_TRACKED_DEAD_HOLDERS { + tracing::warn!( + node_id, + cap = MAX_TRACKED_DEAD_HOLDERS, + "lease holder liveness: dead-holder map full; holder keeps full lease expiry" + ); + return; + } + map.insert(node_id, at); + } + + /// Clear the Dead record of `node_id` after SWIM sees it Alive again. + pub fn record_alive(&self, node_id: u64) { + self.dead_since + .lock() + .unwrap_or_else(|p| p.into_inner()) + .remove(&node_id); + } + + /// When `node_id` was declared Dead, if it still is. + pub fn dead_since(&self, node_id: u64) -> Option { + self.dead_since + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&node_id) + .copied() + } + + /// Whether `node_id` has been Dead for at least [`DEAD_HOLDER_LEASE_GRACE`] at `now`. + pub fn dead_grace_elapsed(&self, node_id: u64, now: Instant) -> bool { + self.dead_since(node_id) + .is_some_and(|since| now.saturating_duration_since(since) >= DEAD_HOLDER_LEASE_GRACE) + } + + /// Whether any holder's dead grace has elapsed at `now`. + pub fn any_dead_grace_elapsed(&self, now: Instant) -> bool { + self.dead_since + .lock() + .unwrap_or_else(|p| p.into_inner()) + .values() + .any(|since| now.saturating_duration_since(*since) >= DEAD_HOLDER_LEASE_GRACE) + } + + /// Drop the Dead records of every node `keep` rejects. + pub fn retain(&self, keep: impl Fn(u64) -> bool) { + self.dead_since + .lock() + .unwrap_or_else(|p| p.into_inner()) + .retain(|id, _| keep(*id)); + } + + /// Fold in the metadata leader's per-peer response counts, sampled at `now`. + pub fn observe_raft_contact(&self, sample: &PeerAckSample, now: Instant) { + self.raft_contact.observe(sample, now); + } + + /// Forget Raft contact samples. Called when this node stops leading the + /// metadata group. + pub fn clear_raft_contact(&self) { + self.raft_contact.clear(); + } + + /// Whether the leases of `holder` count as released: SWIM has held it Dead + /// past the grace, and the metadata leader, in `leader_term`, has seen no + /// Raft response from it for [`DEAD_HOLDER_RAFT_SILENCE`]. + pub fn dead_holder_released( + &self, + holder: u64, + leader_term: Option, + now: Instant, + ) -> bool { + let Some(term) = leader_term else { + return false; + }; + self.dead_grace_elapsed(holder, now) + && self + .raft_contact + .silent_for(holder, term, DEAD_HOLDER_RAFT_SILENCE) + } + + /// Whether a lease held by `holder` still blocks a drain. + /// + /// This node's own lease expires at `expires_at`. Another node's lease + /// stays live until [`MAX_CLOCK_SKEW_NS`] past it, unless + /// [`Self::dead_holder_released`] holds first. + pub fn lease_is_live( + &self, + holder: u64, + local_node: u64, + expires_at_wall_ns: u64, + now: &LeaseNow, + ) -> bool { + if holder == local_node { + return expires_at_wall_ns > now.wall_ns; + } + if self.dead_holder_released(holder, now.metadata_leader_term, now.instant) { + return false; + } + expires_at_wall_ns.saturating_add(MAX_CLOCK_SKEW_NS) > now.wall_ns + } +} + +impl MembershipSubscriber for LeaseHolderLiveness { + fn on_state_change(&self, node_id: &NodeId, _old: Option, new: MemberState) { + // Seed placeholders (`seed:`) carry no numeric id and hold no lease. + let Ok(numeric_id) = node_id.as_str().parse::() else { + return; + }; + match new { + MemberState::Dead | MemberState::Left => { + self.record_dead_at(numeric_id, Instant::now()) + } + MemberState::Alive => self.record_alive(numeric_id), + MemberState::Suspect => {} + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const LOCAL: u64 = 1; + const REMOTE: u64 = 2; + const TERM: u64 = 3; + const SECOND_NS: u64 = 1_000_000_000; + const NOW_WALL: u64 = 100 * SECOND_NS; + + fn node(id: u64) -> NodeId { + NodeId::try_new(id.to_string()).expect("numeric node id") + } + + fn past(ago: Duration) -> Instant { + Instant::now() + .checked_sub(ago) + .expect("monotonic clock far enough from its origin") + } + + fn leader_now(instant: Instant) -> LeaseNow { + LeaseNow { + wall_ns: NOW_WALL, + instant, + metadata_leader_term: Some(TERM), + } + } + + /// Leader samples showing `REMOTE` silent since `from` until `to`. + fn silent_between(liveness: &LeaseHolderLiveness, from: Instant, to: Instant) { + let sample = PeerAckSample { + term: TERM, + acks: vec![(REMOTE, 9)], + }; + liveness.observe_raft_contact(&sample, from); + liveness.observe_raft_contact(&sample, to); + } + + #[test] + fn self_fence_window_exceeds_the_default_election_timeout() { + let max_election = Duration::from_millis( + nodedb_types::config::tuning::ClusterTransportTuning::default() + .effective_election_timeout_max_ms(), + ); + assert!(LEASE_SELF_FENCE_WINDOW > max_election); + } + + #[test] + fn remote_lease_is_live_inside_the_skew_margin() { + let liveness = LeaseHolderLiveness::new(); + let now = leader_now(Instant::now()); + let expires = NOW_WALL - SECOND_NS; + assert!(liveness.lease_is_live(REMOTE, LOCAL, expires, &now)); + assert!(!liveness.lease_is_live(LOCAL, LOCAL, expires, &now)); + let far_past = NOW_WALL - MAX_CLOCK_SKEW_NS - SECOND_NS; + assert!(!liveness.lease_is_live(REMOTE, LOCAL, far_past, &now)); + } + + #[test] + fn dead_and_raft_silent_holder_is_released_only_after_the_grace() { + let liveness = LeaseHolderLiveness::new(); + let expires = NOW_WALL + 60 * SECOND_NS; + let now = Instant::now(); + silent_between( + &liveness, + past(DEAD_HOLDER_RAFT_SILENCE + Duration::from_secs(1)), + now, + ); + + liveness.record_dead_at(REMOTE, now); + assert!(liveness.lease_is_live(REMOTE, LOCAL, expires, &leader_now(now))); + + liveness.record_dead_at( + REMOTE, + past(DEAD_HOLDER_LEASE_GRACE + Duration::from_secs(1)), + ); + assert!(!liveness.lease_is_live(REMOTE, LOCAL, expires, &leader_now(now))); + } + + /// SWIM says Dead, but the leader heard from the holder over Raft + /// recently: the holder has not fenced, so its lease must stay live. + #[test] + fn dead_by_swim_but_recently_acked_by_raft_keeps_the_lease() { + let liveness = LeaseHolderLiveness::new(); + let expires = NOW_WALL + 60 * SECOND_NS; + let now = Instant::now(); + liveness.record_dead_at( + REMOTE, + past(DEAD_HOLDER_LEASE_GRACE + Duration::from_secs(1)), + ); + let earlier = past(DEAD_HOLDER_RAFT_SILENCE + Duration::from_secs(1)); + liveness.observe_raft_contact( + &PeerAckSample { + term: TERM, + acks: vec![(REMOTE, 9)], + }, + earlier, + ); + liveness.observe_raft_contact( + &PeerAckSample { + term: TERM, + acks: vec![(REMOTE, 10)], + }, + now, + ); + + assert!(liveness.lease_is_live(REMOTE, LOCAL, expires, &leader_now(now))); + } + + /// A drainer that does not lead the metadata group never releases early. + #[test] + fn a_non_leader_keeps_a_dead_holders_lease_live() { + let liveness = LeaseHolderLiveness::new(); + let expires = NOW_WALL + 60 * SECOND_NS; + let now = Instant::now(); + silent_between( + &liveness, + past(DEAD_HOLDER_RAFT_SILENCE + Duration::from_secs(1)), + now, + ); + liveness.record_dead_at( + REMOTE, + past(DEAD_HOLDER_LEASE_GRACE + Duration::from_secs(1)), + ); + let follower_now = LeaseNow { + metadata_leader_term: None, + ..leader_now(now) + }; + assert!(liveness.lease_is_live(REMOTE, LOCAL, expires, &follower_now)); + } + + #[test] + fn a_later_dead_report_keeps_the_earlier_record() { + let liveness = LeaseHolderLiveness::new(); + let earlier = past(Duration::from_secs(30)); + liveness.record_dead_at(REMOTE, earlier); + liveness.record_dead_at(REMOTE, Instant::now()); + assert_eq!(liveness.dead_since(REMOTE), Some(earlier)); + } + + #[test] + fn subscriber_records_dead_and_alive_refutation_clears() { + let liveness = LeaseHolderLiveness::new(); + liveness.on_state_change(&node(REMOTE), Some(MemberState::Suspect), MemberState::Dead); + assert!(liveness.dead_since(REMOTE).is_some()); + + liveness.on_state_change(&node(REMOTE), Some(MemberState::Dead), MemberState::Suspect); + assert!(liveness.dead_since(REMOTE).is_some()); + + liveness.on_state_change(&node(REMOTE), Some(MemberState::Dead), MemberState::Alive); + assert!(liveness.dead_since(REMOTE).is_none()); + } + + #[test] + fn subscriber_ignores_seed_placeholders() { + let liveness = LeaseHolderLiveness::new(); + let seed = NodeId::try_new("seed:127.0.0.1:9000").expect("seed placeholder id"); + liveness.on_state_change(&seed, Some(MemberState::Alive), MemberState::Dead); + assert!(liveness.dead_since.lock().expect("unpoisoned").is_empty()); + } + + #[test] + fn map_is_bounded() { + let liveness = LeaseHolderLiveness::new(); + for id in 0..(MAX_TRACKED_DEAD_HOLDERS as u64 + 10) { + liveness.record_dead_at(id, Instant::now()); + } + assert_eq!( + liveness.dead_since.lock().expect("unpoisoned").len(), + MAX_TRACKED_DEAD_HOLDERS + ); + } + + #[test] + fn retain_drops_rejected_holders() { + let liveness = LeaseHolderLiveness::new(); + liveness.record_dead_at(REMOTE, Instant::now()); + liveness.record_dead_at(3, Instant::now()); + liveness.retain(|id| id == REMOTE); + assert!(liveness.dead_since(REMOTE).is_some()); + assert!(liveness.dead_since(3).is_none()); + } +} diff --git a/nodedb-cluster/src/lease_liveness/mod.rs b/nodedb-cluster/src/lease_liveness/mod.rs new file mode 100644 index 000000000..7d3f74535 --- /dev/null +++ b/nodedb-cluster/src/lease_liveness/mod.rs @@ -0,0 +1,29 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Descriptor-lease holder liveness. +//! +//! A holder that SWIM declares Dead stays in topology, so the non-member lease +//! filter never drops its leases. Its leases count as expired early only when +//! both of these hold: +//! +//! - SWIM has held it Dead for [`DEAD_HOLDER_LEASE_GRACE`]. +//! - The metadata leader has seen no Raft response from it for +//! [`DEAD_HOLDER_RAFT_SILENCE`]. +//! +//! Early release relies on the holder's self-fence: a holder refuses its +//! cached lease once its last metadata-leader contact is older than +//! [`LEASE_SELF_FENCE_WINDOW`]. A holder that still hears the leader applies +//! the release before any further use. +//! +//! Every lease expiry is stamped on the holder's own clock. A lease held by +//! another node therefore counts as live until `MAX_CLOCK_SKEW_NS` past its +//! `expires_at`. A lease held by the local node gets no margin. + +pub mod dead_holders; +pub mod raft_contact; + +pub use dead_holders::{ + DEAD_HOLDER_LEASE_GRACE, DEAD_HOLDER_RAFT_SILENCE, LEASE_CLOCK_SKEW, LEASE_SELF_FENCE_WINDOW, + LeaseHolderLiveness, LeaseNow, +}; +pub use raft_contact::RaftContactClock; diff --git a/nodedb-cluster/src/lease_liveness/raft_contact.rs b/nodedb-cluster/src/lease_liveness/raft_contact.rs new file mode 100644 index 000000000..0df80521a --- /dev/null +++ b/nodedb-cluster/src/lease_liveness/raft_contact.rs @@ -0,0 +1,152 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! When each peer last answered this node as metadata-group leader. +//! +//! Raft keeps a per-peer response count, not a time. [`RaftContactClock`] +//! turns successive count samples into a last-change instant per peer. A +//! change is dated to the sample that saw it, which is never earlier than the +//! response itself. Silence is measured only up to the latest sample, so a +//! response that arrived after it cannot be missed. + +use std::collections::HashMap; +use std::sync::Mutex; +use std::time::{Duration, Instant}; + +use crate::multi_raft::PeerAckSample; + +#[derive(Debug, Clone, Copy)] +struct PeerContact { + acks: u64, + changed_at: Instant, +} + +#[derive(Debug)] +struct TermContact { + term: u64, + sampled_at: Instant, + peers: HashMap, +} + +/// Leader-side record of each peer's last observed Raft response. +#[derive(Debug, Default)] +pub struct RaftContactClock { + inner: Mutex>, +} + +impl RaftContactClock { + /// Fold in a sample taken at `now`. A new term restarts every peer's + /// clock at `now`: counts restart with the term, so older history says + /// nothing about the current one. + pub fn observe(&self, sample: &PeerAckSample, now: Instant) { + let mut guard = self.inner.lock().unwrap_or_else(|p| p.into_inner()); + let same_term = guard.as_ref().is_some_and(|t| t.term == sample.term); + if !same_term { + *guard = Some(TermContact { + term: sample.term, + sampled_at: now, + peers: sample + .acks + .iter() + .map(|&(peer, acks)| { + ( + peer, + PeerContact { + acks, + changed_at: now, + }, + ) + }) + .collect(), + }); + return; + } + let Some(term) = guard.as_mut() else { + return; + }; + term.sampled_at = now; + let mut peers = HashMap::with_capacity(sample.acks.len()); + for &(peer, acks) in &sample.acks { + let contact = match term.peers.get(&peer) { + Some(prev) if prev.acks == acks => *prev, + _ => PeerContact { + acks, + changed_at: now, + }, + }; + peers.insert(peer, contact); + } + term.peers = peers; + } + + /// Forget every sample. Called when this node stops leading. + pub fn clear(&self) { + *self.inner.lock().unwrap_or_else(|p| p.into_inner()) = None; + } + + /// Whether samples taken in `term` show `peer` silent for at least + /// `min_silence`. False for an untracked peer or another term. + pub fn silent_for(&self, peer: u64, term: u64, min_silence: Duration) -> bool { + let guard = self.inner.lock().unwrap_or_else(|p| p.into_inner()); + let Some(contact) = guard.as_ref().filter(|t| t.term == term) else { + return false; + }; + contact.peers.get(&peer).is_some_and(|p| { + contact.sampled_at.saturating_duration_since(p.changed_at) >= min_silence + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const PEER: u64 = 2; + const TERM: u64 = 4; + const SILENCE: Duration = Duration::from_secs(15); + + fn sample(term: u64, acks: u64) -> PeerAckSample { + PeerAckSample { + term, + acks: vec![(PEER, acks)], + } + } + + #[test] + fn an_unchanged_count_accumulates_silence() { + let clock = RaftContactClock::default(); + let start = Instant::now(); + clock.observe(&sample(TERM, 7), start); + clock.observe(&sample(TERM, 7), start + SILENCE); + assert!(clock.silent_for(PEER, TERM, SILENCE)); + } + + #[test] + fn a_new_response_resets_silence() { + let clock = RaftContactClock::default(); + let start = Instant::now(); + clock.observe(&sample(TERM, 7), start); + clock.observe(&sample(TERM, 8), start + SILENCE); + assert!(!clock.silent_for(PEER, TERM, SILENCE)); + } + + #[test] + fn a_new_term_restarts_the_clock() { + let clock = RaftContactClock::default(); + let start = Instant::now(); + clock.observe(&sample(TERM, 7), start); + clock.observe(&sample(TERM + 1, 0), start + SILENCE); + assert!(!clock.silent_for(PEER, TERM + 1, SILENCE)); + assert!(!clock.silent_for(PEER, TERM, SILENCE)); + } + + #[test] + fn cleared_or_untracked_is_never_silent() { + let clock = RaftContactClock::default(); + let start = Instant::now(); + clock.observe(&sample(TERM, 7), start); + clock.observe(&sample(TERM, 7), start + SILENCE); + assert!(!clock.silent_for(PEER + 1, TERM, SILENCE)); + clock.clear(); + assert!(!clock.silent_for(PEER, TERM, SILENCE)); + } +} diff --git a/nodedb-cluster/src/lib.rs b/nodedb-cluster/src/lib.rs index 0464926ea..c6e62ea1b 100644 --- a/nodedb-cluster/src/lib.rs +++ b/nodedb-cluster/src/lib.rs @@ -42,8 +42,10 @@ pub mod forward; pub mod ghost; #[doc(hidden)] pub mod ghost_sweeper; +pub mod group_disk; pub mod health; pub mod install_snapshot; +pub mod lease_liveness; pub mod lifecycle; pub mod lifecycle_state; pub mod loop_metrics; @@ -54,6 +56,7 @@ pub mod migration_executor; pub mod mirror; pub mod multi_raft; pub mod quic_transport; +pub mod raft_bootstrap; pub mod raft_loop; pub mod raft_storage; pub mod rdma_transport; @@ -65,6 +68,7 @@ pub mod rebalance_scheduler; pub mod rebalancer; pub mod routing; pub mod routing_liveness; +pub mod routing_membership; pub mod rpc_codec; pub mod shard_split; pub mod subsystem; @@ -78,7 +82,8 @@ pub mod wire_version; pub use applied_watcher::{AppliedIndexWatcher, GroupAppliedWatchers, WaitOutcome}; pub use bootstrap::{ - ClusterConfig, ClusterState, JoinRetryPolicy, start_cluster, start_cluster_subsystems, + ClusterConfig, ClusterState, JoinRetryPolicy, SubsystemHandles, start_cluster, + start_cluster_subsystems, }; #[doc(hidden)] pub use calvin::{EngineKeySet, EpochBatch, ReadWriteSet, SequencedTxn, SortedVec, TxClass}; @@ -100,6 +105,10 @@ pub use error::{ pub use forward::{ChunkSink, NoopPlanExecutor, PlanExecutor}; pub use ghost::{GhostStub, GhostTable}; pub use health::{HealthConfig, HealthMonitor}; +pub use lease_liveness::{ + DEAD_HOLDER_LEASE_GRACE, DEAD_HOLDER_RAFT_SILENCE, LEASE_CLOCK_SKEW, LEASE_SELF_FENCE_WINDOW, + LeaseHolderLiveness, LeaseNow, +}; pub use lifecycle_state::{ClusterLifecycleState, ClusterLifecycleTracker}; pub use loop_metrics::{LoopMetrics, LoopMetricsRegistry}; pub use migration::{MigrationPhase, MigrationState}; @@ -108,7 +117,8 @@ pub use migration_executor::{ }; pub use multi_raft::{GroupStatus, MultiRaft}; pub use raft_loop::{ - AssignRemoteSurrogate, AuthLeaseService, CalvinSubmit, CalvinSubmitInbox, CommitApplier, + ApplyPermit, AssignRemoteSurrogate, AuthLeaseService, BuiltGroupSnapshot, CalvinSubmit, + CalvinSubmitInbox, CommitApplier, GroupApplyGates, InstallPermit, MetadataSnapshotCapture, RaftLoop, ReleaseReservation, ReserveRead, ShuffleAggregator, ShuffleConsumer, ShuffleProducer, ShuffleReceiver, SnapshotApplier, SnapshotBuilder, SnapshotQuarantineHook, VShardEnvelopeHandler, @@ -128,8 +138,9 @@ pub use routing_liveness::{NodeIdResolver, RoutingLivenessHook}; pub use rpc_codec::{ AssignSurrogateRequest, AssignSurrogateResponse, AuthBarrierOutcome, AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewOutcome, AuthLeaseRenewRequest, AuthLeaseRenewResponse, - DataPlaneErrorCode, GroupCoverage, JoinKeyPair, MacKey, PartNodeEntry, RaftRpc, - ReleaseReservationRequest, ReleaseReservationResponse, ReserveReadRequest, ReserveReadResponse, + CalvinPartsRequest, CalvinPartsResponse, DataPlaneErrorCode, GroupCoverage, JoinKeyPair, + MAX_PARTS_BATCH_BYTES, MacKey, PartNodeEntry, RaftRpc, ReleaseReservationRequest, + ReleaseReservationResponse, ReserveReadRequest, ReserveReadResponse, ShuffleAggregateConsumeRequest, ShuffleAggregateConsumeResponse, ShuffleConsumeRequest, ShuffleConsumeResponse, ShuffleProduceRequest, ShuffleProduceResponse, ShufflePushChunk, ShufflePushEnd, ShufflePushRequest, SortKey, SubmitCalvinInboxRequest, @@ -152,12 +163,12 @@ pub use cross_shard_txn::{ }; pub use metadata_group::entry::JoinTokenTransitionKind; pub use metadata_group::{ - CacheApplier, Compensation, DescriptorHeader, DescriptorId, DescriptorKind, DescriptorLease, - DescriptorState, METADATA_GROUP_ID, MetadataApplier, MetadataCache, MetadataEntry, - MigrationCheckpointPayload, MigrationId, MigrationPhaseTag, NoopMetadataApplier, - PendingDdlObject, PersistedMigrationCheckpoint, RoutingChange, SharedMigrationStateTable, - TopologyChange, apply_migration_abort, apply_migration_checkpoint, decode_entry, encode_entry, - new_shared, + CacheApplier, CommittedMetadata, Compensation, DescriptorHeader, DescriptorId, DescriptorKind, + DescriptorLease, DescriptorState, DrainOwner, METADATA_GROUP_ID, MetadataApplier, + MetadataCache, MetadataEntry, MetadataPayload, MigrationCheckpointPayload, MigrationId, + MigrationPhaseTag, NoopMetadataApplier, PendingDdlObject, PersistedMigrationCheckpoint, + RoutingChange, SharedMigrationStateTable, TopologyChange, apply_migration_abort, + apply_migration_checkpoint, decode_entry, encode_entry, entry_stamp, new_shared, stamp_entry, }; pub use migration_executor::recover_in_flight_migrations; pub use quic_transport::{QuicTransport, QuicTransportConfig}; @@ -172,12 +183,12 @@ pub use rebalance_scheduler::{NodeMetrics, RebalanceScheduler, RebalanceTrigger, pub use shard_split::{SplitPlan, SplitStrategy, plan_graph_split, plan_vector_split}; pub use subsystem::{ BootstrapCtx, BootstrapError, ClusterHealth, ClusterSubsystem, RunningCluster, ShutdownError, - SubsystemHandle, SubsystemHealth, SubsystemRegistry, TopoError, topo_sort, + SubsystemHandle, SubsystemHealth, SubsystemRegistry, SwimWiring, TopoError, topo_sort, }; pub use swim::bootstrap::spawn_with_subscribers as spawn_swim_with_subscribers; pub use swim::{ Incarnation, Member, MemberState, MembershipList, MembershipSubscriber, SwimConfig, SwimError, - SwimHandle, UdpTransport, spawn as spawn_swim, + SwimHandle, UdpTransport, bind_swim_listener, default_swim_addr, spawn as spawn_swim, }; pub use auth::{ diff --git a/nodedb-cluster/src/lifecycle.rs b/nodedb-cluster/src/lifecycle.rs index 5de97d63f..a9acdac1c 100644 --- a/nodedb-cluster/src/lifecycle.rs +++ b/nodedb-cluster/src/lifecycle.rs @@ -14,7 +14,7 @@ //! [`MetadataEntry::TopologyChange`] / [`MetadataEntry::RoutingChange`] //! entries and applied through the `MetadataApplier` on every node. -use tracing::{info, warn}; +use tracing::info; use crate::error::{ClusterError, Result}; use crate::metadata_group::{MetadataEntry, TopologyChange}; @@ -82,22 +82,28 @@ pub fn is_safe_to_remove(node_id: u64, topology: &ClusterTopology, routing: &Rou /// Register a joining node in the local topology and produce the /// [`MetadataEntry`] to be proposed on the metadata Raft group. -pub fn handle_node_join(node_id: u64, addr: &str, topology: &mut ClusterTopology) -> MetadataEntry { - use std::net::SocketAddr; - - let socket_addr: SocketAddr = addr.parse().unwrap_or_else(|_| { - warn!(node_id, addr, "invalid address, using default"); - SocketAddr::from(([0, 0, 0, 0], 0)) - }); +/// +/// An `addr` that does not parse as a socket address is refused. The +/// topology is left unchanged and nothing is proposed. +pub fn handle_node_join( + node_id: u64, + addr: &str, + swim_addr: Option, + topology: &mut ClusterTopology, +) -> Result { + let socket_addr: std::net::SocketAddr = addr.parse().map_err(|e| ClusterError::Config { + detail: format!("join: node {node_id} advertises invalid address {addr:?}: {e}"), + })?; - let info = NodeInfo::new(node_id, socket_addr, NodeState::Joining); + let info = NodeInfo::new(node_id, socket_addr, NodeState::Joining).with_swim_addr(swim_addr); topology.join_as_learner(info); info!(node_id, addr, "node joining as learner"); - MetadataEntry::TopologyChange(TopologyChange::Join { + Ok(MetadataEntry::TopologyChange(TopologyChange::Join { node_id, addr: addr.to_string(), - }) + swim_addr: swim_addr.map(|a| a.to_string()), + })) } /// Handle learner promotion after state catch-up validation. @@ -173,17 +179,33 @@ mod tests { #[test] fn node_join_creates_learner() { let mut topo = ClusterTopology::new(); - let entry = handle_node_join(5, "10.0.0.5:9000", &mut topo); + let swim: Option = "10.0.0.5:9001".parse().ok(); + let entry = handle_node_join(5, "10.0.0.5:9000", swim, &mut topo).unwrap(); assert!(topo.contains(5)); assert_eq!(topo.learner_nodes().len(), 1); + assert_eq!(topo.get_node(5).and_then(NodeInfo::swim_socket_addr), swim); match entry { - MetadataEntry::TopologyChange(TopologyChange::Join { node_id, .. }) => { + MetadataEntry::TopologyChange(TopologyChange::Join { + node_id, swim_addr, .. + }) => { assert_eq!(node_id, 5); + assert_eq!(swim_addr.as_deref(), Some("10.0.0.5:9001")); } other => panic!("expected Join, got {other:?}"), } } + /// An invalid join address is refused at propose time. The topology + /// stays unchanged. + #[test] + fn node_join_refuses_invalid_address() { + let mut topo = ClusterTopology::new(); + let err = handle_node_join(5, "not-an-address", None, &mut topo) + .expect_err("an invalid address must be refused"); + assert!(matches!(err, ClusterError::Config { .. }), "got {err:?}"); + assert!(!topo.contains(5)); + } + #[test] fn learner_promotion_checks_lag() { let mut topo = ClusterTopology::new(); diff --git a/nodedb-cluster/src/metadata_group/applier.rs b/nodedb-cluster/src/metadata_group/applier.rs index f45b3f844..beb999b7f 100644 --- a/nodedb-cluster/src/metadata_group/applier.rs +++ b/nodedb-cluster/src/metadata_group/applier.rs @@ -11,6 +11,7 @@ use uuid; use crate::auth::raft_backed_store::apply_token_transition_to_mirror; use crate::auth::token_state::SharedTokenStateMirror; +use crate::error::ClusterError; use crate::metadata_group::cache::{ MetadataCache, apply_migration_abort, apply_migration_checkpoint, }; @@ -28,11 +29,90 @@ use crate::topology::{ClusterTopology, NodeInfo, NodeState}; /// applier in the `nodedb` crate to additionally decode the /// `CatalogDdl` payload as a `CatalogEntry` and write through to /// `SystemCatalog`. +/// +/// The apply is awaited: an entry whose effects await other work completes +/// them before the raft loop advances past it. +#[async_trait::async_trait] pub trait MetadataApplier: Send + Sync + 'static { - /// Apply a batch of committed raft entries. Entries with empty - /// `data` (raft no-ops) are skipped. Returns the highest log - /// index applied. - fn apply(&self, entries: &[(u64, Vec)]) -> u64; + /// Apply a batch of committed raft entries, each decoded once by the + /// caller. Entries with an empty payload (raft no-ops and conf changes) + /// apply nothing. Returns the highest log index applied. + async fn apply_decoded(&self, entries: &[CommittedMetadata<'_>]) -> u64; + + /// Decode each encoded entry once, then apply the batch through + /// [`apply_decoded`](Self::apply_decoded). For callers that hold the + /// encoded bytes. + async fn apply(&self, entries: &[(u64, Vec)]) -> u64 { + let committed: Vec> = entries + .iter() + .map(|(index, data)| CommittedMetadata::decode(*index, data)) + .collect(); + self.apply_decoded(&committed).await + } + + /// Whether every effect of an entry `apply` reports as applied is durable + /// when `apply` returns. When true, the raft loop saves the returned index + /// as the group's applied floor, and a restart resumes delivery above it. + /// When false, a restart replays the whole retained log. + fn durable_effects(&self) -> bool { + false + } +} + +/// One committed metadata entry: its raw payload and its payload decoded +/// once. +#[derive(Debug)] +pub struct CommittedMetadata<'a> { + pub index: u64, + /// The entry's payload. Empty for a raft no-op or a conf change. + pub data: &'a [u8], + pub payload: MetadataPayload, +} + +/// The decoded payload of a [`CommittedMetadata`]. +#[derive(Debug)] +pub enum MetadataPayload { + /// A raft no-op or a conf change. It applies nothing. + Empty, + Decoded(MetadataEntry), + /// The payload does not decode. Every replica fails the same way. + Undecodable(ClusterError), +} + +impl<'a> CommittedMetadata<'a> { + /// Decode `data`, the payload committed at `index`. + pub fn decode(index: u64, data: &'a [u8]) -> Self { + let payload = if data.is_empty() { + MetadataPayload::Empty + } else { + match decode_entry(data) { + Ok(entry) => MetadataPayload::Decoded(entry), + Err(e) => MetadataPayload::Undecodable(e), + } + }; + Self { + index, + data, + payload, + } + } + + /// An entry that applies nothing, such as a conf change. + pub fn empty(index: u64) -> Self { + Self { + index, + data: &[], + payload: MetadataPayload::Empty, + } + } + + /// The decoded entry, if the payload decoded. + pub fn entry(&self) -> Option<&MetadataEntry> { + match &self.payload { + MetadataPayload::Decoded(entry) => Some(entry), + MetadataPayload::Empty | MetadataPayload::Undecodable(_) => None, + } + } } /// Default applier that writes committed entries to an in-memory @@ -123,15 +203,37 @@ impl CacheApplier { }; let mut topo = live.write().unwrap_or_else(|p| p.into_inner()); match change { - TopologyChange::Join { node_id, addr } => { - if topo.contains(*node_id) { + TopologyChange::Join { + node_id, + addr, + swim_addr, + } => { + let swim: Option = swim_addr.as_deref().and_then(|raw| { + raw.parse() + .inspect_err(|_| warn!(node_id, raw, "join: invalid SWIM address, dropped")) + .ok() + }); + if let Some(existing) = topo.get_node(*node_id) { + // A known node re-advertising a new SWIM address updates + // its entry; nothing else about it changes here. + if swim.is_some() && existing.swim_socket_addr() != swim { + let updated = existing.clone().with_swim_addr(swim); + topo.add_node(updated); + } return; } - let parsed: SocketAddr = addr.parse().unwrap_or_else(|_| { - warn!(node_id, addr, "join: invalid address, using placeholder"); - SocketAddr::from(([0, 0, 0, 0], 0)) - }); - topo.join_as_learner(NodeInfo::new(*node_id, parsed, NodeState::Joining)); + // Propose refuses an invalid address. One that still + // commits never enters topology with a placeholder. + let parsed: SocketAddr = match addr.parse() { + Ok(parsed) => parsed, + Err(e) => { + error!(node_id, addr, error = %e, "join: invalid address, join skipped"); + return; + } + }; + topo.join_as_learner( + NodeInfo::new(*node_id, parsed, NodeState::Joining).with_swim_addr(swim), + ); } TopologyChange::PromoteToVoter { node_id } => { topo.promote_to_voter(*node_id); @@ -150,16 +252,16 @@ impl CacheApplier { /// Cascade live-state mutations for a committed entry. Handles /// `Batch` by recursing into each sub-entry. - fn cascade_live_state(&self, entry: &MetadataEntry) { + fn cascade_live_state(&self, index: u64, entry: &MetadataEntry) { match entry { // The applied epoch advances in the raft loop, on every node, // regardless of which applier the host installed. MetadataEntry::ClusterEpochBump { .. } => {} MetadataEntry::TopologyChange(change) => self.apply_topology_change(change), - MetadataEntry::RoutingChange(change) => self.apply_routing_change(change), + MetadataEntry::RoutingChange(change) => self.apply_routing_change(index, change), MetadataEntry::Batch { entries } => { for sub in entries { - self.cascade_live_state(sub); + self.cascade_live_state(index, sub); } } MetadataEntry::MigrationCheckpoint { @@ -232,8 +334,8 @@ impl CacheApplier { } /// Mutate the live routing handle (if attached) in response to - /// a committed `RoutingChange`. - fn apply_routing_change(&self, change: &RoutingChange) { + /// a committed `RoutingChange` at metadata log `index`. + fn apply_routing_change(&self, index: u64, change: &RoutingChange) { let Some(live) = &self.live_routing else { return; }; @@ -244,13 +346,18 @@ impl CacheApplier { new_group_id, new_leaseholder_node_id, } => { - rt.reassign_vshard(*vshard_id, *new_group_id); + rt.reassign_vshard(*vshard_id, *new_group_id, index); + // The entry names a planned leaseholder with no term. It + // fills only a hint that holds no term. rt.set_leader(*new_group_id, *new_leaseholder_node_id); } RoutingChange::LeadershipTransfer { group_id, new_leader_node_id, } => { + // The entry names the transfer target before its election, + // so it carries no term. The election's leader reaches the + // hint with its term from Raft or a redirect. rt.set_leader(*group_id, *new_leader_node_id); } RoutingChange::RemoveMember { group_id, node_id } => { @@ -266,24 +373,26 @@ impl CacheApplier { } } +#[async_trait::async_trait] impl MetadataApplier for CacheApplier { - fn apply(&self, entries: &[(u64, Vec)]) -> u64 { + async fn apply_decoded(&self, entries: &[CommittedMetadata<'_>]) -> u64 { let mut last = 0u64; let mut guard = self .cache .write() .unwrap_or_else(|poison| poison.into_inner()); - for (index, data) in entries { - last = *index; - if data.is_empty() { - continue; - } - match decode_entry(data) { - Ok(entry) => { - guard.apply(*index, &entry); - self.cascade_live_state(&entry); + for committed in entries { + let index = committed.index; + last = index; + match &committed.payload { + MetadataPayload::Empty => {} + MetadataPayload::Decoded(entry) => { + guard.apply(index, entry); + self.cascade_live_state(index, entry); + } + MetadataPayload::Undecodable(e) => { + warn!(index, error = %e, "metadata decode failed") } - Err(e) => warn!(index = *index, error = %e, "metadata decode failed"), } } last @@ -295,9 +404,10 @@ impl MetadataApplier for CacheApplier { /// index so raft can advance its applied watermark. pub struct NoopMetadataApplier; +#[async_trait::async_trait] impl MetadataApplier for NoopMetadataApplier { - fn apply(&self, entries: &[(u64, Vec)]) -> u64 { - entries.last().map(|(idx, _)| *idx).unwrap_or(0) + async fn apply_decoded(&self, entries: &[CommittedMetadata<'_>]) -> u64 { + entries.last().map_or(0, |e| e.index) } } @@ -307,8 +417,8 @@ mod tests { use crate::metadata_group::codec::encode_entry; use crate::metadata_group::entry::{MetadataEntry, TopologyChange}; - #[test] - fn cache_applier_counts_catalog_ddl() { + #[tokio::test] + async fn cache_applier_counts_catalog_ddl() { let cache = Arc::new(RwLock::new(MetadataCache::new())); let applier = CacheApplier::new(cache.clone()); @@ -319,10 +429,11 @@ mod tests { let topo = encode_entry(&MetadataEntry::TopologyChange(TopologyChange::Join { node_id: 7, addr: "10.0.0.7:9000".into(), + swim_addr: None, })) .unwrap(); - let last = applier.apply(&[(1, ddl), (2, topo)]); + let last = applier.apply(&[(1, ddl), (2, topo)]).await; assert_eq!(last, 2); let guard = cache.read().unwrap(); @@ -331,8 +442,8 @@ mod tests { assert_eq!(guard.topology_log.len(), 1); } - #[test] - fn cache_applier_idempotent() { + #[tokio::test] + async fn cache_applier_idempotent() { let cache = Arc::new(RwLock::new(MetadataCache::new())); let applier = CacheApplier::new(cache.clone()); @@ -340,16 +451,16 @@ mod tests { payload: vec![9, 9], }) .unwrap(); - applier.apply(&[(5, bytes.clone())]); - applier.apply(&[(3, bytes)]); // Earlier index — ignored. + applier.apply(&[(5, bytes.clone())]).await; + applier.apply(&[(3, bytes)]).await; // Earlier index — ignored. let guard = cache.read().unwrap(); assert_eq!(guard.applied_index, 5); assert_eq!(guard.catalog_entries_applied, 1); } - #[test] - fn cache_applier_mutates_live_topology_on_start_decommission() { + #[tokio::test] + async fn cache_applier_mutates_live_topology_on_start_decommission() { use crate::topology::{ClusterTopology, NodeInfo, NodeState}; use std::net::SocketAddr; @@ -370,14 +481,83 @@ mod tests { TopologyChange::StartDecommission { node_id: 7 }, )) .unwrap(); - applier.apply(&[(1, bytes)]); + applier.apply(&[(1, bytes)]).await; let topo = topology.read().unwrap(); assert_eq!(topo.get_node(7).unwrap().state, NodeState::Draining); } - #[test] - fn cache_applier_mutates_live_routing_on_remove_member() { + /// A committed `Join` puts the joiner's SWIM address into the live + /// topology, and a later `Join` with a new address updates it. + #[tokio::test] + async fn cache_applier_applies_join_swim_address() { + let cache = Arc::new(RwLock::new(MetadataCache::new())); + let topology = Arc::new(RwLock::new(crate::topology::ClusterTopology::new())); + let routing = Arc::new(RwLock::new(crate::routing::RoutingTable::uniform( + 1, + &[1], + 1, + ))); + let applier = + CacheApplier::new(cache.clone()).with_live_state(topology.clone(), routing.clone()); + let join = |swim: &str| { + encode_entry(&MetadataEntry::TopologyChange(TopologyChange::Join { + node_id: 9, + addr: "10.0.0.9:9400".into(), + swim_addr: Some(swim.into()), + })) + .unwrap() + }; + + applier.apply(&[(1, join("10.0.0.9:9401"))]).await; + assert_eq!( + topology + .read() + .unwrap() + .get_node(9) + .unwrap() + .swim_socket_addr(), + "10.0.0.9:9401".parse().ok() + ); + + applier.apply(&[(2, join("10.0.0.9:9501"))]).await; + assert_eq!( + topology + .read() + .unwrap() + .get_node(9) + .unwrap() + .swim_socket_addr(), + "10.0.0.9:9501".parse().ok() + ); + } + + /// A committed `Join` with an invalid address is skipped. The node + /// never enters topology with a placeholder address. + #[tokio::test] + async fn cache_applier_skips_join_with_invalid_address() { + let cache = Arc::new(RwLock::new(MetadataCache::new())); + let topology = Arc::new(RwLock::new(crate::topology::ClusterTopology::new())); + let routing = Arc::new(RwLock::new(crate::routing::RoutingTable::uniform( + 1, + &[1], + 1, + ))); + let applier = + CacheApplier::new(cache.clone()).with_live_state(topology.clone(), routing.clone()); + let bytes = encode_entry(&MetadataEntry::TopologyChange(TopologyChange::Join { + node_id: 9, + addr: "not-an-address".into(), + swim_addr: None, + })) + .unwrap(); + + assert_eq!(applier.apply(&[(1, bytes)]).await, 1); + assert!(!topology.read().unwrap().contains(9)); + } + + #[tokio::test] + async fn cache_applier_mutates_live_routing_on_remove_member() { use crate::metadata_group::entry::RoutingChange; let cache = Arc::new(RwLock::new(MetadataCache::new())); @@ -395,14 +575,14 @@ mod tests { node_id: 2, })) .unwrap(); - applier.apply(&[(1, bytes)]); + applier.apply(&[(1, bytes)]).await; let rt = routing.read().unwrap(); assert!(!rt.group_info(0).unwrap().members.contains(&2)); } - #[test] - fn cache_applier_mutates_live_routing_on_set_placement() { + #[tokio::test] + async fn cache_applier_mutates_live_routing_on_set_placement() { use crate::metadata_group::entry::RoutingChange; let cache = Arc::new(RwLock::new(MetadataCache::new())); @@ -420,7 +600,7 @@ mod tests { placement: vec![1, 2], })) .unwrap(); - applier.apply(&[(1, bytes)]); + applier.apply(&[(1, bytes)]).await; let rt = routing.read().unwrap(); assert_eq!( @@ -430,8 +610,8 @@ mod tests { ); } - #[test] - fn cache_applier_without_live_state_stays_log_only() { + #[tokio::test] + async fn cache_applier_without_live_state_stays_log_only() { let cache = Arc::new(RwLock::new(MetadataCache::new())); let applier = CacheApplier::new(cache.clone()); let bytes = encode_entry(&MetadataEntry::TopologyChange( @@ -439,14 +619,17 @@ mod tests { )) .unwrap(); // Must not panic and must still advance the applied index. - let last = applier.apply(&[(1, bytes)]); + let last = applier.apply(&[(1, bytes)]).await; assert_eq!(last, 1); } - #[test] - fn noop_applier_advances_watermark() { + #[tokio::test] + async fn noop_applier_advances_watermark() { let noop = NoopMetadataApplier; - assert_eq!(noop.apply(&[(7, b"x".to_vec()), (9, b"y".to_vec())]), 9); - assert_eq!(noop.apply(&[]), 0); + assert_eq!( + noop.apply(&[(7, b"x".to_vec()), (9, b"y".to_vec())]).await, + 9 + ); + assert_eq!(noop.apply(&[]).await, 0); } } diff --git a/nodedb-cluster/src/metadata_group/cache.rs b/nodedb-cluster/src/metadata_group/cache.rs index e9c9bde53..2a46ab931 100644 --- a/nodedb-cluster/src/metadata_group/cache.rs +++ b/nodedb-cluster/src/metadata_group/cache.rs @@ -124,10 +124,8 @@ impl MetadataCache { } // Drain state is host-side (lives in // `nodedb::control::lease::DescriptorDrainTracker`); - // the cluster-side cache only tracks lease state - // directly. These no-op arms keep the exhaustive - // match coverage so adding new variants is a - // compile-time error here too. + // the cluster-side cache only folds the drain expiry into + // `last_applied_hlc`. MetadataEntry::DescriptorDrainStart { expires_at, .. } => { if *expires_at > self.last_applied_hlc { self.last_applied_hlc = *expires_at; @@ -135,7 +133,7 @@ impl MetadataCache { } MetadataEntry::DescriptorDrainEnd { .. } => {} MetadataEntry::DdlPrepareAcquire { .. } | MetadataEntry::DdlPrepareRelease { .. } => { - // Host-side ephemeral coordination state; replay rebuilds it. + // Host-side state: the production applier persists the owner. } MetadataEntry::DdlPrepared { .. } => { // The host-side applier unwraps and fences this entry. The cache @@ -144,10 +142,9 @@ impl MetadataCache { MetadataEntry::DdlPendingPropose { .. } | MetadataEntry::DdlPendingFinalize { .. } | MetadataEntry::DdlPendingCancel { .. } => { - // Host-side only: the production applier owns the pending-DDL - // table (`nodedb::control::pending_ddl::PendingDdlTable`), - // rebuilt entirely by replay. The cluster cache has no state - // to track beyond `applied_index`. + // Host-side only: the production applier owns and persists + // the pending-DDL table. The cluster cache has no state to + // track beyond `applied_index`. } MetadataEntry::CaTrustChange { .. } => { // CA trust mutations are host-side only: the production @@ -181,6 +178,12 @@ impl MetadataCache { } MetadataEntry::EnrollmentPreauthorization { .. } | MetadataEntry::EnrollmentPreauthorizationRevoke { .. } => {} + // Host-side only: the production applier advances the database + // high-watermark and hands the id to the requesting node. + MetadataEntry::DatabaseIdReserve { .. } => {} + // Host-side only: the production applier records the point and + // cuts every group this node hosts. + MetadataEntry::RestorePoint { .. } => {} MetadataEntry::JoinTokenTransition { .. } => { // Token lifecycle transitions are enforced by the bootstrap- // listener handler at apply time. The cluster cache records @@ -198,6 +201,8 @@ impl MetadataCache { // has no migration state to track beyond applied_index. MetadataEntry::MigrationCheckpoint { .. } => {} MetadataEntry::MigrationAbort { .. } => {} + // Advances the archive frontier only. + MetadataEntry::ArchiveMark => {} } } } @@ -295,6 +300,8 @@ fn apply_compensation( rt.remove_group_member(*group_id, *peer_id); } Compensation::RestoreLeaderHint { group_id, peer_id } => { + // The compensation carries no term. It never replaces a hint a + // Raft observation or redirect wrote since. rt.set_leader(*group_id, *peer_id); } Compensation::RemoveGhostStub { vshard_id: _ } => { diff --git a/nodedb-cluster/src/metadata_group/codec.rs b/nodedb-cluster/src/metadata_group/codec.rs index 7244b68f7..52117ea5e 100644 --- a/nodedb-cluster/src/metadata_group/codec.rs +++ b/nodedb-cluster/src/metadata_group/codec.rs @@ -17,12 +17,46 @@ pub fn encode_entry(entry: &MetadataEntry) -> Result, ClusterError> { }) } -/// Decode a [`MetadataEntry`] from bytes. +/// First byte of a stamped entry: `[STAMP_MARKER][hlc u64 BE][envelope]`. +const STAMP_MARKER: u8 = 0xC7; + +/// Length of the stamp prefix. +const STAMP_LEN: usize = 1 + 8; + +/// Prefix the encoded entry `bytes` with the HLC wall time, in nanoseconds, +/// the metadata leader stamped as it appended the entry. A restore keeps a +/// stamped entry only when the stamp is below its watermark. +pub fn stamp_entry(bytes: &[u8], hlc: u64) -> Vec { + let mut stamped = Vec::with_capacity(STAMP_LEN + bytes.len()); + stamped.push(STAMP_MARKER); + stamped.extend_from_slice(&hlc.to_be_bytes()); + stamped.extend_from_slice(bytes); + stamped +} + +/// The leader's HLC stamp of an encoded entry, `None` for an unstamped one. +pub fn entry_stamp(data: &[u8]) -> Option { + if data.first() != Some(&STAMP_MARKER) { + return None; + } + let raw: [u8; 8] = data.get(1..STAMP_LEN)?.try_into().ok()?; + Some(u64::from_be_bytes(raw)) +} + +/// The versioned envelope of `data`, past any stamp. +fn envelope(data: &[u8]) -> &[u8] { + match entry_stamp(data) { + Some(_) => &data[STAMP_LEN..], + None => data, + } +} + +/// Decode a [`MetadataEntry`] from bytes, stamped or not. /// /// Requires a v2 versioned envelope. Rejects bytes without the envelope /// marker and envelopes with unsupported future version numbers. pub fn decode_entry(data: &[u8]) -> Result { - decode_versioned(data).map_err(|e| ClusterError::Codec { + decode_versioned(envelope(data)).map_err(|e| ClusterError::Codec { detail: format!("metadata decode: {e}"), }) } @@ -40,6 +74,7 @@ mod tests { let entry = MetadataEntry::TopologyChange(TopologyChange::Join { node_id: 42, addr: "127.0.0.1:7001".to_string(), + swim_addr: Some("127.0.0.1:7002".to_string()), }); let bytes = encode_entry(&entry).unwrap(); let decoded = decode_entry(&bytes).unwrap(); @@ -89,6 +124,21 @@ mod tests { assert_eq!(entry, decoded); } + #[test] + fn a_stamped_entry_decodes_and_reports_its_stamp() { + let entry = MetadataEntry::DdlPendingFinalize { token: 3 }; + let bytes = encode_entry(&entry).unwrap(); + assert_eq!(entry_stamp(&bytes), None); + let stamped = stamp_entry(&bytes, 0x0102_0304_0506_0708); + assert_eq!(entry_stamp(&stamped), Some(0x0102_0304_0506_0708)); + assert_eq!(decode_entry(&stamped).unwrap(), entry); + assert_eq!( + entry_stamp(&stamped[..4]), + None, + "a short stamp is no stamp" + ); + } + #[test] fn ddl_pending_cancel_roundtrip() { let entry = MetadataEntry::DdlPendingCancel { token: 12 }; diff --git a/nodedb-cluster/src/metadata_group/descriptors/common.rs b/nodedb-cluster/src/metadata_group/descriptors/common.rs index 84e13a82f..22e95c9bf 100644 --- a/nodedb-cluster/src/metadata_group/descriptors/common.rs +++ b/nodedb-cluster/src/metadata_group/descriptors/common.rs @@ -99,6 +99,8 @@ pub enum DescriptorKind { Tenant, ApiKey, AuditRetention, + /// An ND array. Its drain gates array DDL and cell writes. + Array, } /// Common header embedded in every descriptor. diff --git a/nodedb-cluster/src/metadata_group/descriptors/drain_owner.rs b/nodedb-cluster/src/metadata_group/descriptors/drain_owner.rs new file mode 100644 index 000000000..d5587a2bb --- /dev/null +++ b/nodedb-cluster/src/metadata_group/descriptors/drain_owner.rs @@ -0,0 +1,89 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Owner of one descriptor drain. + +use serde::{Deserialize, Serialize}; + +/// Who started a descriptor drain. +/// +/// A descriptor stays drained while any owner's drain remains. Each owner ends +/// only its own drain, so one owner never ends another's early. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + Hash, + Serialize, + Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum DrainOwner { + /// The catalog DDL altering the descriptor. The replicated DDL preparation + /// lease serialises every DDL, so at most one DDL drain exists per + /// descriptor. The apply of that DDL's own catalog entry ends it. + Ddl, + /// A `MOVE TENANT` of `tenant_id` out of `source_db_id`. The apply of its + /// cutover entry ends it. + MoveTenant { tenant_id: u64, source_db_id: u64 }, + /// The clone materializer copying a source into one clone collection. + CloneMaterialize { + clone_database: u64, + tenant_id: u64, + clone_collection: String, + }, +} + +impl std::fmt::Display for DrainOwner { + /// Names the operation the drain belongs to, for a refused statement's + /// error message. + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Ddl => f.write_str("a DDL is altering the collection"), + Self::MoveTenant { + tenant_id, + source_db_id, + } => write!( + f, + "the collection is being moved: MOVE TENANT {tenant_id} out of database \ + {source_db_id}" + ), + Self::CloneMaterialize { + clone_database, + clone_collection, + .. + } => write!( + f, + "clone materializing: '{clone_collection}' in database {clone_database} \ + is copying from this collection" + ), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn display_names_the_owner() { + assert!( + DrainOwner::MoveTenant { + tenant_id: 3, + source_db_id: 1024 + } + .to_string() + .contains("being moved") + ); + assert!( + DrainOwner::CloneMaterialize { + clone_database: 1025, + tenant_id: 1, + clone_collection: "kv".into(), + } + .to_string() + .contains("clone materializing") + ); + } +} diff --git a/nodedb-cluster/src/metadata_group/descriptors/mod.rs b/nodedb-cluster/src/metadata_group/descriptors/mod.rs index d7b2ed3bf..bc90c1752 100644 --- a/nodedb-cluster/src/metadata_group/descriptors/mod.rs +++ b/nodedb-cluster/src/metadata_group/descriptors/mod.rs @@ -7,7 +7,9 @@ //! deliberately opaque to them. pub mod common; +pub mod drain_owner; pub mod lease; pub use common::{DescriptorHeader, DescriptorId, DescriptorKind}; +pub use drain_owner::DrainOwner; pub use lease::DescriptorLease; diff --git a/nodedb-cluster/src/metadata_group/entry.rs b/nodedb-cluster/src/metadata_group/entry.rs index bd6552176..6d3b6be9e 100644 --- a/nodedb-cluster/src/metadata_group/entry.rs +++ b/nodedb-cluster/src/metadata_group/entry.rs @@ -7,7 +7,7 @@ use serde::{Deserialize, Serialize}; use nodedb_types::Hlc; use crate::metadata_group::compensation::Compensation; -use crate::metadata_group::descriptors::{DescriptorId, DescriptorLease}; +use crate::metadata_group::descriptors::{DescriptorId, DescriptorLease, DrainOwner}; use crate::metadata_group::migration_state::{MigrationCheckpointPayload, MigrationPhaseTag}; /// An entry in the replicated metadata log. @@ -103,8 +103,12 @@ pub enum MetadataEntry { /// Acquire the replicated global descriptor-preparation lease. The first /// unexpired owner wins; contenders observe the winner after this entry is /// applied and retry after its matching release or expiry. + /// + /// `node_id` is the proposing node. The metadata leader reclaims the lease + /// once that node is SWIM-Dead and Raft-silent past its grace. DdlPrepareAcquire { token: u64, + node_id: u64, }, /// Release the descriptor-preparation lease iff `token` still owns it. DdlPrepareRelease { @@ -116,7 +120,8 @@ pub enum MetadataEntry { /// pending-DDL table; nothing is written to the catalog yet. A crash /// after this applies but before the matching finalize leaves the /// pending record for a future reconciliation pass, never orphaned - /// catalog state. + /// catalog state. Applies only while `token` owns the preparation lease, + /// like `DdlPrepared`: a reclaimed owner's late propose is a no-op. DdlPendingPropose { token: u64, objects: Vec, @@ -125,7 +130,8 @@ pub enum MetadataEntry { /// Commit the objects `token`'s `DdlPendingPropose` reserved: replay /// each object's host-side effects, then drop the pending record. /// Idempotent — a record already finalized, or absent, is a no-op, so - /// Raft re-delivery is always safe. + /// Raft re-delivery is always safe. Applies only while `token` owns the + /// preparation lease. DdlPendingFinalize { token: u64, }, @@ -171,6 +177,12 @@ pub enum MetadataEntry { descriptor_id: DescriptorId, up_to_version: u64, expires_at: Hlc, + /// Node that proposed the drain. When it leaves the topology, the + /// cluster ends its drains: no proposer is left to end them. + proposer_node_id: u64, + /// Who holds this drain. Other owners' drains on the same descriptor + /// are independent. + owner: DrainOwner, }, /// End draining on a descriptor. Emitted explicitly on drain /// timeout so the cluster can make progress. On the happy @@ -179,6 +191,8 @@ pub enum MetadataEntry { /// hatch for the failure path. DescriptorDrainEnd { descriptor_id: DescriptorId, + /// The owner whose drain ends. Other owners' drains stay. + owner: DrainOwner, }, /// Cluster-wide CA trust mutation (L.4). Proposed by @@ -350,6 +364,33 @@ pub enum MetadataEntry { spki: [u8; 32], expires_at_ms: u64, }, + + /// Reserve one database id for `node_id`. + /// + /// The id is not carried: every node computes it at apply time as the + /// next id past the replicated database high-watermark, in log order, so + /// all nodes agree on it. `request_id` routes the id back to the waiting + /// allocation on `node_id`. + DatabaseIdReserve { + node_id: u64, + request_id: u64, + }, + + /// A cluster restore point: one consistent instant across every Raft + /// group. Its id is this entry's log index, and the metadata group's place + /// at the point is that index. `hlc` is the point's watermark: a + /// point-in-time restore to it keeps every write committed below it. + /// Every node records the point, then cuts each group it hosts at `hlc`. + RestorePoint { + hlc: u64, + created_at_ms: u64, + }, + + /// A stamped entry the metadata log archiver proposes on the leader when + /// the log holds no recent stamp. Stamps rise with the log index, so the + /// archive covers every stamp up to the newest one it holds. This entry + /// moves that frontier forward on an idle log. It applies nothing. + ArchiveMark, } /// The direction of a join-token lifecycle transition. @@ -425,11 +466,25 @@ pub enum PendingDdlObject { zerompk::FromMessagePack, )] pub enum TopologyChange { - Join { node_id: u64, addr: String }, - Leave { node_id: u64 }, - PromoteToVoter { node_id: u64 }, - StartDecommission { node_id: u64 }, - FinishDecommission { node_id: u64 }, + /// `swim_addr` is the joiner's bound SWIM UDP address, if it runs SWIM. + Join { + node_id: u64, + addr: String, + #[serde(default)] + swim_addr: Option, + }, + Leave { + node_id: u64, + }, + PromoteToVoter { + node_id: u64, + }, + StartDecommission { + node_id: u64, + }, + FinishDecommission { + node_id: u64, + }, } /// Routing-table mutations proposed through the metadata group. diff --git a/nodedb-cluster/src/metadata_group/mod.rs b/nodedb-cluster/src/metadata_group/mod.rs index 86267237c..ba84aaaa3 100644 --- a/nodedb-cluster/src/metadata_group/mod.rs +++ b/nodedb-cluster/src/metadata_group/mod.rs @@ -18,11 +18,15 @@ pub mod migration_recovery; pub mod migration_state; pub mod state; -pub use applier::{CacheApplier, MetadataApplier, NoopMetadataApplier}; +pub use applier::{ + CacheApplier, CommittedMetadata, MetadataApplier, MetadataPayload, NoopMetadataApplier, +}; pub use cache::{MetadataCache, apply_migration_abort, apply_migration_checkpoint}; -pub use codec::{decode_entry, encode_entry}; +pub use codec::{decode_entry, encode_entry, entry_stamp, stamp_entry}; pub use compensation::Compensation; -pub use descriptors::{DescriptorHeader, DescriptorId, DescriptorKind, DescriptorLease}; +pub use descriptors::{ + DescriptorHeader, DescriptorId, DescriptorKind, DescriptorLease, DrainOwner, +}; pub use entry::{MetadataEntry, PendingDdlObject, RoutingChange, TopologyChange}; pub use ids::METADATA_GROUP_ID; pub use migration_recovery::{ diff --git a/nodedb-cluster/src/migration_executor/executor.rs b/nodedb-cluster/src/migration_executor/executor.rs index 5f7f2f1d6..7905ac59a 100644 --- a/nodedb-cluster/src/migration_executor/executor.rs +++ b/nodedb-cluster/src/migration_executor/executor.rs @@ -62,6 +62,8 @@ pub struct MigrationExecutor { pub(super) catalog: Option>, pub(super) metadata_proposer: Option>, pub(super) migration_state: Option, + /// Where each migration this executor runs reports its state. + pub(super) tracker: Arc, } impl MigrationExecutor { @@ -80,9 +82,17 @@ impl MigrationExecutor { catalog: None, metadata_proposer: None, migration_state: None, + tracker: Arc::new(super::MigrationTracker::new()), } } + /// Report every migration's state to `tracker`, which `SHOW MIGRATIONS` + /// and the status route read. + pub fn with_tracker(mut self, tracker: Arc) -> Self { + self.tracker = tracker; + self + } + pub fn with_metadata_proposer(mut self, proposer: Arc) -> Self { self.metadata_proposer = Some(proposer); self @@ -161,10 +171,13 @@ impl MigrationExecutor { "starting vShard migration" ); - super::phases::phase1_base_copy(self, &mut state, source_group, &req, migration_id).await?; - super::phases::phase2_wal_catchup(self, &mut state, source_group, &req, migration_id) - .await?; - super::phases::phase3_cutover(self, &mut state, source_group, &req, migration_id).await?; + self.tracker.record(migration_id, &state); + let phases = self + .run_phases(&mut state, source_group, &req, migration_id) + .await; + // The final state, a failure included, stays visible until GC. + self.tracker.record(migration_id, &state); + phases?; let elapsed = state.elapsed(); let phase = state.phase().clone(); @@ -186,6 +199,21 @@ impl MigrationExecutor { }) } + /// Run the three phases, recording the state after each. + async fn run_phases( + &self, + state: &mut MigrationState, + source_group: u64, + req: &MigrationRequest, + migration_id: MigrationId, + ) -> Result<()> { + super::phases::phase1_base_copy(self, state, source_group, req, migration_id).await?; + self.tracker.record(migration_id, state); + super::phases::phase2_wal_catchup(self, state, source_group, req, migration_id).await?; + self.tracker.record(migration_id, state); + super::phases::phase3_cutover(self, state, source_group, req, migration_id).await + } + pub(super) async fn propose_checkpoint( &self, migration_id: MigrationId, diff --git a/nodedb-cluster/src/migration_executor/phases.rs b/nodedb-cluster/src/migration_executor/phases.rs index 63cde1580..ce9d07b59 100644 --- a/nodedb-cluster/src/migration_executor/phases.rs +++ b/nodedb-cluster/src/migration_executor/phases.rs @@ -256,6 +256,8 @@ pub(super) async fn phase3_cutover( proposer.propose_and_wait(entry).await?; } else { let mut routing = ex.routing.write().unwrap_or_else(|p| p.into_inner()); + // The target is named before its election, so the hint carries no + // term and fills only a hint that holds none. routing.set_leader(group_id, req.target_node); } @@ -335,6 +337,8 @@ mod tests { node.election_deadline_override(Instant::now() - Duration::from_millis(1)); } mr.tick().unwrap(); + // The no-op commits once the disk holds it. + mr.wait_all_durable_blocking(); for (gid, ready) in mr.tick().unwrap().groups { if let Some(last) = ready.committed_entries.last() { mr.advance_applied(gid, last.index).unwrap(); diff --git a/nodedb-cluster/src/migration_executor/tracker.rs b/nodedb-cluster/src/migration_executor/tracker.rs index 6d902a8d9..16d0f3be3 100644 --- a/nodedb-cluster/src/migration_executor/tracker.rs +++ b/nodedb-cluster/src/migration_executor/tracker.rs @@ -3,11 +3,13 @@ use std::sync::Mutex; use std::time::Duration; +use crate::metadata_group::migration_state::MigrationId; use crate::migration::MigrationState; -/// Track active migrations across the cluster. +/// The migrations this node's executor ran, by migration id, with the state +/// each reached last. pub struct MigrationTracker { - active: Mutex>, + active: Mutex>, } impl MigrationTracker { @@ -17,21 +19,26 @@ impl MigrationTracker { } } - pub fn add(&self, state: MigrationState) { + /// Record the current state of migration `id`, replacing the state + /// recorded for it before. + pub fn record(&self, id: MigrationId, state: &MigrationState) { let mut active = self.active.lock().unwrap_or_else(|p| p.into_inner()); - active.push(state); + match active.iter_mut().find(|(recorded, _)| *recorded == id) { + Some((_, recorded)) => *recorded = state.clone(), + None => active.push((id, state.clone())), + } } pub fn active_count(&self) -> usize { let active = self.active.lock().unwrap_or_else(|p| p.into_inner()); - active.iter().filter(|s| s.is_active()).count() + active.iter().filter(|(_, s)| s.is_active()).count() } pub fn snapshot(&self) -> Vec { let active = self.active.lock().unwrap_or_else(|p| p.into_inner()); active .iter() - .map(|s| MigrationSnapshot { + .map(|(_, s)| MigrationSnapshot { vshard_id: s.vshard_id(), phase: format!("{:?}", s.phase()), elapsed_ms: s.elapsed().map(|d| d.as_millis() as u64).unwrap_or(0), @@ -42,7 +49,7 @@ impl MigrationTracker { pub fn gc(&self, max_age: Duration) { let mut active = self.active.lock().unwrap_or_else(|p| p.into_inner()); - active.retain(|s| s.is_active() || s.elapsed().map(|d| d < max_age).unwrap_or(true)); + active.retain(|(_, s)| s.is_active() || s.elapsed().map(|d| d < max_age).unwrap_or(true)); } } @@ -71,12 +78,19 @@ mod tests { let tracker = MigrationTracker::new(); assert_eq!(tracker.active_count(), 0); + let id = uuid::Uuid::new_v4(); let mut state = MigrationState::new(0, 0, 1, 1, 2, 500_000); state.start_base_copy(100); - tracker.add(state); + tracker.record(id, &state); assert_eq!(tracker.active_count(), 1); assert_eq!(tracker.snapshot().len(), 1); assert!(tracker.snapshot()[0].is_active); + + // A later record of the same migration replaces its state. + state.complete(10); + tracker.record(id, &state); + assert_eq!(tracker.snapshot().len(), 1); + assert_eq!(tracker.active_count(), 0); } } diff --git a/nodedb-cluster/src/multi_raft/conf_change.rs b/nodedb-cluster/src/multi_raft/conf_change.rs index bb472c4d5..cca910be2 100644 --- a/nodedb-cluster/src/multi_raft/conf_change.rs +++ b/nodedb-cluster/src/multi_raft/conf_change.rs @@ -92,10 +92,11 @@ impl MultiRaft { let data = change.to_entry_data()?; let log_index = node.propose(data)?; - // A single-voter group self-commits inside `propose`: - // its `commit_index` is bumped to the new `log_index` - // before we return. Detecting this is the one safe - // trigger for an inline apply. + // A single-voter group commits once its own disk holds the + // entry. When that is already so inside `propose`, the + // `commit_index` reaches `log_index` before we return, and the + // change applies inline. Otherwise it applies with the + // group's other committed entries on a later tick. let committed_immediately = node.commit_index() >= log_index; (log_index, committed_immediately) }; @@ -187,7 +188,10 @@ impl MultiRaft { } ConfChangeType::RemoveLearner => { // Non-voting removal: safe at any time — learners are not in - // quorum, commit, or election paths. + // quorum, commit, or election paths. The departing learner is + // told its removal committed, as a departing voter is, so its + // routing view stops listing it as a replica of the group. + node.notify_removed_peer(change.node_id); node.remove_learner(change.node_id); self.routing .write() @@ -293,13 +297,15 @@ mod tests { fn propose_promote_requires_learner_caught_up() { let (mut mr, _dir) = new_mr(1, &[0]); // Force election: single-voter group becomes leader on first tick, - // and its no-op commits immediately. + // and its no-op commits once its disk holds it. for node in mr.groups.values_mut() { node.election_deadline_override( std::time::Instant::now() - std::time::Duration::from_millis(1), ); } mr.tick().unwrap(); + mr.wait_all_durable_blocking(); + mr.tick().unwrap(); let node = mr.groups.get(&0).unwrap(); assert_eq!(node.role(), nodedb_raft::NodeRole::Leader); assert!( @@ -350,6 +356,8 @@ mod tests { ); } mr.tick().unwrap(); + mr.wait_all_durable_blocking(); + mr.tick().unwrap(); mr.apply_conf_change( 0, @@ -367,6 +375,8 @@ mod tests { term: mr.groups.get(&0).unwrap().current_term(), success: true, last_log_index: last, + round: nodedb_raft::node::leader_lease::UNTRACKED_ROUND, + needs_snapshot: false, }; mr.handle_append_entries_response(0, 2, &resp).unwrap(); assert_eq!(mr.match_index_for(0, 2), Some(last)); diff --git a/nodedb-cluster/src/multi_raft/core.rs b/nodedb-cluster/src/multi_raft/core.rs index d8c201b96..beaefed7d 100644 --- a/nodedb-cluster/src/multi_raft/core.rs +++ b/nodedb-cluster/src/multi_raft/core.rs @@ -7,13 +7,11 @@ use std::path::PathBuf; use std::sync::{Arc, RwLock}; use std::time::Duration; -use tracing::info; - -use nodedb_raft::node::RaftConfig; use nodedb_raft::{RaftNode, Ready}; -use crate::error::{ClusterError, Result}; -use crate::raft_storage::RedbLogStorage; +use crate::applied_watcher::GroupAppliedWatchers; +use crate::error::Result; +use crate::group_disk::StagedLogStorage; use crate::routing::RoutingTable; /// Multi-Raft coordinator managing multiple Raft groups on a single node. @@ -27,7 +25,10 @@ pub struct MultiRaft { /// This node's ID. pub(super) node_id: u64, /// Raft groups hosted on this node (group_id → RaftNode). - pub(super) groups: HashMap>, + /// + /// Each group's storage stages its writes. The group's own writer thread + /// makes them durable, never under this struct's lock. + pub(super) groups: HashMap>, /// Routing table (vShard → group mapping). /// /// This is the SAME `Arc>` held by @@ -35,6 +36,9 @@ pub struct MultiRaft { /// conf-changes applied here (via `apply_conf_change`) write THROUGH to /// the one table the query/data plane reads. Raft is the convergence /// mechanism on every applying node (leader and follower). + /// + /// Lock order: this struct's mutex first, then this routing guard. + /// See the lock-order section on [`RoutingTable`]. pub(super) routing: Arc>, /// Default election timeout range. pub(super) election_timeout_min: Duration, @@ -43,7 +47,7 @@ pub struct MultiRaft { pub(super) heartbeat_interval: Duration, /// Auto-compaction threshold applied to every group created on this /// node. `None` (default) disables auto-compaction. See - /// [`RaftConfig::log_compaction_threshold`]. + /// [`nodedb_raft::node::RaftConfig::log_compaction_threshold`]. pub(super) log_compaction_threshold: Option, /// Data directory for persistent Raft log storage. pub(super) data_dir: PathBuf, @@ -51,8 +55,31 @@ pub struct MultiRaft { /// Compaction is deferred for any group with an active transfer so the /// snapshot boundary never advances mid-transfer. pub(super) in_flight_snapshots: Arc, + /// Per-group gate that orders committed-entry applies against snapshot + /// installs. + pub(super) apply_gates: Arc, + /// Per-group ceiling on log compaction: the highest index an archiver + /// holds a copy of. A group with a ceiling never compacts past it, so no + /// entry leaves the log before it is archived. + pub(super) compaction_ceilings: HashMap>, + /// The node clock the metadata entries this node stamps as leader take + /// their stamp from. `None` proposes them unstamped. + pub(super) metadata_clock: Option>, + /// The per-group applied watchers. A group mounted here seeds its + /// watcher with the applied index it restored, because entries at or + /// below it are never delivered again. `None` seeds nothing. + pub(super) applied_watchers: Option>, + /// Decides whether a data group mounted here takes a snapshot before any + /// log entry. `None` requires none. + pub(super) snapshot_requirement: Option, } +/// Whether this node's replica of a data group must take a snapshot before +/// it takes log entries. Called with the group id and the first index the +/// Calvin sequencer log still holds here, under the `MultiRaft` lock: it +/// must not take that lock. +pub type SnapshotRequirement = Arc bool + Send + Sync>; + /// Aggregated output from all Raft groups after a tick. #[derive(Debug, Default)] pub struct MultiRaftReady { @@ -107,7 +134,49 @@ impl MultiRaft { in_flight_snapshots: Arc::new( crate::raft_loop::in_flight_snapshots::InFlightSnapshots::default(), ), + apply_gates: Arc::new(crate::raft_loop::apply_gate::GroupApplyGates::new()), + compaction_ceilings: HashMap::new(), + metadata_clock: None, + applied_watchers: None, + snapshot_requirement: None, + } + } + + /// Install the snapshot requirement for data groups, and apply it to + /// every data group mounted already. + pub fn set_snapshot_requirement(&mut self, requirement: SnapshotRequirement) { + let sequencer_first = self.sequencer_first_available(); + for (&group_id, node) in &mut self.groups { + if is_data_group(group_id) { + node.set_snapshot_required(requirement(group_id, sequencer_first)); + } } + self.snapshot_requirement = Some(requirement); + } + + /// The auto-compaction threshold every group on this node runs with, + /// `None` when logs are never compacted. + pub fn log_compaction_threshold(&self) -> Option { + self.log_compaction_threshold + } + + /// The first index the Calvin sequencer log holds here, `1` when this + /// node hosts no sequencer replica. + pub(super) fn sequencer_first_available(&self) -> u64 { + self.groups + .get(&crate::calvin::SEQUENCER_GROUP_ID) + .map_or(1, |node| node.first_available_index()) + } + + /// Install the per-group applied watchers. Every group mounted here, now + /// or later, starts its watcher at the applied index it restored: the + /// durable applied floor, whose entries applied before the restart and + /// are never delivered again. + pub fn set_applied_watchers(&mut self, watchers: Arc) { + for (&group_id, node) in &self.groups { + seed_watcher(&watchers, group_id, node.last_applied()); + } + self.applied_watchers = Some(watchers); } /// Configure election timeout range. @@ -129,6 +198,12 @@ impl MultiRaft { self.election_timeout_min } + /// The longest election timeout a group on this node waits before it + /// campaigns. + pub fn election_timeout_max(&self) -> Duration { + self.election_timeout_max + } + /// How often a leader on this node sends heartbeats. pub fn heartbeat_interval(&self) -> Duration { self.heartbeat_interval @@ -136,7 +211,7 @@ impl MultiRaft { /// Configure the auto-compaction threshold for every group created on /// this node. `None` disables auto-compaction (the default). See - /// [`RaftConfig::log_compaction_threshold`]. + /// [`nodedb_raft::node::RaftConfig::log_compaction_threshold`]. pub fn with_log_compaction_threshold(mut self, threshold: Option) -> Self { self.log_compaction_threshold = threshold; self @@ -167,63 +242,40 @@ impl MultiRaft { self.add_group_inner(group_id, voters, learners, true) } - fn add_group_inner( - &mut self, - group_id: u64, - peers: Vec, - learners: Vec, - starts_as_learner: bool, - ) -> Result<()> { - let config = RaftConfig { - node_id: self.node_id, - group_id, - peers, - learners, - observers: vec![], - starts_as_learner, - starts_as_observer: false, - election_timeout_min: self.election_timeout_min, - election_timeout_max: self.election_timeout_max, - heartbeat_interval: self.heartbeat_interval, - log_compaction_threshold: self.log_compaction_threshold, - }; + /// A ticket for every storage write `group_id` staged so far, or `None` + /// when all of them are durable or the group is not mounted. A caller + /// takes it right after the Raft call whose writes a reply depends on, + /// and awaits it after it releases this struct's lock. + pub fn durability_ticket(&self, group_id: u64) -> Option { + self.groups.get(&group_id)?.storage().ticket() + } - let storage_path = self.data_dir.join(format!("raft/group-{group_id}.redb")); - let storage = RedbLogStorage::open(&storage_path).map_err(|e| ClusterError::Transport { - detail: format!("failed to open raft storage for group {group_id}: {e}"), - })?; - let mut node = RaftNode::new(config, storage); - // Reload durable state (HardState + log) from redb before mounting the - // group. On a restart this recovers the persisted term/voted_for — so - // a restarted voter cannot forget its vote and double-vote — AND the - // persisted log entries, so the node does not depend on full - // re-replication from the leader to recover its log. On a fresh group - // the storage is empty and this is a no-op (default HardState, empty - // log). Also resets the election timeout. - node.restore()?; - self.groups.insert(group_id, node); - - info!( - node = self.node_id, - group = group_id, - as_learner = starts_as_learner, - path = %storage_path.display(), - "added raft group with persistent storage" - ); - Ok(()) + /// Block until every write each group staged so far is durable. For boot + /// and tests, off the async threads: a running node waits on a + /// [`Self::durability_ticket`] instead. + pub fn wait_all_durable_blocking(&self) { + for node in self.groups.values() { + if let Some(ticket) = node.storage().ticket() { + ticket.wait_blocking(); + } + } } /// Tick all Raft groups. Returns aggregated ready output. /// - /// Any HardState staged by a tick (an election term bump + self-vote from - /// an election timeout) is durably persisted BEFORE the aggregated `Ready` - /// — and therefore the vote requests it carries — is returned for - /// dispatch. A persist failure aborts the tick so the caller never sends - /// vote requests for a term that was not made durable. + /// Any HardState a tick changes (an election term bump + self-vote from + /// an election timeout) is staged on the group's disk before the + /// aggregated `Ready` returns. The caller awaits the group's + /// [`Self::durability_ticket`] before it sends the vote requests that + /// `Ready` carries, so no vote request leaves for a term that is not + /// durable. A staging error aborts the tick. pub fn tick(&mut self) -> Result { let mut ready = MultiRaftReady::default(); for (&group_id, node) in &mut self.groups { + // A leader counts the entries its disk made durable since the + // last tick, so a commit the disk completed lands in this Ready. + node.on_storage_progress(); node.tick(); node.persist_hard_state_if_dirty()?; let r = node.take_ready(); @@ -259,6 +311,12 @@ impl MultiRaft { self.in_flight_snapshots.clone() } + /// Clone of the per-group apply gates. Every applier of committed entries + /// and every snapshot install of this node goes through them. + pub fn apply_gates(&self) -> Arc { + self.apply_gates.clone() + } + pub fn group_count(&self) -> usize { self.groups.len() } @@ -277,11 +335,26 @@ impl MultiRaft { } /// Mutable access to the underlying Raft groups (for testing / bootstrap). - pub fn groups_mut(&mut self) -> &mut HashMap> { + pub fn groups_mut(&mut self) -> &mut HashMap> { &mut self.groups } } +/// Start `group_id`'s watcher at `restored`, the applied index the group +/// restored from storage. A fresh group restores 0 and seeds nothing. +pub(super) fn seed_watcher(watchers: &GroupAppliedWatchers, group_id: u64, restored: u64) { + if restored > 0 { + watchers.bump(group_id, restored); + } +} + +/// Whether `group_id` is a data group: neither the metadata group nor the +/// Calvin sequencer group. +pub(super) fn is_data_group(group_id: u64) -> bool { + group_id != crate::metadata_group::METADATA_GROUP_ID + && group_id != crate::calvin::SEQUENCER_GROUP_ID +} + // Re-export LogEntry so callers of `read_committed_entries` can name the type. pub use nodedb_raft::LogEntry; @@ -308,6 +381,10 @@ mod tests { node.election_deadline_override(Instant::now() - Duration::from_millis(1)); } + // The first tick elects every group's leader. Each leader's no-op + // commits once its disk holds it, and the next tick delivers it. + mr.tick().unwrap(); + mr.wait_all_durable_blocking(); let ready = mr.tick().unwrap(); assert_eq!(ready.groups.len(), 5); } @@ -339,6 +416,48 @@ mod tests { assert!(idx > 0); } + #[test] + fn compaction_never_passes_the_ceiling() { + use std::sync::atomic::{AtomicU64, Ordering}; + + let dir = tempfile::tempdir().unwrap(); + let rt = RoutingTable::uniform(1, &[1], 1); + let mut mr = + MultiRaft::new(1, rt, dir.path().to_path_buf()).with_log_compaction_threshold(Some(1)); + mr.add_group(0, vec![]).unwrap(); + for node in mr.groups.values_mut() { + node.election_deadline_override(Instant::now() - Duration::from_millis(1)); + } + let ceiling = std::sync::Arc::new(AtomicU64::new(0)); + mr.set_compaction_ceiling(0, std::sync::Arc::clone(&ceiling)); + let apply = |mr: &mut MultiRaft| { + mr.wait_all_durable_blocking(); + for (gid, ready) in mr.tick().unwrap().groups { + if let Some(last) = ready.committed_entries.last() { + mr.advance_applied(gid, last.index).unwrap(); + mr.save_applied_index(gid, last.index).unwrap(); + } + } + }; + mr.tick().unwrap(); + apply(&mut mr); + for i in 0..5u8 { + mr.propose_to_group(0, vec![i]).unwrap(); + } + for _ in 0..3 { + apply(&mut mr); + } + let applied = mr.groups[&0].last_applied(); + assert!(applied >= 5, "applied {applied}"); + + assert!(!mr.maybe_compact_group(0, applied).unwrap()); + assert_eq!(mr.first_available_index(0), Some(1)); + + ceiling.store(3, Ordering::Release); + assert!(mr.maybe_compact_group(0, applied).unwrap()); + assert_eq!(mr.first_available_index(0), Some(4)); + } + #[test] fn add_group_as_learner_starts_in_learner_role() { use nodedb_raft::NodeRole; diff --git a/nodedb-cluster/src/multi_raft/membership.rs b/nodedb-cluster/src/multi_raft/membership.rs index d2936b240..59c83fa5e 100644 --- a/nodedb-cluster/src/multi_raft/membership.rs +++ b/nodedb-cluster/src/multi_raft/membership.rs @@ -15,7 +15,9 @@ //! RaftNode state, used by the join flow to decide redirect vs admit. //! - `group_role_is_leader(group)`: cheap leader-check helper. -use nodedb_raft::NodeRole; +use nodedb_raft::{NodeRole, RaftNode}; + +use crate::group_disk::StagedLogStorage; use crate::error::{ClusterError, Result}; @@ -62,6 +64,15 @@ impl MultiRaft { .unwrap_or(0) } + /// The leader of `group_id` this node knows, and this node's term for + /// the group: `(leader, term)`. `(0, 0)` when the group is not hosted + /// here. The leader is `0` while none is known. + pub fn group_leader_at_term(&self, group_id: u64) -> (u64, u64) { + self.groups + .get(&group_id) + .map_or((0, 0), |n| (n.leader_id(), n.current_term())) + } + /// Whether this node is currently the leader of `group_id`. pub fn group_role_is_leader(&self, group_id: u64) -> bool { self.groups @@ -70,6 +81,25 @@ impl MultiRaft { .unwrap_or(false) } + /// `AppendEntries` responses this leader counted from `peer` in + /// `group_id` in the current term. `None` when this node does not lead + /// the group. + pub fn peer_ack_count(&self, group_id: u64, peer: u64) -> Option { + self.groups.get(&group_id)?.peer_ack_count(peer) + } + + /// Whether `peer` holds every entry this leader has committed in + /// `group_id`, and did not ask for a snapshot last. `false` when this + /// node does not lead the group. + pub fn peer_caught_up(&self, group_id: u64, peer: u64) -> bool { + let Some(node) = self.groups.get(&group_id) else { + return false; + }; + node.role() == NodeRole::Leader + && node.match_index_for(peer).unwrap_or(0) >= node.commit_index() + && !node.peer_awaits_snapshot(peer) + } + /// Initiate a leadership transfer for `group_id` to `target`. /// /// Delegates to `RaftNode::transfer_leadership`. Returns @@ -83,6 +113,35 @@ impl MultiRaft { .ok_or(ClusterError::GroupNotFound { group_id })?; node.transfer_leadership(target).map_err(ClusterError::Raft) } + + /// Hand this node's leadership of `group_id` to another voter that holds + /// every committed entry and did not ask for a snapshot last. Returns the + /// target. `None` when this node does not lead the group, no voter + /// qualifies, or the transfer did not start. + pub fn hand_off_leadership(&mut self, group_id: u64) -> Option { + let membership = self.group_membership(group_id)?; + if membership.leader_id != self.node_id { + return None; + } + let target = membership + .voters + .iter() + .copied() + .find(|&voter| voter != self.node_id && self.peer_caught_up(group_id, voter))?; + self.transfer_leadership(group_id, target).ok()?; + Some(target) + } + + /// Stop hosting `group_id`: take its replica out, and return it. + /// + /// Dropping the returned replica closes its log, which can write to + /// disk, so the caller drops it off the async threads. The log file stays + /// on disk with the applied index the Data Plane state matches, so a + /// later mount of the group resumes from it. The group's apply gate stays + /// too. `None` when the group is not hosted here. + pub fn unmount_group(&mut self, group_id: u64) -> Option> { + self.groups.remove(&group_id) + } } #[cfg(test)] diff --git a/nodedb-cluster/src/multi_raft/membership_sync.rs b/nodedb-cluster/src/multi_raft/membership_sync.rs new file mode 100644 index 000000000..9d92af0e0 --- /dev/null +++ b/nodedb-cluster/src/multi_raft/membership_sync.rs @@ -0,0 +1,96 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Bring a group's Raft membership in line with the routing table. +//! +//! A snapshot install covers conf changes this node never applied. The +//! routing table the install wrote holds their outcome, so the group's voter +//! and learner sets are set from it. + +use crate::error::{ClusterError, Result}; + +use super::core::MultiRaft; + +impl MultiRaft { + /// Set `group_id`'s voters and learners to its routing entry. + /// + /// A voter the routing lists and this node holds as a learner is + /// promoted. This node's own membership is never removed here: a node + /// the routing drops learns that from the next committed conf change. + pub fn sync_group_membership_from_routing(&mut self, group_id: u64) -> Result<()> { + let self_id = self.node_id; + let (members, learners) = { + let routing = self.routing.read().unwrap_or_else(|p| p.into_inner()); + let Some(info) = routing.group_info(group_id) else { + return Ok(()); + }; + (info.members.clone(), info.learners.clone()) + }; + let node = self + .groups + .get_mut(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + + if members.contains(&self_id) { + node.promote_self_to_voter(); + } + for &member in members.iter().filter(|&&id| id != self_id) { + if node.learners().contains(&member) { + node.promote_learner(member); + } else { + node.add_peer(member); + } + } + let stale_voters: Vec = node + .peers() + .iter() + .copied() + .filter(|id| !members.contains(id)) + .collect(); + for voter in stale_voters { + node.remove_peer(voter); + } + for &learner in learners.iter().filter(|&&id| id != self_id) { + node.add_learner(learner); + } + let stale_learners: Vec = node + .learners() + .iter() + .copied() + .filter(|id| !learners.contains(id) && !members.contains(id)) + .collect(); + for learner in stale_learners { + node.remove_learner(learner); + } + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use std::path::Path; + + use super::*; + use crate::routing::RoutingTable; + + fn multi_raft(dir: &Path) -> MultiRaft { + let mut mr = MultiRaft::new(1, RoutingTable::uniform(1, &[1, 2], 2), dir.to_path_buf()); + mr.add_group(0, vec![2]).unwrap(); + mr + } + + #[test] + fn voters_and_learners_follow_the_routing_entry() { + let dir = tempfile::tempdir().unwrap(); + let mut mr = multi_raft(dir.path()); + { + let routing = mr.routing(); + let mut rt = routing.write().unwrap(); + rt.set_group_members(0, vec![1, 3]); + rt.set_group_learners(0, vec![4]); + } + mr.sync_group_membership_from_routing(0).unwrap(); + let node = mr.groups_mut().get(&0).unwrap(); + assert_eq!(node.peers(), &[3]); + assert_eq!(node.learners(), &[4]); + } +} diff --git a/nodedb-cluster/src/multi_raft/mod.rs b/nodedb-cluster/src/multi_raft/mod.rs index e2c3ddc55..e285def53 100644 --- a/nodedb-cluster/src/multi_raft/mod.rs +++ b/nodedb-cluster/src/multi_raft/mod.rs @@ -16,15 +16,22 @@ //! proper voter / learner / promotion semantics. //! - [`membership`]: learner catch-up / promotion helpers //! (`commit_index_for`, `ready_learners`, etc.) driven by the tick loop. +//! - [`membership_sync`]: set a group's voters and learners from routing +//! after a snapshot install. //! - [`read_index`]: leadership confirmation for linearizable reads. pub mod conf_change; pub mod core; pub mod membership; +pub mod membership_sync; +pub mod mount; +pub mod peer_contact; pub mod proposals; pub mod read_index; pub mod rpc_dispatch; pub mod status; -pub use core::{MultiRaft, MultiRaftReady}; +pub use core::{MultiRaft, MultiRaftReady, SnapshotRequirement}; +pub use mount::{GroupMountSpec, OpenedGroup}; +pub use peer_contact::PeerAckSample; pub use status::{GroupMembership, GroupStatus}; diff --git a/nodedb-cluster/src/multi_raft/mount.rs b/nodedb-cluster/src/multi_raft/mount.rs new file mode 100644 index 000000000..3f277a650 --- /dev/null +++ b/nodedb-cluster/src/multi_raft/mount.rs @@ -0,0 +1,156 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Mounting a Raft group: the disk work apart from the `MultiRaft` lock. +//! +//! A mount opens the group's redb log and reloads its hard state and log. +//! That is disk work, so a mount on a running node takes three steps: +//! 1. [`MultiRaft::mount_spec`] under the lock copies what the group needs; +//! 2. [`GroupMountSpec::open`] off the async threads and without the lock +//! opens and restores the group; +//! 3. [`MultiRaft::insert_opened`] under the lock mounts it. +//! +//! Boot and tests mount through [`MultiRaft::add_group`], which runs the +//! three steps in a row. + +use std::path::PathBuf; +use std::sync::Arc; +use std::sync::atomic::AtomicU64; + +use nodedb_raft::RaftNode; +use nodedb_raft::node::RaftConfig; +use tracing::info; + +use crate::error::{ClusterError, Result}; +use crate::group_disk::StagedLogStorage; + +use super::core::{MultiRaft, seed_watcher}; + +/// What a group's mount needs from its `MultiRaft`. +pub struct GroupMountSpec { + config: RaftConfig, + storage_path: PathBuf, + compaction_ceiling: Option>, +} + +/// A group opened and restored from its disk, ready to mount. +pub struct OpenedGroup { + node: RaftNode, + storage_path: PathBuf, +} + +impl GroupMountSpec { + /// The group this spec mounts. + pub fn group_id(&self) -> u64 { + self.config.group_id + } + + /// Open the group's log and reload its hard state and entries. Blocks on + /// disk: call it off the async threads. + pub fn open(self) -> Result { + let group_id = self.config.group_id; + let storage = StagedLogStorage::open(group_id, &self.storage_path).map_err(|e| { + ClusterError::Transport { + detail: format!("failed to open raft storage for group {group_id}: {e}"), + } + })?; + let mut node = RaftNode::new(self.config, storage); + if let Some(ceiling) = self.compaction_ceiling { + node.set_compaction_ceiling(ceiling); + } + // Reload durable state (HardState + log) before mounting the group. + // On a restart this recovers the persisted term/voted_for, so a + // restarted voter cannot forget its vote and double-vote, and the + // persisted log entries, so the node does not depend on full + // re-replication from the leader. On a fresh group the storage is + // empty and this is a no-op. Also resets the election timeout. + node.restore()?; + Ok(OpenedGroup { + node, + storage_path: self.storage_path, + }) + } +} + +impl MultiRaft { + /// What mounting `group_id` needs: its Raft config and its log path. + /// Touches no disk. + /// + /// `peers` are the other voters. A learner-start group lists every voter + /// in `peers` and the other learners in `learners`. + pub fn mount_spec( + &self, + group_id: u64, + peers: Vec, + learners: Vec, + starts_as_learner: bool, + ) -> GroupMountSpec { + GroupMountSpec { + config: RaftConfig { + node_id: self.node_id, + group_id, + peers, + learners, + observers: vec![], + starts_as_learner, + starts_as_observer: false, + election_timeout_min: self.election_timeout_min, + election_timeout_max: self.election_timeout_max, + heartbeat_interval: self.heartbeat_interval, + log_compaction_threshold: self.log_compaction_threshold, + }, + storage_path: crate::raft_bootstrap::group_log_path(&self.data_dir, group_id), + compaction_ceiling: self.compaction_ceilings.get(&group_id).map(Arc::clone), + } + } + + /// Mount an opened group. Touches no disk. Returns the group back when + /// it is mounted already: the caller drops it off the async threads. + pub fn insert_opened(&mut self, opened: OpenedGroup) -> Option { + let group_id = opened.node.group_id(); + if self.groups.contains_key(&group_id) { + return Some(opened); + } + if let Some(watchers) = self.applied_watchers.as_ref() { + seed_watcher(watchers, group_id, opened.node.last_applied()); + } + let as_learner = opened.node.role() == nodedb_raft::NodeRole::Learner; + let mut node = opened.node; + // Decided before the group takes its first entry: a replica with no + // Calvin state to resume from would otherwise replay a log that does + // not hold the Calvin transactions the sequencer compacted away. + if super::core::is_data_group(group_id) + && let Some(requirement) = self.snapshot_requirement.as_ref() + { + node.set_snapshot_required(requirement(group_id, self.sequencer_first_available())); + } + self.groups.insert(group_id, node); + self.apply_gates.mount(group_id); + info!( + node = self.node_id, + group = group_id, + as_learner, + path = %opened.storage_path.display(), + "added raft group with persistent storage" + ); + None + } + + /// Open and mount a group in one call. Blocks on disk: boot and tests + /// only. A running node mounts through [`Self::mount_spec`], + /// [`GroupMountSpec::open`] and [`Self::insert_opened`]. + pub(super) fn add_group_inner( + &mut self, + group_id: u64, + peers: Vec, + learners: Vec, + starts_as_learner: bool, + ) -> Result<()> { + let opened = self + .mount_spec(group_id, peers, learners, starts_as_learner) + .open()?; + // A group mounted already keeps its replica; the one opened here + // closes as it drops. + drop(self.insert_opened(opened)); + Ok(()) + } +} diff --git a/nodedb-cluster/src/multi_raft/peer_contact.rs b/nodedb-cluster/src/multi_raft/peer_contact.rs new file mode 100644 index 000000000..d858ebca4 --- /dev/null +++ b/nodedb-cluster/src/multi_raft/peer_contact.rs @@ -0,0 +1,58 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Leader-side peer response counts for a hosted group. + +use std::time::Duration; + +use nodedb_raft::{NodeRole, StalenessVerdict}; + +use crate::multi_raft::core::MultiRaft; + +/// One sample of a leader's per-peer `AppendEntries` response counts. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct PeerAckSample { + /// Term the counts belong to. Counts restart at zero in every term. + pub term: u64, + /// `(peer, responses received this term)` for every voter and learner. + pub acks: Vec<(u64, u64)>, +} + +impl MultiRaft { + /// This node's term in `group_id` while it leads the group, else `None`. + /// Reads one group only. + pub fn leader_term(&self, group_id: u64) -> Option { + self.groups + .get(&group_id) + .filter(|node| node.role() == NodeRole::Leader) + .map(|node| node.current_term()) + } + + /// Whether this node heard from the leader of `group_id` within + /// `window`. The leader itself always has. Unlike a staleness bound, a + /// replica that is merely behind on apply still counts as in contact. + /// False for a group not hosted here. + pub fn leader_contact_within(&self, group_id: u64, window: Duration) -> bool { + self.groups + .get(&group_id) + .is_some_and(|node| node.staleness_verdict(window) != StalenessVerdict::NoRecentContact) + } + + /// Response counts from every peer of `group_id`. `None` when the group + /// is not hosted here or this node does not lead it. + pub fn leader_peer_acks(&self, group_id: u64) -> Option { + let node = self.groups.get(&group_id)?; + if node.role() != NodeRole::Leader { + return None; + } + let acks = node + .voters() + .iter() + .chain(node.learners()) + .filter_map(|&peer| node.peer_ack_count(peer).map(|count| (peer, count))) + .collect(); + Some(PeerAckSample { + term: node.current_term(), + acks, + }) + } +} diff --git a/nodedb-cluster/src/multi_raft/proposals.rs b/nodedb-cluster/src/multi_raft/proposals.rs index 5dc69b9c0..6c5e4ef0e 100644 --- a/nodedb-cluster/src/multi_raft/proposals.rs +++ b/nodedb-cluster/src/multi_raft/proposals.rs @@ -2,7 +2,10 @@ //! Leadership checks, proposals and log access on hosted Raft groups. +use nodedb_raft::RaftNode; + use crate::error::{ClusterError, Result}; +use crate::group_disk::StagedLogStorage; use crate::multi_raft::core::MultiRaft; impl MultiRaft { @@ -122,6 +125,65 @@ impl MultiRaft { .map(|n| n.first_available_index()) } + /// Make this node's replica of `group_id` refuse log entries until a + /// snapshot installs, or lift that. Its leader answers the refusal with + /// a snapshot. Returns whether this node hosts the group. + pub fn set_snapshot_required(&mut self, group_id: u64, required: bool) -> bool { + match self.groups.get_mut(&group_id) { + Some(node) => { + node.set_snapshot_required(required); + true + } + None => false, + } + } + + /// Decide again whether this node's replica of data group `group_id` + /// requires a snapshot, by the installed requirement. Run after a + /// snapshot of `group_id` installs: the host recorded the state it + /// brought. A sequencer snapshot moves the first index the sequencer log + /// holds, so every data group mounted here is decided again. + pub fn refresh_snapshot_requirement(&mut self, group_id: u64) { + let Some(requirement) = self.snapshot_requirement.clone() else { + return; + }; + let groups: Vec = if group_id == crate::calvin::SEQUENCER_GROUP_ID { + self.groups + .keys() + .copied() + .filter(|&g| super::core::is_data_group(g)) + .collect() + } else if super::core::is_data_group(group_id) { + vec![group_id] + } else { + return; + }; + let sequencer_first = self.sequencer_first_available(); + for group_id in groups { + if let Some(node) = self.groups.get_mut(&group_id) { + node.set_snapshot_required(requirement(group_id, sequencer_first)); + } + } + } + + /// The first index this node's sequencer log holds, once the log has a + /// known start: it holds an entry or the boundary of a snapshot. `None` + /// while it holds neither, or the group is not mounted here. A replica + /// with an empty log learns only from its leader whether the log + /// reaches it from index 1 or from a snapshot. + pub fn sequencer_log_start(&self) -> Option { + let node = self.groups.get(&crate::calvin::SEQUENCER_GROUP_ID)?; + (node.last_log_index() >= 1).then(|| node.first_available_index()) + } + + /// Whether this node's replica of `group_id` refuses log entries until a + /// snapshot installs. + pub fn snapshot_required(&self, group_id: u64) -> bool { + self.groups + .get(&group_id) + .is_some_and(|node| node.snapshot_required()) + } + /// Auto-compact a group's log if its configured threshold has been /// reached, given the DATA-PLANE applied watermark `applied_index`. /// @@ -144,6 +206,157 @@ impl MultiRaft { let Some(node) = self.groups.get_mut(&group_id) else { return Ok(false); }; + // A sequencer replica that installs a snapshot never receives the + // Calvin inputs the snapshot covers, and its schedulers lose them. + // When every voter does, the inputs are gone from the cluster. So no + // replica compacts past an entry a voter's log may lack: no voter + // ever takes a sequencer snapshot, under this leader or a later one. + let applied_index = if group_id == crate::calvin::SEQUENCER_GROUP_ID { + applied_index.min(node.replicated_floor()) + } else { + applied_index + }; Ok(node.maybe_compact_log(applied_index)?) } + + /// Stamp every metadata entry this node appends as leader from `clock`, + /// the node HLC. + pub fn set_metadata_clock(&mut self, clock: std::sync::Arc) { + self.metadata_clock = Some(clock); + } + + /// Propose the encoded metadata entry `data` to group 0, stamped with + /// the metadata clock when one is set. Returns its log index. + /// + /// The stamp is taken under the `MultiRaft` lock, on the leader, as the + /// entry is appended. It is above the metadata clock's reading and above + /// every stamp the log holds. Every node folds each applied stamp into + /// its clock, and restores the folded high-water at boot and at a + /// snapshot install. So stamps rise with the log index, and every entry + /// stamped at or below a committed entry's stamp sits at a lower index. + pub fn propose_stamped_metadata(&mut self, data: &[u8]) -> Result { + let group_id = crate::metadata_group::METADATA_GROUP_ID; + let data = match &self.metadata_clock { + Some(clock) => { + let node = self + .groups + .get(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + let floor = newest_log_stamp(node).map_or(0, |stamp| stamp.saturating_add(1)); + let stamp = clock.now().wall_ns.max(floor); + crate::metadata_group::codec::stamp_entry(data, stamp) + } + None => data.to_vec(), + }; + // A node that does not lead refuses here, and its stamp stays unused. + self.propose_to_group(group_id, data) + } + + /// The stamp of the newest stamped entry the metadata log holds in + /// memory, committed or not. `None` when it holds none. + pub fn newest_metadata_stamp(&self) -> Option { + self.groups + .get(&crate::metadata_group::METADATA_GROUP_ID) + .and_then(newest_log_stamp) + } + + /// Bound `group_id`'s log compaction by `ceiling`: the log never + /// compacts past the index it holds, on every path. An archiver raises + /// it once it holds a copy of every entry at or below the new value. The + /// bound also holds for the group when it is added again. + pub fn set_compaction_ceiling( + &mut self, + group_id: u64, + ceiling: std::sync::Arc, + ) { + if let Some(node) = self.groups.get_mut(&group_id) { + node.set_compaction_ceiling(std::sync::Arc::clone(&ceiling)); + } + self.compaction_ceilings.insert(group_id, ceiling); + } +} + +/// The stamp of the newest stamped entry `node` holds in memory. Stamps rise +/// with the index, so the walk stops at the first stamped entry from the end. +fn newest_log_stamp(node: &RaftNode) -> Option { + let first = node.log_snapshot_index().saturating_add(1); + (first..=node.last_log_index()) + .rev() + .filter_map(|index| node.log_entry_at(index)) + .find_map(|entry| crate::metadata_group::codec::entry_stamp(&entry.data)) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + use std::time::{Duration, Instant}; + + use super::*; + use crate::metadata_group::codec::{entry_stamp, stamp_entry}; + use crate::routing::RoutingTable; + + /// A one-node metadata group that leads, stamping from a fresh clock. + fn leading_group(dir: &std::path::Path) -> MultiRaft { + let rt = RoutingTable::uniform(1, &[1], 1); + let mut mr = MultiRaft::new(1, rt, dir.to_path_buf()); + mr.add_group(0, vec![]).unwrap(); + for node in mr.groups.values_mut() { + node.election_deadline_override(Instant::now() - Duration::from_millis(1)); + } + for _ in 0..3 { + if mr.is_group_leader(0) { + break; + } + mr.tick().unwrap(); + } + mr.set_metadata_clock(Arc::new(nodedb_types::HlcClock::new())); + assert!(mr.is_group_leader(0)); + mr + } + + fn stamp_at(mr: &MultiRaft, index: u64) -> Option { + mr.groups[&0] + .log_entry_at(index) + .and_then(|entry| entry_stamp(&entry.data)) + } + + /// An entry stamped ahead of the clock reaches the log, as a previous + /// leader with a faster clock appends one. Every entry this leader stamps + /// after it lands above it, so no entry stamped below a stamp the archive + /// holds can follow that stamp in the log. + #[test] + fn stamps_rise_with_the_index_when_the_clock_lags_the_log() { + let dir = tempfile::tempdir().unwrap(); + let mut mr = leading_group(dir.path()); + let first = mr.propose_stamped_metadata(b"a").unwrap(); + let ahead = stamp_at(&mr, first).unwrap() + 3_600_000_000_000; + let future = mr.propose_to_group(0, stamp_entry(b"b", ahead)).unwrap(); + // An unstamped entry between them hides nothing: the walk goes past it. + mr.propose_to_group(0, b"no stamp".to_vec()).unwrap(); + let mut indexes = vec![first, future]; + for data in [b"c", b"d", b"e"] { + indexes.push(mr.propose_stamped_metadata(data).unwrap()); + } + let stamps: Vec = indexes + .iter() + .map(|index| stamp_at(&mr, *index).unwrap()) + .collect(); + assert!( + stamps.windows(2).all(|pair| pair[0] < pair[1]), + "stamps must rise with the index: {stamps:?}" + ); + assert_eq!(mr.newest_metadata_stamp(), stamps.last().copied()); + } + + /// A follower takes no stamp into the log: its propose is refused. + #[test] + fn a_follower_refuses_a_stamped_proposal() { + let dir = tempfile::tempdir().unwrap(); + let rt = RoutingTable::uniform(1, &[1, 2], 2); + let mut mr = MultiRaft::new(1, rt, dir.path().to_path_buf()); + mr.add_group(0, vec![2]).unwrap(); + mr.set_metadata_clock(Arc::new(nodedb_types::HlcClock::new())); + assert!(mr.propose_stamped_metadata(b"a").is_err()); + assert_eq!(mr.newest_metadata_stamp(), None); + } } diff --git a/nodedb-cluster/src/multi_raft/read_index.rs b/nodedb-cluster/src/multi_raft/read_index.rs index ae0443b38..92edcdc36 100644 --- a/nodedb-cluster/src/multi_raft/read_index.rs +++ b/nodedb-cluster/src/multi_raft/read_index.rs @@ -2,17 +2,37 @@ //! Read placement questions answered from a group hosted on this node. //! -//! Every call is non-blocking. The read-index probe is taken under the -//! coordinator lock and confirmed under a later one, so the caller polls -//! between ticks instead of holding the lock while a quorum answers. +//! Every call is non-blocking. A leader holding a valid lease answers at once. +//! Otherwise the read-index probe is taken under the coordinator lock and +//! confirmed under a later one, so the caller polls between ticks instead of +//! holding the lock while a quorum answers. -use std::time::Duration; +use std::time::{Duration, Instant}; use nodedb_raft::{ReadIndexProbe, ReadIndexStatus}; use crate::multi_raft::core::MultiRaft; impl MultiRaft { + /// The read index `group_id` can serve now under its leader lease. + /// + /// `None` when the group is not hosted here or holds no valid lease. The + /// caller then runs a read-index probe. + pub fn lease_read_index(&self, group_id: u64) -> Option { + self.groups.get(&group_id)?.lease_read_index(Instant::now()) + } + + /// The term of this node's valid leader lease on `group_id`. + /// + /// `None` when the group is not hosted here or holds no valid lease. A + /// later leader of the group leads at a higher term, so the term fences + /// work a previous leaseholder started. + pub fn lease_term(&self, group_id: u64) -> Option { + let node = self.groups.get(&group_id)?; + node.lease_read_index(Instant::now())?; + Some(node.current_term()) + } + /// Begin a linearizable read on `group_id`. /// /// `None` when the group is not hosted here or this node does not lead @@ -59,6 +79,16 @@ mod tests { let dir = tempfile::tempdir().expect("tempdir"); let mut mr = coordinator(&dir); assert!(mr.start_read_index(7).is_none()); + assert!(mr.lease_read_index(7).is_none()); + } + + #[test] + fn a_follower_holds_no_lease() { + let dir = tempfile::tempdir().expect("tempdir"); + let mut mr = coordinator(&dir); + mr.add_group(7, vec![1, 2, 3]).expect("add group"); + assert!(mr.lease_read_index(7).is_none()); + assert!(mr.lease_term(7).is_none()); } #[test] diff --git a/nodedb-cluster/src/multi_raft/rpc_dispatch.rs b/nodedb-cluster/src/multi_raft/rpc_dispatch.rs index 8d765479e..8edbb59f4 100644 --- a/nodedb-cluster/src/multi_raft/rpc_dispatch.rs +++ b/nodedb-cluster/src/multi_raft/rpc_dispatch.rs @@ -8,7 +8,8 @@ use nodedb_raft::{ AppendEntriesRequest, AppendEntriesResponse, InstallSnapshotRequest, InstallSnapshotResponse, - PreVoteRequest, PreVoteResponse, RequestVoteRequest, RequestVoteResponse, TimeoutNowRequest, + LogEntry, PreVoteRequest, PreVoteResponse, RequestVoteRequest, RequestVoteResponse, + TimeoutNowRequest, }; use crate::error::{ClusterError, Result}; @@ -66,6 +67,46 @@ impl MultiRaft { Ok(node.handle_install_snapshot(req)?) } + /// Whether a snapshot at `last_included_index`, sent at `term`, must be + /// applied to the local state machine. + /// + /// False for a stale leader's snapshot, and for one this node already + /// applied past. Installing either would move the state machine behind + /// the log it continues to apply from. A replica that requires a snapshot + /// also takes one at exactly its applied index: its log is whole there, + /// and the state beside the log is what it lacks. + pub fn snapshot_install_needed( + &self, + group_id: u64, + term: u64, + last_included_index: u64, + ) -> Result { + let node = self + .groups + .get(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + let applied = node.last_applied(); + let ahead = last_included_index > applied + || (node.snapshot_required() && last_included_index == applied); + Ok(term >= node.current_term() && ahead) + } + + /// Adopt a snapshot the local state machine already holds as the group's + /// log boundary and durable applied floor. See + /// [`nodedb_raft::RaftNode::adopt_snapshot_boundary`]. + pub fn adopt_snapshot_boundary( + &mut self, + group_id: u64, + last_included_index: u64, + last_included_term: u64, + ) -> Result<()> { + let node = self + .groups + .get_mut(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + Ok(node.adopt_snapshot_boundary(last_included_index, last_included_term)?) + } + /// Route a TimeoutNow RPC to the correct group. /// /// One-way — no response is produced. Silently ignored if the group is @@ -78,10 +119,12 @@ impl MultiRaft { } } - /// Durably persist a group's HardState (current_term/voted_for) if it - /// changed since the last persist. Must run under the `MultiRaft` lock - /// before an RPC reply that granted a vote or bumped the term leaves this - /// node, so a restart cannot forget the vote and let two leaders form. + /// Stage a group's HardState (current_term/voted_for) on its disk if it + /// changed since the last persist. Runs under the `MultiRaft` lock and + /// never waits on disk. Before an RPC reply that granted a vote or bumped + /// the term leaves this node, the caller awaits the group's + /// [`MultiRaft::reply_ticket`], so a restart cannot forget the vote and + /// let two leaders form. /// /// No-op when the group is not mounted on this node. pub fn persist_group_hard_state(&mut self, group_id: u64) -> Result<()> { @@ -91,6 +134,27 @@ impl MultiRaft { Ok(()) } + /// The sequence number of the last write `group_id` staged, or 0 when the + /// group is not mounted. A reply handler reads it before its Raft call + /// and passes it to [`Self::reply_ticket`]. + pub fn staged_through(&self, group_id: u64) -> u64 { + self.groups + .get(&group_id) + .map_or(0, |node| node.storage().staged_through()) + } + + /// A ticket for the writes a reply depends on: the group's latest hard + /// state, and every write staged after `mark` when there is one. `None` + /// when those are durable or the group is not mounted. Writes the apply + /// loop staged before `mark` hold no reply back. + pub fn reply_ticket( + &self, + group_id: u64, + mark: u64, + ) -> Option { + self.groups.get(&group_id)?.storage().reply_ticket(mark) + } + /// Get the current term and snapshot metadata for a group (for building /// InstallSnapshot RPCs). pub fn snapshot_metadata(&self, group_id: u64) -> Result<(u64, u64, u64)> { @@ -105,6 +169,25 @@ impl MultiRaft { )) } + /// Record that `peer` installed a snapshot of `group_id` through + /// `last_included_index`. No-op when the group is not mounted here. + pub fn record_snapshot_installed( + &mut self, + group_id: u64, + peer: u64, + last_included_index: u64, + ) { + if let Some(node) = self.groups.get_mut(&group_id) { + node.record_snapshot_installed(peer, last_included_index); + } + } + + /// Term of the entry at `index` in a group's log, or `None` when the + /// group is not mounted here or the log no longer holds that entry. + pub fn log_term_at(&self, group_id: u64, index: u64) -> Option { + self.groups.get(&group_id)?.log_term_at(index) + } + /// Handle AppendEntries response for a specific group. pub fn handle_append_entries_response( &mut self, @@ -163,6 +246,17 @@ impl MultiRaft { Ok(()) } + /// Return a committed batch the tick took but did not apply. See + /// [`nodedb_raft::RaftNode::requeue_committed`]. + pub fn requeue_committed(&mut self, group_id: u64, batch: Vec) -> Result<()> { + let node = self + .groups + .get_mut(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + node.requeue_committed(batch); + Ok(()) + } + /// Durably record `applied_to` as the group's applied floor. /// /// `applied_to` MUST name an entry whose state-machine effects are already diff --git a/nodedb-cluster/src/multi_raft/status.rs b/nodedb-cluster/src/multi_raft/status.rs index 081fbff1e..2a019f7bc 100644 --- a/nodedb-cluster/src/multi_raft/status.rs +++ b/nodedb-cluster/src/multi_raft/status.rs @@ -61,16 +61,31 @@ impl MultiRaft { }) } + /// Every hosted group's live leader as this node's Raft knows it, with + /// the term it knows it at: `(group_id, leader_id, term)`. + /// + /// `leader_id` is this node when it leads, or the leader whose contact + /// is fresher than `election_timeout_min`. It is `0` otherwise, as during + /// an election or once a crashed leader's contact went stale. + pub fn observed_leaders(&self) -> Vec<(u64, u64, u64)> { + let now = std::time::Instant::now(); + self.groups + .iter() + .map(|(&group_id, node)| (group_id, node.live_leader(now), node.current_term())) + .collect() + } + /// Snapshot of all Raft group states for observability. + /// + /// The caller holds the `MultiRaft` lock, so this takes the routing read + /// guard second. That follows the crate lock order: `MultiRaft`, then routing. + /// One guard covers the whole loop so a queued routing writer cannot + /// interleave between groups. pub fn group_statuses(&self) -> Vec { + let routing = self.routing.read().unwrap_or_else(|p| p.into_inner()); let mut statuses = Vec::with_capacity(self.groups.len()); for (&group_id, node) in &self.groups { - let vshard_count = self - .routing - .read() - .unwrap_or_else(|p| p.into_inner()) - .vshards_for_group(group_id) - .len(); + let vshard_count = routing.vshards_for_group(group_id).len(); let self_is_voter = !matches!( node.role(), nodedb_raft::NodeRole::Learner | nodedb_raft::NodeRole::Observer @@ -95,3 +110,88 @@ impl MultiRaft { statuses } } + +#[cfg(test)] +mod tests { + use std::sync::atomic::{AtomicBool, Ordering}; + use std::sync::{Arc, Mutex, mpsc}; + use std::thread; + use std::time::Duration; + + use super::*; + use crate::routing::RoutingTable; + + const DATA_GROUPS: u64 = 4; + const CALLERS: usize = 4; + const ROUNDS: usize = 500; + const DEADLINE: Duration = Duration::from_secs(30); + + /// Status callers that follow the lock order finish under a contending routing writer. + /// + /// Each caller takes the `MultiRaft` lock for `group_statuses`, drops it, then reads routing. + /// A writer thread queues on the routing lock without pause. + /// std's Linux `RwLock` blocks a new reader while a writer waits. + /// A caller that holds a routing read guard across `group_statuses` deadlocks here. + /// Its nested routing read waits behind the writer. + /// The writer waits behind its outer guard. + /// It also holds the `MultiRaft` lock, so every other caller stalls and the deadline fails. + #[test] + fn status_callers_finish_under_a_contending_routing_writer() { + let dir = tempfile::tempdir().expect("tempdir"); + let mut multi_raft = MultiRaft::new( + 1, + RoutingTable::uniform(DATA_GROUPS, &[1], 1), + dir.path().to_path_buf(), + ); + for group_id in 0..=DATA_GROUPS { + multi_raft.add_group(group_id, vec![]).expect("add group"); + } + let routing = multi_raft.routing(); + let multi_raft = Arc::new(Mutex::new(multi_raft)); + let stop = Arc::new(AtomicBool::new(false)); + + let writer = { + let routing = Arc::clone(&routing); + let stop = Arc::clone(&stop); + thread::spawn(move || { + while !stop.load(Ordering::Relaxed) { + drop(routing.write().unwrap_or_else(|p| p.into_inner())); + thread::yield_now(); + } + }) + }; + + let (done_tx, done_rx) = mpsc::channel(); + for _ in 0..CALLERS { + let multi_raft = Arc::clone(&multi_raft); + let routing = Arc::clone(&routing); + let done_tx = done_tx.clone(); + thread::spawn(move || { + let mut consistent = true; + for _ in 0..ROUNDS { + let statuses = multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .group_statuses(); + let table = routing.read().unwrap_or_else(|p| p.into_inner()); + consistent &= statuses.len() as u64 == DATA_GROUPS + 1 + && statuses.iter().all(|status| { + status.vshard_count == table.vshards_for_group(status.group_id).len() + }); + } + // The receiver can be gone after a failed deadline. Nothing then reads this. + let _ = done_tx.send(consistent); + }); + } + drop(done_tx); + + for _ in 0..CALLERS { + let consistent = done_rx + .recv_timeout(DEADLINE) + .expect("a status caller deadlocked on the routing lock"); + assert!(consistent, "a status disagreed with the routing table"); + } + stop.store(true, Ordering::Relaxed); + writer.join().expect("routing writer"); + } +} diff --git a/nodedb-cluster/src/raft_bootstrap.rs b/nodedb-cluster/src/raft_bootstrap.rs new file mode 100644 index 000000000..5a883beba --- /dev/null +++ b/nodedb-cluster/src/raft_bootstrap.rs @@ -0,0 +1,175 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Start a Raft group's log at a chosen point, with no prior log. +//! +//! A cluster restore rebuilds every group's state machine from backups and +//! then starts each group's log at the restored point: the snapshot marker +//! at `(snapshot_index, snapshot_term)`, the durable applied index there, +//! and the entries after it that every replica writes alike. The next boot +//! loads the file as it loads any group log. + +use std::path::{Path, PathBuf}; + +use nodedb_raft::message::LogEntry; +use nodedb_raft::state::HardState; +use nodedb_raft::storage::LogStorage; + +use crate::error::ClusterError; +use crate::raft_storage::RedbLogStorage; + +/// Where group `group_id` keeps its log under `data_dir`. +pub fn group_log_path(data_dir: &Path, group_id: u64) -> PathBuf { + data_dir.join(format!("raft/group-{group_id}.redb")) +} + +/// The log a group starts from. +#[derive(Debug, Clone, Copy)] +pub struct GroupLogStart<'a> { + /// Index the state machine holds every entry through. + pub snapshot_index: u64, + /// Term recorded for `snapshot_index`. + pub snapshot_term: u64, + /// The term the group's hard state starts in, with no vote cast. + pub current_term: u64, + /// Entries after `snapshot_index`, consecutive from `snapshot_index + 1`, + /// not yet applied. + pub entries: &'a [LogEntry], +} + +fn refuse(path: &Path, why: String) -> ClusterError { + ClusterError::Storage { + detail: format!("start raft log {}: {why}", path.display()), + } +} + +/// Write group log `path` so it starts at `start`. Refuses a path that +/// exists: the start never merges into an earlier log. +pub fn start_group_log(path: &Path, start: &GroupLogStart<'_>) -> crate::Result<()> { + if path.exists() { + return Err(refuse(path, "a log already exists there".into())); + } + let mut expected = start.snapshot_index; + let mut last_term = start.snapshot_term; + for entry in start.entries { + expected += 1; + if entry.index != expected { + return Err(refuse( + path, + format!("entry {} follows {}", entry.index, expected - 1), + )); + } + if entry.term < last_term { + return Err(refuse( + path, + format!( + "entry {} has term {} below {last_term}", + entry.index, entry.term + ), + )); + } + last_term = entry.term; + } + if start.current_term < last_term { + return Err(refuse( + path, + format!( + "current term {} is below the log's last term {last_term}", + start.current_term + ), + )); + } + + let mut storage = RedbLogStorage::open(path)?; + storage.compact(start.snapshot_index, start.snapshot_term)?; + storage.append(start.entries)?; + storage.save_hard_state(&HardState { + current_term: start.current_term, + voted_for: 0, + })?; + storage.save_applied_index(start.snapshot_index)?; + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::multi_raft::MultiRaft; + use crate::routing::RoutingTable; + + fn entry(index: u64, term: u64) -> LogEntry { + LogEntry { + term, + index, + data: index.to_le_bytes().to_vec(), + } + } + + #[test] + fn a_started_log_reloads_at_its_start() { + let dir = tempfile::tempdir().unwrap(); + let path = group_log_path(dir.path(), 0); + let entries = [entry(11, 7), entry(12, 7)]; + start_group_log( + &path, + &GroupLogStart { + snapshot_index: 10, + snapshot_term: 7, + current_term: 9, + entries: &entries, + }, + ) + .unwrap(); + + let storage = RedbLogStorage::open(&path).unwrap(); + assert_eq!(storage.snapshot_metadata(), (10, 7)); + assert_eq!(storage.load_entries_after(10).unwrap(), entries); + assert_eq!(storage.load_hard_state().unwrap().current_term, 9); + assert_eq!(storage.load_hard_state().unwrap().voted_for, 0); + assert_eq!(storage.load_applied_index().unwrap(), 10); + drop(storage); + + let rt = RoutingTable::uniform(1, &[1], 1); + let mut mr = MultiRaft::new(1, rt, dir.path().to_path_buf()); + mr.add_group(0, vec![]).unwrap(); + assert_eq!(mr.first_available_index(0), Some(11)); + assert_eq!(mr.log_term_at(0, 10), Some(7)); + assert_eq!(mr.log_term_at(0, 12), Some(7)); + assert_eq!(mr.log_term_at(0, 13), None); + } + + fn start(entries: &[LogEntry], current_term: u64) -> GroupLogStart<'_> { + GroupLogStart { + snapshot_index: 5, + snapshot_term: 3, + current_term, + entries, + } + } + + #[test] + fn a_bad_start_is_refused() { + let dir = tempfile::tempdir().unwrap(); + let gap = [entry(7, 3)]; + let falling = [entry(6, 2)]; + let behind = [entry(6, 4)]; + for (name, entries, term) in [ + ("gap", &gap[..], 3), + ("falling", &falling[..], 3), + ("behind", &behind[..], 3), + ] { + let path = dir.path().join(name); + assert!( + start_group_log(&path, &start(entries, term)).is_err(), + "{name}" + ); + assert!(!path.exists(), "{name}: nothing was written"); + } + + let path = group_log_path(dir.path(), 4); + start_group_log(&path, &start(&[], 3)).unwrap(); + assert!( + start_group_log(&path, &start(&[], 3)).is_err(), + "an existing log is never overwritten" + ); + } +} diff --git a/nodedb-cluster/src/raft_loop/apply_gate.rs b/nodedb-cluster/src/raft_loop/apply_gate.rs new file mode 100644 index 000000000..ab5b1a36d --- /dev/null +++ b/nodedb-cluster/src/raft_loop/apply_gate.rs @@ -0,0 +1,202 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-group gate between applying committed entries and installing a +//! snapshot. +//! +//! An install restores the state machine to a snapshot, then adopts the +//! snapshot as the group's Raft boundary. An entry the snapshot covers must +//! never reach the state machine after the restore starts. Appliers hold the +//! gate shared from their floor check until the entry is handed to the state +//! machine. An install holds it exclusive from its need check through the +//! boundary advance. Each check therefore sees either no install started, or +//! the adopted snapshot index. +//! +//! The gate is a `tokio::sync::RwLock` for two reasons: +//! - The install holds it across the Data-Plane restore await. That await is +//! local (every core's SPSC queue), never a network round-trip. +//! - The host apply loop holds it across a write's enqueue await. +//! +//! Appliers share the gate, so the tick and the host apply loop never block +//! each other. The sync tick never waits: it takes the gate with `try_read` +//! and requeues a batch the gate refuses. A refused applier retries when the +//! release generation changes. + +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; + +use tokio::sync::{OwnedRwLockReadGuard, OwnedRwLockWriteGuard, RwLock, watch}; + +/// One gate per Raft group. The protected value is the highest snapshot index +/// an install adopted through the gate. +/// +/// A group gets its gate entry when it mounts. The entry stays when the group +/// unmounts, so a later mount keeps its installed index. The entries stay +/// bounded by the groups of the cluster. +/// +/// The `gates` mutex is a LEAF lock: no other lock is taken while it is held. +#[derive(Debug)] +pub struct GroupApplyGates { + gates: Mutex>>>, + released: watch::Sender, +} + +impl Default for GroupApplyGates { + fn default() -> Self { + Self::new() + } +} + +impl GroupApplyGates { + pub fn new() -> Self { + let (released, _) = watch::channel(0); + Self { + gates: Mutex::new(HashMap::new()), + released, + } + } + + /// Create the gate of `group_id` when the group mounts. A gate that + /// exists is kept, with its installed index. + pub fn mount(&self, group_id: u64) { + self.gates + .lock() + .unwrap_or_else(|p| p.into_inner()) + .entry(group_id) + .or_insert_with(|| Arc::new(RwLock::new(0))); + } + + /// The gate of `group_id`. A group this node does not mount has no gate + /// entry: it gets an open gate of its own, and no entry is created. + fn gate(&self, group_id: u64) -> Arc> { + let gates = self.gates.lock().unwrap_or_else(|p| p.into_inner()); + match gates.get(&group_id) { + Some(gate) => Arc::clone(gate), + None => Arc::new(RwLock::new(0)), + } + } + + /// Number of gate entries. + #[cfg(test)] + fn len(&self) -> usize { + self.gates.lock().unwrap_or_else(|p| p.into_inner()).len() + } + + /// Admit an apply of `group_id`. `None` while an install holds or awaits + /// the gate. + pub fn try_apply(&self, group_id: u64) -> Option { + self.gate(group_id) + .try_read_owned() + .ok() + .map(|guard| ApplyPermit { guard }) + } + + /// Wait for exclusive hold of `group_id`'s gate for a snapshot install. + pub async fn install(self: &Arc, group_id: u64) -> InstallPermit { + let guard = self.gate(group_id).write_owned().await; + InstallPermit { + guard: Some(guard), + gates: Arc::clone(self), + } + } + + /// A receiver whose value changes each time an install releases a gate. + pub fn subscribe_released(&self) -> watch::Receiver { + self.released.subscribe() + } +} + +/// Shared hold of one group's gate. +#[derive(Debug)] +pub struct ApplyPermit { + guard: OwnedRwLockReadGuard, +} + +impl ApplyPermit { + /// Highest snapshot index an install adopted for this group. An entry at + /// or below it must not apply. + pub fn installed_through(&self) -> u64 { + *self.guard + } +} + +/// Exclusive hold of one group's gate for a snapshot install. +#[derive(Debug)] +pub struct InstallPermit { + guard: Option>, + gates: Arc, +} + +impl InstallPermit { + /// Record that the group adopted a snapshot at `index`. + pub fn adopted(&mut self, index: u64) { + if let Some(guard) = self.guard.as_mut() { + **guard = (**guard).max(index); + } + } +} + +impl Drop for InstallPermit { + fn drop(&mut self) { + // Release before the bump, so a woken applier finds the gate open. + drop(self.guard.take()); + self.gates + .released + .send_modify(|generation| *generation = generation.wrapping_add(1)); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn appliers_share_and_an_install_excludes() { + let gates = Arc::new(GroupApplyGates::new()); + gates.mount(1); + let first = gates.try_apply(1).expect("open gate admits"); + let second = gates.try_apply(1).expect("appliers share the gate"); + drop((first, second)); + + let mut install = gates.install(1).await; + assert!(gates.try_apply(1).is_none(), "install excludes appliers"); + assert!(gates.try_apply(2).is_some(), "other groups stay open"); + + let released = gates.subscribe_released(); + install.adopted(9); + drop(install); + assert!(released.has_changed().expect("sender alive")); + + let permit = gates.try_apply(1).expect("released gate admits"); + assert_eq!(permit.installed_through(), 9); + } + + #[tokio::test] + async fn installed_index_never_moves_back() { + let gates = Arc::new(GroupApplyGates::new()); + gates.mount(1); + gates.install(1).await.adopted(9); + gates.install(1).await.adopted(4); + let permit = gates.try_apply(1).expect("open gate admits"); + assert_eq!(permit.installed_through(), 9); + } + + /// Only a mounted group holds a gate entry. Applying or installing for + /// another group creates none. + #[tokio::test] + async fn gate_entries_follow_mounted_groups() { + let gates = Arc::new(GroupApplyGates::new()); + assert!(gates.try_apply(5).is_some(), "an unmounted group is open"); + drop(gates.install(6).await); + assert_eq!(gates.len(), 0, "no entry for unmounted groups"); + + gates.mount(1); + gates.install(1).await.adopted(3); + gates.mount(1); + assert_eq!( + gates.try_apply(1).expect("open").installed_through(), + 3, + "a remount keeps the installed index" + ); + assert_eq!(gates.len(), 1); + } +} diff --git a/nodedb-cluster/src/raft_loop/auth_lease.rs b/nodedb-cluster/src/raft_loop/auth_lease.rs index be6e87e6f..ad10f817d 100644 --- a/nodedb-cluster/src/raft_loop/auth_lease.rs +++ b/nodedb-cluster/src/raft_loop/auth_lease.rs @@ -19,7 +19,10 @@ impl RaftLoop { let response = match &self.auth_lease { Some(service) => service.renew(req).await, None => AuthLeaseRenewResponse { - outcome: AuthLeaseRenewOutcome::NotLeader { leader_hint: None }, + outcome: AuthLeaseRenewOutcome::NotLeader { + leader_hint: None, + term: 0, + }, }, }; Ok(RaftRpc::AuthLeaseRenewResponse(response)) @@ -29,7 +32,10 @@ impl RaftLoop { let response = match &self.auth_lease { Some(service) => service.barrier(req).await, None => AuthBarrierResponse { - outcome: AuthBarrierOutcome::NotLeader { leader_hint: None }, + outcome: AuthBarrierOutcome::NotLeader { + leader_hint: None, + term: 0, + }, }, }; Ok(RaftRpc::AuthBarrierResponse(response)) diff --git a/nodedb-cluster/src/raft_loop/builder.rs b/nodedb-cluster/src/raft_loop/builder.rs index e3729119a..c04245500 100644 --- a/nodedb-cluster/src/raft_loop/builder.rs +++ b/nodedb-cluster/src/raft_loop/builder.rs @@ -38,6 +38,7 @@ impl RaftLoop { tick_interval: self.tick_interval, vshard_handler: self.vshard_handler, catalog: self.catalog, + routing_persister: self.routing_persister, shutdown_watch: self.shutdown_watch, ready_watch: self.ready_watch, loop_metrics: self.loop_metrics, @@ -65,10 +66,12 @@ impl RaftLoop { // `with_plan_executor` is a construction-time builder, called before // `run()` ever ticks — `tick_count` is still 0, so reset is exact. tick_count: std::sync::atomic::AtomicU64::new(0), + tick_state: self.tick_state, // Construction-time builder, before `run()` and any join kick — a // fresh `Notify` has no pending permit, so this loses nothing. reconcile_notify: tokio::sync::Notify::new(), metadata_cache: self.metadata_cache, + lease_holder_liveness: self.lease_holder_liveness, } } @@ -78,7 +81,16 @@ impl RaftLoop { /// `SharedState` so proposers and consistent-read paths share /// one registry with the tick loop's bump points. Defaults to a /// fresh empty registry when not set. + /// + /// Every group mounted on this node, now or later, starts its watcher at + /// the applied index it restored: its durable applied floor. Every entry + /// at or below it applied before the restart and is never delivered + /// again, so no apply moves the watcher there. pub fn with_group_watchers(mut self, watchers: Arc) -> Self { + self.multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .set_applied_watchers(Arc::clone(&watchers)); self.group_watchers = watchers; self } @@ -231,8 +243,8 @@ impl RaftLoop { /// /// The supplied implementation (backed by `nodedb`'s Data Plane restore /// dispatch) is called by the install-snapshot finalize path to apply a - /// received per-group snapshot to the local state machine after the atomic - /// `.partial`→`.snap` rename and before advancing Raft. When not set, the + /// received per-group snapshot to the local state machine after staging + /// it and before advancing Raft. When not set, the /// follower advances Raft without restoring engine state. pub fn with_snapshot_applier(mut self, applier: Arc) -> Self { self.snapshot_applier = Some(applier); @@ -282,11 +294,34 @@ impl RaftLoop { /// Wire the metadata cache used by the periodic lease-GC sweep. The /// host passes the same `Arc` the production metadata applier holds. + /// + /// The cache's applied index starts at the restored applied floor, so + /// replay resumes above it and the cache agrees with the group watcher. pub fn with_metadata_cache(mut self, cache: Arc>) -> Self { + let floor = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .last_applied(crate::metadata_group::METADATA_GROUP_ID) + .unwrap_or(0); + cache + .write() + .unwrap_or_else(|p| p.into_inner()) + .advance_applied_index(floor); self.metadata_cache = Some(cache); self } + /// Share the SWIM Dead records the lease-GC sweep reads. The host passes + /// the same `Arc` it registers as a SWIM subscriber. + pub fn with_lease_holder_liveness( + mut self, + liveness: Arc, + ) -> Self { + self.lease_holder_liveness = liveness; + self + } + pub fn with_tick_interval(mut self, interval: Duration) -> Self { self.tick_interval = interval; self @@ -317,6 +352,15 @@ impl RaftLoop { tracing::warn!(error = %e, "could not load the persisted cluster epoch; starting at 0") } } + let routing = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .routing(); + self.routing_persister = Some(Arc::new(super::routing_persist::RoutingPersister::new( + Arc::clone(&catalog), + routing, + ))); self.catalog = Some(catalog); self } diff --git a/nodedb-cluster/src/raft_loop/group_unmount.rs b/nodedb-cluster/src/raft_loop/group_unmount.rs new file mode 100644 index 000000000..6e38c87f2 --- /dev/null +++ b/nodedb-cluster/src/raft_loop/group_unmount.rs @@ -0,0 +1,166 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Unmount the replica of a data group this node has left. +//! +//! Placement convergence removes a node from a group it is no longer placed +//! in: `RemoveNode` for a voter, `RemoveLearner` for a learner. The leader +//! tells the departing node its removal committed, so the node applies it +//! and its routing view stops listing it. When that one message is lost, the +//! leader probe learns the group's membership from its leader and the +//! routing view drops the node the same way (see the leader probe in +//! [`super::tick`]). Its replica then only holds a log the group no longer sends to. This step drops that replica, so a node +//! outside a group's placement hosts no replica of it. +//! +//! A replica is dropped only once all of these hold: +//! - the group's placement is authored and names other nodes only; +//! - this node's routing view lists it as neither voter nor learner; +//! - this node does not lead the group. +//! +//! `mount_entering_groups` mounts only a group whose placement names this +//! node, so the two steps never undo each other. The metadata group and +//! the sequencer group are never unmounted. + +use std::collections::HashSet; + +use tracing::debug; + +use crate::forward::PlanExecutor; +use crate::routing::RoutingTable; + +use super::loop_core::{CommitApplier, RaftLoop}; + +/// Hosted data groups this node has left, per the rules in the module docs. +/// `leading` holds the hosted groups this node leads. Returned sorted +/// ascending. Pure and deterministic. +pub(super) fn plan_unmounts( + self_id: u64, + hosted: &HashSet, + leading: &HashSet, + routing: &RoutingTable, +) -> Vec { + let mut out: Vec = hosted + .iter() + .copied() + .filter(|gid| { + *gid != crate::metadata_group::METADATA_GROUP_ID + && *gid != crate::calvin::sequencer::SEQUENCER_GROUP_ID + && !leading.contains(gid) + }) + .filter(|gid| { + routing.group_info(*gid).is_some_and(|info| { + info.placement + .as_ref() + .is_some_and(|placement| !placement.contains(&self_id)) + && !info.members.contains(&self_id) + && !info.learners.contains(&self_id) + }) + }) + .collect(); + out.sort_unstable(); + out +} + +impl RaftLoop { + /// Drop the replica of every data group this node has left. + /// + /// Never waits on disk. A dropped group's Raft storage closes on a + /// blocking thread. The routing table is saved by the routing persister, + /// so a restart mounts only the groups the table still lists this node + /// in. The persister retries a failed save. + pub(super) fn unmount_left_groups(&self) { + let dropped = { + let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let group_ids = mr.group_ids(); + let hosted: HashSet = group_ids.iter().copied().collect(); + let leading: HashSet = group_ids + .iter() + .copied() + .filter(|gid| mr.group_role_is_leader(*gid)) + .collect(); + let left = { + let routing = mr.routing(); + let table = routing.read().unwrap_or_else(|p| p.into_inner()); + plan_unmounts(self.node_id, &hosted, &leading, &table) + }; + let mut dropped = Vec::new(); + for group_id in left { + if let Some(node) = mr.unmount_group(group_id) { + self.tick_state.clear_conf_save(group_id); + debug!( + group_id, + node_id = self.node_id, + "unmount: dropped the replica of a group this node left" + ); + dropped.push(node); + } + } + dropped + }; + if dropped.is_empty() { + return; + } + // Closing a group's log storage can write to disk. + tokio::task::spawn_blocking(move || drop(dropped)); + if let Some(persister) = self.routing_persister.as_ref() { + persister.request(); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn table() -> RoutingTable { + // Data groups 1 and 2 over nodes 1..=3, metadata group 0. + RoutingTable::uniform(2, &[1, 2, 3], 3) + } + + fn set(ids: &[u64]) -> HashSet { + ids.iter().copied().collect() + } + + #[test] + fn a_left_group_outside_the_placement_unmounts() { + let mut rt = table(); + rt.set_placement(1, vec![2, 3]); + rt.set_group_members(1, vec![2, 3]); + assert_eq!(plan_unmounts(1, &set(&[0, 1, 2]), &set(&[]), &rt), vec![1]); + } + + #[test] + fn a_group_that_still_lists_this_node_stays() { + let mut rt = table(); + // Placement excludes node 1, but its removal has not applied yet. + rt.set_placement(1, vec![2, 3]); + assert!(plan_unmounts(1, &set(&[1]), &set(&[]), &rt).is_empty()); + // A learner entry keeps it too. + rt.set_group_members(1, vec![2, 3]); + rt.set_group_learners(1, vec![1]); + assert!(plan_unmounts(1, &set(&[1]), &set(&[]), &rt).is_empty()); + } + + #[test] + fn an_unauthored_or_including_placement_keeps_the_group() { + let mut rt = table(); + rt.set_group_members(1, vec![2, 3]); + // No placement authored yet. + assert!(plan_unmounts(1, &set(&[1]), &set(&[]), &rt).is_empty()); + // The placement names this node: it is entering, not leaving. + rt.set_placement(1, vec![1, 2]); + assert!(plan_unmounts(1, &set(&[1]), &set(&[]), &rt).is_empty()); + } + + #[test] + fn a_led_metadata_or_sequencer_group_is_never_unmounted() { + let mut rt = table(); + rt.set_placement(1, vec![2, 3]); + rt.set_group_members(1, vec![2, 3]); + assert!(plan_unmounts(1, &set(&[1]), &set(&[1]), &rt).is_empty()); + rt.set_placement(0, vec![2, 3]); + rt.set_group_members(0, vec![2, 3]); + assert!(plan_unmounts(1, &set(&[0]), &set(&[]), &rt).is_empty()); + let seq = crate::calvin::sequencer::SEQUENCER_GROUP_ID; + assert!(plan_unmounts(1, &set(&[seq]), &set(&[]), &rt).is_empty()); + } +} diff --git a/nodedb-cluster/src/raft_loop/handle_rpc/consensus.rs b/nodedb-cluster/src/raft_loop/handle_rpc/consensus.rs index 672177c4f..b427e057a 100644 --- a/nodedb-cluster/src/raft_loop/handle_rpc/consensus.rs +++ b/nodedb-cluster/src/raft_loop/handle_rpc/consensus.rs @@ -15,21 +15,47 @@ use super::super::loop_core::{CommitApplier, RaftLoop}; use super::membership::TOPOLOGY_GROUP_ID; impl RaftLoop { - pub(super) fn handle_append_entries_rpc(&self, req: AppendEntriesRequest) -> Result { - let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - let resp = mr.handle_append_entries(&req)?; - // Persist any term bump (become_follower) durably before the - // reply leaves this node, so a restart cannot forget it. - mr.persist_group_hard_state(req.group_id)?; + /// The leader counts a successful answer as this node holding the + /// claimed entries durably. The answer waits only for what it claims: + /// - the latest term and vote, whoever staged them + /// - the entries and truncations this request staged, with every write + /// staged before them + /// + /// A request that staged no entries claims only the durable prefix. It + /// never waits on writes the apply loop staged. The disk wait runs + /// without the `MultiRaft` lock. + pub(super) async fn handle_append_entries_rpc( + &self, + req: AppendEntriesRequest, + ) -> Result { + let (resp, ticket) = { + let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let mark = mr.staged_through(req.group_id); + let resp = mr.handle_append_entries(&req)?; + mr.persist_group_hard_state(req.group_id)?; + (resp, mr.reply_ticket(req.group_id, mark)) + }; + if let Some(ticket) = ticket { + ticket.durable().await?; + } Ok(RaftRpc::AppendEntriesResponse(resp)) } - pub(super) fn handle_request_vote_rpc(&self, req: RequestVoteRequest) -> Result { - let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - let resp = mr.handle_request_vote(&req)?; - // Persist voted_for/current_term to stable storage BEFORE the - // grant leaves this node, so a restart cannot double-vote. - mr.persist_group_hard_state(req.group_id)?; + /// `voted_for` and `current_term` are durable before the answer leaves + /// this node, so a restart cannot double-vote. A repeated request that + /// staged nothing still waits for a vote an earlier one staged. The disk + /// wait runs without the `MultiRaft` lock. + pub(super) async fn handle_request_vote_rpc(&self, req: RequestVoteRequest) -> Result { + let (resp, ticket) = { + let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let mark = mr.staged_through(req.group_id); + let resp = mr.handle_request_vote(&req)?; + mr.persist_group_hard_state(req.group_id)?; + (resp, mr.reply_ticket(req.group_id, mark)) + }; + if let Some(ticket) = ticket { + ticket.durable().await?; + } Ok(RaftRpc::RequestVoteResponse(resp)) } @@ -108,6 +134,12 @@ impl RaftLoop { let last_included_index = req.last_included_index; let group_id = req.group_id; + // The snapshot covers conf changes this node never applies. The + // final chunk carries the membership they produced. + if req.done { + self.adopt_snapshot_membership(&req).await?; + } + // Route through the chunk accumulator when a data directory is // configured. The accumulator writes chunks to a `.partial` file, // validates the full CRC on the final chunk, and then calls @@ -131,12 +163,18 @@ impl RaftLoop { ) .await { - Ok(crate::install_snapshot::ChunkOutcome::Committed(snap_resp)) => { - // Final chunk committed — bump watcher for metadata group. - if group_id == TOPOLOGY_GROUP_ID { + Ok(crate::install_snapshot::ChunkOutcome::Committed(committed)) => { + // The watcher means "state visible through N". It moves + // only when the host state machine holds the snapshot. + // Data-group watchers are bumped by the host apply loop. + if group_id == TOPOLOGY_GROUP_ID && committed.state_installed { self.group_watchers.bump(group_id, last_included_index); } - return Ok(RaftRpc::InstallSnapshotResponse(snap_resp)); + // The leader resumes replication after the snapshot index + // once this answer arrives, so the new boundary and any + // term bump are durable first. + self.await_group_durable(group_id).await?; + return Ok(RaftRpc::InstallSnapshotResponse(committed.response)); } Ok(crate::install_snapshot::ChunkOutcome::Pending) => { // Non-final chunk — pass a done=false stub to MultiRaft so @@ -151,14 +189,17 @@ impl RaftLoop { done: false, group_id, total_size: 0, + voters: Vec::new(), + learners: Vec::new(), }; let resp = { let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); let resp = mr.handle_install_snapshot(&pending_req)?; - // Persist any term bump before replying. mr.persist_group_hard_state(group_id)?; resp }; + // Any term bump is durable before the reply. + self.await_group_durable(group_id).await?; return Ok(RaftRpc::InstallSnapshotResponse(resp)); } Err(e @ crate::error::ClusterError::SnapshotOffsetRegression { .. }) => { @@ -188,52 +229,42 @@ impl RaftLoop { let resp = { let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); let resp = mr.handle_install_snapshot(&req)?; - // Persist any term bump before replying. mr.persist_group_hard_state(group_id)?; resp }; - // Watcher contract: `applied_index` means "state visible - // on this node up to index N", NOT "raft has advanced to - // N". Bumping the watcher must therefore mirror actual - // state-machine progress. - // - // - Metadata group: `mr.handle_install_snapshot` restores - // the metadata state machine synchronously before - // returning, so the watcher can be bumped here — state - // IS visible at `last_included_index`. - // - // - Data groups: snapshot install fast-forwards raft's - // `last_applied` but does NOT restore the data-plane - // state machine (no committed entries are produced for - // `run_apply_loop`, and there is currently no - // data-group state-machine snapshot restore path). - // Bumping the watcher here would wake waiters that - // then read missing state — silent data-loss-shaped - // bug. The data-group watcher is bumped only by the - // host crate's apply loop after the SPSC round-trip - // completes; that path is the single source of truth - // for "state visible". - // - // When data-group state-machine snapshots are - // implemented, the restore path must bump the watcher - // itself — not this handler. - if group_id == TOPOLOGY_GROUP_ID { - self.group_watchers.bump(group_id, last_included_index); - } + // Any term bump is durable before the reply. + self.await_group_durable(group_id).await?; + // No host state machine restores anything on this path, so no + // watcher moves: the watcher means "state visible through N". Ok(RaftRpc::InstallSnapshotResponse(resp)) } + /// Wait until every write `group_id` staged so far is durable. Takes the + /// ticket under the `MultiRaft` lock and waits without it. + async fn await_group_durable(&self, group_id: u64) -> Result<()> { + let ticket = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .durability_ticket(group_id); + match ticket { + Some(ticket) => ticket.durable().await, + None => Ok(()), + } + } + + /// A TimeoutNow triggers an immediate election: a term bump and a + /// self-vote. The hard state is staged here. The tick loop sends the + /// resulting vote requests only once the group's staged writes are + /// durable, so a restart cannot forget the term. pub(super) async fn on_timeout_now_impl(&self, req: TimeoutNowRequest) { let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); mr.handle_timeout_now(&req); - // A TimeoutNow triggers an immediate election (term bump + self-vote); - // persist that HardState before the resulting vote requests are - // dispatched by the tick loop, so a restart cannot forget the term. if let Err(e) = mr.persist_group_hard_state(req.group_id) { tracing::error!( group_id = req.group_id, error = %e, - "failed to persist hard state after timeout-now election trigger" + "failed to stage hard state after timeout-now election trigger" ); } } diff --git a/nodedb-cluster/src/raft_loop/handle_rpc/dispatch.rs b/nodedb-cluster/src/raft_loop/handle_rpc/dispatch.rs index 775dbe7a1..8439b9128 100644 --- a/nodedb-cluster/src/raft_loop/handle_rpc/dispatch.rs +++ b/nodedb-cluster/src/raft_loop/handle_rpc/dispatch.rs @@ -24,8 +24,8 @@ impl RaftRpcHandler for RaftLoop { async fn handle_rpc(&self, rpc: RaftRpc) -> Result { match rpc { // Raft consensus RPCs — lock MultiRaft (sync, never across await). - RaftRpc::AppendEntriesRequest(req) => self.handle_append_entries_rpc(req), - RaftRpc::RequestVoteRequest(req) => self.handle_request_vote_rpc(req), + RaftRpc::AppendEntriesRequest(req) => self.handle_append_entries_rpc(req).await, + RaftRpc::RequestVoteRequest(req) => self.handle_request_vote_rpc(req).await, RaftRpc::PreVoteRequest(req) => self.handle_pre_vote_rpc(req), RaftRpc::InstallSnapshotRequest(req) => self.handle_install_snapshot_rpc(req).await, // Cluster join — full orchestration in `super::join`. @@ -43,9 +43,17 @@ impl RaftRpcHandler for RaftLoop { RaftRpc::DataProposeRequest(req) => self.handle_data_propose_rpc(req), // Read index for a node that does not lead the group. RaftRpc::ReadIndexRequest(req) => self.handle_read_index_rpc(req).await, + // The leader this node knows for a group, from its own Raft state. + RaftRpc::LeaderStatusRequest(req) => Ok(RaftRpc::LeaderStatusResponse( + self.leader_status(req.group_id), + )), // Authorization lease renewal and barrier, answered by the host hook. RaftRpc::AuthLeaseRenewRequest(req) => self.handle_auth_lease_renew_rpc(req).await, RaftRpc::AuthBarrierRequest(req) => self.handle_auth_barrier_rpc(req).await, + // Streamed parts of a multi-part Calvin transaction. + RaftRpc::CalvinPartsRequest(req) => Ok(RaftRpc::CalvinPartsResponse( + self.on_calvin_parts_impl(req).await, + )), // VShardEnvelope — dispatch to registered handler (Event Plane, etc.). RaftRpc::VShardEnvelope(bytes) => self.handle_vshard_envelope_rpc(bytes).await, other => Err(ClusterError::Transport { @@ -184,7 +192,7 @@ mod tests { let topo = Arc::new(RwLock::new(ClusterTopology::new())); let raft_loop = RaftLoop::new(mr, transport, topo, NoopApplier); - raft_loop.do_tick(); + raft_loop.do_tick().await; tokio::time::sleep(Duration::from_millis(20)).await; let req = RaftRpc::AppendEntriesRequest(nodedb_raft::AppendEntriesRequest { @@ -195,6 +203,8 @@ mod tests { entries: vec![], leader_commit: 0, group_id: 0, + round: 1, + replicated_floor: 0, }); let resp = raft_loop.handle_rpc(req).await.unwrap(); @@ -214,6 +224,9 @@ mod tests { let rt = RoutingTable::uniform(1, &[1, 2, 3], 3); let mut mr = MultiRaft::new(1, rt, dir.path().to_path_buf()); mr.add_group(0, vec![2, 3]).unwrap(); + for node in mr.groups_mut().values_mut() { + node.expire_boot_vote_fence(); + } let topo = Arc::new(RwLock::new(ClusterTopology::new())); let raft_loop = RaftLoop::new(mr, transport, topo, NoopApplier); @@ -224,6 +237,7 @@ mod tests { last_log_index: 0, last_log_term: 0, group_id: 0, + transfer: false, }); let resp = raft_loop.handle_rpc(req).await.unwrap(); @@ -265,19 +279,36 @@ mod tests { let topo = Arc::new(RwLock::new(topology)); let raft_loop = RaftLoop::new(mr, transport, topo.clone(), NoopApplier); - raft_loop.do_tick(); + raft_loop.do_tick().await; tokio::time::sleep(Duration::from_millis(20)).await; let req = RaftRpc::JoinRequest(crate::rpc_codec::JoinRequest { node_id: 2, listen_addr: "127.0.0.1:9401".into(), wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), spiffe_id: None, spki_pin: None, + swim_addr: None, }); - let resp = raft_loop.handle_rpc(req).await.unwrap(); - match resp { + // A leader commits an entry only once its own disk holds it, and the + // tick is what observes that write. The run loop ticks beside every + // RPC handler, so this test does the same while the join waits. + let done = std::sync::atomic::AtomicBool::new(false); + let join = async { + let resp = raft_loop.handle_rpc(req).await; + done.store(true, std::sync::atomic::Ordering::Release); + resp + }; + let ticking = async { + while !done.load(std::sync::atomic::Ordering::Acquire) { + raft_loop.do_tick().await; + tokio::time::sleep(Duration::from_millis(5)).await; + } + }; + let (resp, ()) = tokio::join!(join, ticking); + match resp.unwrap() { RaftRpc::JoinResponse(r) => { assert!( r.success, @@ -288,18 +319,22 @@ mod tests { // uniform(2, ...) creates 3 groups (metadata + 2 data). assert_eq!(r.groups.len(), 3); assert_eq!(r.vshard_to_group.len(), 1024); - // The new node should appear as a learner on every group, - // not as a voter — voter promotion happens asynchronously - // via the tick loop's promotion phase. - for g in &r.groups { - assert!( - g.learners.contains(&2), - "expected node 2 as learner in group {}, got learners={:?} members={:?}", - g.group_id, - g.learners, - g.members - ); - } + // The join admits the new node to the metadata group as a + // learner. Voter promotion happens later in the tick loop. + // Data groups take it only where their placement names it, + // through the group leader's convergence step, so the + // response makes no claim about them. + let metadata = r + .groups + .iter() + .find(|g| g.group_id == 0) + .expect("the response lists the metadata group"); + assert!( + metadata.learners.contains(&2), + "metadata group: learners={:?} members={:?}", + metadata.learners, + metadata.members + ); } other => panic!("expected JoinResponse, got {other:?}"), } diff --git a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs index 9d6058d43..af6cb2e2a 100644 --- a/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs +++ b/nodedb-cluster/src/raft_loop/handle_rpc/plan_dispatch.rs @@ -30,12 +30,18 @@ impl RaftLoop { &self, req: MetadataProposeRequest, ) -> Result { - let resp = match self.propose_to_metadata_group(req.bytes) { + let proposed = if req.stamp { + self.propose_stamped_to_metadata_group(&req.bytes) + } else { + self.propose_to_metadata_group(req.bytes) + }; + let resp = match proposed { Ok(log_index) => crate::rpc_codec::MetadataProposeResponse::ok(log_index), Err(crate::error::ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint, - })) => crate::rpc_codec::MetadataProposeResponse::err("not leader", leader_hint), - Err(e) => crate::rpc_codec::MetadataProposeResponse::err(e.to_string(), None), + term, + })) => crate::rpc_codec::MetadataProposeResponse::err("not leader", leader_hint, term), + Err(e) => crate::rpc_codec::MetadataProposeResponse::err(e.to_string(), None, 0), }; Ok(RaftRpc::MetadataProposeResponse(resp)) } diff --git a/nodedb-cluster/src/raft_loop/handle_rpc/shuffle_calvin.rs b/nodedb-cluster/src/raft_loop/handle_rpc/shuffle_calvin.rs index 3b97eda21..fee16a61b 100644 --- a/nodedb-cluster/src/raft_loop/handle_rpc/shuffle_calvin.rs +++ b/nodedb-cluster/src/raft_loop/handle_rpc/shuffle_calvin.rs @@ -196,6 +196,23 @@ impl RaftLoop { } } + // Streamed parts of a multi-part Calvin transaction: offered to this + // node's sequencer leader queue through the host-crate Calvin-inbox + // hook. Without the hook, the stream learns it cannot run here. + pub(super) async fn on_calvin_parts_impl( + &self, + req: crate::rpc_codec::CalvinPartsRequest, + ) -> crate::rpc_codec::CalvinPartsResponse { + match &self.calvin_submit_inbox { + Some(submit) => submit.on_calvin_parts(req).await, + None => crate::rpc_codec::CalvinPartsResponse { + status: crate::calvin::PartsOfferStatus::Unknown as u8, + next_index: 0, + detail: Some("calvin-parts not configured (no CalvinSubmitInbox installed)".into()), + }, + } + } + // Routed reserve-read (Calvin OLLP) — delegate to the host-crate // `ReserveRead`. This node is the sequencer-group leader; reserving // through it is correct because the leader's scheduler holds the diff --git a/nodedb-cluster/src/raft_loop/hooks.rs b/nodedb-cluster/src/raft_loop/hooks.rs index 6009210de..ce03e3ca4 100644 --- a/nodedb-cluster/src/raft_loop/hooks.rs +++ b/nodedb-cluster/src/raft_loop/hooks.rs @@ -10,6 +10,10 @@ use crate::error::Result; +pub use super::hooks_routed::{ + AssignRemoteSurrogate, CalvinSubmit, CalvinSubmitInbox, ReleaseReservation, ReserveRead, +}; + /// Hook for building per-group snapshot payloads on the Raft snapshot SEND path. /// /// `nodedb-cluster` cannot depend on `nodedb` (circular), so the snapshot @@ -20,7 +24,7 @@ use crate::error::Result; /// before framing the chunked `InstallSnapshot` RPC. /// /// Cluster-only tests leave the `RaftLoop` field `None`, which makes the sender -/// fall back to the stub (empty) chunk — exactly the pre-builder behaviour. +/// fall back to the stub (empty) chunk. /// /// The hook is **async** because the host-crate implementation dispatches the /// per-vshard snapshot build to the Data Plane through the existing SPSC bridge @@ -29,13 +33,67 @@ use crate::error::Result; #[async_trait::async_trait] pub trait SnapshotBuilder: Send + Sync + 'static { /// Build the per-group snapshot payload (serialized engine state for the - /// group's vshards) to ship to a lagging/new follower. Empty Vec is a valid - /// "nothing to send" result (caller falls back to the stub chunk). + /// group's vshards) to ship to a lagging/new follower. + /// + /// The capture holds every entry of the group through a cut at or above + /// `last_included_index`, and nothing above it. The returned + /// [`BuiltGroupSnapshot::cut_index`] names that cut, and the snapshot is sent + /// labelled with it. So the follower resumes the log right after the cut, + /// with no gap of entries the state already holds. + /// + /// Empty bytes are a valid "nothing to send" result: the caller sends the + /// stub chunk, with the cut the builder names. async fn build_group_snapshot( &self, group_id: u64, last_included_index: u64, last_included_term: u64, + ) -> std::result::Result>; + + /// Capture metadata group 0's state machine at `applied_index`, whose + /// entry has term `applied_term`. + /// + /// Called on the tick thread, between apply batches, so the capture holds + /// exactly the entries applied through `applied_index`. The capture must + /// only open read views: the caller serializes it on another task. + fn capture_metadata( + &self, + applied_index: u64, + applied_term: u64, + ) -> std::result::Result< + Box, + Box, + >; + + /// Capture the Calvin sequencer group's state machine at + /// `applied_index`, encoded as the snapshot payload. + /// + /// Called on the tick thread, between apply batches, so the capture holds + /// exactly the entries applied through `applied_index`. The payload is a + /// few scalars and the open multi-part transactions, so it is encoded in + /// place. + fn capture_sequencer( + &self, + applied_index: u64, + ) -> std::result::Result, Box>; +} + +/// A data group snapshot built by [`SnapshotBuilder::build_group_snapshot`]. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct BuiltGroupSnapshot { + /// The serialized payload. Empty when there is nothing to send. + pub bytes: Vec, + /// The highest log index the payload's state holds. At or above the + /// `last_included_index` the build was asked for. + pub cut_index: u64, +} + +/// A group 0 state machine capture taken by +/// [`SnapshotBuilder::capture_metadata`]. +pub trait MetadataSnapshotCapture: Send + 'static { + /// Serialize the captured state into the snapshot payload. + fn serialize( + self: Box, ) -> std::result::Result, Box>; } @@ -48,13 +106,14 @@ pub trait SnapshotBuilder: Send + Sync + 'static { /// SPSC bridge — lives in the host crate (`nodedb`) behind this `Send + Sync` /// hook. The install-snapshot finalize path (see /// [`crate::install_snapshot::finalize::commit`]) calls -/// [`apply_snapshot`](Self::apply_snapshot) AFTER the atomic `.partial`→`.snap` -/// rename and BEFORE advancing Raft, so the data is visible on this node before -/// the Raft log boundary moves. +/// [`apply_snapshot`](Self::apply_snapshot) AFTER staging the snapshot and +/// BEFORE advancing Raft, so the data is visible on this node before the Raft +/// log boundary moves. Boot recovery calls it again for a staged install that +/// did not finish. /// /// Cluster-only tests leave the `RaftLoop` field `None`, which makes the -/// follower advance Raft WITHOUT restoring engine state — the pre-applier -/// behaviour (correct for tests that ship only the empty bootstrap stub). +/// follower advance Raft WITHOUT restoring engine state — correct for tests +/// that ship only the empty bootstrap stub. /// /// The hook is **async** because the host-crate implementation dispatches the /// per-tenant restore to the Data Plane through the SPSC bridge (an awaited @@ -62,15 +121,23 @@ pub trait SnapshotBuilder: Send + Sync + 'static { /// storage directly. #[async_trait::async_trait] pub trait SnapshotApplier: Send + Sync + 'static { - /// Apply a per-group snapshot to the local data-plane state machine. - /// Called AFTER the atomic .partial→.snap rename, BEFORE handle_install_snapshot - /// advances Raft. Err MUST prevent the raft advance (follower retries). - /// group_id 0 (metadata) is a no-op (metadata restored inline). + /// Apply a per-group snapshot to the local state machine. Called after + /// staging, before Raft advances. `Ok` MUST mean the install is durable + /// without the WAL: the caller then moves the durable floor past it and + /// keeps no copy. Err MUST prevent the raft advance (follower retries). + /// For group 0 the bytes are the host's metadata image, captured by + /// [`SnapshotBuilder::capture_metadata`]. async fn apply_snapshot( &self, group_id: u64, snapshot_bytes: &[u8], ) -> std::result::Result<(), Box>; + + /// Called after the group adopted a snapshot at `last_included_index`, + /// while its apply gate still excludes every applier. No entry at or below + /// that index applies on this node afterwards, so nothing produces those + /// entries' results here. + fn snapshot_adopted(&self, _group_id: u64, _last_included_index: u64) {} } /// Hook for quarantine integration on the Raft snapshot receive path. @@ -99,11 +166,11 @@ pub trait SnapshotQuarantineHook: Send + Sync + 'static { fn record_failure(&self, group_id: u64, last_included_index: u64, error: &str) -> bool; } -/// Hook for the cross-node streaming-shuffle receiver registry (E1). +/// Hook for the cross-node streaming-shuffle receiver registry. /// /// `nodedb-cluster` cannot depend on `nodedb` (circular), so the receiver /// registry — which is owned by `nodedb`'s `SharedState` and consumed by the -/// `!Send` Data Plane in a later unit — lives behind this `Send + Sync` hook. +/// `!Send` Data Plane — lives behind this `Send + Sync` hook. /// The transport read-loop drives a `ShufflePush` stream and calls these /// methods; the host crate's implementation deposits payloads into the /// per-`(shuffle_id, part, side)` inbox and advances the per-part build @@ -113,7 +180,7 @@ pub trait SnapshotQuarantineHook: Send + Sync + 'static { /// against a node with no receiver installed returns a typed error. /// /// The hook is **async** because the host-crate implementation stages arriving -/// rows to a Control-Plane scratch file (E3b: receive-to-spill) and must NOT +/// rows to a Control-Plane scratch file (receive-to-spill) and must NOT /// block the transport reactor thread on a synchronous `std::fs` write. The /// awaited `tokio::fs` write inside `on_shuffle_chunk` is what lets QUIC flow /// control back-pressure the producer — the chunk is staged inline, never @@ -149,7 +216,7 @@ pub trait ShuffleReceiver: Send + Sync + 'static { ); } -/// Hook for the cross-node shuffle PRODUCER (E4a). +/// Hook for the cross-node shuffle PRODUCER. /// /// Sibling of [`ShuffleReceiver`]: `nodedb-cluster` cannot depend on `nodedb` /// (circular), so the produce logic — decode the local scan plan, run it through @@ -182,7 +249,7 @@ pub trait ShuffleProducer: Send + Sync + 'static { ) -> crate::rpc_codec::ShuffleProduceResponse; } -/// Hook for the cross-node shuffle CONSUMER (E4b). +/// Hook for the cross-node shuffle CONSUMER. /// /// Sibling of [`ShuffleProducer`]: `nodedb-cluster` cannot depend on `nodedb` /// (circular), so the consume logic — wait for both staged sides of the part to @@ -215,7 +282,7 @@ pub trait ShuffleConsumer: Send + Sync + 'static { ) -> crate::rpc_codec::ShuffleConsumeResponse; } -/// Hook for the cross-node distributed GROUP BY shuffle CONSUMER (E5b). +/// Hook for the cross-node distributed GROUP BY shuffle CONSUMER. /// /// SINGLE-SIDED aggregate sibling of [`ShuffleConsumer`]: `nodedb-cluster` cannot /// depend on `nodedb` (circular), so the aggregate-consume logic — wait for the @@ -250,185 +317,3 @@ pub trait ShuffleAggregator: Send + Sync + 'static { req: crate::rpc_codec::ShuffleAggregateConsumeRequest, ) -> crate::rpc_codec::ShuffleAggregateConsumeResponse; } - -/// Hook for routed-surrogate-exchange (F1b). -/// -/// `nodedb-cluster` cannot depend on `nodedb` (circular), so the assign logic — -/// run a LOCAL `SurrogateAssigner::assign` for the `(collection, pk)` endpoint -/// key carried by the request — lives in `nodedb` behind this `Send + Sync` hook. -/// The transport read-loop calls [`on_assign_surrogate`](Self::on_assign_surrogate) -/// when an `AssignSurrogateRequest` arrives at the home vShard's LEADER and writes -/// the returned [`AssignSurrogateResponse`](crate::rpc_codec::AssignSurrogateResponse) -/// back to the coordinator. -/// -/// Because the handler runs on the home node (the vShard leader), a LOCAL assign -/// yields the AUTHORITATIVE surrogate: the first call allocates it and every later -/// call for the same key returns the same value (idempotent, first-wins). The -/// coordinator routes here precisely so the value it carries is the one the home -/// node will store under. -/// -/// Cluster-only tests leave the `RaftLoop` field `None`; an `AssignSurrogate` -/// request against a node with no assigner installed returns a typed "not -/// configured" error. -/// -/// The hook is **async** for signature symmetry with the other one-shot hooks; -/// the host-crate implementation performs a synchronous local assign (the -/// `SurrogateAssigner` is a sync `Send + Sync` facade) and never touches -/// io_uring or the Data Plane directly. -#[async_trait::async_trait] -pub trait AssignRemoteSurrogate: Send + Sync + 'static { - /// Assign-or-return the authoritative surrogate for the `(collection, pk)` - /// endpoint key carried by `req`. Returns an [`AssignSurrogateResponse`] with - /// the surrogate on success or a typed error on failure (never a silent - /// drop). - async fn on_assign_surrogate( - &self, - req: crate::rpc_codec::AssignSurrogateRequest, - ) -> crate::rpc_codec::AssignSurrogateResponse; -} - -/// Hook for routed Calvin-submit (Cv1). -/// -/// `nodedb-cluster` cannot depend on `nodedb` (circular), so the submit logic — -/// decode the `TxClass`, submit it to THIS node's Calvin sequencer inbox, and -/// await assignment + completion through the node-local `CalvinCompletionRegistry` -/// — lives in `nodedb` behind this `Send + Sync` hook. The transport read-loop -/// calls [`on_submit_calvin_txn`](Self::on_submit_calvin_txn) when a -/// `SubmitCalvinTxnRequest` arrives at the SEQUENCER-GROUP leader and writes the -/// returned [`SubmitCalvinTxnResponse`](crate::rpc_codec::SubmitCalvinTxnResponse) -/// back to the coordinator. -/// -/// Because the handler runs on the sequencer-group leader, the submit-and-await -/// is correct: only the leader's sequencer service assigns transactions -/// (`note_assigned`), and only the leader's registry receives BOTH the -/// assignment and the replicated completion ack. The coordinator routes here -/// precisely so the submit lands where it will actually be sequenced and acked. -/// -/// Cluster-only tests leave the `RaftLoop` field `None`; a `SubmitCalvinTxn` -/// request against a node with no Calvin-submit hook installed returns a typed -/// "not configured" error. -/// -/// The hook is **async** because the submit-and-await blocks on the assignment -/// and completion oneshot channels (bounded by the request deadline) on the -/// Tokio transport reactor. The actual transaction execution happens on the Data -/// Plane via the sequencer service / per-vshard schedulers; this hook never -/// touches io_uring or storage directly. -#[async_trait::async_trait] -pub trait CalvinSubmit: Send + Sync + 'static { - /// Submit the `TxClass` carried by `req` (msgpack-encoded) to this node's - /// Calvin sequencer inbox and await its completion. Returns a - /// [`SubmitCalvinTxnResponse`](crate::rpc_codec::SubmitCalvinTxnResponse) - /// with `error: None` on commit or a typed error on failure (never a silent - /// drop). - async fn on_submit_calvin_txn( - &self, - req: crate::rpc_codec::SubmitCalvinTxnRequest, - ) -> crate::rpc_codec::SubmitCalvinTxnResponse; -} - -/// Hook for routed Calvin-INBOX submit (Cv1). -/// -/// OLLP dependent sibling of [`CalvinSubmit`]: `nodedb-cluster` cannot depend on -/// `nodedb` (circular), so the submit logic — decode the `TxClass`, submit it to -/// THIS node's Calvin sequencer inbox, and await only the ASSIGNMENT (NOT -/// completion) through the node-local `CalvinCompletionRegistry` — lives in -/// `nodedb` behind this `Send + Sync` hook. The transport read-loop calls -/// [`on_submit_calvin_inbox`](Self::on_submit_calvin_inbox) when a -/// `SubmitCalvinInboxRequest` arrives at the SEQUENCER-GROUP leader and writes the -/// returned [`SubmitCalvinInboxResponse`](crate::rpc_codec::SubmitCalvinInboxResponse) -/// back to the coordinator. -/// -/// Because the handler runs on the sequencer-group leader, the submit-and-assign -/// is correct: only the leader's sequencer service assigns transactions -/// (`note_assigned`). Unlike [`CalvinSubmit`] it returns AS SOON AS the -/// assignment is observed — the OLLP coordinator loop drives the dependent -/// transaction to completion itself in a later unit, so this hook must NOT block -/// until completion. -/// -/// Cluster-only tests leave the `RaftLoop` field `None`; a `SubmitCalvinInbox` -/// request against a node with no Calvin-inbox hook installed returns a typed -/// "not configured" error. -/// -/// The hook is **async** because the submit-and-assign blocks on the assignment -/// oneshot channel (bounded by the request deadline) on the Tokio transport -/// reactor. The actual transaction execution happens on the Data Plane via the -/// sequencer service / per-vshard schedulers; this hook never touches io_uring or -/// storage directly. -#[async_trait::async_trait] -pub trait CalvinSubmitInbox: Send + Sync + 'static { - /// Submit the `TxClass` carried by `req` (msgpack-encoded) to this node's - /// Calvin sequencer inbox and await its ASSIGNMENT (not completion). Returns - /// a [`SubmitCalvinInboxResponse`](crate::rpc_codec::SubmitCalvinInboxResponse) - /// with `error: None` carrying the assignment on success or a typed error on - /// failure (never a silent drop). - async fn on_submit_calvin_inbox( - &self, - req: crate::rpc_codec::SubmitCalvinInboxRequest, - ) -> crate::rpc_codec::SubmitCalvinInboxResponse; -} - -/// Hook for routed reserve-read (Calvin OLLP). -/// -/// `nodedb-cluster` cannot depend on `nodedb` (circular), so the reserve -/// logic — decode the `LockKey` and assign-only reserve the read lock through -/// THIS node's Calvin sequencer scheduler — lives in `nodedb` behind this -/// `Send + Sync` hook. The transport read-loop calls -/// [`on_reserve_read`](Self::on_reserve_read) when a `ReserveReadRequest` -/// arrives at the SEQUENCER-GROUP leader and writes the returned -/// [`ReserveReadResponse`](crate::rpc_codec::ReserveReadResponse) back to the -/// coordinator. -/// -/// Because the handler runs on the sequencer-group leader, the reserve is -/// correct: only the leader's scheduler holds the authoritative lock table for -/// its local sequencer inbox. The coordinator routes here precisely so the -/// reservation lands where it will actually be enforced. -/// -/// Cluster-only tests leave the `RaftLoop` field `None`; a `ReserveRead` -/// request against a node with no reserve-read hook installed returns a typed -/// "not configured" error. -/// -/// The hook is **async** for signature symmetry with the other one-shot -/// hooks; the reserve itself is bounded by the request deadline on the Tokio -/// transport reactor. It never touches io_uring or storage directly. -#[async_trait::async_trait] -pub trait ReserveRead: Send + Sync + 'static { - /// Assign-only reserve the read lock for the `LockKey` carried by `req`. - /// Returns a [`ReserveReadResponse`](crate::rpc_codec::ReserveReadResponse) - /// with the minted (or confirmed) owner on success or a typed error on - /// failure (never a silent drop). - async fn on_reserve_read( - &self, - req: crate::rpc_codec::ReserveReadRequest, - ) -> crate::rpc_codec::ReserveReadResponse; -} - -/// Hook for routed release-reservation (Calvin OLLP). -/// -/// Ack-only sibling of [`ReserveRead`]: `nodedb-cluster` cannot depend on -/// `nodedb` (circular), so the release logic — decode the owner and release -/// reason, and release the reservation through THIS node's Calvin sequencer -/// scheduler — lives in `nodedb` behind this `Send + Sync` hook. The transport -/// read-loop calls [`on_release_reservation`](Self::on_release_reservation) -/// when a `ReleaseReservationRequest` arrives at the SEQUENCER-GROUP leader -/// and writes the returned -/// [`ReleaseReservationResponse`](crate::rpc_codec::ReleaseReservationResponse) -/// back to the coordinator. -/// -/// Cluster-only tests leave the `RaftLoop` field `None`; a -/// `ReleaseReservation` request against a node with no release-reservation -/// hook installed returns a typed "not configured" error. -/// -/// The hook is **async** for signature symmetry with the other one-shot -/// hooks; the release itself is bounded by the request deadline on the Tokio -/// transport reactor. It never touches io_uring or storage directly. -#[async_trait::async_trait] -pub trait ReleaseReservation: Send + Sync + 'static { - /// Release the reservation held by the owner carried by `req`. Returns a - /// [`ReleaseReservationResponse`](crate::rpc_codec::ReleaseReservationResponse) - /// with `error: None` on success (ack) or a typed error on failure (never - /// a silent drop). - async fn on_release_reservation( - &self, - req: crate::rpc_codec::ReleaseReservationRequest, - ) -> crate::rpc_codec::ReleaseReservationResponse; -} diff --git a/nodedb-cluster/src/raft_loop/hooks_routed.rs b/nodedb-cluster/src/raft_loop/hooks_routed.rs new file mode 100644 index 000000000..1535c46ab --- /dev/null +++ b/nodedb-cluster/src/raft_loop/hooks_routed.rs @@ -0,0 +1,195 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Host-crate hooks for requests routed to a vShard or sequencer-group leader. +//! +//! Covers surrogate assignment, Calvin submit, Calvin inbox submit, and the +//! reserve-read and release-reservation pair. Re-exported through [`super::hooks`]. + +/// Hook for routed-surrogate-exchange. +/// +/// `nodedb-cluster` cannot depend on `nodedb` (circular), so the assign logic — +/// run a LOCAL `SurrogateAssigner::assign` for the `(collection, pk)` endpoint +/// key carried by the request — lives in `nodedb` behind this `Send + Sync` hook. +/// The transport read-loop calls [`on_assign_surrogate`](Self::on_assign_surrogate) +/// when an `AssignSurrogateRequest` arrives at the home vShard's LEADER and writes +/// the returned [`AssignSurrogateResponse`](crate::rpc_codec::AssignSurrogateResponse) +/// back to the coordinator. +/// +/// Because the handler runs on the home node (the vShard leader), a LOCAL assign +/// yields the AUTHORITATIVE surrogate: the first call allocates it and every later +/// call for the same key returns the same value (idempotent, first-wins). The +/// coordinator routes here precisely so the value it carries is the one the home +/// node will store under. +/// +/// Cluster-only tests leave the `RaftLoop` field `None`; an `AssignSurrogate` +/// request against a node with no assigner installed returns a typed "not +/// configured" error. +/// +/// The hook is **async** for signature symmetry with the other one-shot hooks; +/// the host-crate implementation performs a synchronous local assign (the +/// `SurrogateAssigner` is a sync `Send + Sync` facade) and never touches +/// io_uring or the Data Plane directly. +#[async_trait::async_trait] +pub trait AssignRemoteSurrogate: Send + Sync + 'static { + /// Assign-or-return the authoritative surrogate for the `(collection, pk)` + /// endpoint key carried by `req`. Returns an [`AssignSurrogateResponse`] with + /// the surrogate on success or a typed error on failure (never a silent + /// drop). + async fn on_assign_surrogate( + &self, + req: crate::rpc_codec::AssignSurrogateRequest, + ) -> crate::rpc_codec::AssignSurrogateResponse; +} + +/// Hook for routed Calvin-submit. +/// +/// `nodedb-cluster` cannot depend on `nodedb` (circular), so the submit logic — +/// decode the `TxClass`, submit it to THIS node's Calvin sequencer inbox, and +/// await assignment + completion through the node-local `CalvinCompletionRegistry` +/// — lives in `nodedb` behind this `Send + Sync` hook. The transport read-loop +/// calls [`on_submit_calvin_txn`](Self::on_submit_calvin_txn) when a +/// `SubmitCalvinTxnRequest` arrives at the SEQUENCER-GROUP leader and writes the +/// returned [`SubmitCalvinTxnResponse`](crate::rpc_codec::SubmitCalvinTxnResponse) +/// back to the coordinator. +/// +/// Because the handler runs on the sequencer-group leader, the submit-and-await +/// is correct: only the leader's sequencer service assigns transactions +/// (`note_assigned`), and only the leader's registry receives BOTH the +/// assignment and the replicated completion ack. The coordinator routes here +/// precisely so the submit lands where it will actually be sequenced and acked. +/// +/// Cluster-only tests leave the `RaftLoop` field `None`; a `SubmitCalvinTxn` +/// request against a node with no Calvin-submit hook installed returns a typed +/// "not configured" error. +/// +/// The hook is **async** because the submit-and-await blocks on the assignment +/// and completion oneshot channels (bounded by the request deadline) on the +/// Tokio transport reactor. The actual transaction execution happens on the Data +/// Plane via the sequencer service / per-vshard schedulers; this hook never +/// touches io_uring or storage directly. +#[async_trait::async_trait] +pub trait CalvinSubmit: Send + Sync + 'static { + /// Submit the `TxClass` carried by `req` (msgpack-encoded) to this node's + /// Calvin sequencer inbox and await its completion. Returns a + /// [`SubmitCalvinTxnResponse`](crate::rpc_codec::SubmitCalvinTxnResponse) + /// with `error: None` on commit or a typed error on failure (never a silent + /// drop). + async fn on_submit_calvin_txn( + &self, + req: crate::rpc_codec::SubmitCalvinTxnRequest, + ) -> crate::rpc_codec::SubmitCalvinTxnResponse; +} + +/// Hook for routed Calvin-INBOX submit. +/// +/// OLLP dependent sibling of [`CalvinSubmit`]: `nodedb-cluster` cannot depend on +/// `nodedb` (circular), so the submit logic — decode the `TxClass`, submit it to +/// THIS node's Calvin sequencer inbox, and await only the ASSIGNMENT (NOT +/// completion) through the node-local `CalvinCompletionRegistry` — lives in +/// `nodedb` behind this `Send + Sync` hook. The transport read-loop calls +/// [`on_submit_calvin_inbox`](Self::on_submit_calvin_inbox) when a +/// `SubmitCalvinInboxRequest` arrives at the SEQUENCER-GROUP leader and writes the +/// returned [`SubmitCalvinInboxResponse`](crate::rpc_codec::SubmitCalvinInboxResponse) +/// back to the coordinator. +/// +/// Because the handler runs on the sequencer-group leader, the submit-and-assign +/// is correct: only the leader's sequencer service assigns transactions +/// (`note_assigned`). Unlike [`CalvinSubmit`] it returns AS SOON AS the +/// assignment is observed — the OLLP coordinator loop drives the dependent +/// transaction to completion itself, so this hook must NOT block +/// until completion. +/// +/// Cluster-only tests leave the `RaftLoop` field `None`; a `SubmitCalvinInbox` +/// request against a node with no Calvin-inbox hook installed returns a typed +/// "not configured" error. +/// +/// The hook is **async** because the submit-and-assign blocks on the assignment +/// oneshot channel (bounded by the request deadline) on the Tokio transport +/// reactor. The actual transaction execution happens on the Data Plane via the +/// sequencer service / per-vshard schedulers; this hook never touches io_uring or +/// storage directly. +#[async_trait::async_trait] +pub trait CalvinSubmitInbox: Send + Sync + 'static { + /// Submit the `TxClass` carried by `req` (msgpack-encoded) to this node's + /// Calvin sequencer inbox and await its ASSIGNMENT (not completion). Returns + /// a [`SubmitCalvinInboxResponse`](crate::rpc_codec::SubmitCalvinInboxResponse) + /// with `error: None` carrying the assignment on success or a typed error on + /// failure (never a silent drop). + async fn on_submit_calvin_inbox( + &self, + req: crate::rpc_codec::SubmitCalvinInboxRequest, + ) -> crate::rpc_codec::SubmitCalvinInboxResponse; + + /// Offer a batch of a multi-part transaction's streamed parts to this + /// node's sequencer leader queue, and answer how far the stream got. + async fn on_calvin_parts( + &self, + req: crate::rpc_codec::CalvinPartsRequest, + ) -> crate::rpc_codec::CalvinPartsResponse; +} + +/// Hook for routed reserve-read (Calvin OLLP). +/// +/// `nodedb-cluster` cannot depend on `nodedb` (circular), so the reserve +/// logic — decode the `LockKey` and assign-only reserve the read lock through +/// THIS node's Calvin sequencer scheduler — lives in `nodedb` behind this +/// `Send + Sync` hook. The transport read-loop calls +/// [`on_reserve_read`](Self::on_reserve_read) when a `ReserveReadRequest` +/// arrives at the SEQUENCER-GROUP leader and writes the returned +/// [`ReserveReadResponse`](crate::rpc_codec::ReserveReadResponse) back to the +/// coordinator. +/// +/// Because the handler runs on the sequencer-group leader, the reserve is +/// correct: only the leader's scheduler holds the authoritative lock table for +/// its local sequencer inbox. The coordinator routes here precisely so the +/// reservation lands where it will actually be enforced. +/// +/// Cluster-only tests leave the `RaftLoop` field `None`; a `ReserveRead` +/// request against a node with no reserve-read hook installed returns a typed +/// "not configured" error. +/// +/// The hook is **async** for signature symmetry with the other one-shot +/// hooks; the reserve itself is bounded by the request deadline on the Tokio +/// transport reactor. It never touches io_uring or storage directly. +#[async_trait::async_trait] +pub trait ReserveRead: Send + Sync + 'static { + /// Assign-only reserve the read lock for the `LockKey` carried by `req`. + /// Returns a [`ReserveReadResponse`](crate::rpc_codec::ReserveReadResponse) + /// with the minted (or confirmed) owner on success or a typed error on + /// failure (never a silent drop). + async fn on_reserve_read( + &self, + req: crate::rpc_codec::ReserveReadRequest, + ) -> crate::rpc_codec::ReserveReadResponse; +} + +/// Hook for routed release-reservation (Calvin OLLP). +/// +/// Ack-only sibling of [`ReserveRead`]: `nodedb-cluster` cannot depend on +/// `nodedb` (circular), so the release logic — decode the owner and release +/// reason, and release the reservation through THIS node's Calvin sequencer +/// scheduler — lives in `nodedb` behind this `Send + Sync` hook. The transport +/// read-loop calls [`on_release_reservation`](Self::on_release_reservation) +/// when a `ReleaseReservationRequest` arrives at the SEQUENCER-GROUP leader +/// and writes the returned +/// [`ReleaseReservationResponse`](crate::rpc_codec::ReleaseReservationResponse) +/// back to the coordinator. +/// +/// Cluster-only tests leave the `RaftLoop` field `None`; a +/// `ReleaseReservation` request against a node with no release-reservation +/// hook installed returns a typed "not configured" error. +/// +/// The hook is **async** for signature symmetry with the other one-shot +/// hooks; the release itself is bounded by the request deadline on the Tokio +/// transport reactor. It never touches io_uring or storage directly. +#[async_trait::async_trait] +pub trait ReleaseReservation: Send + Sync + 'static { + /// Release the reservation held by the owner carried by `req`. Returns a + /// [`ReleaseReservationResponse`](crate::rpc_codec::ReleaseReservationResponse) + /// with `error: None` on success (ack) or a typed error on failure (never + /// a silent drop). + async fn on_release_reservation( + &self, + req: crate::rpc_codec::ReleaseReservationRequest, + ) -> crate::rpc_codec::ReleaseReservationResponse; +} diff --git a/nodedb-cluster/src/raft_loop/join.rs b/nodedb-cluster/src/raft_loop/join.rs index a5d94a793..9b0371b83 100644 --- a/nodedb-cluster/src/raft_loop/join.rs +++ b/nodedb-cluster/src/raft_loop/join.rs @@ -28,14 +28,12 @@ //! step 1 is intentionally *not* reused for the final response; a //! fresh clone is taken after step 6 so the response reflects the //! post-AddLearner routing state. -//! 6. **Propose AddLearner on each missing group.** For each Raft group -//! that does not already contain the node, take -//! the `MultiRaft` lock, propose -//! `ConfChange::AddLearner(new_node_id)`, and record the resulting -//! log index. Drop the lock between groups. Metadata-group admission -//! remains mandatory. Once that group already has three voters, -//! `NotLeader` on an independently led non-metadata group is deferred; -//! a later same-address join can reconcile the missing membership. +//! 6. **Propose AddLearner on the metadata and sequencer groups** where +//! they do not contain the node yet. Metadata-group admission is +//! mandatory. The sequencer group defers only once the metadata group +//! has three voters. Data groups follow placement: reconcile, kicked in +//! step 9, names the joiner in the placements that take it, and each +//! group's leader adds it there. //! 7. **Wait for each conf-change to apply.** Poll actual group membership //! every 20 ms with a 5-second deadline. A //! single-voter group (the bootstrap seed before any voters have @@ -46,8 +44,8 @@ //! attached). Order matters: Raft log → catalog → response. //! 9. **Broadcast TopologyUpdate** to every currently-active peer so //! followers learn the new node's address. Fire-and-forget. -//! 10. **Build and return JoinResponse** with the updated routing -//! (which now includes the new node as a learner on every group). +//! 10. **Build and return JoinResponse** with every group of the routing +//! view. A group this node hosts carries its Raft membership. //! //! The Raft-level promotion from learner to voter happens asynchronously //! in the tick loop (`super::tick::promote_ready_learners`) once the @@ -56,6 +54,7 @@ //! two-phase single-server add. use std::net::SocketAddr; +use std::sync::Arc; use std::time::{Duration, Instant}; use tracing::{debug, info, warn}; @@ -80,14 +79,25 @@ const CONF_CHANGE_COMMIT_TIMEOUT: Duration = Duration::from_secs(5); /// Polling interval for the commit-wait loop. const CONF_CHANGE_POLL_INTERVAL: Duration = Duration::from_millis(20); +/// The groups the join admits `node_id` to: the metadata group and the +/// sequencer group, where they do not contain it yet. +/// +/// A data group's members follow its placement alone. Reconcile authors the +/// placement with the joiner, each group's leader adds the joiner where the +/// placement names it, and the joiner mounts those groups. A join-time +/// learner in every data group would outrun that placement: the leader drops +/// each learner the placement it holds does not name, then adds it back +/// once the new placement applies. fn groups_requiring_admission(multi_raft: &MultiRaft, node_id: u64) -> Vec { multi_raft .group_ids() .into_iter() .filter(|group_id| { - !multi_raft - .group_contains_node(*group_id, node_id) - .unwrap_or(false) + (*group_id == TOPOLOGY_GROUP_ID + || *group_id == crate::calvin::sequencer::SEQUENCER_GROUP_ID) + && !multi_raft + .group_contains_node(*group_id, node_id) + .unwrap_or(false) }) .collect() } @@ -174,24 +184,13 @@ impl RaftLoop { // 4. Register transport peer so the leader can reach it. self.transport.register_peer(req.node_id, new_addr); - // Read the local cluster id from the catalog and echo it - // on every successful `JoinResponse`. The joining node - // persists this value so its next boot takes the - // `restart()` path instead of re-bootstrapping. - // - // Strict contract: - // - // - If a catalog is attached and is missing a cluster_id, - // the server is lying about being bootstrapped — this - // is an invariant violation, so we reject the join - // loudly instead of papering over it with a sentinel - // zero that would silently collapse two different - // clusters into one "cluster 0". - // - If a catalog is not attached (unit-test path), we - // fall back to `self.node_id`. This is a test-only - // affordance: it keeps the response well-formed without - // inventing a cross-cluster identity, because in tests - // every node id is locally unique by construction. + // Every successful `JoinResponse` echoes the catalog's cluster id. + // The joiner persists it, so its next boot takes `restart()`. + // - An attached catalog with no cluster id is an invariant + // violation: the join is rejected, never answered with a + // sentinel id. + // - With no catalog attached (unit tests), `self.node_id` stands + // in: test node ids are unique by construction. let cluster_id = match self.catalog.as_ref() { Some(catalog) => match catalog.load_cluster_id() { Ok(Some(id)) => id, @@ -221,20 +220,24 @@ impl RaftLoop { } } - // 6. Propose AddLearner on every group that does not already contain - // this node. This makes retries resume a partial join instead of - // treating topology insertion as proof that Raft admission finished. - let (group_ids, can_defer_non_metadata) = { + // 6. Propose AddLearner on the metadata group and the sequencer + // group where they do not contain this node yet, so a retry + // resumes a partial join. Data groups are left to placement (see + // `groups_requiring_admission`). Metadata-group admission is + // mandatory. The sequencer group defers only once the metadata + // group has three voters. + let (group_ids, metadata_voters) = { let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); let metadata_voters = mr .group_membership(TOPOLOGY_GROUP_ID) - .map(|membership| membership.voters.len()) - .unwrap_or(0); + .map_or(0, |membership| membership.voters.len()); ( groups_requiring_admission(&mr, req.node_id), - metadata_voters >= 3, + metadata_voters, ) }; + let deferrable = + |gid: u64| gid == crate::calvin::sequencer::SEQUENCER_GROUP_ID && metadata_voters >= 3; let mut pending: Vec<(u64, u64)> = Vec::with_capacity(group_ids.len()); // (group_id, log_index) for gid in &group_ids { @@ -248,9 +251,9 @@ impl RaftLoop { }; match propose_result { Ok((_, log_index)) => pending.push((*gid, log_index)), - Err(ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint })) - if can_defer_non_metadata && *gid != TOPOLOGY_GROUP_ID => - { + Err(ClusterError::Raft(nodedb_raft::RaftError::NotLeader { + leader_hint, .. + })) if deferrable(*gid) => { debug!( group_id = *gid, joining_node = req.node_id, @@ -259,16 +262,8 @@ impl RaftLoop { ); } Err(ClusterError::Transport { detail }) - if can_defer_non_metadata - && *gid != TOPOLOGY_GROUP_ID - && detail.contains("not leader") => + if deferrable(*gid) && detail.contains("not leader") => { - // Join routing is anchored on the metadata leader, but - // independent Raft groups can elect different leaders. - // Do not make topology admission unavailable merely - // because this node cannot propose a non-metadata group - // change. A later same-address join retries only the - // still-missing groups. debug!( group_id = *gid, joining_node = req.node_id, @@ -287,12 +282,8 @@ impl RaftLoop { } } - // 7. Wait for every conf change to actually *apply* to - // routing. Earlier versions of this flow polled - // `commit_index_for` and relied on an unconditional inline apply. - // The semantic signal is actual Raft membership: the node appears - // as either learner or voter. This works for non-routing groups and - // also tolerates promotion racing the polling interval. + // 7. Wait for every conf change to apply: the node appears in the + // group's Raft membership as a learner or a voter. let deadline = Instant::now() + CONF_CHANGE_COMMIT_TIMEOUT; for (gid, log_index) in &pending { if let Err(err) = self @@ -303,28 +294,35 @@ impl RaftLoop { } } - // 8. Persist catalog (topology + post-AddLearner routing). + // 8. Persist catalog (topology + post-AddLearner routing), off the + // async threads. The routing table goes through the one routing + // writer, so this save never lands after a newer one. if let Some(catalog) = self.catalog.as_ref() { let topo_snapshot = self .topology .read() .unwrap_or_else(|p| p.into_inner()) .clone(); - let routing_snapshot = { - let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - mr.routing() - .read() - .unwrap_or_else(|p| p.into_inner()) - .clone() + let catalog = Arc::clone(catalog); + let saved = + tokio::task::spawn_blocking(move || catalog.save_topology(&topo_snapshot)).await; + let error = match saved { + Ok(Ok(())) => None, + Ok(Err(e)) => Some(e.to_string()), + Err(e) => Some(format!("save task: {e}")), }; - if let Err(e) = catalog.save_topology(&topo_snapshot) { + if let Some(e) = error { warn!(error = %e, "failed to persist topology after join"); return reject(format!("catalog save_topology failed: {e}")); } - if let Err(e) = catalog.save_routing(&routing_snapshot) { - warn!(error = %e, "failed to persist routing after join"); - return reject(format!("catalog save_routing failed: {e}")); - } + } + if let Some(persister) = self.routing_persister.as_ref() + && !persister.wait(persister.request()).await + { + warn!("failed to persist routing after join"); + return reject( + "catalog save_routing failed; see the routing persister's warning".into(), + ); } // 9. Broadcast topology to everyone so peers learn the new addr. @@ -418,7 +416,7 @@ impl RaftLoop { .clone(); let (routing_clone, raft_groups) = { let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - let groups = mr + let groups: Vec = mr .group_ids() .into_iter() .filter_map(|group_id| mr.group_membership(group_id)) @@ -443,11 +441,31 @@ impl RaftLoop { // wire response. let mut topo = topology_clone; let mut response = handle_join_request(req, &mut topo, &routing_clone, cluster_id); - response.groups = raft_groups; + overlay_hosted_groups(&mut response.groups, raft_groups); response } } +/// Replace each routing-view group in `groups` with the Raft membership this +/// node hosts for it, and add the hosted groups routing does not list. +/// +/// `groups` starts as every group of this node's routing view. The joiner +/// builds its whole routing table from the response. A group missing from it +/// has no routing entry on the joiner, and nothing adds one later: the leader +/// probe and the placement apply only move entries that exist. A replication +/// factor below the node count leaves groups this node hosts no replica of, +/// so the response cannot list only the hosted groups. A hosted group's Raft +/// membership holds every conf change this node applied, so it replaces the +/// routing entry. +fn overlay_hosted_groups(groups: &mut Vec, hosted: Vec) { + for group in hosted { + match groups.iter_mut().find(|g| g.group_id == group.group_id) { + Some(slot) => *slot = group, + None => groups.push(group), + } + } +} + /// Build a failure `JoinResponse` with the given error message. fn reject(error: String) -> JoinResponse { JoinResponse { @@ -491,5 +509,46 @@ mod tests { groups_requiring_admission(&multi_raft, 2), vec![SEQUENCER_GROUP_ID] ); + // A new node is admitted to the metadata and sequencer groups only. + // Data group 1 takes it as its placement names it. + let mut fresh = groups_requiring_admission(&multi_raft, 3); + fresh.sort_unstable(); + assert_eq!(fresh, vec![0, SEQUENCER_GROUP_ID]); + } + + fn info(group_id: u64, leader: u64, members: &[u64], learners: &[u64]) -> JoinGroupInfo { + JoinGroupInfo { + group_id, + leader, + members: members.to_vec(), + learners: learners.to_vec(), + } + } + + #[test] + fn the_response_keeps_the_groups_this_node_hosts_no_replica_of() { + // Routing lists groups 0, 1 and 2. This node hosts 0, 1 and the + // sequencer; group 2 lives on node 2 only. + let mut groups = vec![ + info(0, 1, &[1, 2], &[]), + info(1, 1, &[1], &[]), + info(2, 2, &[2], &[]), + ]; + let hosted = vec![ + info(0, 1, &[1, 2], &[3]), + info(1, 1, &[1], &[3]), + info(SEQUENCER_GROUP_ID, 1, &[1, 2], &[3]), + ]; + overlay_hosted_groups(&mut groups, hosted); + groups.sort_by_key(|g| g.group_id); + assert_eq!( + groups, + vec![ + info(0, 1, &[1, 2], &[3]), + info(1, 1, &[1], &[3]), + info(2, 2, &[2], &[]), + info(SEQUENCER_GROUP_ID, 1, &[1, 2], &[3]), + ] + ); } } diff --git a/nodedb-cluster/src/raft_loop/leader_balance.rs b/nodedb-cluster/src/raft_loop/leader_balance.rs new file mode 100644 index 000000000..6e4561563 --- /dev/null +++ b/nodedb-cluster/src/raft_loop/leader_balance.rs @@ -0,0 +1,346 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Leader balance: each data group's leader hands leadership to the group's +//! preferred leader (see [`crate::rebalancer::leader_preference`]). +//! +//! The bootstrap node leads every group, and nothing in Raft moves a leader +//! that stays healthy. This phase moves every group's leadership to its +//! preferred leader, so a formed cluster spreads its leaders over the voters. +//! +//! A leader transfers when all of these hold: +//! - this node is not leaving the group (the step-aside phase moves that +//! leader to a placement voter); +//! - the preferred leader is another current voter of the group; +//! - it was healthy for [`PREFERRED_HEALTHY_PASSES`] balance passes in a +//! row, this pass included. Healthy means it answered a heartbeat since +//! the previous pass and holds every committed entry. +//! +//! The streak damps failover. A preferred leader that just came back, or +//! that missed one heartbeat round, earns its groups back only after it +//! stayed healthy for the whole streak. How this node won the group plays +//! no part. A check-quorum step-down that this node wins back keeps the +//! streak: the pass after the new term counts again (see [`PeerHealth`]). +//! Every pass rechecks every group, so the balance converges whenever the +//! rules hold, however the voter set changed. + +use std::collections::HashSet; + +use tracing::debug; + +use crate::forward::PlanExecutor; +use crate::rebalancer::preferred_leaders; + +use super::loop_core::{CommitApplier, RaftLoop}; + +/// Ticks between leader-balance passes (about 0.5 s at the 10 ms tick). A +/// pass apart spans several heartbeats, so a live voter always answers one. +pub(super) const LEADER_BALANCE_TICK_INTERVAL: u64 = 50; + +/// Consecutive healthy passes a preferred leader needs before a transfer to +/// it (about 2 s). A node that failed one pass starts the streak over. +pub(super) const PREFERRED_HEALTHY_PASSES: u32 = 4; + +/// Passes a peer's record outlives its last sample. One unsampled pass +/// covers a brief loss of leadership. A longer gap starts the streak over. +pub(super) const PEER_HEALTH_GAP_PASSES: u64 = 2; + +/// One pass's view of a peer from the group's leader. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct PeerSample { + /// The leader's term. Response counts restart at zero in every term. + pub term: u64, + /// `AppendEntries` responses from the peer in `term`. + pub acks: u64, + /// Whether the peer holds every committed entry. + pub caught_up: bool, +} + +/// What the balance knows of one peer of one group across passes. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct PeerHealth { + /// Term of the last sample. + term: u64, + /// Response count of the last sample. + acks: u64, + /// Healthy passes in a row up to the last sample. + streak: u32, + /// Whether the last sample itself showed the peer healthy. + healthy_now: bool, + /// Pass of the last sample. + seen_pass: u64, +} + +impl PeerHealth { + /// The record after `sample`, taken at balance pass `pass`. + /// + /// - A sample in the same term counts the pass healthy when the response + /// count rose and the peer is caught up. Any other sample ends the + /// streak. + /// - A sample in a new term cannot compare counts. The pass sets a new + /// baseline, keeps the streak and does not count as healthy. + /// - A record older than [`PEER_HEALTH_GAP_PASSES`] starts over. + pub(super) fn next(previous: Option, sample: PeerSample, pass: u64) -> Self { + let recent = previous.filter(|prev| prev.is_current(pass)); + let (streak, healthy_now) = match recent { + Some(prev) if prev.term == sample.term => { + if sample.acks > prev.acks && sample.caught_up { + (prev.streak.saturating_add(1), true) + } else { + (0, false) + } + } + Some(prev) => (prev.streak, false), + None => (0, false), + }; + Self { + term: sample.term, + acks: sample.acks, + streak, + healthy_now, + seen_pass: pass, + } + } + + /// Whether the record still counts at balance pass `pass`. + pub(super) fn is_current(&self, pass: u64) -> bool { + pass.saturating_sub(self.seen_pass) <= PEER_HEALTH_GAP_PASSES + } + + /// Whether the peer earned leadership: healthy at the last sample, and + /// for [`PREFERRED_HEALTHY_PASSES`] passes in a row. + pub(super) fn earned_leadership(&self) -> bool { + self.healthy_now && self.streak >= PREFERRED_HEALTHY_PASSES + } +} + +impl RaftLoop { + /// Transfer each data group this node leads to its preferred leader, + /// when the rules above allow it. `pass` numbers the balance passes and + /// rises by one each pass. + pub(super) fn balance_leadership(&self, pass: u64) { + self.tick_state.forget_stale_peer_health(pass); + let transfers: Vec<(u64, u64)> = { + let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let leading: Vec = mr + .group_ids() + .into_iter() + .filter(|gid| is_data_group(*gid) && mr.group_role_is_leader(*gid)) + .collect(); + let (preferred, leaving): (_, HashSet) = { + let routing = mr.routing(); + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let leaving = leading + .iter() + .copied() + .filter(|gid| { + routing + .group_info(*gid) + .and_then(|info| info.placement.as_ref()) + .is_some_and(|placement| !placement.contains(&self.node_id)) + }) + .collect(); + (preferred_leaders(&routing), leaving) + }; + let mut out = Vec::new(); + for gid in leading { + let Some(&target) = preferred.get(&gid) else { + continue; + }; + if target == self.node_id || leaving.contains(&gid) { + continue; + } + let is_voter = mr + .group_membership(gid) + .is_some_and(|m| m.voters.contains(&target)); + let (Some(term), Some(acks)) = + (mr.leader_term(gid), mr.peer_ack_count(gid, target)) + else { + continue; + }; + let sample = PeerSample { + term, + acks, + caught_up: mr.peer_caught_up(gid, target), + }; + // Sampled on every pass, so the streak is known once it is long enough. + let health = self + .tick_state + .observe_peer_health(gid, target, sample, pass); + if is_voter && health.earned_leadership() { + out.push((gid, target)); + } + } + out + }; + + for (group_id, target) in transfers { + let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + match mr.transfer_leadership(group_id, target) { + Ok(()) => debug!( + group_id, + target, "leader balance: transferring to the preferred leader" + ), + Err(e) => debug!( + group_id, + target, + error = %e, + "leader balance: transfer refused" + ), + } + } + } +} + +/// Whether `group_id` is a data group: neither the metadata group nor the +/// sequencer group. +fn is_data_group(group_id: u64) -> bool { + group_id != crate::metadata_group::METADATA_GROUP_ID + && group_id != crate::calvin::sequencer::SEQUENCER_GROUP_ID +} + +#[cfg(test)] +mod tests { + use super::*; + + fn sample(term: u64, acks: u64, caught_up: bool) -> PeerSample { + PeerSample { + term, + acks, + caught_up, + } + } + + /// Feed `samples` to a fresh record, one per pass from `first_pass`. + fn run(first_pass: u64, samples: &[Option]) -> Vec> { + let mut record: Option = None; + let mut out = Vec::new(); + for (offset, sample) in samples.iter().enumerate() { + let pass = first_pass + offset as u64; + if let Some(sample) = sample { + record = Some(PeerHealth::next(record, *sample, pass)); + } + out.push(record.filter(|r| r.is_current(pass))); + } + out + } + + #[test] + fn a_steadily_healthy_peer_earns_leadership_after_the_streak() { + let samples: Vec<_> = (0..=u64::from(PREFERRED_HEALTHY_PASSES)) + .map(|i| Some(sample(1, 10 + i, true))) + .collect(); + let records = run(0, &samples); + let earned: Vec = records + .iter() + .map(|r| r.is_some_and(|r| r.earned_leadership())) + .collect(); + let last = earned.len() - 1; + assert!(earned[..last].iter().all(|e| !e), "{earned:?}"); + assert!(earned[last], "{earned:?}"); + } + + /// The leader steps down on check-quorum and wins the group back every + /// few passes. The preferred voter answers throughout, so the streak + /// survives each new term and the transfer still happens. + #[test] + fn a_check_quorum_flap_among_live_voters_does_not_block_the_transfer() { + let mut samples = Vec::new(); + let mut term = 1; + let mut acks = 0; + let mut earned_at = None; + for pass in 0..40u64 { + // Every third pass this node is a follower, then wins a new term. + if pass % 3 == 2 { + samples.push(None); + term += 1; + acks = 0; + continue; + } + acks += 5; + samples.push(Some(sample(term, acks, true))); + let records = run(0, &samples); + if records + .last() + .copied() + .flatten() + .is_some_and(|r| r.earned_leadership()) + { + earned_at = Some(pass); + break; + } + } + let earned_at = earned_at.expect("flapping leadership blocked the transfer"); + assert!(earned_at < 12, "earned at pass {earned_at}"); + } + + /// A preferred leader that stopped answering for one pass is not handed + /// leadership until it answered a full streak again. + #[test] + fn a_recently_failed_peer_waits_for_a_full_streak() { + let streak = u64::from(PREFERRED_HEALTHY_PASSES); + let mut samples: Vec<_> = (0..=streak) + .map(|i| Some(sample(1, 10 + i, true))) + .collect(); + // The peer misses a heartbeat round: its count does not rise. + let stalled = 10 + streak; + samples.push(Some(sample(1, stalled, true))); + for i in 1..=streak { + samples.push(Some(sample(1, stalled + i, true))); + } + let records = run(0, &samples); + let failed_at = streak as usize + 1; + assert!(records[failed_at - 1].is_some_and(|r| r.earned_leadership())); + for (offset, record) in records[failed_at..].iter().enumerate() { + let earned = record.is_some_and(|r| r.earned_leadership()); + assert_eq!( + earned, + offset as u64 == streak, + "pass {} after the failure", + offset + ); + } + } + + /// A peer that is not caught up breaks the streak like a missed answer. + #[test] + fn a_lagging_peer_starts_the_streak_over() { + let first = PeerHealth::next(None, sample(1, 1, true), 0); + let healthy = PeerHealth::next(Some(first), sample(1, 2, true), 1); + assert_eq!(healthy.streak, 1); + let lagging = PeerHealth::next(Some(healthy), sample(1, 3, false), 2); + assert_eq!(lagging.streak, 0); + assert!(!lagging.healthy_now); + } + + /// A record unsampled for longer than the gap starts over, so a node + /// that left and came back earns its streak again. + #[test] + fn a_stale_record_starts_over() { + let mut record = PeerHealth::next(None, sample(1, 0, true), 0); + for pass in 1..=u64::from(PREFERRED_HEALTHY_PASSES) { + record = PeerHealth::next(Some(record), sample(1, pass, true), pass); + } + assert!(record.earned_leadership()); + let late = u64::from(PREFERRED_HEALTHY_PASSES) + PEER_HEALTH_GAP_PASSES + 1; + assert!(!record.is_current(late)); + let fresh = PeerHealth::next(Some(record), sample(2, 50, true), late); + assert_eq!(fresh.streak, 0); + assert!(!fresh.earned_leadership()); + } + + /// A new term sets a baseline: it keeps the streak but never transfers + /// on a sample whose count it cannot compare. + #[test] + fn a_new_term_keeps_the_streak_without_counting_the_pass() { + let mut record = PeerHealth::next(None, sample(1, 0, true), 0); + for pass in 1..=u64::from(PREFERRED_HEALTHY_PASSES) { + record = PeerHealth::next(Some(record), sample(1, pass, true), pass); + } + let pass = u64::from(PREFERRED_HEALTHY_PASSES) + 2; + let baseline = PeerHealth::next(Some(record), sample(2, 1, true), pass); + assert_eq!(baseline.streak, record.streak); + assert!(!baseline.earned_leadership()); + let next = PeerHealth::next(Some(baseline), sample(2, 3, true), pass + 1); + assert!(next.earned_leadership()); + } +} diff --git a/nodedb-cluster/src/raft_loop/leadership_transfer.rs b/nodedb-cluster/src/raft_loop/leadership_transfer.rs index e77c797a0..4a0e4963d 100644 --- a/nodedb-cluster/src/raft_loop/leadership_transfer.rs +++ b/nodedb-cluster/src/raft_loop/leadership_transfer.rs @@ -34,6 +34,9 @@ impl RaftLoop { let transfers: Vec<(u64, u64)> = { let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); let group_ids = mr.group_ids(); + let preferred_map = crate::rebalancer::preferred_leaders( + &mr.routing().read().unwrap_or_else(|p| p.into_inner()), + ); let mut out = Vec::new(); for gid in group_ids { if gid == crate::metadata_group::METADATA_GROUP_ID @@ -66,11 +69,14 @@ impl RaftLoop { continue; } // Pick an in-placement node that is currently a voter (so it can - // win the election) and is not this node. - let target = placement - .iter() - .copied() - .find(|t| *t != m.leader_id && m.voters.contains(t)); + // win the election) and is not this node. The group's preferred + // leader goes first, so the leader balance phase does not move + // the leadership again. + let preferred = preferred_map.get(&gid).copied(); + let eligible = |t: &u64| *t != m.leader_id && m.voters.contains(t); + let target = preferred + .filter(|t| placement.contains(t) && eligible(t)) + .or_else(|| placement.iter().copied().find(|t| eligible(t))); if let Some(target) = target { out.push((gid, target)); } else { diff --git a/nodedb-cluster/src/raft_loop/lease_gc.rs b/nodedb-cluster/src/raft_loop/lease_gc.rs index 6a497e4ff..bbb6184a2 100644 --- a/nodedb-cluster/src/raft_loop/lease_gc.rs +++ b/nodedb-cluster/src/raft_loop/lease_gc.rs @@ -4,13 +4,17 @@ //! //! Mirrors `placement_reconcile`: leader-gated, throttled by tick count in //! `tick::core::do_tick`. Sweeps `MetadataCache.leases` and proposes -//! `DescriptorLeaseRelease` for every holder no longer in the cluster -//! topology. This is the safety net behind the Leave apply hook. +//! `DescriptorLeaseRelease` for every holder that is no longer in the cluster +//! topology, or that is both SWIM-Dead and Raft-silent past its grace (see +//! [`crate::lease_liveness`]). This is the safety net behind the Leave apply +//! hook. use std::collections::HashMap; +use std::time::Instant; use tracing::{debug, warn}; use crate::forward::PlanExecutor; +use crate::lease_liveness::LeaseHolderLiveness; use crate::metadata_group::cache::MetadataCache; use crate::metadata_group::descriptors::DescriptorId; use crate::topology::ClusterTopology; @@ -18,16 +22,25 @@ use crate::topology::ClusterTopology; use super::loop_core::{CommitApplier, RaftLoop}; /// Pure collection: `(node_id, descriptor_ids)` for every lease holder that -/// is not in `topology`. Sorted by `node_id` for deterministic proposal -/// order. Extracted so the sweep's decision logic is unit-testable without -/// a full `RaftLoop`. -pub(super) fn collect_non_member_lease_releases( +/// is not in `topology`, or whose leases `liveness` counts as released in +/// `leader_term` at `now`. `local_node` is never collected. Sorted by +/// `node_id` for deterministic proposal order. +pub(super) fn collect_stale_lease_releases( topology: &ClusterTopology, cache: &MetadataCache, + liveness: &LeaseHolderLiveness, + local_node: u64, + leader_term: u64, + now: Instant, ) -> Vec<(u64, Vec)> { let mut by_holder: HashMap> = HashMap::new(); for (id, holder) in cache.leases.keys() { - if !topology.contains(*holder) { + if *holder == local_node { + continue; + } + if !topology.contains(*holder) + || liveness.dead_holder_released(*holder, Some(leader_term), now) + { by_holder.entry(*holder).or_default().push(id.clone()); } } @@ -37,20 +50,34 @@ pub(super) fn collect_non_member_lease_releases( } impl RaftLoop { - /// On the metadata-group leader, propose `DescriptorLeaseRelease` for - /// every lease whose holder is no longer in `ClusterTopology`. + /// On the metadata-group leader, sample each peer's Raft contact, then + /// propose `DescriptorLeaseRelease` for every lease whose holder left the + /// topology, or stayed SWIM-Dead and Raft-silent past its grace. pub(super) fn gc_stale_node_leases(&self) { let Some(cache) = &self.metadata_cache else { return; // not wired (some tests) — nothing to sweep }; let to_release: Vec<(u64, Vec)> = { let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - if !mr.group_role_is_leader(crate::metadata_group::METADATA_GROUP_ID) { + let Some(sample) = mr.leader_peer_acks(crate::metadata_group::METADATA_GROUP_ID) else { + self.lease_holder_liveness.clear_raft_contact(); return; - } + }; + let now = Instant::now(); + self.lease_holder_liveness + .observe_raft_contact(&sample, now); let topo = self.topology.read().unwrap_or_else(|p| p.into_inner()); + // A holder outside topology is released by membership alone. + self.lease_holder_liveness.retain(|id| topo.contains(id)); let cache = cache.read().unwrap_or_else(|p| p.into_inner()); - collect_non_member_lease_releases(&topo, &cache) + collect_stale_lease_releases( + &topo, + &cache, + &self.lease_holder_liveness, + self.node_id, + sample.term, + now, + ) }; for (node_id, descriptor_ids) in to_release { @@ -65,11 +92,11 @@ impl RaftLoop { continue; } }; - match self.propose_to_metadata_group(bytes) { + match self.propose_stamped_to_metadata_group(&bytes) { Ok(idx) => debug!( node_id, log_index = idx, - "lease GC: released leases of non-member node" + "lease GC: released leases of non-member or dead node" ), Err(e) => { warn!(node_id, error = %e, "lease GC: proposal failed; will be retried on next sweep") @@ -83,9 +110,31 @@ impl RaftLoop { mod tests { use crate::metadata_group::cache::MetadataCache; use crate::metadata_group::descriptors::{DescriptorId, DescriptorKind}; + use std::time::{Duration, Instant}; + + use crate::lease_liveness::{DEAD_HOLDER_LEASE_GRACE, LeaseHolderLiveness}; + use crate::multi_raft::PeerAckSample; use crate::topology::{ClusterTopology, NodeInfo, NodeState}; - use super::collect_non_member_lease_releases; + use super::collect_stale_lease_releases; + + const LOCAL: u64 = 1; + const TERM: u64 = 2; + + /// Collect with no SWIM Dead records. + fn collect_non_member( + topo: &ClusterTopology, + cache: &MetadataCache, + ) -> Vec<(u64, Vec)> { + collect_stale_lease_releases( + topo, + cache, + &LeaseHolderLiveness::new(), + LOCAL, + TERM, + Instant::now(), + ) + } fn topo_with(ids: &[u64]) -> ClusterTopology { let mut t = ClusterTopology::new(); @@ -132,7 +181,7 @@ mod tests { .leases .insert((metrics.clone(), 3), lease(&metrics, 3)); - let collected = collect_non_member_lease_releases(&topo, &cache); + let collected = collect_non_member(&topo, &cache); assert_eq!(collected.len(), 2); assert_eq!(collected[0].0, 2); assert_eq!(collected[0].1, vec![orders.clone()]); @@ -153,13 +202,97 @@ mod tests { .insert((orders.clone(), holder), lease(&orders, holder)); } - assert!(collect_non_member_lease_releases(&topo, &cache).is_empty()); + assert!(collect_non_member(&topo, &cache).is_empty()); } #[test] fn gc_collects_empty_cache() { let topo = topo_with(&[1]); let cache = MetadataCache::new(); - assert!(collect_non_member_lease_releases(&topo, &cache).is_empty()); + assert!(collect_non_member(&topo, &cache).is_empty()); + } + + /// Leader samples showing holder 2 Raft-silent from `from` to `to`. + fn raft_silent(liveness: &LeaseHolderLiveness, from: Instant, to: Instant) { + let sample = PeerAckSample { + term: TERM, + acks: vec![(2, 5)], + }; + liveness.observe_raft_contact(&sample, from); + liveness.observe_raft_contact(&sample, to); + } + + #[test] + fn dead_member_holder_is_collected_only_after_the_grace() { + let topo = topo_with(&[1, 2]); + let mut cache = MetadataCache::new(); + let orders = DescriptorId::new(0, 1, DescriptorKind::Collection, "orders".to_string()); + cache.leases.insert((orders.clone(), 2), lease(&orders, 2)); + let liveness = LeaseHolderLiveness::new(); + + let dead_at = Instant::now(); + let after_grace = dead_at + DEAD_HOLDER_LEASE_GRACE + Duration::from_millis(1); + raft_silent(&liveness, dead_at, after_grace); + liveness.record_dead_at(2, dead_at); + assert!( + collect_stale_lease_releases(&topo, &cache, &liveness, LOCAL, TERM, dead_at).is_empty() + ); + + let collected = + collect_stale_lease_releases(&topo, &cache, &liveness, LOCAL, TERM, after_grace); + assert_eq!(collected, vec![(2, vec![orders])]); + } + + /// SWIM says Dead, but holder 2 answered the leader over Raft recently: + /// it has not fenced, so its lease is not released. + #[test] + fn dead_holder_recently_acked_by_raft_is_kept() { + let topo = topo_with(&[1, 2]); + let mut cache = MetadataCache::new(); + let orders = DescriptorId::new(0, 1, DescriptorKind::Collection, "orders".to_string()); + cache.leases.insert((orders.clone(), 2), lease(&orders, 2)); + let liveness = LeaseHolderLiveness::new(); + + let dead_at = Instant::now(); + let after_grace = dead_at + DEAD_HOLDER_LEASE_GRACE + Duration::from_millis(1); + liveness.record_dead_at(2, dead_at); + liveness.observe_raft_contact( + &PeerAckSample { + term: TERM, + acks: vec![(2, 5)], + }, + dead_at, + ); + liveness.observe_raft_contact( + &PeerAckSample { + term: TERM, + acks: vec![(2, 6)], + }, + after_grace, + ); + + assert!( + collect_stale_lease_releases(&topo, &cache, &liveness, LOCAL, TERM, after_grace) + .is_empty() + ); + } + + #[test] + fn local_node_is_never_collected() { + let topo = topo_with(&[2]); + let mut cache = MetadataCache::new(); + let orders = DescriptorId::new(0, 1, DescriptorKind::Collection, "orders".to_string()); + cache + .leases + .insert((orders.clone(), LOCAL), lease(&orders, LOCAL)); + let liveness = LeaseHolderLiveness::new(); + let dead_at = Instant::now(); + liveness.record_dead_at(LOCAL, dead_at); + + let after_grace = dead_at + DEAD_HOLDER_LEASE_GRACE + Duration::from_millis(1); + assert!( + collect_stale_lease_releases(&topo, &cache, &liveness, LOCAL, TERM, after_grace) + .is_empty() + ); } } diff --git a/nodedb-cluster/src/raft_loop/loop_core.rs b/nodedb-cluster/src/raft_loop/loop_core.rs index 6c1bf03b0..65061f2c1 100644 --- a/nodedb-cluster/src/raft_loop/loop_core.rs +++ b/nodedb-cluster/src/raft_loop/loop_core.rs @@ -8,9 +8,7 @@ use std::pin::Pin; use std::sync::{Arc, Mutex, RwLock}; -use std::time::{Duration, Instant}; - -use tracing::debug; +use std::time::Duration; use nodedb_raft::message::LogEntry; @@ -85,6 +83,9 @@ pub struct RaftLoop { /// from the join flow. When `None`, persistence is skipped — useful /// for unit tests that don't care about durability. pub(super) catalog: Option>, + /// The single writer of the routing table to `catalog`, off the async + /// threads. Set with the catalog. `None` saves nothing. + pub(super) routing_persister: Option>, /// Cooperative shutdown signal observed by every detached /// `tokio::spawn` task in [`super::tick`]. `run()` flips it on /// its own shutdown, and [`Self::begin_shutdown`] provides a @@ -249,8 +250,8 @@ pub struct RaftLoop { /// /// When set (by the `nodedb` binary via `with_snapshot_applier`), the /// install-snapshot finalize path applies the received per-group snapshot to - /// the local Data-Plane state machine AFTER the atomic `.partial`→`.snap` - /// rename and BEFORE advancing Raft. Cluster-only tests leave this `None`, + /// the local Data-Plane state machine AFTER staging it and BEFORE advancing + /// Raft. Cluster-only tests leave this `None`, /// which makes the follower advance Raft without restoring engine state /// (correct for the empty bootstrap stub shipped by those tests). pub(super) snapshot_applier: Option>, @@ -274,16 +275,15 @@ pub struct RaftLoop { /// Orphan partial-snapshot max age for the GC sweeper (seconds). pub(super) orphan_partial_max_age_secs: u64, - /// Cluster replication factor (target voters per group), loaded once from - /// `ClusterSettings` at startup; immutable for the loop's lifetime. Used - /// to cap voter promotion at min(RF, N). Defaults to 1 (single-node, no - /// extra replicas) when not overridden via `with_replication_factor`. + /// Target voters per group, loaded from `ClusterSettings` at startup. + /// Defaults to 1 when not overridden via `with_replication_factor`. pub(super) replication_factor: u32, - /// Monotonic tick counter; throttles periodic maintenance (placement - /// reconcile) to a coarse cadence off the 10ms tick. `AtomicU64` because - /// [`super::tick::do_tick`] runs against `&self`. + /// Monotonic tick counter. It throttles periodic maintenance off the + /// 10ms tick. `AtomicU64` because `do_tick` runs against `&self`. pub(super) tick_count: std::sync::atomic::AtomicU64, + /// Samples and in-flight markers the periodic tick phases carry. + pub(super) tick_state: Arc, /// Notification channel for kicking placement reconcile immediately /// when a node joins, instead of waiting up to ~1 s for the throttled @@ -296,16 +296,20 @@ pub struct RaftLoop { /// periodic lease-GC sweep can read committed lease state directly. /// `None` in cluster-only tests that don't wire it. pub(super) metadata_cache: Option>>, + /// SWIM Dead records and Raft contact samples for the lease-GC sweep. + pub(super) lease_holder_liveness: Arc, } impl RaftLoop { pub fn new( - multi_raft: MultiRaft, + mut multi_raft: MultiRaft, transport: Arc, topology: Arc>, applier: A, ) -> Self { let node_id = multi_raft.node_id(); + let group_watchers = Arc::new(GroupAppliedWatchers::new()); + multi_raft.set_applied_watchers(Arc::clone(&group_watchers)); // Share the transport's epoch state rather than making a second one: // the generation this node applies and the generation it stamps must // be the same fact. @@ -323,10 +327,11 @@ impl RaftLoop { tick_interval: DEFAULT_TICK_INTERVAL, vshard_handler: None, catalog: None, + routing_persister: None, shutdown_watch, ready_watch, loop_metrics: LoopMetrics::new("raft_tick_loop"), - group_watchers: Arc::new(GroupAppliedWatchers::new()), + group_watchers, prev_metadata_leader: std::sync::atomic::AtomicBool::new(false), cluster_epoch, snapshot_quarantine_hook: None, @@ -348,8 +353,10 @@ impl RaftLoop { orphan_partial_max_age_secs: 300, replication_factor: 1, tick_count: std::sync::atomic::AtomicU64::new(0), + tick_state: Arc::new(super::tick_state::TickState::new()), reconcile_notify: tokio::sync::Notify::new(), metadata_cache: None, + lease_holder_liveness: Arc::new(crate::lease_liveness::LeaseHolderLiveness::new()), } } } @@ -420,65 +427,6 @@ impl RaftLoop { self.replication_factor } - /// Run the event loop until shutdown. - /// - /// This drives Raft elections, heartbeats, and message dispatch. - /// Call [`NexarTransport::serve`] separately with `Arc` as the handler. - /// - /// When the externally-supplied `shutdown` receiver fires, - /// the loop also propagates the signal to the internal - /// cooperative-shutdown channel so every detached task - /// spawned inside `do_tick` exits promptly and drops its - /// `Arc>` clone. - pub async fn run(&self, mut shutdown: tokio::sync::watch::Receiver) { - let mut interval = tokio::time::interval(self.tick_interval); - interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); - self.loop_metrics.set_up(true); - - // Startup GC sweep: remove orphaned partial-snapshot files from - // previous runs that did not complete. - if let Some(ref dir) = self.data_dir { - match crate::install_snapshot::gc::sweep_orphans(dir, self.orphan_partial_max_age_secs) - { - Ok((removed, errs)) => { - if removed > 0 { - tracing::info!(removed, "startup: removed orphaned partial snapshot files"); - } - for e in errs { - tracing::warn!(error = %e, "startup: partial snapshot GC error"); - } - } - Err(e) => { - tracing::warn!(error = %e, "startup: failed to sweep partial snapshot directory"); - } - } - } - - loop { - tokio::select! { - _ = interval.tick() => { - let started = Instant::now(); - self.do_tick(); - self.loop_metrics.observe(started.elapsed()); - } - _ = self.reconcile_notify.notified() => { - if *shutdown.borrow() { - return; - } - self.reconcile_placement(); - } - _ = shutdown.changed() => { - if *shutdown.borrow() { - debug!("raft loop shutting down"); - self.begin_shutdown(); - break; - } - } - } - } - self.loop_metrics.set_up(false); - } - /// Returns the inner multi-raft handle. Exposed for tests and for /// the host crate's metadata proposer so it can hold a second /// reference to the same underlying mutex without pulling the @@ -540,11 +488,15 @@ mod tests { applied: Arc, } + #[async_trait::async_trait] impl MetadataApplier for CountingMetadataApplier { - fn apply(&self, entries: &[(u64, Vec)]) -> u64 { + async fn apply_decoded( + &self, + entries: &[crate::metadata_group::CommittedMetadata<'_>], + ) -> u64 { self.applied .fetch_add(entries.len() as u64, Ordering::Relaxed); - entries.last().map(|(idx, _)| *idx).unwrap_or(0) + entries.last().map_or(0, |e| e.index) } } @@ -688,6 +640,18 @@ mod tests { mr3.add_group(0, vec![1, 2]).unwrap(); mr3.add_group(1, vec![1, 2]).unwrap(); + // Nodes 2 and 3 keep the production election timeouts so node 1 + // campaigns first. A node refuses every vote until its boot fence, + // `election_timeout_max` after it starts, passes: that is the whole + // wait of this test. These nodes start fresh and vote at once. + for node in mr2 + .groups_mut() + .values_mut() + .chain(mr3.groups_mut().values_mut()) + { + node.expire_boot_vote_fence(); + } + let a1 = CountingApplier::new(); let m1 = a1.metadata_applier(); let a2 = CountingApplier::new(); @@ -764,4 +728,162 @@ mod tests { shutdown_tx.send(true).unwrap(); } + + /// Metadata applier whose effects count as durable. Records every + /// delivered index and stops before `halt_at`. + struct DurableRecordingMetadataApplier { + delivered: Arc>>, + halt_at: Option, + } + + #[async_trait::async_trait] + impl MetadataApplier for DurableRecordingMetadataApplier { + async fn apply_decoded( + &self, + entries: &[crate::metadata_group::CommittedMetadata<'_>], + ) -> u64 { + let mut delivered = self.delivered.lock().unwrap_or_else(|p| p.into_inner()); + let mut last = 0; + for committed in entries { + if Some(committed.index) == self.halt_at { + break; + } + delivered.push(committed.index); + last = committed.index; + } + last + } + + fn durable_effects(&self) -> bool { + true + } + } + + /// Open group 0 over `dir`, retrying while a previous session still + /// holds the log file. + async fn open_metadata_group(dir: &std::path::Path) -> MultiRaft { + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + loop { + let mut mr = MultiRaft::new(1, RoutingTable::uniform(1, &[1], 1), dir.to_path_buf()); + match mr.add_group(0, vec![]) { + Ok(()) => return mr, + Err(e) => { + assert!( + tokio::time::Instant::now() < deadline, + "group 0 log never released: {e}" + ); + tokio::time::sleep(Duration::from_millis(20)).await; + } + } + } + } + + /// One node session over `dir`: propose `proposals` metadata entries, + /// wait for their delivery, shut down. Returns the delivered indices. + async fn metadata_session( + dir: &std::path::Path, + proposals: usize, + halt_at: Option, + ) -> Vec { + let mut mr = open_metadata_group(dir).await; + for node in mr.groups_mut().values_mut() { + node.election_deadline_override(Instant::now() - Duration::from_millis(1)); + } + let delivered = Arc::new(Mutex::new(Vec::new())); + let applier = Arc::new(DurableRecordingMetadataApplier { + delivered: Arc::clone(&delivered), + halt_at, + }); + let topo = Arc::new(RwLock::new(ClusterTopology::new())); + let raft_loop = Arc::new( + RaftLoop::new(mr, make_transport(1), topo, CountingApplier::new()) + .with_metadata_applier(applier), + ); + let (shutdown_tx, shutdown_rx) = tokio::sync::watch::channel(false); + let rl = Arc::clone(&raft_loop); + let run = tokio::spawn(async move { rl.run(shutdown_rx).await }); + + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + let mut last_proposed = 0; + let mut proposed = 0; + while proposed < proposals { + assert!(tokio::time::Instant::now() < deadline, "no metadata leader"); + match raft_loop.propose_to_metadata_group(b"entry".to_vec()) { + Ok(index) => { + last_proposed = index; + proposed += 1; + } + Err(_) => tokio::time::sleep(Duration::from_millis(10)).await, + } + } + let target = halt_at.map_or(last_proposed, |k| k - 1); + loop { + let reached = delivered + .lock() + .unwrap_or_else(|p| p.into_inner()) + .iter() + .any(|index| *index >= target); + if reached { + break; + } + assert!(tokio::time::Instant::now() < deadline, "delivery stalled"); + tokio::time::sleep(Duration::from_millis(10)).await; + } + // Let the ticks that follow retry the halted entry. + tokio::time::sleep(Duration::from_millis(50)).await; + + shutdown_tx.send(true).unwrap(); + raft_loop.begin_shutdown(); + assert!( + tokio::time::timeout(Duration::from_secs(2), run) + .await + .is_ok(), + "raft loop did not stop" + ); + drop(raft_loop); + delivered.lock().unwrap_or_else(|p| p.into_inner()).clone() + } + + /// A durable-effects applier's delivered index becomes group 0's applied + /// floor. After a restart only entries above the floor are delivered. + #[tokio::test] + async fn restart_delivers_only_entries_above_the_metadata_floor() { + let dir = tempfile::tempdir().unwrap(); + let first = metadata_session(dir.path(), 3, None).await; + let floor = open_metadata_group(dir.path()) + .await + .last_applied(0) + .unwrap_or(0); + assert_eq!( + Some(&floor), + first.iter().max(), + "floor = highest delivered" + ); + + let second = metadata_session(dir.path(), 2, None).await; + assert!(!second.is_empty()); + assert!( + second.iter().all(|index| *index > floor), + "restart re-delivered an entry at or below floor {floor}: {second:?}" + ); + } + + /// An applier that halts at entry k keeps the floor below k. + #[tokio::test] + async fn applier_halt_keeps_the_floor_below_the_entry() { + let dir = tempfile::tempdir().unwrap(); + metadata_session(dir.path(), 1, None).await; + let base = open_metadata_group(dir.path()) + .await + .last_applied(0) + .unwrap_or(0); + let halt_at = base + 3; + let delivered = metadata_session(dir.path(), 4, Some(halt_at)).await; + assert!(delivered.iter().all(|index| *index < halt_at)); + let floor = open_metadata_group(dir.path()) + .await + .last_applied(0) + .unwrap_or(0); + assert_eq!(floor, halt_at - 1); + } } diff --git a/nodedb-cluster/src/raft_loop/membership_convergence.rs b/nodedb-cluster/src/raft_loop/membership_convergence.rs index e49bf839d..2a27fd8c2 100644 --- a/nodedb-cluster/src/raft_loop/membership_convergence.rs +++ b/nodedb-cluster/src/raft_loop/membership_convergence.rs @@ -139,11 +139,12 @@ impl RaftLoop { /// here lets the existing replication → snapshot → promotion machinery /// converge the group to `members == placement`. /// - /// At most ONE group is mounted per tick. `add_group_as_learner` opens a - /// per-group redb file synchronously while holding the `multi_raft` lock; - /// capping at one bounds that work so a burst of newly-placed groups cannot - /// stall the reactor past an election timeout. The phase is idempotent and - /// runs every tick, so remaining groups mount on subsequent ticks. + /// The tick never touches the disk here. The mount opens the group's redb + /// log and restores it on a blocking thread, without the `multi_raft` + /// lock, and then mounts it under the lock. One mount of a group runs at + /// a time. At most one group starts per tick, so a burst of newly placed + /// groups spreads its disk work. The phase is idempotent and runs every + /// tick, so remaining groups mount on subsequent ticks. pub(super) fn mount_entering_groups(&self) { // Phase 1: snapshot the hosted set and the first planned mount under one // lock acquisition. Lock order is multi_raft THEN routing (the @@ -175,31 +176,55 @@ impl RaftLoop { }) }; - // Phase 2: mount under a re-acquired lock, re-checking `contains_group` - // to stay idempotent against a race with the join-time mount. - if let Some((gid, voters, other_learners)) = mount { - let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - if mr.contains_group(gid) { - return; - } - match mr.add_group_as_learner(gid, voters, other_learners) { - Ok(()) => { - debug!( - group_id = gid, - node_id = self.node_id, - "mount: added local learner replica for placed group" - ); - } - Err(e) => { - debug!( - group_id = gid, - node_id = self.node_id, - error = %e, - "mount: add_group_as_learner failed; retrying next tick" - ); + // Phase 2: open the group's disk off the async threads, then mount it + // under a re-acquired lock. `insert_opened` re-checks the group, so a + // race with the join-time mount keeps the replica mounted first. + let Some((gid, voters, other_learners)) = mount else { + return; + }; + if !self.tick_state.begin_mount(gid) { + return; + } + let spec = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .mount_spec(gid, voters, other_learners, true); + let multi_raft = std::sync::Arc::clone(&self.multi_raft); + let tick_state = std::sync::Arc::clone(&self.tick_state); + let node_id = self.node_id; + tokio::spawn(async move { + let opened = tokio::task::spawn_blocking(move || spec.open()).await; + match opened { + Ok(Ok(opened)) => { + let unused = multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .insert_opened(opened); + match unused { + // Closing the unused replica's disk blocks. + Some(unused) => drop(tokio::task::spawn_blocking(move || drop(unused))), + None => debug!( + group_id = gid, + node_id, "mount: added local learner replica for placed group" + ), + } } + Ok(Err(e)) => debug!( + group_id = gid, + node_id, + error = %e, + "mount: opening the group failed; retrying next tick" + ), + Err(e) => debug!( + group_id = gid, + node_id, + error = %e, + "mount: the open task failed; retrying next tick" + ), } - } + tick_state.end_mount(gid); + }); } /// For each group this node leads that has an authored placement set, @@ -455,6 +480,7 @@ mod tests { fn group(members: Vec, learners: Vec, placement: Option>) -> GroupInfo { GroupInfo { leader: members.first().copied().unwrap_or(0), + leader_term: 0, members, learners, placement, diff --git a/nodedb-cluster/src/raft_loop/mod.rs b/nodedb-cluster/src/raft_loop/mod.rs index c4c6119b8..ebfc4b702 100644 --- a/nodedb-cluster/src/raft_loop/mod.rs +++ b/nodedb-cluster/src/raft_loop/mod.rs @@ -15,13 +15,17 @@ //! peer, propose `AddLearner` on every group, wait for commit, //! broadcast topology, persist catalog, build the wire response. +pub mod apply_gate; mod auth_lease; pub mod auth_lease_hook; mod builder; +mod group_unmount; pub mod handle_rpc; pub mod hooks; +mod hooks_routed; pub mod in_flight_snapshots; pub mod join; +mod leader_balance; mod leadership_transfer; mod lease_gc; pub mod loop_core; @@ -29,13 +33,18 @@ mod membership_convergence; mod placement_reconcile; pub mod proposals; mod read_index; +mod routing_persist; +mod run; +mod snapshot_membership; pub mod tick; +mod tick_state; +pub use apply_gate::{ApplyPermit, GroupApplyGates, InstallPermit}; pub use auth_lease_hook::AuthLeaseService; pub use hooks::{ - AssignRemoteSurrogate, CalvinSubmit, CalvinSubmitInbox, ReleaseReservation, ReserveRead, - ShuffleAggregator, ShuffleConsumer, ShuffleProducer, ShuffleReceiver, SnapshotApplier, - SnapshotBuilder, SnapshotQuarantineHook, + AssignRemoteSurrogate, BuiltGroupSnapshot, CalvinSubmit, CalvinSubmitInbox, + MetadataSnapshotCapture, ReleaseReservation, ReserveRead, ShuffleAggregator, ShuffleConsumer, + ShuffleProducer, ShuffleReceiver, SnapshotApplier, SnapshotBuilder, SnapshotQuarantineHook, }; pub use in_flight_snapshots::{InFlightSnapshotGuard, InFlightSnapshots}; pub use loop_core::{CommitApplier, RaftLoop, VShardEnvelopeHandler}; diff --git a/nodedb-cluster/src/raft_loop/placement_reconcile.rs b/nodedb-cluster/src/raft_loop/placement_reconcile.rs index 7c253b579..b4fea3ec7 100644 --- a/nodedb-cluster/src/raft_loop/placement_reconcile.rs +++ b/nodedb-cluster/src/raft_loop/placement_reconcile.rs @@ -38,7 +38,13 @@ impl RaftLoop { if active_nodes.is_empty() { return; } - let data_group_ids: Vec = mr + // Every data group of the routing view, hosted here or not. With a + // replication factor below the node count this node hosts only + // some groups. A group left out would keep its placement through + // every topology change, and never take a joining node. + let routing = mr.routing(); + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let data_group_ids: Vec = routing .group_ids() .into_iter() .filter(|g| { @@ -51,8 +57,6 @@ impl RaftLoop { &data_group_ids, self.replication_factor(), ); - let routing = mr.routing(); - let routing = routing.read().unwrap_or_else(|p| p.into_inner()); crate::rebalancer::placement::compute_placement_changes(&routing, &target) }; @@ -70,7 +74,7 @@ impl RaftLoop { continue; } }; - match self.propose_to_metadata_group(bytes) { + match self.propose_stamped_to_metadata_group(&bytes) { Ok(idx) => { debug!( group_id, diff --git a/nodedb-cluster/src/raft_loop/proposals.rs b/nodedb-cluster/src/raft_loop/proposals.rs index 50df25f26..88ef3d8e7 100644 --- a/nodedb-cluster/src/raft_loop/proposals.rs +++ b/nodedb-cluster/src/raft_loop/proposals.rs @@ -64,6 +64,14 @@ impl RaftLoop { mr.propose_to_group(crate::metadata_group::METADATA_GROUP_ID, data) } + /// Propose the encoded metadata entry `data` to the metadata group, + /// stamped with the node clock (see + /// [`MultiRaft::set_metadata_clock`](crate::multi_raft::MultiRaft::set_metadata_clock)). + pub fn propose_stamped_to_metadata_group(&self, data: &[u8]) -> Result { + let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + mr.propose_stamped_metadata(data) + } + /// Propose to the metadata Raft group, transparently forwarding /// to the current leader if this node is not it. /// @@ -74,7 +82,7 @@ impl RaftLoop { /// receiving leader applies the proposal locally and returns /// the log index. /// - /// On `NotLeader { leader_hint: None }` (election in progress, + /// On `NotLeader { leader_hint: None, .. }` (election in progress, /// no observed leader yet) the call returns the original /// `NotLeader` error so the caller can decide whether to retry. /// We deliberately do not implement a wait-and-retry loop here @@ -85,15 +93,35 @@ impl RaftLoop { /// the bare `propose_to_metadata_group` — the only extra cost is /// an `is_leader_locally` check before the local propose. pub async fn propose_to_metadata_group_via_leader(&self, data: Vec) -> Result { + self.propose_metadata_via_leader(data, false).await + } + + /// [`Self::propose_to_metadata_group_via_leader`] for an unstamped + /// metadata entry the leader stamps as it appends it (see + /// [`MultiRaft::propose_stamped_metadata`](crate::multi_raft::MultiRaft::propose_stamped_metadata)). + pub async fn propose_stamped_to_metadata_group_via_leader(&self, data: Vec) -> Result { + self.propose_metadata_via_leader(data, true).await + } + + async fn propose_metadata_via_leader(&self, data: Vec, stamp: bool) -> Result { // First, try a local propose. - match self.propose_to_metadata_group(data.clone()) { + let local = if stamp { + self.propose_stamped_to_metadata_group(&data) + } else { + self.propose_to_metadata_group(data.clone()) + }; + match local { Ok(idx) => Ok(idx), Err(crate::error::ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint, + term, })) => { let Some(leader_id) = leader_hint else { return Err(crate::error::ClusterError::Raft( - nodedb_raft::RaftError::NotLeader { leader_hint: None }, + nodedb_raft::RaftError::NotLeader { + leader_hint: None, + term, + }, )); }; if leader_id == self.node_id { @@ -104,11 +132,12 @@ impl RaftLoop { return Err(crate::error::ClusterError::Raft( nodedb_raft::RaftError::NotLeader { leader_hint: Some(leader_id), + term, }, )); } // Otherwise forward to the hinted leader. - self.forward_metadata_propose(leader_id, data).await + self.forward_metadata_propose(leader_id, data, stamp).await } Err(other) => Err(other), } @@ -117,7 +146,12 @@ impl RaftLoop { /// Send a `MetadataProposeRequest` to `leader_id`. Looks up the /// leader's listen address via the local topology snapshot and /// dispatches through the existing peer transport. - async fn forward_metadata_propose(&self, leader_id: u64, data: Vec) -> Result { + async fn forward_metadata_propose( + &self, + leader_id: u64, + data: Vec, + stamp: bool, + ) -> Result { // Resolve and register the leader's address with the // transport so `send_rpc` has a destination. Topology is // updated by the membership / health subsystem; if the @@ -146,7 +180,7 @@ impl RaftLoop { } let req = crate::rpc_codec::RaftRpc::MetadataProposeRequest( - crate::rpc_codec::MetadataProposeRequest { bytes: data }, + crate::rpc_codec::MetadataProposeRequest { bytes: data, stamp }, ); let resp = self.transport.send_rpc(leader_id, req).await?; match resp { @@ -154,6 +188,11 @@ impl RaftLoop { if r.success { Ok(r.log_index) } else if let Some(hint) = r.leader_hint { + self.observe_redirect( + crate::metadata_group::METADATA_GROUP_ID, + Some(hint), + r.leader_term, + ); // The receiving node was also not the leader // (rare: leader changed between our local check // and the forwarded RPC). Surface as NotLeader @@ -161,6 +200,7 @@ impl RaftLoop { Err(crate::error::ClusterError::Raft( nodedb_raft::RaftError::NotLeader { leader_hint: Some(hint), + term: r.leader_term, }, )) } else { @@ -183,8 +223,8 @@ impl RaftLoop { /// a `DataProposeRequest` over QUIC. The receiving leader applies the /// proposal locally and returns `(group_id, log_index)`. /// - /// On `NotLeader { leader_hint: None }` (election in progress) the call - /// returns the original `NotLeader` error so the caller can retry. + /// On `NotLeader { leader_hint: None, .. }` (election in progress) the + /// call returns the original `NotLeader` error so the caller can retry. pub async fn propose_via_data_leader( &self, vshard_id: u32, @@ -195,16 +235,21 @@ impl RaftLoop { Ok(pair) => Ok(pair), Err(crate::error::ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint, + term, })) => { let Some(leader_id) = leader_hint else { return Err(crate::error::ClusterError::Raft( - nodedb_raft::RaftError::NotLeader { leader_hint: None }, + nodedb_raft::RaftError::NotLeader { + leader_hint: None, + term, + }, )); }; if leader_id == self.node_id { return Err(crate::error::ClusterError::Raft( nodedb_raft::RaftError::NotLeader { leader_hint: Some(leader_id), + term, }, )); } @@ -253,6 +298,17 @@ impl RaftLoop { if r.success { Ok((r.group_id, r.log_index)) } else { + let group_id = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .routing() + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(vshard_id); + if let Ok(group_id) = group_id { + self.observe_redirect(group_id, r.leader_hint, r.leader_term); + } Err(r.refusal_error()) } } diff --git a/nodedb-cluster/src/raft_loop/read_index.rs b/nodedb-cluster/src/raft_loop/read_index.rs index e6d6ca56f..e2e3098a9 100644 --- a/nodedb-cluster/src/raft_loop/read_index.rs +++ b/nodedb-cluster/src/raft_loop/read_index.rs @@ -6,13 +6,19 @@ //! A read index is the leader's commit index at a moment a quorum confirmed //! its leadership. A node whose state machine has applied a group through //! that index observes every entry committed before the read index was taken. +//! +//! The leader status a node answers for another node's routing hint is also +//! here. It reads this node's own Raft state and runs no quorum round. use std::time::Duration; use crate::error::{ClusterError, Result}; use crate::forward::PlanExecutor; use crate::read_index_wait::confirm_read_index; -use crate::rpc_codec::{RaftRpc, ReadIndexOutcome, ReadIndexRequest, ReadIndexResponse}; +use crate::rpc_codec::{ + LeaderMembership, LeaderStatusResponse, RaftRpc, ReadIndexOutcome, ReadIndexRequest, + ReadIndexResponse, +}; use super::loop_core::{CommitApplier, RaftLoop}; @@ -23,17 +29,25 @@ impl RaftLoop { /// Obtain a read index for `group_id`. /// /// On the group leader, confirms leadership against a quorum. On any other - /// node, asks the known leader. Fails with - /// [`ClusterError::ReadIndexNotLeader`] when no leader is known or the - /// asked node no longer leads, and with [`ClusterError::ReadIndexTimeout`] - /// when no quorum answered within `timeout`. + /// node, asks the known leader: the leader this node's Raft knows when it + /// hosts a replica of the group, else the leader its routing table names. + /// A refusal that names a leader at a newer term moves the routing hint + /// to that leader, and so does a confirmation from the asked leader. + /// Fails with [`ClusterError::ReadIndexNotLeader`] when + /// no leader is known or the asked node no longer leads, and with + /// [`ClusterError::ReadIndexTimeout`] when no quorum answered within + /// `timeout`. pub async fn read_index_via_leader(&self, group_id: u64, timeout: Duration) -> Result { let leader = { let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); if mr.is_group_leader(group_id) { None - } else { + } else if mr.contains_group(group_id) { Some(mr.group_leader(group_id)) + } else { + let routing = mr.routing(); + let table = routing.read().unwrap_or_else(|p| p.into_inner()); + Some(table.group_info(group_id).map_or(0, |info| info.leader)) } }; let leader_id = match leader { @@ -50,8 +64,12 @@ impl RaftLoop { }); match self.transport.send_rpc(leader_id, request).await? { RaftRpc::ReadIndexResponse(ReadIndexResponse { outcome }) => match outcome { - ReadIndexOutcome::Confirmed { read_index } => Ok(read_index), - ReadIndexOutcome::NotLeader { .. } => { + ReadIndexOutcome::Confirmed { read_index, term } => { + self.observe_redirect(group_id, Some(leader_id), term); + Ok(read_index) + } + ReadIndexOutcome::NotLeader { leader_hint, term } => { + self.observe_redirect(group_id, leader_hint, term); Err(ClusterError::ReadIndexNotLeader { group_id }) } ReadIndexOutcome::Timeout { waited_ms } => Err(ClusterError::ReadIndexTimeout { @@ -70,24 +88,67 @@ impl RaftLoop { pub(super) async fn handle_read_index_rpc(&self, req: ReadIndexRequest) -> Result { let timeout = Duration::from_millis(req.timeout_ms).min(MAX_REMOTE_TIMEOUT); let outcome = match confirm_read_index(&self.multi_raft, req.group_id, timeout).await { - Ok(read_index) => ReadIndexOutcome::Confirmed { read_index }, + Ok(read_index) => { + // The term this node leads at. A node that stepped down since + // the quorum confirmed it reports term 0: it names no term. + let (leader, term) = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .group_leader_at_term(req.group_id); + let term = if leader == self.node_id { term } else { 0 }; + ReadIndexOutcome::Confirmed { read_index, term } + } Err(ClusterError::ReadIndexTimeout { waited_ms, .. }) => { ReadIndexOutcome::Timeout { waited_ms } } Err(_) => { - let hint = self + let (hint, term) = self .multi_raft .lock() .unwrap_or_else(|p| p.into_inner()) - .group_leader(req.group_id); + .group_leader_at_term(req.group_id); ReadIndexOutcome::NotLeader { leader_hint: (hint != 0).then_some(hint), + term, } } }; Ok(RaftRpc::ReadIndexResponse(ReadIndexResponse { outcome })) } + /// The leader this node knows for `group_id` and the term it leads, + /// from this node's own Raft state with no quorum round. `(0, 0)` when + /// this node knows no leader or hosts no replica of the group. A node + /// that leads the group also answers its Raft voters and learners. + /// + /// Raft clears a node's leader whenever its term moves, so a named + /// leader always leads the reported term. + pub(super) fn leader_status(&self, group_id: u64) -> LeaderStatusResponse { + let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let (leader, term) = mr.group_leader_at_term(group_id); + if leader == 0 { + return LeaderStatusResponse { + leader: 0, + term: 0, + membership: None, + }; + } + let membership = if leader == self.node_id { + mr.group_membership(group_id).map(|m| LeaderMembership { + voters: m.voters, + learners: m.learners, + }) + } else { + None + }; + LeaderStatusResponse { + leader, + term, + membership, + } + } + /// Register `node_id`'s listen address with the transport, from the local /// topology. fn register_peer_addr(&self, node_id: u64) -> Result<()> { diff --git a/nodedb-cluster/src/raft_loop/routing_persist.rs b/nodedb-cluster/src/raft_loop/routing_persist.rs new file mode 100644 index 000000000..a62467334 --- /dev/null +++ b/nodedb-cluster/src/raft_loop/routing_persist.rs @@ -0,0 +1,220 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Durable saves of the routing table, off the async threads. +//! +//! A routing save is a catalog write with an fsync. Run on the tick, one +//! stalled fsync would hold up the heartbeats and elections of every group on +//! this node. So every routing save goes through one persister: +//! - A caller changes the in-memory table, then calls +//! [`RoutingPersister::request`], which returns a sequence number. +//! - The persister runs beside the tick loop. It copies the table as it +//! stands, saves the copy on a blocking thread, and reports the highest +//! sequence number the save covers. +//! - A caller that must not go on before the save is durable either polls +//! [`RoutingPersister::is_durable`] each tick or awaits +//! [`RoutingPersister::wait`]. +//! +//! One writer saves, and it copies the table after it reads the requested +//! sequence number. So a save covers every change made before the request +//! it reports, and an older copy never lands after a newer one. +//! +//! A failed save is reported, and the persister retries it after +//! [`RETRY_DELAY`]. + +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::{Arc, RwLock}; +use std::time::Duration; + +use tokio::sync::{Notify, watch}; +use tracing::warn; + +use crate::catalog::ClusterCatalog; +use crate::routing::RoutingTable; + +/// The wait before a failed save is tried again. +const RETRY_DELAY: Duration = Duration::from_millis(100); + +/// The highest sequence numbers a save covered. +#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)] +struct SaveOutcome { + /// Every request at or below it is durable. + durable: u64, + /// The last failed save covered the requests at or below it. + failed: u64, + /// The persister ended. A request not yet durable is never saved. + stopped: bool, +} + +/// The single writer of the routing table to the catalog. +pub(in crate::raft_loop) struct RoutingPersister { + catalog: Arc, + routing: Arc>, + requested: AtomicU64, + wake: Notify, + outcome: watch::Sender, +} + +impl RoutingPersister { + pub(in crate::raft_loop) fn new( + catalog: Arc, + routing: Arc>, + ) -> Self { + Self { + catalog, + routing, + requested: AtomicU64::new(0), + wake: Notify::new(), + outcome: watch::Sender::new(SaveOutcome::default()), + } + } + + /// Ask for a save of the table as it stands now. Returns the sequence + /// number the save reports. Never waits. + pub(in crate::raft_loop) fn request(&self) -> u64 { + let seq = self.requested.fetch_add(1, Ordering::AcqRel) + 1; + self.wake.notify_one(); + seq + } + + /// Whether the save of request `seq` is durable. + pub(in crate::raft_loop) fn is_durable(&self, seq: u64) -> bool { + self.outcome.borrow().durable >= seq + } + + /// Wait until the save of request `seq` is durable, a save that covers it + /// failed, or the persister ended. Returns whether it is durable. + pub(in crate::raft_loop) async fn wait(&self, seq: u64) -> bool { + let mut rx = self.outcome.subscribe(); + loop { + let outcome = *rx.borrow_and_update(); + if outcome.durable >= seq { + return true; + } + if outcome.failed >= seq || outcome.stopped { + return false; + } + if rx.changed().await.is_err() { + return false; + } + } + } + + /// Save the table whenever a request is not yet durable, until shutdown + /// begins. A request made before shutdown gets one last save attempt. + pub(in crate::raft_loop) async fn run(&self, shutdown: watch::Receiver) { + self.save_until_shutdown(shutdown).await; + self.save_pending().await; + self.outcome.send_modify(|o| o.stopped = true); + } + + async fn save_until_shutdown(&self, mut shutdown: watch::Receiver) { + loop { + if self.save_pending().await == Some(false) + && wait_or_shutdown(&mut shutdown, RETRY_DELAY).await + { + return; + } + if *shutdown.borrow() { + return; + } + tokio::select! { + _ = self.wake.notified() => {} + changed = shutdown.changed() => { + if changed.is_err() || *shutdown.borrow() { + return; + } + } + } + } + } + + /// Save the table when a request is not yet durable. `None` when nothing + /// waited, else whether the save succeeded. + async fn save_pending(&self) -> Option { + let target = self.requested.load(Ordering::Acquire); + if target <= self.outcome.borrow().durable { + return None; + } + // Copied after `target` was read: the copy holds every change made + // before each request up to `target`. + let table = self + .routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .clone(); + let catalog = Arc::clone(&self.catalog); + let saved = tokio::task::spawn_blocking(move || catalog.save_routing(&table)).await; + let error = match saved { + Ok(Ok(())) => { + self.outcome + .send_modify(|o| o.durable = o.durable.max(target)); + return Some(true); + } + Ok(Err(e)) => e.to_string(), + Err(e) => format!("save task: {e}"), + }; + warn!( + requested = target, + error = %error, + "could not save the routing table; the save is retried" + ); + self.outcome + .send_modify(|o| o.failed = o.failed.max(target)); + Some(false) + } +} + +/// Wait `wait`, or less when shutdown begins. Returns whether it began. +async fn wait_or_shutdown(shutdown: &mut watch::Receiver, wait: Duration) -> bool { + if *shutdown.borrow() { + return true; + } + tokio::select! { + _ = tokio::time::sleep(wait) => false, + changed = shutdown.changed() => changed.is_err() || *shutdown.borrow(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn persister(dir: &tempfile::TempDir) -> Arc { + let catalog = Arc::new( + ClusterCatalog::open(&dir.path().join("cluster.redb")).expect("open the catalog"), + ); + let routing = Arc::new(RwLock::new(RoutingTable::uniform(2, &[1, 2, 3], 3))); + Arc::new(RoutingPersister::new(catalog, routing)) + } + + /// A requested save becomes durable, and the saved table holds the + /// change made before the request. + #[tokio::test] + async fn a_request_is_saved_off_the_caller() { + let dir = tempfile::tempdir().expect("tempdir"); + let persister = persister(&dir); + let (stop_tx, stop_rx) = watch::channel(false); + let runner = { + let persister = Arc::clone(&persister); + tokio::spawn(async move { persister.run(stop_rx).await }) + }; + + persister + .routing + .write() + .unwrap_or_else(|p| p.into_inner()) + .set_group_members(1, vec![2]); + let seq = persister.request(); + assert!(persister.wait(seq).await); + assert!(persister.is_durable(seq)); + let saved = persister + .catalog + .load_routing() + .expect("load the routing table") + .expect("a saved routing table"); + assert_eq!(saved.group_info(1).expect("group 1").members, vec![2]); + + stop_tx.send(true).expect("signal shutdown"); + runner.await.expect("the persister ends"); + } +} diff --git a/nodedb-cluster/src/raft_loop/run.rs b/nodedb-cluster/src/raft_loop/run.rs new file mode 100644 index 000000000..6cecc81b3 --- /dev/null +++ b/nodedb-cluster/src/raft_loop/run.rs @@ -0,0 +1,103 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The Raft event loop: the tick loop and the metadata apply lane, run side +//! by side in one task until shutdown. + +use std::time::Instant; + +use tracing::debug; + +use crate::forward::PlanExecutor; + +use super::loop_core::{CommitApplier, RaftLoop}; +use super::tick::metadata_lane::METADATA_LANE_DEPTH; + +impl RaftLoop { + /// Run the event loop until shutdown. + /// + /// This drives Raft elections, heartbeats, and message dispatch. + /// Call [`crate::transport::NexarTransport::serve`] separately with + /// `Arc` as the handler. + /// + /// The metadata apply lane runs beside the tick loop in this task (see + /// `tick::metadata_lane`). A metadata apply that awaits the host never + /// holds up a tick. When shutdown begins, the tick loop stops, closes + /// the lane and propagates the signal to the internal cooperative + /// shutdown channel, so every detached task spawned inside `do_tick` + /// exits promptly and drops its `Arc>` clone. `run` + /// returns once the lane ended too. + pub async fn run(&self, shutdown: tokio::sync::watch::Receiver) { + let (tx, rx) = tokio::sync::mpsc::channel(METADATA_LANE_DEPTH); + self.tick_state.open_metadata_lane(tx); + let lane = self.run_metadata_lane(rx, shutdown.clone()); + let persister = async { + if let Some(persister) = self.routing_persister.as_ref() { + persister.run(shutdown.clone()).await; + } + }; + let ticks = async { + self.run_ticks(shutdown.clone()).await; + self.tick_state.close_metadata_lane(); + }; + tokio::join!(ticks, lane, persister); + } + + /// Tick until shutdown begins. + async fn run_ticks(&self, mut shutdown: tokio::sync::watch::Receiver) { + let mut interval = tokio::time::interval(self.tick_interval); + interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + self.loop_metrics.set_up(true); + + // Startup GC sweep: remove orphaned partial-snapshot files from + // previous runs that did not complete. It walks the disk, so it runs + // on a blocking thread. + if let Some(dir) = self.data_dir.clone() { + let max_age = self.orphan_partial_max_age_secs; + let swept = tokio::task::spawn_blocking(move || { + crate::install_snapshot::gc::sweep_orphans(&dir, max_age) + }) + .await; + match swept { + Err(e) => { + tracing::warn!(error = %e, "startup: partial snapshot sweep task failed"); + } + Ok(Ok((removed, errs))) => { + if removed > 0 { + tracing::info!(removed, "startup: removed orphaned partial snapshot files"); + } + for e in errs { + tracing::warn!(error = %e, "startup: partial snapshot GC error"); + } + } + Ok(Err(e)) => { + tracing::warn!(error = %e, "startup: failed to sweep partial snapshot directory"); + } + } + } + + loop { + tokio::select! { + _ = interval.tick() => { + let started = Instant::now(); + self.do_tick().await; + self.loop_metrics.observe(started.elapsed()); + } + _ = self.reconcile_notify.notified() => { + if *shutdown.borrow() { + break; + } + self.reconcile_placement(); + } + changed = shutdown.changed() => { + // A dropped sender is a shutdown too: no signal can come. + if changed.is_err() || *shutdown.borrow() { + debug!("raft loop shutting down"); + self.begin_shutdown(); + break; + } + } + } + } + self.loop_metrics.set_up(false); + } +} diff --git a/nodedb-cluster/src/raft_loop/snapshot_membership.rs b/nodedb-cluster/src/raft_loop/snapshot_membership.rs new file mode 100644 index 000000000..a06341041 --- /dev/null +++ b/nodedb-cluster/src/raft_loop/snapshot_membership.rs @@ -0,0 +1,91 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A data group's membership, adopted from the final chunk of an +//! `InstallSnapshot`. +//! +//! A snapshot replaces the log through its index, conf changes included. +//! This node never applies the conf changes it covers, so its Raft and +//! routing view of the group's voters and learners would stay as they were +//! before the gap. The leader sends its membership on the final chunk, and +//! this node takes it before the install. +//! +//! The membership is taken only from a leader at or above this node's term +//! for the group. A leader's membership moves only by applying a committed +//! conf change, so it holds committed changes only, and taking it ahead of +//! the install is safe: a conf change above the snapshot index applies again +//! from the log, and applying a change twice is a no-op. +//! +//! The metadata group takes its membership from the routing table its +//! snapshot image carries (see [`crate::install_snapshot::finalize`]). + +use nodedb_raft::InstallSnapshotRequest; +use tracing::debug; + +use crate::error::{ClusterError, Result}; +use crate::forward::PlanExecutor; +use crate::metadata_group::METADATA_GROUP_ID; + +use super::loop_core::{CommitApplier, RaftLoop}; + +impl RaftLoop { + /// Set a data group's voters and learners to the ones the final chunk + /// `req` carries, in this node's routing view and Raft, and save the + /// routing table. + /// + /// The save comes before the install moves the durable applied floor + /// past the covered conf changes, so a restart mounts the group with this + /// membership. It runs through the routing persister, off the async + /// threads, and this handler awaits it. A failed save fails the install, + /// and the leader sends the snapshot again. + /// + /// A no-op for the metadata group, for a chunk that carries no voters, + /// for a group this node does not host, and for a sender below this + /// node's term for the group. + pub(super) async fn adopt_snapshot_membership( + &self, + req: &InstallSnapshotRequest, + ) -> Result<()> { + let group_id = req.group_id; + if group_id == METADATA_GROUP_ID || req.voters.is_empty() { + return Ok(()); + } + { + let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + if !mr.contains_group(group_id) { + return Ok(()); + } + let (_, term) = mr.group_leader_at_term(group_id); + if req.term < term { + return Ok(()); + } + let routing = mr.routing(); + { + let mut table = routing.write().unwrap_or_else(|p| p.into_inner()); + if table.group_info(group_id).is_none() { + return Ok(()); + } + table.set_group_members(group_id, req.voters.clone()); + table.set_group_learners(group_id, req.learners.clone()); + } + mr.sync_group_membership_from_routing(group_id)?; + } + debug!( + group_id, + voters = ?req.voters, + learners = ?req.learners, + "install snapshot: membership adopted from the leader" + ); + let Some(persister) = self.routing_persister.as_ref() else { + return Ok(()); + }; + if persister.wait(persister.request()).await { + return Ok(()); + } + Err(ClusterError::Storage { + detail: format!( + "could not save the routing table with group {group_id}'s snapshot membership; \ + the leader sends the snapshot again" + ), + }) + } +} diff --git a/nodedb-cluster/src/raft_loop/tick/apply_committed.rs b/nodedb-cluster/src/raft_loop/tick/apply_committed.rs index 123d0fd55..5a7121b78 100644 --- a/nodedb-cluster/src/raft_loop/tick/apply_committed.rs +++ b/nodedb-cluster/src/raft_loop/tick/apply_committed.rs @@ -1,14 +1,19 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Apply a group's committed entries: detect + apply conf-changes, dispatch -//! to the metadata or data applier, advance the applied watermark, flip the -//! boot-time readiness watch, and bump the cluster epoch on metadata-group -//! leadership acquisition. +//! Apply a group's committed entries: detect + apply conf-changes, hand the +//! metadata group's entries to the metadata lane or a data group's to the +//! data applier, advance the applied watermark, and bump the cluster epoch +//! on metadata-group leadership acquisition. -use tracing::warn; +use std::sync::Mutex; + +use nodedb_raft::LogEntry; +use tracing::{debug, error, warn}; use crate::conf_change::ConfChange; use crate::forward::PlanExecutor; +use crate::multi_raft::MultiRaft; +use crate::raft_loop::apply_gate::ApplyPermit; use super::super::loop_core::{CommitApplier, RaftLoop}; @@ -19,68 +24,184 @@ use super::super::loop_core::{CommitApplier, RaftLoop}; /// a non-increasing pair is a producer regression. The applier's delivery guard /// stays the boundary that absorbs a repeat; this makes the regression /// observable instead of silent. -fn first_non_increasing_committed_index(entries: &[nodedb_raft::LogEntry]) -> Option { +fn first_non_increasing_committed_index(entries: &[LogEntry]) -> Option { entries .windows(2) .find(|pair| pair[1].index <= pair[0].index) .map(|pair| pair[1].index) } +/// Admit `batch` for apply: take the group's apply gate, then return the +/// entries above the group's `last_applied`. +/// +/// The batch leaves `Ready` before the apply runs. A snapshot adopted since +/// raises `last_applied`, and the entries it covers must not apply on top of +/// it. The permit keeps a new install from starting until the caller has +/// handed the entries to the applier. `None` when an install holds or awaits +/// the gate: the batch then goes back to `Ready` for the next tick. An +/// unmounted group applies the batch as taken. +fn admit_batch<'a>( + multi_raft: &Mutex, + group_id: u64, + batch: &'a [LogEntry], +) -> Option<(ApplyPermit, &'a [LogEntry])> { + let mut mr = multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let Some(permit) = mr.apply_gates().try_apply(group_id) else { + if let Err(e) = mr.requeue_committed(group_id, batch.to_vec()) { + warn!(group_id, error = %e, "failed to requeue a batch deferred by a snapshot install"); + } + return None; + }; + let entries = match mr.last_applied(group_id) { + Some(applied) => { + let start = batch + .iter() + .position(|entry| entry.index > applied) + .unwrap_or(batch.len()); + &batch[start..] + } + None => batch, + }; + Some((permit, entries)) +} + impl RaftLoop { + /// Log a committed range the group's log no longer holds. + /// + /// Apply for the group halts: the missing entries exist only in a + /// snapshot. The node rejects `AppendEntries` at its applied index until + /// the leader installs one. + pub(super) fn surface_committed_read_error(&self, group_id: u64, err: &nodedb_raft::RaftError) { + let last_applied = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .last_applied(group_id); + error!( + group_id, + ?last_applied, + error = %err, + "committed entries are below the retained log; apply is halted until a snapshot covers the gap" + ); + } + + /// Apply the conf changes among `entries` to this node's Raft and + /// routing view. Returns whether the batch goes on to the applier now. + /// + /// A data group's durable applied floor passes a conf change once a + /// later entry applies, and a restart never delivers that entry again. + /// So no entry of the batch reaches the applier before the routing table + /// the change produced is durable. A restart then mounts the group with + /// the membership the change left. + /// + /// The tick never waits on the save. It applies the changes in memory + /// once, asks the routing persister for a save, and puts the batch back + /// in `Ready`. Each later tick finds the batch again and lets it go on + /// once the save is durable. The persister retries a failed save, and + /// the batch waits meanwhile. + /// + /// The metadata group's changes apply here too. The metadata lane saves + /// the routing table before its applied floor moves. + fn apply_conf_changes(&self, group_id: u64, entries: &[LogEntry]) -> bool { + let is_data_group = group_id != crate::metadata_group::METADATA_GROUP_ID; + let persister = self.routing_persister.as_ref().filter(|_| is_data_group); + let waiting = persister.and_then(|_| self.tick_state.conf_save(group_id)); + let applied_through = waiting.map_or(0, |(through, _)| through); + let mut last_change = None; + for entry in entries { + let Some(cc) = ConfChange::from_entry_data(&entry.data) else { + continue; + }; + last_change = Some(entry.index); + if entry.index <= applied_through { + continue; + } + let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + if let Err(e) = mr.apply_conf_change(group_id, &cc) { + warn!(group_id, error = %e, "failed to apply conf change"); + } + } + let (Some(persister), Some(through)) = (persister, last_change) else { + return true; + }; + let seq = match waiting { + Some((waited_through, seq)) if waited_through >= through => seq, + _ => { + let seq = persister.request(); + self.tick_state.set_conf_save(group_id, through, seq); + seq + } + }; + if persister.is_durable(seq) { + self.tick_state.clear_conf_save(group_id); + return true; + } + let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + if let Err(e) = mr.requeue_committed(group_id, entries.to_vec()) { + warn!(group_id, error = %e, "failed to requeue a batch whose conf change is not saved"); + } + false + } + /// Apply one group's committed entries from this tick's `Ready` output. /// Called only when `!group_ready.committed_entries.is_empty()`. + /// + /// Never waits for an apply. A data group's applier only enqueues. The + /// metadata group's entries go to the metadata lane (see + /// [`super::metadata_lane`]), which applies them in order off the tick + /// and reports the applied index back to Raft. pub(super) fn apply_group_commits(&self, group_id: u64, group_ready: &nodedb_raft::Ready) { - if let Some(index) = first_non_increasing_committed_index(&group_ready.committed_entries) { + let batch = &group_ready.committed_entries; + if let Some(index) = first_non_increasing_committed_index(batch) { warn!( group_id, index, "committed batch is not strictly increasing; the applier guard absorbs a repeat" ); } - for entry in &group_ready.committed_entries { - if let Some(cc) = ConfChange::from_entry_data(&entry.data) { + // Held until this function returns: from the floor check through the + // applier call and the watermark advance. + let Some((_permit, entries)) = admit_batch(&self.multi_raft, group_id, batch) else { + debug!( + group_id, + "snapshot install in progress; committed batch deferred" + ); + return; + }; + if entries.len() < batch.len() { + debug!( + group_id, + skipped = batch.len() - entries.len(), + "skipped committed entries an installed snapshot already covers" + ); + } + if !self.apply_conf_changes(group_id, entries) { + return; + } + + let last_applied = if entries.is_empty() { + 0 + } else if group_id == crate::metadata_group::METADATA_GROUP_ID { + // Metadata group (0): the lane applies the entries, with cluster + // epochs adopted in log order and the durable applied floor + // saved before the watcher moves. Entries the lane cannot take + // go back to `Ready` for the next tick. + let back = self.tick_state.send_to_metadata_lane(entries); + if !back.is_empty() { let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - if let Err(e) = mr.apply_conf_change(group_id, &cc) { - warn!(group_id, error = %e, "failed to apply conf change"); + if let Err(e) = mr.requeue_committed(group_id, back) { + warn!(group_id, error = %e, "failed to requeue metadata entries"); } } - } - - let last_applied = if group_id == crate::metadata_group::METADATA_GROUP_ID { - // Metadata group (0): dispatch to the metadata applier. - // Raft no-op entries and conf-changes are already - // handled above; data entries carry a serialized - // `MetadataEntry` and are decoded by the applier. - let pairs: Vec<(u64, Vec)> = group_ready - .committed_entries - .iter() - .filter(|e| ConfChange::from_entry_data(&e.data).is_none()) - .map(|e| (e.index, e.data.clone())) - .collect(); - // The applied cluster epoch advances here, in the cluster crate, - // rather than inside whichever applier the host installed: the - // epoch is what this node stamps on its own frames, so it must - // move on every node that applies the entry, not only on nodes - // whose host applier happens to know about it. - self.adopt_committed_cluster_epochs(&pairs); - self.metadata_applier.apply(&pairs) + 0 } else { - self.applier - .apply_committed(group_id, &group_ready.committed_entries) + self.applier.apply_committed(group_id, entries) }; if last_applied > 0 { let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); if let Err(e) = mr.advance_applied(group_id, last_applied) { warn!(group_id, error = %e, "failed to advance applied index"); - } else if group_id == crate::metadata_group::METADATA_GROUP_ID - || group_id == crate::calvin::SEQUENCER_GROUP_ID - { - // Metadata group: the metadata applier - // applied entries synchronously to redb - // before returning, so the apply - // watermark is data-visible at this - // point. Bump the watcher. - // + } else if group_id == crate::calvin::SEQUENCER_GROUP_ID { // Sequencer group: the host applies each // entry to the sequencer state machine // inline, before returning. The watcher @@ -102,23 +223,12 @@ impl RaftLoop { // Snapshot-install path also bumps // (in `super::handle_rpc`) — covers // jump-on-snapshot for both group - // kinds. + // kinds. The metadata lane bumps the + // metadata group's watcher. self.group_watchers.bump(group_id, last_applied); } } - // Boot-time readiness: the first time the metadata - // group (0) applies any entry on this node — which - // is the leader-election no-op or a replayed entry - // — flip the ready watch. The host crate's - // `start_raft` returns the receiver; `main.rs` - // awaits it before binding client-facing - // listeners. Idempotent: subsequent ticks are a - // no-op once the latch is set. - if group_id == crate::metadata_group::METADATA_GROUP_ID && !*self.ready_watch.borrow() { - let _ = self.ready_watch.send(true); - } - // On acquiring metadata-group leadership, PROPOSE a new cluster // generation rather than bumping a local counter. Going through the // log is what makes the epoch an agreed fact: every node advances by @@ -143,8 +253,10 @@ impl RaftLoop { #[cfg(test)] mod tests { + use std::time::{Duration, Instant}; + use super::*; - use nodedb_raft::LogEntry; + use crate::routing::RoutingTable; fn entry(index: u64) -> LogEntry { LogEntry { @@ -173,4 +285,95 @@ mod tests { Some(1) ); } + + fn indices(entries: &[LogEntry]) -> Vec { + entries.iter().map(|e| e.index).collect() + } + + /// A single-voter group 1 with three committed entries taken from + /// `Ready` and not yet applied. + fn group_with_taken_batch(dir: &tempfile::TempDir) -> (MultiRaft, Vec) { + let rt = RoutingTable::uniform(1, &[1], 1); + let mut mr = MultiRaft::new(1, rt, dir.path().to_path_buf()); + mr.add_group(1, vec![]).expect("mount group 1"); + for node in mr.groups_mut().values_mut() { + node.election_deadline_override(Instant::now() - Duration::from_millis(1)); + } + // Elect, then deliver the no-op once the disk holds it. + mr.tick().expect("tick"); + mr.wait_all_durable_blocking(); + for (gid, ready) in mr.tick().expect("tick").groups { + if let Some(last) = ready.committed_entries.last() { + mr.advance_applied(gid, last.index).expect("advance"); + } + } + for _ in 0..3 { + mr.propose(0, b"write".to_vec()) + .expect("single voter commits"); + } + mr.wait_all_durable_blocking(); + let batch: Vec = mr + .tick() + .expect("tick") + .groups + .into_iter() + .find(|(gid, _)| *gid == 1) + .map(|(_, ready)| ready.committed_entries) + .unwrap_or_default(); + assert_eq!(batch.len(), 3, "batch: {:?}", indices(&batch)); + (mr, batch) + } + + /// A batch taken before a snapshot install skips the entries the + /// snapshot covers and applies only the rest. + #[test] + fn batch_taken_before_install_skips_snapshot_covered_entries() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mut mr, batch) = group_with_taken_batch(&dir); + + // The snapshot lands between taking the batch and applying it. + mr.adopt_snapshot_boundary(1, batch[1].index, batch[1].term) + .expect("adopt snapshot"); + let mr = Mutex::new(mr); + + let (_permit, applying) = admit_batch(&mr, 1, &batch).expect("gate open"); + assert_eq!(indices(applying), vec![batch[2].index]); + } + + /// While an install holds the gate, the tick applies nothing and the + /// batch returns to `Ready`. After the release it applies the entries + /// the install did not cover. + #[tokio::test] + async fn batch_is_deferred_while_an_install_holds_the_gate() { + let dir = tempfile::tempdir().expect("tempdir"); + let (mr, batch) = group_with_taken_batch(&dir); + let gates = mr.apply_gates(); + let mr = Mutex::new(mr); + + let mut install = gates.install(1).await; + assert!(admit_batch(&mr, 1, &batch).is_none()); + { + let mut guard = mr.lock().unwrap_or_else(|p| p.into_inner()); + guard + .adopt_snapshot_boundary(1, batch[0].index, batch[0].term) + .expect("adopt snapshot"); + } + install.adopted(batch[0].index); + drop(install); + + let requeued: Vec = mr + .lock() + .unwrap_or_else(|p| p.into_inner()) + .tick() + .expect("tick") + .groups + .into_iter() + .find(|(gid, _)| *gid == 1) + .map(|(_, ready)| ready.committed_entries) + .unwrap_or_default(); + assert_eq!(indices(&requeued), indices(&batch[1..])); + + let (_permit, applying) = admit_batch(&mr, 1, &requeued).expect("gate open"); + assert_eq!(indices(applying), indices(&batch[1..])); + } } diff --git a/nodedb-cluster/src/raft_loop/tick/core.rs b/nodedb-cluster/src/raft_loop/tick/core.rs index 7c270fece..c4649848d 100644 --- a/nodedb-cluster/src/raft_loop/tick/core.rs +++ b/nodedb-cluster/src/raft_loop/tick/core.rs @@ -12,7 +12,9 @@ //! all messages targeting the same peer (see [`super::dispatch_outbound`]). //! 3. **Dispatch RequestVote**: same batching strategy. //! 4. **Apply committed entries**: feed them to the user-supplied -//! `CommitApplier`. Conf-change entries are detected and applied to +//! `CommitApplier`, or hand group 0's entries to the metadata lane, which +//! applies them off the tick (see [`super::metadata_lane`]). The tick +//! never waits for an apply. Conf-change entries are detected and applied to //! `MultiRaft` before the user applier sees them (see //! [`super::apply_committed`]). //! 5. **Install snapshots**: send `InstallSnapshot` RPCs to peers that @@ -34,6 +36,11 @@ //! 9. **Remove leaving learners**: for each group this node leads that has an //! authored placement set, propose `RemoveLearner` for non-voting learners //! not in the placement. Inert while placement is `None` (bootstrap window). +//! 10. **Unmount left groups**: drop the replica of a data group this node +//! was removed from and is not placed in (see [`super::super::group_unmount`]). +//! 11. **Throttled passes**: placement reconcile, leader balance (see +//! [`super::super::leader_balance`]), the unhosted-group leader probe +//! (see [`super::leader_probe`]), orphan snapshot GC, and lease GC. use tracing::{debug, error}; @@ -67,7 +74,7 @@ const LEASE_GC_TICK_INTERVAL: u64 = 200; impl RaftLoop { /// Execute a single tick: drive Raft, dispatch outbound messages, /// apply commits, promote caught-up learners. - pub(in crate::raft_loop) fn do_tick(&self) { + pub(in crate::raft_loop) async fn do_tick(&self) { // Tick under lock and extract Ready. `tick` durably persists any // HardState staged this tick (election term bump + self-vote) before // returning the vote requests it carries. A persist failure is a @@ -81,7 +88,7 @@ impl RaftLoop { Err(e) => { error!( error = %e, - "raft tick failed to persist hard state durably; \ + "raft tick failed to stage hard state; \ skipping message/vote dispatch for this tick" ); return; @@ -89,6 +96,11 @@ impl RaftLoop { } }; + // Every leader change this node's Raft saw since the last tick, + // including its own election and a leader learned from an + // AppendEntries or heartbeat, reaches the routing table's leader hint. + self.sync_leader_hints(); + // Dispatch outgoing messages and persist log/HardState first (even if // ready looks "empty" we still want to run the learner-promotion step // each tick so a just-caught-up learner is promoted promptly). @@ -98,6 +110,9 @@ impl RaftLoop { // Apply committed entries and conf-changes, then dispatch any // needed install-snapshot RPCs, per group. for (group_id, group_ready) in ready.groups { + if let Some(err) = &group_ready.committed_read_error { + self.surface_committed_read_error(group_id, err); + } if !group_ready.committed_entries.is_empty() { self.apply_group_commits(group_id, &group_ready); } @@ -145,6 +160,11 @@ impl RaftLoop { // while placement is None (bootstrap window — guard is inside the fn). self.converge_leaving_learners(); + // Drop the replica of every data group this node has left, once its + // own removal applied. A node outside a group's placement then hosts + // no replica of it. + self.unmount_left_groups(); + // Placement reconcile is throttled well above the tick rate: SetPlacement // is a normal metadata entry (not a conf-change), so Raft would not dedup // per-tick re-proposals before they commit. Running ~1s apart lets each @@ -155,6 +175,17 @@ impl RaftLoop { if tick.is_multiple_of(PLACEMENT_RECONCILE_TICK_INTERVAL) { self.reconcile_placement(); } + // Move each data group's leadership to its preferred leader. The + // bootstrap node leads every group otherwise. + let balance_interval = super::super::leader_balance::LEADER_BALANCE_TICK_INTERVAL; + if tick.is_multiple_of(balance_interval) { + self.balance_leadership(tick / balance_interval); + } + // Find the leader of an unhosted data group whose hint names none. + let probe_interval = super::leader_probe::LEADER_PROBE_TICK_INTERVAL; + if tick.is_multiple_of(probe_interval) { + self.probe_unhosted_leaders(tick / probe_interval); + } if tick.is_multiple_of(ORPHAN_PARTIAL_GC_TICK_INTERVAL) && let Some(ref dir) = self.data_dir { diff --git a/nodedb-cluster/src/raft_loop/tick/dispatch_outbound.rs b/nodedb-cluster/src/raft_loop/tick/dispatch_outbound.rs index 677c7df8d..1378d85cc 100644 --- a/nodedb-cluster/src/raft_loop/tick/dispatch_outbound.rs +++ b/nodedb-cluster/src/raft_loop/tick/dispatch_outbound.rs @@ -14,6 +14,7 @@ use nodedb_raft::transport::RaftTransport; use crate::forward::PlanExecutor; use super::super::loop_core::{CommitApplier, RaftLoop}; +use super::peer_batch::drive_peer_batch; impl RaftLoop { /// Batch and dispatch AppendEntries / RequestVote / TimeoutNow messages @@ -21,6 +22,7 @@ impl RaftLoop { pub(super) fn dispatch_outbound_messages(&self, groups: &[(u64, nodedb_raft::Ready)]) { let mut ae_batches: BatchMap> = BatchMap::new(); + // Keyed by group: a group's vote requests wait for its own disk. let mut vote_batches: BatchMap> = BatchMap::new(); let mut pre_vote_batches: BatchMap> = @@ -36,9 +38,9 @@ impl RaftLoop { } for (peer, req) in &group_ready.vote_requests { vote_batches - .entry(*peer) + .entry(*group_id) .or_default() - .push((*group_id, req.clone())); + .push((*peer, req.clone())); } for (peer, req) in &group_ready.pre_vote_requests { pre_vote_batches @@ -69,76 +71,48 @@ impl RaftLoop { if *shutdown_rx.borrow() { return; } - for (group_id, req) in messages { - tokio::select! { - biased; - _ = shutdown_rx.changed() => return, - rpc = transport.append_entries(peer, req) => { - match rpc { - Ok(resp) => { - let mut mr = - mr.lock().unwrap_or_else(|p| p.into_inner()); - if let Err(e) = mr - .handle_append_entries_response(group_id, peer, &resp) - { - debug!(group_id, peer, error = %e, "handle ae response"); - } - // A response can bump the term (step - // down to follower); persist it durably. - if let Err(e) = mr.persist_group_hard_state(group_id) { - error!(group_id, peer, error = %e, "persist hard state after ae response"); - } - } - Err(e) => { - warn!(group_id, peer, error = %e, "append_entries RPC failed"); - break; // Peer is down — skip remaining groups. - } - } + drive_peer_batch( + peer, + "append_entries", + messages, + &mut shutdown_rx, + |req| transport.send_append_entries(peer, req), + |group_id, resp| { + let mut mr = mr.lock().unwrap_or_else(|p| p.into_inner()); + if let Err(e) = mr.handle_append_entries_response(group_id, peer, &resp) { + debug!(group_id, peer, error = %e, "handle ae response"); } - } - } + // A response can bump the term and step this node + // down to follower. Persist that durably. + if let Err(e) = mr.persist_group_hard_state(group_id) { + error!(group_id, peer, error = %e, "persist hard state after ae response"); + } + }, + ) + .await; }); } - // Dispatch batched RequestVote — one task per peer. - for (peer, votes) in vote_batches { - let transport = self.transport.clone(); - let mr = self.multi_raft.clone(); - let mut shutdown_rx = self.shutdown_watch.subscribe(); - tokio::spawn(async move { - if *shutdown_rx.borrow() { - return; - } - for (group_id, req) in votes { - tokio::select! { - biased; - _ = shutdown_rx.changed() => return, - rpc = transport.request_vote(peer, req) => { - match rpc { - Ok(resp) => { - let mut mr = - mr.lock().unwrap_or_else(|p| p.into_inner()); - if let Err(e) = mr - .handle_request_vote_response(group_id, peer, &resp) - { - debug!(group_id, peer, error = %e, "handle vote response"); - } - // A higher-term response steps this - // candidate down to follower; persist - // that term bump durably. - if let Err(e) = mr.persist_group_hard_state(group_id) { - error!(group_id, peer, error = %e, "persist hard state after vote response"); - } - } - Err(e) => { - warn!(group_id, peer, error = %e, "request_vote RPC failed"); - break; - } - } - } - } - } - }); + // Dispatch RequestVote — one task per group, one request per peer. + // A vote request leaves only once the candidate's term and self-vote + // are durable, so a restart cannot forget them. The group waits for + // its own disk, never under the `MultiRaft` lock, and no other + // group's election waits with it. + if !vote_batches.is_empty() { + let tickets: BatchMap = { + let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + vote_batches + .keys() + .filter_map(|group_id| { + mr.durability_ticket(*group_id) + .map(|ticket| (*group_id, ticket)) + }) + .collect() + }; + for (group_id, votes) in vote_batches { + let ticket = tickets.get(&group_id).cloned(); + self.dispatch_group_votes(group_id, ticket, votes); + } } // Dispatch batched PreVote — one task per peer. No hard-state persist @@ -153,40 +127,28 @@ impl RaftLoop { if *shutdown_rx.borrow() { return; } - for (group_id, req) in probes { - tokio::select! { - biased; - _ = shutdown_rx.changed() => return, - rpc = transport.pre_vote(peer, req) => { - match rpc { - Ok(resp) => { - let mut mr = - mr.lock().unwrap_or_else(|p| p.into_inner()); - if let Err(e) = mr - .handle_pre_vote_response(group_id, peer, &resp) - { - debug!(group_id, peer, error = %e, "handle pre-vote response"); - } - // A pre-vote grants nothing and persists - // nothing, but a higher-term response - // still steps this node down — that term - // bump must be durable before it acts on - // it. Cheap otherwise: the persist is - // skipped when nothing changed. - if let Err(e) = mr.persist_group_hard_state(group_id) { - error!(group_id, peer, error = %e, "persist hard state after pre-vote response"); - } - } - Err(e) => { - // Treat as "no grant" and move on — a failed - // peer must not abort the round for the rest. - warn!(group_id, peer, error = %e, "pre_vote RPC failed"); - break; // Peer is down — skip remaining groups. - } - } + drive_peer_batch( + peer, + "pre_vote", + probes, + &mut shutdown_rx, + |req| transport.send_pre_vote(peer, req), + |group_id, resp| { + let mut mr = mr.lock().unwrap_or_else(|p| p.into_inner()); + if let Err(e) = mr.handle_pre_vote_response(group_id, peer, &resp) { + debug!(group_id, peer, error = %e, "handle pre-vote response"); } - } - } + // A pre-vote grants nothing and persists nothing, but + // a higher-term response still steps this node down — + // that term bump must be durable before it acts on + // it. Cheap otherwise: the persist is skipped when + // nothing changed. + if let Err(e) = mr.persist_group_hard_state(group_id) { + error!(group_id, peer, error = %e, "persist hard state after pre-vote response"); + } + }, + ) + .await; }); } diff --git a/nodedb-cluster/src/raft_loop/tick/epoch_bump.rs b/nodedb-cluster/src/raft_loop/tick/epoch_bump.rs index 023cfa5c7..40b26a9cd 100644 --- a/nodedb-cluster/src/raft_loop/tick/epoch_bump.rs +++ b/nodedb-cluster/src/raft_loop/tick/epoch_bump.rs @@ -2,8 +2,11 @@ //! Proposing a new cluster generation on metadata-group leadership acquisition. +use crate::catalog::ClusterCatalog; +use crate::cluster_epoch::ClusterEpochState; +use crate::error::Result; use crate::forward::PlanExecutor; -use crate::metadata_group::codec::{decode_entry, encode_entry}; +use crate::metadata_group::codec::encode_entry; use crate::metadata_group::entry::MetadataEntry; use crate::raft_loop::loop_core::{CommitApplier, RaftLoop}; @@ -57,57 +60,82 @@ impl RaftLoop { } impl RaftLoop { - /// Advance this node's applied epoch for every committed - /// [`MetadataEntry::ClusterEpochBump`] in `pairs`. + /// Adopt the epoch bumps `entry` carries, committed at `index`. /// - /// Applying the entry is the moment the generation becomes this node's - /// own: before it, the number was something a peer asserted; after it, - /// this node has processed the same committed fact everyone else has and - /// may stamp it outbound. + /// The caller runs this only after every earlier entry of the batch has + /// applied, so no epoch lands ahead of an entry the applier stopped at. /// - /// Entries that fail to decode are skipped rather than fatal — they belong - /// to variants this node's build does not know, and the epoch is not among - /// them. - pub(super) fn adopt_committed_cluster_epochs(&self, pairs: &[(u64, Vec)]) { - for (index, data) in pairs { - let Ok(entry) = decode_entry(data) else { - continue; - }; - self.adopt_entry_epoch(&entry, *index); - } + /// The epoch is persisted to the catalog, so the adoption runs on a + /// blocking thread and the metadata lane awaits it. + pub(super) async fn adopt_cluster_epoch( + &self, + entry: &MetadataEntry, + index: u64, + ) -> Result<()> { + let state = std::sync::Arc::clone(&self.cluster_epoch); + let catalog = self.catalog.clone(); + let node_id = self.node_id; + let entry = entry.clone(); + tokio::task::spawn_blocking(move || { + adopt_entry_epoch(&state, catalog.as_deref(), node_id, &entry, index) + }) + .await + .map_err(|e| crate::error::ClusterError::Storage { + detail: format!("cluster epoch adoption task at log index {index}: {e}"), + })? } +} - /// Recurse into batches so a bump packed inside one still lands. - fn adopt_entry_epoch(&self, entry: &MetadataEntry, index: u64) { - match entry { - MetadataEntry::ClusterEpochBump { epoch } => { - self.cluster_epoch.advance_applied(*epoch); - if let Some(catalog) = self.catalog.as_ref() - && let Err(e) = crate::cluster_epoch::persist_applied_epoch(catalog, *epoch) - { - tracing::warn!( - node = self.node_id, - epoch = *epoch, - error = %e, - "applied the cluster epoch but could not persist it; \ - it is re-learned from the log after a restart" - ); +/// Whether `entry` carries a [`MetadataEntry::ClusterEpochBump`], directly or +/// nested in a batch or a prepared DDL. +pub(super) fn carries_epoch(entry: &MetadataEntry) -> bool { + match entry { + MetadataEntry::ClusterEpochBump { .. } => true, + MetadataEntry::Batch { entries } => entries.iter().any(carries_epoch), + MetadataEntry::DdlPrepared { entry, .. } => carries_epoch(entry), + _ => false, + } +} + +/// Adopt every epoch bump `entry` carries. +/// +/// Applying the entry is the moment the generation becomes this node's own: +/// before it, the number was something a peer asserted. The epoch is persisted +/// before the in-memory mark advances, so the applied mark is always durable. +/// A bump at or below the applied mark is already durable and is not written. +fn adopt_entry_epoch( + state: &ClusterEpochState, + catalog: Option<&ClusterCatalog>, + node_id: u64, + entry: &MetadataEntry, + index: u64, +) -> Result<()> { + match entry { + MetadataEntry::ClusterEpochBump { epoch } => { + if *epoch > state.applied() { + if let Some(catalog) = catalog { + crate::cluster_epoch::persist_applied_epoch(catalog, *epoch)?; } - tracing::info!( - node = self.node_id, - epoch = *epoch, - log_index = index, - "applied cluster epoch" - ); + state.advance_applied(*epoch); } - MetadataEntry::Batch { entries } => { - for sub in entries { - self.adopt_entry_epoch(sub, index); - } + tracing::info!( + node = node_id, + epoch = *epoch, + log_index = index, + "applied cluster epoch" + ); + Ok(()) + } + MetadataEntry::Batch { entries } => { + for sub in entries { + adopt_entry_epoch(state, catalog, node_id, sub, index)?; } - MetadataEntry::DdlPrepared { entry, .. } => self.adopt_entry_epoch(entry, index), - _ => {} + Ok(()) } + MetadataEntry::DdlPrepared { entry, .. } => { + adopt_entry_epoch(state, catalog, node_id, entry, index) + } + _ => Ok(()), } } @@ -122,8 +150,7 @@ impl RaftLoop { #[cfg(test)] mod tests { use super::*; - use crate::cluster_epoch::ClusterEpochState; - use crate::metadata_group::codec::encode_entry; + use crate::metadata_group::codec::decode_entry; /// Two nodes in one process must hold independent generations. A single /// process-wide counter would alias them and no disagreement could ever be @@ -185,4 +212,50 @@ mod tests { } assert_eq!(found, Some(8), "a nested bump must be reachable"); } + + fn bump(epoch: u64) -> MetadataEntry { + MetadataEntry::ClusterEpochBump { epoch } + } + + /// A failed epoch persist leaves the in-memory mark and the stored epoch + /// where they were. The retry persists and advances. + #[test] + fn epoch_persist_error_does_not_advance() { + let dir = tempfile::tempdir().unwrap(); + let catalog = ClusterCatalog::open(&dir.path().join("cluster.redb")).unwrap(); + let state = ClusterEpochState::new(1); + + adopt_entry_epoch(&state, Some(&catalog), 1, &bump(2), 10).unwrap(); + catalog.fail_next_epoch_write_for_test(); + assert!(adopt_entry_epoch(&state, Some(&catalog), 1, &bump(3), 11).is_err()); + assert_eq!(state.applied(), 2); + assert_eq!(catalog.load_cluster_epoch().unwrap(), Some(2)); + + adopt_entry_epoch(&state, Some(&catalog), 1, &bump(3), 11).unwrap(); + assert_eq!(state.applied(), 3); + assert_eq!(catalog.load_cluster_epoch().unwrap(), Some(3)); + } + + /// Replaying an older bump never lowers the persisted epoch. + #[test] + fn replayed_older_bump_keeps_the_persisted_epoch() { + let dir = tempfile::tempdir().unwrap(); + let catalog = ClusterCatalog::open(&dir.path().join("cluster.redb")).unwrap(); + let state = ClusterEpochState::new(0); + adopt_entry_epoch(&state, Some(&catalog), 1, &bump(7), 5).unwrap(); + adopt_entry_epoch(&state, Some(&catalog), 1, &bump(3), 2).unwrap(); + assert_eq!(state.applied(), 7); + assert_eq!(catalog.load_cluster_epoch().unwrap(), Some(7)); + } + + #[test] + fn nested_bumps_are_detected() { + let batch = MetadataEntry::Batch { + entries: vec![MetadataEntry::CatalogDdl { payload: vec![] }, bump(4)], + }; + assert!(carries_epoch(&batch)); + assert!(!carries_epoch(&MetadataEntry::CatalogDdl { + payload: vec![] + })); + } } diff --git a/nodedb-cluster/src/raft_loop/tick/leader_hints.rs b/nodedb-cluster/src/raft_loop/tick/leader_hints.rs new file mode 100644 index 000000000..62890454d --- /dev/null +++ b/nodedb-cluster/src/raft_loop/tick/leader_hints.rs @@ -0,0 +1,97 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The routing table's leader hints, kept in step with this node's Raft. +//! +//! Five writers touch a group's leader hint: +//! - this node's Raft, here, with the term it saw the leader at; +//! - a leader redirect from another node, with the term that node knows the +//! leader at, here and in the gateway; +//! - the leader probe of a group this node does not host (see +//! [`super::leader_probe`]); +//! - the SWIM liveness hook, which clears a suspected leader and keeps the +//! term, and fills the clear back when the node answers again; +//! - the metadata log and placement, which name a planned leader and carry +//! no term. +//! +//! A termed hint applies above the hint's term. It also fills a hint cleared +//! at its own term, because a clear is only a suspicion: this node's Raft +//! reports a leader only while it leads or the leader's contact is fresh, +//! and a redirect names the leader the redirecting node follows. A term-less +//! hint applies only while the hint holds no term. A term-less hint never +//! replaces an observed one, and a new election's higher term always wins. + +use tracing::debug; + +use crate::forward::PlanExecutor; + +use super::super::loop_core::{CommitApplier, RaftLoop}; + +impl RaftLoop { + /// Write every hosted group's live Raft leader into the routing table, + /// where it is newer than the hint or fills a cleared hint at its term. + /// + /// The Raft state is read under the `MultiRaft` lock, and the routing + /// table is written after that lock is released. The write lock is taken + /// only when a hint changes. + pub(super) fn sync_leader_hints(&self) { + let (observed, routing) = { + let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + (mr.observed_leaders(), mr.routing()) + }; + let changed: Vec<(u64, u64, u64)> = { + let table = routing.read().unwrap_or_else(|p| p.into_inner()); + observed + .into_iter() + .filter(|&(group_id, leader, term)| { + table.leader_confirmation_is_new(group_id, leader, term) + }) + .collect() + }; + if changed.is_empty() { + return; + } + let mut table = routing.write().unwrap_or_else(|p| p.into_inner()); + for (group_id, leader, term) in changed { + if table.confirm_leader(group_id, leader, term) { + debug!( + node = self.node_id, + group_id, leader, term, "routing leader hint follows the Raft leader" + ); + } + } + } + + /// Record the leader a redirect names for `group_id`, at the term the + /// redirecting node knows it at. A redirect without a leader leaves the + /// hint as it is. A redirect below the hint's term is stale and leaves + /// the hint as it is. A redirect at the hint's term fills only a cleared + /// hint. A redirect at term 0 comes from a node that holds no term for + /// the group, and fills only a hint that holds none. + pub(in crate::raft_loop) fn observe_redirect( + &self, + group_id: u64, + leader_hint: Option, + term: u64, + ) { + let Some(leader) = leader_hint else { + return; + }; + let routing = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .routing(); + let mut table = routing.write().unwrap_or_else(|p| p.into_inner()); + let changed = if term == 0 { + table.set_leader(group_id, leader) + } else { + table.confirm_leader(group_id, leader, term) + }; + if changed { + debug!( + node = self.node_id, + group_id, leader, term, "routing leader hint follows a leader redirect" + ); + } + } +} diff --git a/nodedb-cluster/src/raft_loop/tick/leader_probe.rs b/nodedb-cluster/src/raft_loop/tick/leader_probe.rs new file mode 100644 index 000000000..dfaacb3f2 --- /dev/null +++ b/nodedb-cluster/src/raft_loop/tick/leader_probe.rs @@ -0,0 +1,365 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Leader probe for data groups whose Raft does not keep this node's view. +//! +//! Two kinds of data group are probed: +//! - a group this node hosts no replica of; +//! - a group this node hosts but has left: its authored placement names +//! other nodes only, and this node does not lead it. +//! +//! This node's Raft sees no leader of a group it hosts no replica of. Its +//! hint for such a group moves by redirects, and a redirect needs a node to +//! ask. A hint that names no leader, or names a node outside the group's +//! placement, leaves nothing to ask: every read, write and surrogate +//! exchange for the group would fail until an election somewhere happened +//! to reach this node. +//! +//! A hint that names a placement node goes stale too: a transfer or a +//! failover in the group moves its leadership, and nothing tells this node. +//! Nor does anything tell it the group's voters and learners: this node +//! applies none of the group's conf changes. A node that left a group learns +//! of its removal from one `AppendEntries` the leader sends as it applies +//! the removal. A lost send leaves the node listing itself as a replica. +//! +//! Each pass asks one node of every probed group for its leader status: the +//! leader it knows and the term that leader leads. The receiver answers from +//! its own Raft state with no quorum round, so a probe costs the leader +//! nothing but the answer. A routing hint needs no linearizable +//! confirmation: a request sent to a node that no longer leads is redirected. +//! The answer confirms the hint (see +//! [`crate::routing::RoutingTable::confirm_leader`]), so a newer term always +//! replaces it, and a stale answer never moves it back. +//! - A hint that names a placement node is checked at that node. A transfer +//! or a failover reaches the hint within one pass. +//! - A hint that names no leader, or a node outside the placement, is +//! probed at a placement node that rotates each pass, so a dead placement +//! node does not stall the probe. +//! +//! An answer from the leader itself carries the group's voters and learners. +//! This node's routing view adopts them (see +//! [`crate::routing::RoutingTable::adopt_leader_membership`]). A node that +//! left a group then stops listing itself, and the unmount step drops its +//! replica. + +use std::collections::HashSet; +use std::sync::{Arc, RwLock}; + +use tracing::debug; + +use crate::forward::PlanExecutor; +use crate::routing::RoutingTable; +use crate::rpc_codec::{LeaderStatusRequest, LeaderStatusResponse, RaftRpc}; + +use super::super::loop_core::{CommitApplier, RaftLoop}; + +/// Ticks between leader-probe passes (about 0.5 s at the 10 ms tick). +pub(in crate::raft_loop) const LEADER_PROBE_TICK_INTERVAL: u64 = 50; + +/// The `(leader, term)` a leader-status answer names, when it names one. +fn named_leader(status: &LeaderStatusResponse) -> Option<(u64, u64)> { + (status.leader != 0 && status.term != 0).then_some((status.leader, status.term)) +} + +/// Apply `target`'s leader-status answer for `group_id` to `routing`: the +/// named leader confirms the hint, and a membership `target` answers as the +/// leader replaces the view of the group's voters and learners. +fn adopt_answer( + routing: &RwLock, + group_id: u64, + target: u64, + status: &LeaderStatusResponse, +) { + let Some((leader, term)) = named_leader(status) else { + return; + }; + let mut table = routing.write().unwrap_or_else(|p| p.into_inner()); + if table.confirm_leader(group_id, leader, term) { + debug!( + group_id, + leader, term, "leader probe: routing hint names the group's leader" + ); + } + let Some(membership) = status.membership.as_ref().filter(|_| leader == target) else { + return; + }; + if table.adopt_leader_membership( + group_id, + leader, + term, + &membership.voters, + &membership.learners, + ) { + debug!( + group_id, + leader, + term, + voters = ?membership.voters, + learners = ?membership.learners, + "leader probe: routing view adopts the leader's membership" + ); + } +} + +/// The hosted groups whose Raft keeps this node's view: those this node +/// leads, and those whose placement is not authored or names this node. +/// Every other hosted group is one this node has left. +fn kept_by_raft( + self_id: u64, + hosted: &HashSet, + leading: &HashSet, + routing: &RoutingTable, +) -> HashSet { + hosted + .iter() + .copied() + .filter(|gid| { + leading.contains(gid) + || routing + .group_info(*gid) + .and_then(|info| info.placement.as_ref()) + .is_none_or(|placement| placement.contains(&self_id)) + }) + .collect() +} + +/// `(group_id, target)` for each data group not in `kept` that the probe +/// asks on pass `pass`, per the rules in the module docs. `kept` holds the +/// hosted groups whose Raft keeps this node's view (see [`kept_by_raft`]). +/// Pure and deterministic. +pub(super) fn plan_probes( + self_id: u64, + kept: &HashSet, + routing: &RoutingTable, + pass: u64, +) -> Vec<(u64, u64)> { + let mut out: Vec<(u64, u64)> = routing + .group_members() + .iter() + .filter(|(gid, _)| { + **gid != crate::metadata_group::METADATA_GROUP_ID + && **gid != crate::calvin::sequencer::SEQUENCER_GROUP_ID + && !kept.contains(*gid) + }) + .filter_map(|(&gid, info)| { + let mut candidates: Vec = routing + .effective_placement(gid) + .into_iter() + .filter(|n| *n != self_id) + .collect(); + candidates.sort_unstable(); + candidates.dedup(); + if candidates.contains(&info.leader) { + return Some((gid, info.leader)); + } + if candidates.is_empty() { + return None; + } + let idx = usize::try_from(pass % candidates.len() as u64).unwrap_or(0); + candidates.get(idx).map(|&target| (gid, target)) + }) + .collect(); + out.sort_unstable(); + out +} + +impl RaftLoop { + /// Probe the data groups [`plan_probes`] names for `pass`. Each probe + /// runs on its own task, one per group at a time. + pub(in crate::raft_loop) fn probe_unhosted_leaders(&self, pass: u64) { + let (probes, routing) = { + let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let group_ids = mr.group_ids(); + let hosted: HashSet = group_ids.iter().copied().collect(); + let leading: HashSet = group_ids + .iter() + .copied() + .filter(|gid| mr.group_role_is_leader(*gid)) + .collect(); + let routing = mr.routing(); + let probes = { + let table = routing.read().unwrap_or_else(|p| p.into_inner()); + let kept = kept_by_raft(self.node_id, &hosted, &leading, &table); + plan_probes(self.node_id, &kept, &table, pass) + }; + (probes, routing) + }; + for (group_id, target) in probes { + let addr = self + .topology + .read() + .unwrap_or_else(|p| p.into_inner()) + .get_node(target) + .and_then(|node| node.socket_addr()); + let Some(addr) = addr else { + continue; + }; + if !self.tick_state.begin_probe(group_id) { + continue; + } + self.transport.register_peer(target, addr); + let transport = Arc::clone(&self.transport); + let routing = Arc::clone(&routing); + let tick_state = Arc::clone(&self.tick_state); + tokio::spawn(async move { + let request = RaftRpc::LeaderStatusRequest(LeaderStatusRequest { group_id }); + if let Ok(RaftRpc::LeaderStatusResponse(status)) = + transport.send_rpc(target, request).await + { + adopt_answer(&routing, group_id, target, &status); + } + tick_state.end_probe(group_id); + }); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn set(ids: &[u64]) -> HashSet { + ids.iter().copied().collect() + } + + /// An answer that names `leader` at `term`, with no membership. + fn answer(leader: u64, term: u64) -> LeaderStatusResponse { + LeaderStatusResponse { + leader, + term, + membership: None, + } + } + + #[test] + fn an_answer_names_a_leader_only_with_its_term() { + assert_eq!(named_leader(&answer(3, 5)), Some((3, 5))); + assert_eq!(named_leader(&answer(0, 0)), None); + assert_eq!(named_leader(&answer(3, 0)), None); + } + + #[test] + fn every_probed_group_is_probed_every_pass() { + let mut rt = RoutingTable::uniform(3, &[1, 2, 3], 1); + rt.set_placement(1, vec![2]); + rt.set_placement(2, vec![3]); + rt.set_placement(3, vec![2]); + // Group 1: the hint names node 1, outside its placement. + assert!(rt.observe_leader(1, 1, 3)); + // Group 2: the hint names node 3, its placement node. + assert!(rt.observe_leader(2, 3, 3)); + // Group 3: no leader known. + rt.clear_leader(3); + // A hint on a placement node is checked at that node; the others go + // to the rotating placement node. + assert_eq!( + plan_probes(1, &set(&[0]), &rt, 0), + vec![(1, 2), (2, 3), (3, 2)] + ); + assert_eq!( + plan_probes(1, &set(&[0]), &rt, 1), + vec![(1, 2), (2, 3), (3, 2)] + ); + // A group this node's Raft keeps is never probed. + assert_eq!(plan_probes(1, &set(&[0, 1, 3]), &rt, 0), vec![(2, 3)]); + assert_eq!(plan_probes(1, &set(&[0, 1, 2, 3]), &rt, 0), Vec::new()); + } + + /// A transfer in a group this node does not host leaves its hint on the + /// old leader, a placement node. The probe asks that node, whose answer + /// moves the hint to the new leader at the new term. A stale answer from + /// an older term never moves it back. + #[test] + fn a_hint_left_on_the_old_leader_after_a_transfer_follows_the_answer() { + let mut rt = RoutingTable::uniform(1, &[1, 2, 3], 3); + rt.set_placement(1, vec![2, 3]); + assert!(rt.observe_leader(1, 2, 4)); + assert_eq!(plan_probes(1, &set(&[]), &rt, 7), vec![(1, 2)]); + // Node 2 answers that node 3 leads term 5. + let (leader, term) = named_leader(&answer(3, 5)).expect("a named leader"); + assert!(rt.confirm_leader(1, leader, term)); + assert_eq!(rt.leader_at_term_for_vshard(0).unwrap(), (3, 5)); + assert!(!rt.confirm_leader(1, 2, 4)); + assert_eq!(plan_probes(1, &set(&[]), &rt, 8), vec![(1, 3)]); + } + + /// A hosted group whose placement names other nodes only is probed + /// unless this node leads it. A hosted group whose placement names this + /// node, or has none authored, is kept by this node's Raft. + #[test] + fn a_hosted_group_this_node_left_is_probed() { + let mut rt = RoutingTable::uniform(3, &[1, 2, 3], 2); + rt.set_placement(1, vec![2, 3]); + rt.set_placement(2, vec![1, 2]); + rt.set_placement(3, vec![2, 3]); + let hosted = set(&[0, 1, 2, 3]); + // Node 1 leads group 3, which it left too. + let kept = kept_by_raft(1, &hosted, &set(&[3]), &rt); + assert_eq!(kept, set(&[0, 2, 3])); + assert_eq!( + plan_probes(1, &kept, &rt, 0) + .into_iter() + .map(|(gid, _)| gid) + .collect::>(), + vec![1] + ); + // A group with no authored placement is kept. + let unplaced = RoutingTable::uniform(1, &[1, 2, 3], 2); + assert_eq!(kept_by_raft(1, &set(&[1]), &set(&[]), &unplaced), set(&[1])); + } + + /// The leader's answer confirms the hint and replaces the membership + /// view. The same membership from a node that does not lead is ignored. + #[test] + fn a_leaders_answer_carries_the_membership_into_the_view() { + let mut rt = RoutingTable::uniform(1, &[1, 2, 3], 3); + rt.set_placement(1, vec![2]); + let routing = RwLock::new(rt); + let membership = Some(crate::rpc_codec::LeaderMembership { + voters: vec![2], + learners: vec![], + }); + + // Node 3 names node 2 and answers a membership it has no right to. + adopt_answer( + &routing, + 1, + 3, + &LeaderStatusResponse { + leader: 2, + term: 4, + membership: membership.clone(), + }, + ); + { + let table = routing.read().unwrap_or_else(|p| p.into_inner()); + let info = table.group_info(1).expect("group 1"); + assert_eq!((info.leader, info.leader_term), (2, 4)); + assert_eq!(info.members.len(), 3); + } + + // Node 2 answers as the leader. + adopt_answer( + &routing, + 1, + 2, + &LeaderStatusResponse { + leader: 2, + term: 4, + membership, + }, + ); + let table = routing.read().unwrap_or_else(|p| p.into_inner()); + let info = table.group_info(1).expect("group 1"); + assert_eq!(info.members, vec![2]); + assert!(info.learners.is_empty()); + } + + #[test] + fn the_target_rotates_over_the_placement() { + let mut rt = RoutingTable::uniform(1, &[1, 2, 3], 3); + rt.set_placement(1, vec![2, 3]); + rt.clear_leader(1); + assert_eq!(plan_probes(1, &set(&[]), &rt, 0), vec![(1, 2)]); + assert_eq!(plan_probes(1, &set(&[]), &rt, 1), vec![(1, 3)]); + } +} diff --git a/nodedb-cluster/src/raft_loop/tick/metadata_apply.rs b/nodedb-cluster/src/raft_loop/tick/metadata_apply.rs new file mode 100644 index 000000000..027cee8ed --- /dev/null +++ b/nodedb-cluster/src/raft_loop/tick/metadata_apply.rs @@ -0,0 +1,177 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Metadata group (0) apply: cluster epochs adopted in log order between +//! applier calls, then the durable applied floor and log compaction. + +use nodedb_raft::LogEntry; +use tracing::warn; + +use crate::conf_change::ConfChange; +use crate::forward::PlanExecutor; +use crate::metadata_group::METADATA_GROUP_ID; +use crate::metadata_group::applier::CommittedMetadata; +use crate::metadata_group::entry::MetadataEntry; + +use super::super::loop_core::{CommitApplier, RaftLoop}; +use super::epoch_bump::carries_epoch; + +/// Group 0's committed entries, each decoded once. +struct MetadataBatch<'a> { + /// A conf change is an empty entry: the applier advances its watermark + /// past it and applies nothing. + commits: Vec>, + /// The index of the first entry whose apply changes the live routing + /// table. + first_routing_change: Option, +} + +/// Whether `entry` changes the live routing table when it applies. +fn touches_routing(entry: &MetadataEntry) -> bool { + match entry { + MetadataEntry::RoutingChange(_) => true, + MetadataEntry::Batch { entries } => entries.iter().any(touches_routing), + MetadataEntry::DdlPrepared { entry, .. } => touches_routing(entry), + _ => false, + } +} + +/// Decode each of `entries` once. +fn decode_batch(entries: &[LogEntry]) -> MetadataBatch<'_> { + let mut first_routing_change = None; + let commits = entries + .iter() + .map(|entry| { + let (commit, touches) = if ConfChange::from_entry_data(&entry.data).is_some() { + (CommittedMetadata::empty(entry.index), true) + } else { + let commit = CommittedMetadata::decode(entry.index, &entry.data); + let touches = commit.entry().is_some_and(touches_routing); + (commit, touches) + }; + if touches && first_routing_change.is_none() { + first_routing_change = Some(entry.index); + } + commit + }) + .collect(); + MetadataBatch { + commits, + first_routing_change, + } +} + +impl RaftLoop { + /// Apply group 0's committed `entries`. Conf changes are already applied + /// to the membership. Returns the highest index delivered, or 0. + /// + /// When the applier reports durable effects, the returned index is also + /// saved as the group's applied floor, after the routing table it changed. + pub(super) async fn apply_metadata_commits(&self, entries: &[LogEntry]) -> u64 { + let batch = decode_batch(entries); + let delivered = self.apply_in_epoch_order(&batch.commits).await; + if self.metadata_applier.durable_effects() { + self.save_metadata_floor(batch.first_routing_change, delivered) + .await; + } + delivered + } + + /// Hand `commits` to the applier in runs that end before each epoch + /// bump. A bump is adopted only after every earlier entry applied, so no + /// epoch lands past an entry the applier stopped at. + async fn apply_in_epoch_order(&self, commits: &[CommittedMetadata<'_>]) -> u64 { + let mut delivered = 0u64; + let mut run_start = 0usize; + for (position, commit) in commits.iter().enumerate() { + let Some(epoch_entry) = commit.entry().filter(|e| carries_epoch(e)) else { + continue; + }; + if !self + .apply_run(&commits[run_start..position], &mut delivered) + .await + { + return delivered; + } + // The bump entry itself opens the next run. + run_start = position; + if let Err(e) = self.adopt_cluster_epoch(epoch_entry, commit.index).await { + warn!( + node = self.node_id, + log_index = commit.index, + error = %e, + "could not persist a committed cluster epoch; the entry is re-delivered" + ); + return delivered; + } + } + self.apply_run(&commits[run_start..], &mut delivered).await; + delivered + } + + /// Apply `run` and raise `delivered` to what the applier reports. + /// Returns whether every entry of the run applied. + async fn apply_run(&self, run: &[CommittedMetadata<'_>], delivered: &mut u64) -> bool { + let Some(through) = run.last().map(|commit| commit.index) else { + return true; + }; + let applied = self.metadata_applier.apply_decoded(run).await; + if applied > *delivered { + *delivered = applied; + } + applied == through + } + + /// Save the durable applied floor for group 0, then compact the log up + /// to it when the configured threshold is reached. + /// + /// The floor is `delivered`, lowered below the first routing change of + /// the batch when the routing table cannot be persisted. It never passes + /// an entry whose effects are not durable. + /// + /// Every disk write runs off the async threads: the routing save through + /// the routing persister, the floor and the compaction on a blocking + /// thread. The lane awaits them. The tick never does. + async fn save_metadata_floor(&self, first_routing_change: Option, delivered: u64) { + if delivered == 0 { + return; + } + let mut floor = delivered; + if let Some(first) = first_routing_change.filter(|index| *index <= delivered) + && let Some(persister) = self.routing_persister.as_ref() + && !persister.wait(persister.request()).await + { + warn!( + node = self.node_id, + log_index = first, + "could not persist the routing table; the applied floor stays below the change" + ); + floor = first.saturating_sub(1); + } + if floor == 0 { + return; + } + let multi_raft = std::sync::Arc::clone(&self.multi_raft); + let node_id = self.node_id; + let saved = tokio::task::spawn_blocking(move || { + let mut mr = multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + if let Err(e) = mr.save_applied_index(METADATA_GROUP_ID, floor) { + warn!( + node = node_id, + floor, + error = %e, + "could not save the metadata applied floor; a restart replays from the previous floor" + ); + return; + } + // Entries at or below the floor are durable in the host state, + // and a peer that needs them catches up from a group 0 snapshot. + if let Err(e) = mr.maybe_compact_group(METADATA_GROUP_ID, floor) { + warn!(node = node_id, floor, error = %e, "group 0 log compaction failed"); + } + }) + .await; + if let Err(e) = saved { + warn!(node = self.node_id, floor, error = %e, "metadata floor save task failed"); + } + } +} diff --git a/nodedb-cluster/src/raft_loop/tick/metadata_lane.rs b/nodedb-cluster/src/raft_loop/tick/metadata_lane.rs new file mode 100644 index 000000000..74e5c32e1 --- /dev/null +++ b/nodedb-cluster/src/raft_loop/tick/metadata_lane.rs @@ -0,0 +1,188 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Metadata apply lane: group 0's committed entries applied in log order, +//! off the tick. +//! +//! The metadata applier awaits host effects, a Data-Plane dispatch among +//! them. Awaiting it inside the tick would stall heartbeats and elections of +//! every group on this node for as long as one apply takes. The tick hands +//! each committed batch to this lane through a bounded channel instead, and +//! moves on. A batch the full channel refuses goes back to `Ready`. +//! +//! The lane runs in the loop's own task, beside the tick loop, and ends with +//! it. For each run it: +//! 1. takes group 0's apply gate, so a snapshot install never interleaves; +//! 2. drops the entries at or below `last_applied`, which a snapshot covers; +//! 3. awaits the applier, which adopts epochs in log order and saves the +//! durable applied floor before it returns (see [`super::metadata_apply`]); +//! 4. advances Raft's `last_applied` and the group's apply watcher to the +//! index the applier reached. +//! +//! An entry the applier stops at stays in the lane, with every entry after +//! it, and the lane retries it after [`RETRY_TICKS`] ticks. Nothing past it +//! applies first. Log compaction reads the durable floor, which never passes +//! an applied entry. + +use std::collections::VecDeque; +use std::time::Duration; + +use nodedb_raft::LogEntry; +use tokio::sync::{mpsc, watch}; + +use crate::forward::PlanExecutor; +use crate::metadata_group::METADATA_GROUP_ID; +use crate::raft_loop::apply_gate::ApplyPermit; + +use super::super::loop_core::{CommitApplier, RaftLoop}; + +/// Batches the lane holds before the tick queues further entries back into +/// `Ready`. +pub(in crate::raft_loop) const METADATA_LANE_DEPTH: usize = 64; + +/// Ticks the lane waits before it retries an entry the applier stopped at. +const RETRY_TICKS: u32 = 10; + +/// Append the entries of `batch` above the last queued index. +fn enqueue(pending: &mut VecDeque, batch: Vec) { + let through = pending.back().map_or(0, |entry| entry.index); + pending.extend(batch.into_iter().filter(|entry| entry.index > through)); +} + +/// Wait `wait`, or less when shutdown begins. Returns whether it began. +async fn wait_or_shutdown(shutdown: &mut watch::Receiver, wait: Duration) -> bool { + if *shutdown.borrow() { + return true; + } + tokio::select! { + _ = tokio::time::sleep(wait) => false, + changed = shutdown.changed() => changed.is_err() || *shutdown.borrow(), + } +} + +impl RaftLoop { + /// Apply group 0's batches from `rx` in log order until the tick loop + /// closes the lane or shutdown begins. + pub(in crate::raft_loop) async fn run_metadata_lane( + &self, + mut rx: mpsc::Receiver>, + mut shutdown: watch::Receiver, + ) { + let mut pending: VecDeque = VecDeque::new(); + loop { + if pending.is_empty() { + match rx.recv().await { + Some(batch) => enqueue(&mut pending, batch), + None => return, + } + } + while let Ok(batch) = rx.try_recv() { + enqueue(&mut pending, batch); + } + // A node shutting down applies nothing more. Raft delivers the + // entries again above the durable floor on the next boot. + if *shutdown.borrow() { + return; + } + let Some(permit) = self.metadata_apply_permit(&mut shutdown).await else { + return; + }; + let applied = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .last_applied(METADATA_GROUP_ID) + .unwrap_or(0); + pending.retain(|entry| entry.index > applied); + let Some(through) = pending.back().map(|entry| entry.index) else { + continue; + }; + let batch: Vec = pending.iter().cloned().collect(); + let delivered = self.apply_metadata_commits(&batch).await; + drop(permit); + self.finish_metadata_apply(delivered); + pending.retain(|entry| entry.index > delivered); + if delivered < through + && wait_or_shutdown(&mut shutdown, self.tick_interval * RETRY_TICKS).await + { + return; + } + } + } + + /// Group 0's apply gate, once no snapshot install holds it. `None` when + /// shutdown begins first. + async fn metadata_apply_permit( + &self, + shutdown: &mut watch::Receiver, + ) -> Option { + let gates = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .apply_gates(); + loop { + if let Some(permit) = gates.try_apply(METADATA_GROUP_ID) { + return Some(permit); + } + if wait_or_shutdown(shutdown, self.tick_interval).await { + return None; + } + } + } + + /// Report an apply that reached `delivered` back to Raft and to the + /// group's watcher, and open the boot-time readiness watch. + /// + /// The first apply of group 0 on this node, the election no-op or a + /// replayed entry, flips the ready watch. The host awaits it before it + /// binds client-facing listeners. + fn finish_metadata_apply(&self, delivered: u64) { + if delivered > 0 { + let advanced = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .advance_applied(METADATA_GROUP_ID, delivered); + match advanced { + Ok(()) => self.group_watchers.bump(METADATA_GROUP_ID, delivered), + Err(e) => tracing::warn!( + error = %e, + delivered, + "metadata lane: failed to advance the applied index" + ), + } + } + if !*self.ready_watch.borrow() { + let _ = self.ready_watch.send(true); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn entry(index: u64) -> LogEntry { + LogEntry { + term: 1, + index, + data: Vec::new(), + } + } + + #[test] + fn a_repeated_range_is_queued_once() { + let mut pending = VecDeque::new(); + enqueue(&mut pending, vec![entry(1), entry(2)]); + enqueue(&mut pending, vec![entry(2), entry(3)]); + let indices: Vec = pending.iter().map(|e| e.index).collect(); + assert_eq!(indices, vec![1, 2, 3]); + } + + #[tokio::test] + async fn a_closed_shutdown_channel_ends_the_wait() { + let (tx, mut rx) = watch::channel(false); + drop(tx); + assert!(wait_or_shutdown(&mut rx, Duration::from_secs(60)).await); + } +} diff --git a/nodedb-cluster/src/raft_loop/tick/mod.rs b/nodedb-cluster/src/raft_loop/tick/mod.rs index d0484c9d2..779f87fb3 100644 --- a/nodedb-cluster/src/raft_loop/tick/mod.rs +++ b/nodedb-cluster/src/raft_loop/tick/mod.rs @@ -7,10 +7,26 @@ //! RequestVote / TimeoutNow messages. //! - [`apply_committed`]: apply a group's committed entries (conf-changes, //! metadata/data applier dispatch, watermark advance, epoch bump). +//! - [`metadata_apply`]: group 0 apply with epochs in log order and the +//! durable applied floor. //! - [`snapshot_dispatch`]: dispatch `InstallSnapshot` RPCs to lagging peers. +//! - [`leader_hints`]: write each hosted group's Raft-observed leader into +//! the routing table's leader hint. +//! - [`metadata_lane`]: apply group 0's committed entries in log order, +//! off the tick. +//! - [`leader_probe`]: find the leader of a data group this node does not +//! host when its hint names none. +//! - [`peer_batch`]: send one peer's per-group messages in order, past a +//! refusal and up to a link failure. mod apply_committed; mod core; mod dispatch_outbound; mod epoch_bump; +mod leader_hints; +mod leader_probe; +mod metadata_apply; +pub mod metadata_lane; +mod peer_batch; mod snapshot_dispatch; +mod vote_dispatch; diff --git a/nodedb-cluster/src/raft_loop/tick/peer_batch.rs b/nodedb-cluster/src/raft_loop/tick/peer_batch.rs new file mode 100644 index 000000000..01c667681 --- /dev/null +++ b/nodedb-cluster/src/raft_loop/tick/peer_batch.rs @@ -0,0 +1,236 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Send one peer's batch of per-group Raft messages concurrently. +//! +//! Sends start in group order and run side by side. One group whose answer +//! waits on the peer's disk never delays another group's heartbeat. +//! +//! A link failure stops the sends not yet started, since they fail the same +//! way. Sends already in flight can still complete. An answer settles before +//! the next send starts when it is ready by then, so a failure that the +//! transport reports at once, such as an open circuit, stops the rest. +//! +//! A typed refusal answers one group only, for example a group the peer does +//! not host yet. The other groups go on. A refusal is not a Raft response, so +//! it never counts as contact with the peer. + +use std::future::Future; + +use futures::FutureExt; +use futures::stream::{FuturesUnordered, StreamExt}; +use tokio::sync::watch; +use tracing::{debug, warn}; + +use crate::error::Result; + +/// Send each `(group_id, message)` to `peer` through `send`, concurrently. +/// +/// `on_response` runs for each Raft response, one at a time, in arrival +/// order. `kind` names the message in logs. Returns on shutdown, or once +/// every started send has finished. +pub(super) async fn drive_peer_batch( + peer: u64, + kind: &'static str, + messages: Vec<(u64, M)>, + shutdown_rx: &mut watch::Receiver, + mut send: impl FnMut(M) -> Fut, + mut on_response: impl FnMut(u64, R), +) where + Fut: Future>, +{ + let mut in_flight = FuturesUnordered::new(); + let mut link_failed = false; + for (group_id, message) in messages { + // Answers that are ready settle before the next send starts. + while let Some(Some((answered, result))) = in_flight.next().now_or_never() { + link_failed |= settle(peer, kind, answered, result, &mut on_response); + } + if link_failed { + break; + } + in_flight.push(send(message).map(move |result| (group_id, result))); + } + loop { + tokio::select! { + biased; + _ = shutdown_rx.changed() => return, + next = in_flight.next() => match next { + Some((group_id, result)) => { + settle(peer, kind, group_id, result, &mut on_response); + } + None => return, + }, + } + } +} + +/// Handle the outcome of one group's send. Returns whether the link to the +/// peer failed. +fn settle( + peer: u64, + kind: &'static str, + group_id: u64, + result: Result, + on_response: &mut impl FnMut(u64, R), +) -> bool { + match result { + Ok(response) => { + on_response(group_id, response); + false + } + Err(e) if e.is_link_failure() => { + warn!(group_id, peer, kind, error = %e, "raft RPC failed; no further groups go to the peer this round"); + true + } + Err(e) => { + debug!(group_id, peer, kind, error = %e, "raft RPC refused"); + false + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Mutex; + + use super::*; + use crate::error::ClusterError; + + /// Drive a batch whose sends answer with `answer(message)`. Returns the + /// groups that got a Raft response. + async fn run(messages: Vec<(u64, u64)>, answer: impl Fn(u64) -> Result) -> Vec { + let (_tx, mut rx) = watch::channel(false); + let answered = Mutex::new(Vec::new()); + drive_peer_batch( + 2, + "append_entries", + messages, + &mut rx, + |message| std::future::ready(answer(message)), + |group_id, _response| answered.lock().unwrap().push(group_id), + ) + .await; + answered.into_inner().unwrap() + } + + #[tokio::test] + async fn a_refusal_does_not_skip_the_next_groups_heartbeat() { + let answered = run(vec![(4, 4), (0, 0)], |group| { + if group == 4 { + Err(ClusterError::GroupNotFound { group_id: 4 }) + } else { + Ok(group) + } + }) + .await; + assert_eq!(answered, vec![0], "group 0's heartbeat still goes out"); + } + + #[tokio::test] + async fn a_link_failure_ends_the_batch() { + let answered = run(vec![(1, 1), (4, 4), (0, 0)], |group| { + if group == 4 { + Err(ClusterError::Transport { + detail: "connection lost".into(), + }) + } else { + Ok(group) + } + }) + .await; + assert_eq!(answered, vec![1]); + } + + type Pending = std::pin::Pin> + Send>>; + + /// Group 1's answer waits until group 2 has answered. Sent one after the + /// other, the batch never finishes. + #[tokio::test] + async fn a_slow_group_does_not_delay_another_groups_send() { + let (_tx, mut rx) = watch::channel(false); + let gate = std::sync::Arc::new(tokio::sync::Notify::new()); + let answered = Mutex::new(Vec::new()); + let send = |group: u64| -> Pending { + let gate = std::sync::Arc::clone(&gate); + Box::pin(async move { + if group == 1 { + gate.notified().await; + } + Ok(group) + }) + }; + let batch = drive_peer_batch( + 2, + "append_entries", + vec![(1, 1), (2, 2)], + &mut rx, + send, + |group_id, _response| { + answered.lock().unwrap().push(group_id); + if group_id == 2 { + gate.notify_one(); + } + }, + ); + tokio::time::timeout(std::time::Duration::from_secs(5), batch) + .await + .expect("group 2 answers while group 1 waits"); + assert_eq!(answered.into_inner().unwrap(), vec![2, 1]); + } + + /// A link failure stops the sends not yet started. A send already in + /// flight still completes. + #[tokio::test] + async fn a_link_failure_lets_the_sends_in_flight_finish() { + let (_tx, mut rx) = watch::channel(false); + let gate = std::sync::Arc::new(tokio::sync::Notify::new()); + let sent = Mutex::new(Vec::new()); + let answered = Mutex::new(Vec::new()); + let send = |group: u64| -> Pending { + sent.lock().unwrap().push(group); + let gate = std::sync::Arc::clone(&gate); + Box::pin(async move { + match group { + 1 => { + gate.notified().await; + Ok(group) + } + 4 => { + gate.notify_one(); + Err(ClusterError::Transport { + detail: "connection lost".into(), + }) + } + _ => Ok(group), + } + }) + }; + drive_peer_batch( + 2, + "append_entries", + vec![(1, 1), (4, 4), (0, 0)], + &mut rx, + send, + |group_id, _response| answered.lock().unwrap().push(group_id), + ) + .await; + assert_eq!( + sent.into_inner().unwrap(), + vec![1, 4], + "group 0 is never sent" + ); + assert_eq!(answered.into_inner().unwrap(), vec![1]); + } + + #[tokio::test] + async fn an_open_circuit_ends_the_batch() { + let answered = run(vec![(4, 4), (0, 0)], |_| { + Err(ClusterError::CircuitOpen { + node_id: 2, + failures: 5, + }) + }) + .await; + assert!(answered.is_empty()); + } +} diff --git a/nodedb-cluster/src/raft_loop/tick/snapshot_dispatch.rs b/nodedb-cluster/src/raft_loop/tick/snapshot_dispatch.rs index 476c335f3..dfd6c0be1 100644 --- a/nodedb-cluster/src/raft_loop/tick/snapshot_dispatch.rs +++ b/nodedb-cluster/src/raft_loop/tick/snapshot_dispatch.rs @@ -2,18 +2,59 @@ //! Install-snapshot dispatch for peers that have fallen behind the leader's //! snapshot boundary (`group_ready.snapshots_needed`). +//! +//! A data group's build is asked for the log's snapshot boundary. The host +//! captures a cut at or above it and names the cut, and the snapshot ships +//! labelled with the cut and the term of its entry. Metadata +//! group 0 and the Calvin sequencer group ship their state machine at the +//! applied index instead: the host captures it on the tick thread, between +//! apply batches. Group 0's capture is serialized off the tick. + +use std::sync::{Arc, Mutex}; use tracing::{debug, warn}; +use crate::calvin::SEQUENCER_GROUP_ID; use crate::forward::PlanExecutor; +use crate::metadata_group::METADATA_GROUP_ID; +use crate::multi_raft::MultiRaft; +use crate::raft_loop::MetadataSnapshotCapture; +use crate::transport::NexarTransport; use super::super::loop_core::{CommitApplier, RaftLoop}; +/// A state machine capture taken on the tick thread. +enum AppliedCapture { + /// Group 0's capture, serialized off the tick. + Metadata(Box), + /// A payload encoded at capture. + Encoded(Vec), +} + +/// Everything one `InstallSnapshot` transfer to one peer needs. +struct SnapshotSend { + transport: Arc, + multi_raft: Arc>, + peer: u64, + group_id: u64, + term: u64, + leader_id: u64, + last_included_index: u64, + last_included_term: u64, + chunk_bytes: u64, +} + impl RaftLoop { /// Dispatch `InstallSnapshot` RPCs for every peer this group's `Ready` /// output flagged as needing one. Called only when /// `!group_ready.snapshots_needed.is_empty()`. pub(super) fn dispatch_group_snapshots(&self, group_id: u64, snapshots_needed: Vec) { + if (group_id == METADATA_GROUP_ID || group_id == SEQUENCER_GROUP_ID) + && self.snapshot_builder.is_some() + { + self.dispatch_applied_snapshots(group_id, snapshots_needed); + return; + } let (snapshot_meta, in_flight_snapshots) = { let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); ( @@ -21,94 +62,284 @@ impl RaftLoop { mr.in_flight_snapshots(), ) }; + let Some((term, snap_index, snap_term)) = snapshot_meta else { + return; + }; + // A lagging peer is flagged on every heartbeat until its transfer + // lands. One transfer per group runs at a time; the next heartbeat + // after it ends flags any peer still behind. + if in_flight_snapshots.is_active(group_id) { + return; + } + for peer in snapshots_needed { + let mut send = self.snapshot_send(peer, group_id, term, snap_index, snap_term); + let mut shutdown_rx = self.shutdown_watch.subscribe(); + let snapshot_builder = self.snapshot_builder.clone(); + // Taken here, before the spawn, so the next tick sees the + // transfer. Held for the whole task (build + send), including + // the error and shutdown paths. Every compaction path + // (`MultiRaft::maybe_compact_group`) defers while it is held, so + // the log keeps every entry above the boundary, the cut's term + // included, until the transfer ends. + let inflight_guard = in_flight_snapshots.begin(group_id); + tokio::spawn(async move { + let _inflight_guard = inflight_guard; + if *shutdown_rx.borrow() { + return; + } + // A `None` builder (cluster-only tests) sends the stub + // (empty) chunk at the log boundary. A failed build sends + // nothing: an empty chunk would move the peer's boundary + // without its state. The next tick flags the peer again. + let snapshot = match &snapshot_builder { + Some(b) => match b + .build_group_snapshot(group_id, snap_index, snap_term) + .await + { + Ok(snapshot) => snapshot, + Err(e) => { + warn!(group_id, peer, error = %e, "snapshot build failed; nothing sent"); + return; + } + }, + None => crate::raft_loop::BuiltGroupSnapshot { + bytes: Vec::new(), + cut_index: snap_index, + }, + }; + // The snapshot is labelled with its cut: the state it holds + // ends there, so the peer resumes the log right after it. + let Some(cut_term) = cut_term(&send.multi_raft, group_id, snap_index, &snapshot) + else { + return; + }; + send.last_included_index = snapshot.cut_index; + send.last_included_term = cut_term; + send_snapshot(send, &snapshot.bytes, &mut shutdown_rx).await; + }); + } + } - if let Some((term, snap_index, snap_term)) = snapshot_meta { - for peer in snapshots_needed { - let transport = self.transport.clone(); - let mr = self.multi_raft.clone(); - let mut shutdown_rx = self.shutdown_watch.subscribe(); - let node_id = self.node_id; - let chunk_bytes = self.snapshot_chunk_bytes; - let snapshot_builder = self.snapshot_builder.clone(); - let inflight = in_flight_snapshots.clone(); - tokio::spawn(async move { - if *shutdown_rx.borrow() { - return; + /// Capture `group_id`'s state machine at its applied index and ship it to + /// every peer in `peers`. Metadata group 0 and the Calvin sequencer group + /// hold their state in a state machine the tick thread applies, so the + /// capture holds exactly the entries through the applied index. + fn dispatch_applied_snapshots(&self, group_id: u64, peers: Vec) { + let Some(builder) = self.snapshot_builder.clone() else { + return; + }; + let (term, applied, applied_term, inflight) = { + let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let Ok((term, _, _)) = mr.snapshot_metadata(group_id) else { + return; + }; + let applied = mr.last_applied(group_id).unwrap_or(0); + ( + term, + applied, + mr.log_term_at(group_id, applied), + mr.in_flight_snapshots(), + ) + }; + // One capture at a time: a peer is flagged on every heartbeat until + // its transfer lands. + if inflight.is_active(group_id) { + return; + } + let Some(applied_term) = applied_term else { + warn!( + group_id, + applied, + "state machine snapshot: the log no longer holds the applied entry's term; \ + nothing sent" + ); + return; + }; + // Taken before the capture so no compaction passes the captured + // index while the image is serialized and sent. + let inflight_guard = inflight.begin(group_id); + let capture = if group_id == SEQUENCER_GROUP_ID { + builder + .capture_sequencer(applied) + .map(AppliedCapture::Encoded) + } else { + builder + .capture_metadata(applied, applied_term) + .map(AppliedCapture::Metadata) + }; + let capture = match capture { + Ok(capture) => capture, + Err(e) => { + warn!(group_id, applied, error = %e, "state machine snapshot capture failed; nothing sent"); + return; + } + }; + let sends: Vec = peers + .into_iter() + .map(|peer| self.snapshot_send(peer, group_id, term, applied, applied_term)) + .collect(); + let mut shutdown_rx = self.shutdown_watch.subscribe(); + tokio::spawn(async move { + let _inflight_guard = inflight_guard; + let bytes = match capture { + AppliedCapture::Encoded(bytes) => bytes, + AppliedCapture::Metadata(capture) => { + match tokio::task::spawn_blocking(move || capture.serialize()).await { + Ok(Ok(bytes)) => bytes, + Ok(Err(e)) => { + warn!(applied, error = %e, "group 0 snapshot serialize failed; nothing sent"); + return; + } + Err(e) => { + warn!(applied, error = %e, "group 0 snapshot serialize task failed"); + return; + } } - // Mark this group as having a snapshot transfer in - // flight for the whole task (build + send). Held - // until the task ends — including error/shutdown - // paths — so compaction cannot advance the snapshot - // boundary mid-transfer. - let _inflight_guard = inflight.begin(group_id); - // Build the real per-group snapshot payload on the - // leader before framing the chunked RPC. A `None` - // builder (cluster-only tests) or a build failure - // falls back to the stub (empty) chunk, which the - // sender already handles. - let snapshot_bytes: Vec = match &snapshot_builder { - Some(b) => b - .build_group_snapshot(group_id, snap_index, snap_term) - .await - .unwrap_or_else(|e| { - warn!( - group_id, peer, error = %e, - "snapshot build failed; sending stub" - ); - Vec::new() - }), - None => Vec::new(), - }; - tokio::select! { - biased; - _ = shutdown_rx.changed() => {} - result = crate::install_snapshot::sender::send_chunked( - &transport, - crate::install_snapshot::sender::SendChunkedParams { - peer, - group_id, - term, - leader_id: node_id, - last_included_index: snap_index, - last_included_term: snap_term, - snapshot_bytes: &snapshot_bytes, - chunk_bytes, + } + }; + for send in sends { + if *shutdown_rx.borrow() { + return; + } + send_snapshot(send, &bytes, &mut shutdown_rx).await; + } + }); + } + + fn snapshot_send( + &self, + peer: u64, + group_id: u64, + term: u64, + last_included_index: u64, + last_included_term: u64, + ) -> SnapshotSend { + SnapshotSend { + transport: self.transport.clone(), + multi_raft: self.multi_raft.clone(), + peer, + group_id, + term, + leader_id: self.node_id, + last_included_index, + last_included_term, + chunk_bytes: self.snapshot_chunk_bytes, + } + } +} + +/// The term of the entry at `snapshot`'s cut, from the leader's log. `None`, +/// with a warning, when the cut is below the log boundary `snap_index` or the +/// log no longer holds its term: nothing is sent, and the next tick flags the +/// peer again. +fn cut_term( + multi_raft: &Mutex, + group_id: u64, + snap_index: u64, + snapshot: &crate::raft_loop::BuiltGroupSnapshot, +) -> Option { + let cut = snapshot.cut_index; + if cut < snap_index { + warn!( + group_id, + cut, snap_index, "snapshot build: the cut is below the log boundary; nothing sent" + ); + return None; + } + let term = multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .log_term_at(group_id, cut); + if term.is_none() { + warn!( + group_id, + cut, "snapshot build: the log no longer holds the cut's term; nothing sent" + ); + } + term +} + +/// Send `snapshot_bytes` to one peer in chunks, stepping down on a higher +/// term in the reply. +async fn send_snapshot( + send: SnapshotSend, + snapshot_bytes: &[u8], + shutdown_rx: &mut tokio::sync::watch::Receiver, +) { + let SnapshotSend { + transport, + multi_raft, + peer, + group_id, + term, + leader_id, + last_included_index, + last_included_term, + chunk_bytes, + } = send; + // The membership holds every conf change applied through the snapshot + // index. A conf change applied since sits above that index, and the + // peer applies it from the log after the install. + let (voters, learners) = multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .group_membership(group_id) + .map(|m| (m.voters, m.learners)) + .unwrap_or_default(); + tokio::select! { + biased; + _ = shutdown_rx.changed() => {} + result = crate::install_snapshot::sender::send_chunked( + &transport, + crate::install_snapshot::sender::SendChunkedParams { + peer, + group_id, + term, + leader_id, + last_included_index, + last_included_term, + snapshot_bytes, + chunk_bytes, + voters: &voters, + learners: &learners, + }, + ) => { + match result { + Ok(resp_term) if resp_term <= term => { + // The peer holds the state through the snapshot index, so + // replication to it resumes after that index. + multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .record_snapshot_installed(group_id, peer, last_included_index); + debug!(group_id, peer, "install_snapshot sent"); + } + Ok(resp_term) => { + { + let mut mr = multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + // Higher term: the tick loop handles the step-down. + if let Err(e) = mr.handle_append_entries_response( + group_id, + peer, + &nodedb_raft::AppendEntriesResponse { + term: resp_term, + success: false, + last_log_index: 0, + round: nodedb_raft::node::leader_lease::UNTRACKED_ROUND, + needs_snapshot: false, }, - ) => { - match result { - Ok(resp_term) => { - if resp_term > term { - let mut mr = - mr.lock().unwrap_or_else(|p| p.into_inner()); - // Higher term — let the tick loop handle step-down. - let _ = mr.handle_append_entries_response( - group_id, - peer, - &nodedb_raft::AppendEntriesResponse { - term: resp_term, - success: false, - last_log_index: 0, - }, - ); - // Persist the term bump durably. - if let Err(e) = - mr.persist_group_hard_state(group_id) - { - tracing::error!(group_id, peer, error = %e, "persist hard state after snapshot step-down"); - } - } - debug!(group_id, peer, "install_snapshot sent"); - } - Err(e) => { - warn!( - group_id, peer, error = %e, - "install_snapshot RPC failed" - ); - } - } + ) { + tracing::error!(group_id, peer, resp_term, error = %e, "apply higher term from install_snapshot response"); + } + if let Err(e) = mr.persist_group_hard_state(group_id) { + tracing::error!(group_id, peer, error = %e, "persist hard state after snapshot step-down"); } } - }); + debug!(group_id, peer, resp_term, "install_snapshot answered with a higher term"); + } + Err(e) => { + warn!(group_id, peer, error = %e, "install_snapshot RPC failed"); + } } } } diff --git a/nodedb-cluster/src/raft_loop/tick/vote_dispatch.rs b/nodedb-cluster/src/raft_loop/tick/vote_dispatch.rs new file mode 100644 index 000000000..1d757fd49 --- /dev/null +++ b/nodedb-cluster/src/raft_loop/tick/vote_dispatch.rs @@ -0,0 +1,80 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RequestVote dispatch for one group, gated on the group's disk. +//! +//! A candidate's term bump and self-vote are staged on the group's disk by +//! the tick. Its vote requests leave only once those writes are durable. The +//! wait runs on the group's own task, without the `MultiRaft` lock, so no +//! other group's heartbeats or elections wait on this group's disk. + +use nodedb_raft::transport::RaftTransport; +use tracing::{debug, error, warn}; + +use crate::forward::PlanExecutor; +use crate::group_disk::DurabilityTicket; + +use super::super::loop_core::{CommitApplier, RaftLoop}; + +impl RaftLoop { + /// Send `group_id`'s vote requests, one task per peer, once `ticket` is + /// durable. A ticket that fails, because the group was unmounted, sends + /// nothing. + pub(super) fn dispatch_group_votes( + &self, + group_id: u64, + ticket: Option, + votes: Vec<(u64, nodedb_raft::RequestVoteRequest)>, + ) { + let transport = self.transport.clone(); + let multi_raft = self.multi_raft.clone(); + let mut shutdown_rx = self.shutdown_watch.subscribe(); + // One shutdown receiver per request, taken here: the watch sender + // stays with the loop. + let votes: Vec<_> = votes + .into_iter() + .map(|vote| (vote, self.shutdown_watch.subscribe())) + .collect(); + tokio::spawn(async move { + if let Some(ticket) = ticket { + tokio::select! { + biased; + _ = shutdown_rx.changed() => return, + durable = ticket.durable() => { + if let Err(e) = durable { + warn!(group_id, error = %e, "vote requests not sent: the term is not durable"); + return; + } + } + } + } + for ((peer, req), mut shutdown_rx) in votes { + let transport = transport.clone(); + let multi_raft = multi_raft.clone(); + tokio::spawn(async move { + if *shutdown_rx.borrow() { + return; + } + tokio::select! { + biased; + _ = shutdown_rx.changed() => {} + rpc = transport.request_vote(peer, req) => match rpc { + Ok(resp) => { + let mut mr = multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + if let Err(e) = mr.handle_request_vote_response(group_id, peer, &resp) { + debug!(group_id, peer, error = %e, "handle vote response"); + } + // A higher-term response steps this candidate + // down to follower. The term bump is staged, + // and no message depends on it. + if let Err(e) = mr.persist_group_hard_state(group_id) { + error!(group_id, peer, error = %e, "stage hard state after vote response"); + } + } + Err(e) => warn!(group_id, peer, error = %e, "request_vote RPC failed"), + }, + } + }); + } + }); + } +} diff --git a/nodedb-cluster/src/raft_loop/tick_state.rs b/nodedb-cluster/src/raft_loop/tick_state.rs new file mode 100644 index 000000000..b831a08ae --- /dev/null +++ b/nodedb-cluster/src/raft_loop/tick_state.rs @@ -0,0 +1,246 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! State the tick phases carry from one pass to the next, and the handle of +//! the metadata apply lane. + +use std::collections::{HashMap, HashSet}; +use std::sync::Mutex; +use std::sync::atomic::{AtomicU64, Ordering}; + +use nodedb_raft::LogEntry; +use tokio::sync::mpsc; + +use super::leader_balance::{PeerHealth, PeerSample}; + +/// Samples, records and in-flight markers of the leader-balance and +/// leader-probe phases, and the metadata lane's sender. Every mutex is a +/// leaf lock: no other lock is taken while one is held. +#[derive(Debug, Default)] +pub struct TickState { + /// `(group_id, peer) -> health record` of each preferred leader the + /// balance sampled within the last few passes. + peer_health: Mutex>, + /// Groups with a leader probe in flight. + probes_in_flight: Mutex>, + /// Groups whose mount is opening their disk. + mounts_in_flight: Mutex>, + /// Sender of the metadata apply lane while the loop runs. + metadata_tx: Mutex>>>, + /// Highest metadata index handed to the lane. + metadata_sent_through: AtomicU64, + /// `group_id -> (through, seq)` for a data group whose conf changes up + /// to index `through` are applied in memory and wait for the routing + /// save of request `seq`. + conf_saves: Mutex>, +} + +impl TickState { + pub fn new() -> Self { + Self::default() + } + + /// The `(through, seq)` conf-change save `group_id` waits for, if any. + pub(super) fn conf_save(&self, group_id: u64) -> Option<(u64, u64)> { + self.conf_saves + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&group_id) + .copied() + } + + /// Record that `group_id`'s conf changes up to `through` wait for the + /// routing save of request `seq`. + pub(super) fn set_conf_save(&self, group_id: u64, through: u64, seq: u64) { + self.conf_saves + .lock() + .unwrap_or_else(|p| p.into_inner()) + .insert(group_id, (through, seq)); + } + + /// Forget `group_id`'s conf-change save once its batch went on. + pub(super) fn clear_conf_save(&self, group_id: u64) { + self.conf_saves + .lock() + .unwrap_or_else(|p| p.into_inner()) + .remove(&group_id); + } + + /// Record the balance pass `pass` sample of `peer` in `group_id`, and + /// return the peer's updated health record. + pub(super) fn observe_peer_health( + &self, + group_id: u64, + peer: u64, + sample: PeerSample, + pass: u64, + ) -> PeerHealth { + let mut records = self.peer_health.lock().unwrap_or_else(|p| p.into_inner()); + let previous = records.get(&(group_id, peer)).copied(); + let next = PeerHealth::next(previous, sample, pass); + records.insert((group_id, peer), next); + next + } + + /// Forget every record too old to count at balance pass `pass`. The map + /// stays bounded by the groups this node led recently and their voters. + pub(super) fn forget_stale_peer_health(&self, pass: u64) { + self.peer_health + .lock() + .unwrap_or_else(|p| p.into_inner()) + .retain(|_, record| record.is_current(pass)); + } + + /// Mark a probe of `group_id` in flight. Returns `false` when one is. + pub(super) fn begin_probe(&self, group_id: u64) -> bool { + self.probes_in_flight + .lock() + .unwrap_or_else(|p| p.into_inner()) + .insert(group_id) + } + + /// Clear the in-flight marker of `group_id`'s probe. + pub(super) fn end_probe(&self, group_id: u64) { + self.probes_in_flight + .lock() + .unwrap_or_else(|p| p.into_inner()) + .remove(&group_id); + } + + /// Mark a mount of `group_id` in flight. Returns `false` when one is. + pub(super) fn begin_mount(&self, group_id: u64) -> bool { + self.mounts_in_flight + .lock() + .unwrap_or_else(|p| p.into_inner()) + .insert(group_id) + } + + /// Clear the in-flight marker of `group_id`'s mount. + pub(super) fn end_mount(&self, group_id: u64) { + self.mounts_in_flight + .lock() + .unwrap_or_else(|p| p.into_inner()) + .remove(&group_id); + } + + /// Install the sender of the metadata lane. + pub(super) fn open_metadata_lane(&self, tx: mpsc::Sender>) { + *self.metadata_tx.lock().unwrap_or_else(|p| p.into_inner()) = Some(tx); + } + + /// Drop the sender of the metadata lane. The lane ends once it applied + /// or gave up every batch it holds. + pub(super) fn close_metadata_lane(&self) { + self.metadata_tx + .lock() + .unwrap_or_else(|p| p.into_inner()) + .take(); + } + + /// Hand the entries of `batch` above the highest index handed so far to + /// the metadata lane. Returns the entries the lane did not take: the + /// lane is closed or full. The caller queues them again. + pub(super) fn send_to_metadata_lane(&self, batch: &[LogEntry]) -> Vec { + let sent = self.metadata_sent_through.load(Ordering::Acquire); + let fresh: Vec = batch + .iter() + .filter(|entry| entry.index > sent) + .cloned() + .collect(); + let Some(through) = fresh.last().map(|entry| entry.index) else { + return Vec::new(); + }; + let guard = self.metadata_tx.lock().unwrap_or_else(|p| p.into_inner()); + let Some(tx) = guard.as_ref() else { + return fresh; + }; + match tx.try_send(fresh) { + Ok(()) => { + self.metadata_sent_through.store(through, Ordering::Release); + Vec::new() + } + Err(mpsc::error::TrySendError::Full(fresh)) + | Err(mpsc::error::TrySendError::Closed(fresh)) => fresh, + } + } +} + +#[cfg(test)] +mod tests { + use super::super::leader_balance::{PEER_HEALTH_GAP_PASSES, PREFERRED_HEALTHY_PASSES}; + use super::*; + + fn entry(index: u64) -> LogEntry { + LogEntry { + term: 1, + index, + data: Vec::new(), + } + } + + fn healthy(term: u64, acks: u64) -> PeerSample { + PeerSample { + term, + acks, + caught_up: true, + } + } + + #[test] + fn a_peer_record_carries_from_pass_to_pass() { + let state = TickState::new(); + let first = state.observe_peer_health(1, 2, healthy(1, 5), 0); + assert!(!first.earned_leadership(), "no earlier pass"); + let mut last = first; + for pass in 1..=u64::from(PREFERRED_HEALTHY_PASSES) { + last = state.observe_peer_health(1, 2, healthy(1, 5 + pass), pass); + } + assert!(last.earned_leadership()); + // Another group's record is separate. + assert!( + !state + .observe_peer_health(3, 2, healthy(1, 100), 5) + .earned_leadership() + ); + } + + #[test] + fn stale_peer_records_are_forgotten() { + let state = TickState::new(); + state.observe_peer_health(1, 2, healthy(1, 5), 0); + state.forget_stale_peer_health(PEER_HEALTH_GAP_PASSES); + assert_eq!(state.peer_health.lock().unwrap().len(), 1); + state.forget_stale_peer_health(PEER_HEALTH_GAP_PASSES + 1); + assert!(state.peer_health.lock().unwrap().is_empty()); + } + + #[test] + fn one_probe_per_group_at_a_time() { + let state = TickState::new(); + assert!(state.begin_probe(4)); + assert!(!state.begin_probe(4)); + state.end_probe(4); + assert!(state.begin_probe(4)); + } + + #[test] + fn the_lane_takes_each_index_once_and_returns_what_it_cannot_take() { + let state = TickState::new(); + assert_eq!(state.send_to_metadata_lane(&[entry(1)]).len(), 1, "closed"); + let (tx, mut rx) = mpsc::channel(1); + state.open_metadata_lane(tx); + assert!( + state + .send_to_metadata_lane(&[entry(1), entry(2)]) + .is_empty() + ); + // A repeat of handed entries sends nothing; the lane is full. + assert!(state.send_to_metadata_lane(&[entry(2)]).is_empty()); + let back = state.send_to_metadata_lane(&[entry(2), entry(3)]); + assert_eq!(back.iter().map(|e| e.index).collect::>(), vec![3]); + let got = rx.try_recv().expect("the first batch"); + assert_eq!(got.iter().map(|e| e.index).collect::>(), vec![1, 2]); + assert!(state.send_to_metadata_lane(&[entry(3)]).is_empty()); + state.close_metadata_lane(); + assert_eq!(state.send_to_metadata_lane(&[entry(4)]).len(), 1); + } +} diff --git a/nodedb-cluster/src/raft_storage.rs b/nodedb-cluster/src/raft_storage.rs index 615a10600..6bfbd12f8 100644 --- a/nodedb-cluster/src/raft_storage.rs +++ b/nodedb-cluster/src/raft_storage.rs @@ -711,9 +711,17 @@ mod tests { ], leader_commit: 0, group_id: 7, + round: 1, + replicated_floor: 0, }; assert!(node.handle_append_entries(&ae).success); node.persist_hard_state_if_dirty().unwrap(); + // Age the leader contact and end the boot fence, so the vote is + // not refused as disruptive. + node.leader_contact_at_override( + std::time::Instant::now() - std::time::Duration::from_secs(1), + ); + node.expire_boot_vote_fence(); // Grant a vote to candidate 2 in TERM, then persist it durably. let rv = RequestVoteRequest { @@ -722,6 +730,7 @@ mod tests { last_log_index: 2, last_log_term: TERM, group_id: 7, + transfer: false, }; assert!( node.handle_request_vote(&rv).vote_granted, @@ -735,6 +744,10 @@ mod tests { let mut node = RaftNode::new(config(), storage); node.restore().unwrap(); + // The restarted node's boot fence would refuse any vote. End it, so + // only the restored `voted_for` can refuse the second candidate. + node.expire_boot_vote_fence(); + // Log-reload half: the durably-appended entries survived the restart. assert_eq!(node.last_log_index(), 2, "log entries must survive restart"); // Vote-persistence half: the term survived. @@ -750,6 +763,7 @@ mod tests { last_log_index: 2, last_log_term: TERM, group_id: 7, + transfer: false, }; assert!( !node.handle_request_vote(&rv2).vote_granted, diff --git a/nodedb-cluster/src/reachability/driver.rs b/nodedb-cluster/src/reachability/driver.rs index b14ec14a9..86e487417 100644 --- a/nodedb-cluster/src/reachability/driver.rs +++ b/nodedb-cluster/src/reachability/driver.rs @@ -7,9 +7,9 @@ //! injected [`ReachabilityProber`]. Probes run in parallel via //! `tokio::spawn` so a slow peer never blocks the next one. Probe //! results are intentionally ignored: the production `TransportProber` -//! routes through `NexarTransport::send_rpc`, which already walks the -//! circuit breaker's `check → record_success|record_failure` path, so -//! the driver does not need to bookkeep anything itself. +//! routes through `NexarTransport::send_probe_rpc`, which admits the +//! probe past the open circuit and records its outcome on the breaker, +//! so the driver does not need to bookkeep anything itself. //! //! Shutdown is cooperative via `tokio::sync::watch`. On `true` the //! run loop breaks at the next tick or immediately if it is waiting. @@ -121,7 +121,7 @@ impl ReachabilityDriver { #[cfg(test)] mod tests { use super::*; - use crate::circuit_breaker::CircuitBreakerConfig; + use crate::circuit_breaker::{Admission, CircuitBreakerConfig}; use async_trait::async_trait; use std::sync::Mutex; @@ -161,9 +161,9 @@ mod tests { #[tokio::test] async fn sweep_probes_every_open_peer() { let breaker = open_breaker(); - breaker.record_failure(1); - breaker.record_failure(2); - breaker.record_failure(3); + breaker.record_failure(1, Admission::Normal); + breaker.record_failure(2, Admission::Normal); + breaker.record_failure(3, Admission::Normal); let prober = RecordingProber::new(); let driver = Arc::new(ReachabilityDriver::new( @@ -186,8 +186,8 @@ mod tests { #[tokio::test] async fn sweep_skips_closed_peers() { let breaker = open_breaker(); - breaker.record_success(1); // Registers 1 as Closed. - breaker.record_failure(2); // Opens 2. + breaker.record_success(1, Admission::Normal); // Registers 1 as Closed. + breaker.record_failure(2, Admission::Normal); // Opens 2. let prober = RecordingProber::new(); let driver = Arc::new(ReachabilityDriver::new( Arc::clone(&breaker), @@ -204,7 +204,7 @@ mod tests { #[tokio::test(start_paused = true)] async fn run_loop_fires_sweeps_on_interval_and_shuts_down() { let breaker = open_breaker(); - breaker.record_failure(7); + breaker.record_failure(7, Admission::Normal); let prober = RecordingProber::new(); let driver = Arc::new(ReachabilityDriver::new( Arc::clone(&breaker), diff --git a/nodedb-cluster/src/reachability/mod.rs b/nodedb-cluster/src/reachability/mod.rs index 96716f78d..11b30d673 100644 --- a/nodedb-cluster/src/reachability/mod.rs +++ b/nodedb-cluster/src/reachability/mod.rs @@ -8,8 +8,8 @@ //! the peer has recovered. This module closes that blind spot: //! //! - [`ReachabilityDriver`] periodically walks the breaker's open set -//! and sends a lightweight probe RPC to each peer via the existing -//! `send_rpc` path, which drives the normal HalfOpen → Closed / +//! and sends a lightweight probe RPC to each peer via the +//! `send_probe_rpc` path, which drives the HalfOpen → Closed / //! HalfOpen → Open transitions. //! - [`ReachabilityProber`] is the injection seam: production wraps //! [`crate::transport::NexarTransport`], tests use a mock. diff --git a/nodedb-cluster/src/reachability/prober.rs b/nodedb-cluster/src/reachability/prober.rs index 8bacd939f..bfc13e913 100644 --- a/nodedb-cluster/src/reachability/prober.rs +++ b/nodedb-cluster/src/reachability/prober.rs @@ -5,10 +5,9 @@ //! Implementations: //! //! - [`TransportProber`] wraps an `Arc` and sends a -//! `RaftRpc::Ping` to the peer. `send_rpc` already handles the -//! circuit-breaker check, the QUIC dial, retries, and -//! `record_success` / `record_failure` — the prober is a one-line -//! adapter. +//! `RaftRpc::Ping` to the peer as a recovery probe. +//! `send_probe_rpc` goes past the open circuit, dials, and records the +//! outcome on the breaker. The prober is a one-line adapter. //! - [`NoopProber`] always succeeds. Useful for tests that only want //! to verify the loop's tick cadence and shutdown. //! @@ -31,9 +30,9 @@ pub trait ReachabilityProber: Send + Sync { async fn probe(&self, peer: u64) -> Result<()>; } -/// Production prober: sends a `Ping` via the live transport. The -/// transport's internal circuit breaker records success/failure -/// automatically — the driver does not need to bookkeep anything. +/// Production prober: sends a `Ping` via the live transport as a recovery +/// probe. An open circuit lets it through, and its outcome closes or +/// reopens the circuit. The driver does not need to bookkeep anything. pub struct TransportProber { transport: Arc, self_node_id: u64, @@ -55,7 +54,7 @@ impl ReachabilityProber for TransportProber { sender_id: self.self_node_id, topology_version: 0, }); - self.transport.send_rpc(peer, rpc).await.map(|_| ()) + self.transport.send_probe_rpc(peer, rpc).await.map(|_| ()) } } diff --git a/nodedb-cluster/src/read_index_wait.rs b/nodedb-cluster/src/read_index_wait.rs index ecdcc18f4..8f75ffc44 100644 --- a/nodedb-cluster/src/read_index_wait.rs +++ b/nodedb-cluster/src/read_index_wait.rs @@ -2,6 +2,9 @@ //! Waiting for a read index to be confirmed by a quorum. //! +//! A leader with a valid lease skips the wait: no successor can exist before +//! the lease ends, so its commit index is already the read index. +//! //! The coordinator lock is taken to start the probe, released, then retaken //! for each poll. It is never held across an await: the tick loop needs the //! same lock to send the very responses being waited on, so holding it would @@ -22,9 +25,10 @@ const POLL_INTERVAL: Duration = Duration::from_millis(2); /// Confirm this node still leads `group_id`, and return the index the read /// may be served at. /// -/// Refuses rather than waiting out `timeout` once leadership is known to be -/// lost — a deposed leader can never confirm, and the caller retries against -/// the new one. +/// A valid leader lease answers without a quorum round. Otherwise a probe +/// runs. It refuses rather than waiting out `timeout` once leadership is known +/// to be lost: a deposed leader can never confirm, and the caller retries +/// against the new one. pub async fn confirm_read_index( multi_raft: &Arc>, group_id: u64, @@ -32,6 +36,9 @@ pub async fn confirm_read_index( ) -> Result { let probe = { let mut mr = multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + if let Some(read_index) = mr.lease_read_index(group_id) { + return Ok(read_index); + } mr.start_read_index(group_id) .ok_or(ClusterError::ReadIndexNotLeader { group_id })? }; @@ -112,4 +119,31 @@ mod tests { "got {err}" ); } + + /// A leader holding its lease answers at its commit index with no + /// quorum round and no poll. + #[tokio::test] + async fn a_leased_leader_answers_without_a_probe() { + let dir = tempfile::tempdir().expect("tempdir"); + let mr = coordinator(&dir); + let commit = { + let mut guard = mr.lock().expect("lock"); + guard.add_group(7, vec![]).expect("add group"); + let node = guard.groups_mut().get_mut(&7).expect("group 7"); + node.election_deadline_override(Instant::now() - Duration::from_millis(1)); + guard.tick().expect("tick"); + assert!(guard.is_group_leader(7), "a single voter elects itself"); + // The no-op commits once the disk holds it. + guard.wait_all_durable_blocking(); + guard.tick().expect("tick"); + guard + .lease_read_index(7) + .expect("the sole voter holds the lease") + }; + + let read_index = confirm_read_index(&mr, 7, Duration::ZERO) + .await + .expect("the lease serves without waiting"); + assert_eq!(read_index, commit); + } } diff --git a/nodedb-cluster/src/rebalancer/leader_preference.rs b/nodedb-cluster/src/rebalancer/leader_preference.rs new file mode 100644 index 000000000..087e7d13b --- /dev/null +++ b/nodedb-cluster/src/rebalancer/leader_preference.rs @@ -0,0 +1,100 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Deterministic preferred leader per data group. +//! +//! Every node that bootstraps a cluster starts as the sole voter of every +//! group, so it leads them all. Nothing in Raft moves a leader that stays +//! healthy, and one node would keep serving every write. This module names +//! one preferred leader per data group, spread over the group placements. +//! The group's leader transfers leadership to it (see the leader-balance +//! tick phase). +//! +//! The assignment walks the data groups in ascending id. Each group takes +//! the node of its effective placement that has the fewest groups so far. +//! A tie goes to the higher rendezvous score for `(group_id, node_id)`, then +//! the lower node id. Every node that holds the same placements computes +//! the same map. + +use std::collections::BTreeMap; + +use crate::routing::RoutingTable; + +use super::placement::hrw_score; + +/// The preferred leader of every data group in `routing`: `group_id -> +/// node_id`. The metadata group and the sequencer group have none. A group +/// with an empty effective placement has none. +pub fn preferred_leaders(routing: &RoutingTable) -> BTreeMap { + let mut group_ids: Vec = routing + .group_ids() + .into_iter() + .filter(|gid| { + *gid != crate::metadata_group::METADATA_GROUP_ID + && *gid != crate::calvin::sequencer::SEQUENCER_GROUP_ID + }) + .collect(); + group_ids.sort_unstable(); + + let mut led: BTreeMap = BTreeMap::new(); + let mut out = BTreeMap::new(); + for gid in group_ids { + let mut candidates = routing.effective_placement(gid); + candidates.sort_unstable(); + candidates.dedup(); + let chosen = candidates.into_iter().min_by(|&a, &b| { + let count_a = led.get(&a).copied().unwrap_or(0); + let count_b = led.get(&b).copied().unwrap_or(0); + count_a + .cmp(&count_b) + .then(hrw_score(gid, b).cmp(&hrw_score(gid, a))) + .then(a.cmp(&b)) + }); + if let Some(node) = chosen { + *led.entry(node).or_default() += 1; + out.insert(gid, node); + } + } + out +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn leaders_spread_over_the_voters() { + let rt = RoutingTable::uniform(6, &[1, 2, 3], 3); + let preferred = preferred_leaders(&rt); + assert_eq!(preferred.len(), 6, "every data group has one"); + assert!(!preferred.contains_key(&crate::metadata_group::METADATA_GROUP_ID)); + let mut per_node: BTreeMap = BTreeMap::new(); + for node in preferred.values() { + *per_node.entry(*node).or_default() += 1; + } + assert_eq!(per_node.len(), 3, "every voter leads a group"); + assert!(per_node.values().all(|&n| n == 2), "{per_node:?}"); + } + + #[test] + fn two_groups_never_share_a_preferred_leader_on_three_voters() { + let rt = RoutingTable::uniform(2, &[1, 2, 3], 3); + let preferred = preferred_leaders(&rt); + assert_ne!(preferred.get(&1), preferred.get(&2)); + } + + #[test] + fn the_preferred_leader_is_in_the_placement() { + let mut rt = RoutingTable::uniform(2, &[1, 2, 3], 3); + rt.set_placement(1, vec![3]); + rt.set_placement(2, vec![2]); + let preferred = preferred_leaders(&rt); + assert_eq!(preferred.get(&1), Some(&3)); + assert_eq!(preferred.get(&2), Some(&2)); + } + + #[test] + fn the_map_is_deterministic() { + let rt = RoutingTable::uniform(8, &[1, 2, 3, 4], 3); + assert_eq!(preferred_leaders(&rt), preferred_leaders(&rt.clone())); + } +} diff --git a/nodedb-cluster/src/rebalancer/mod.rs b/nodedb-cluster/src/rebalancer/mod.rs index 9d3c2310b..f189f5a5f 100644 --- a/nodedb-cluster/src/rebalancer/mod.rs +++ b/nodedb-cluster/src/rebalancer/mod.rs @@ -25,6 +25,7 @@ pub mod driver; pub mod elastic; +pub mod leader_preference; pub mod metrics; pub mod placement; pub mod plan; @@ -33,5 +34,6 @@ pub use driver::{ AlwaysReadyGate, ElectionGate, MigrationDispatcher, RebalancerLoop, RebalancerLoopConfig, }; pub use elastic::RebalancerKickHook; +pub use leader_preference::preferred_leaders; pub use metrics::{LoadMetrics, LoadMetricsProvider, LoadWeights, normalized_score}; pub use plan::{RebalancerPlanConfig, compute_load_based_plan}; diff --git a/nodedb-cluster/src/rebalancer/placement.rs b/nodedb-cluster/src/rebalancer/placement.rs index 9d34ef955..dbb48bf3d 100644 --- a/nodedb-cluster/src/rebalancer/placement.rs +++ b/nodedb-cluster/src/rebalancer/placement.rs @@ -60,7 +60,7 @@ fn select_nodes_for_group(group_id: u64, active_nodes: &[u64], rf: usize) -> Vec /// /// Uses a fixed 16-byte little-endian layout so the hash value is /// identical on every node and platform. -fn hrw_score(group_id: u64, node_id: u64) -> u64 { +pub(crate) fn hrw_score(group_id: u64, node_id: u64) -> u64 { let mut key = [0u8; 16]; key[..8].copy_from_slice(&group_id.to_le_bytes()); key[8..].copy_from_slice(&node_id.to_le_bytes()); diff --git a/nodedb-cluster/src/routing.rs b/nodedb-cluster/src/routing.rs index 78b93e31f..87c97784d 100644 --- a/nodedb-cluster/src/routing.rs +++ b/nodedb-cluster/src/routing.rs @@ -23,6 +23,16 @@ pub const VSHARD_COUNT: u32 = VShardId::COUNT; /// - A shard migration completes (Phase 3 atomic cut-over) /// - A Raft group membership changes /// - A node joins or decommissions +/// +/// # Lock order +/// +/// The shared `Arc>` ranks below the `MultiRaft` mutex. +/// Take the `MultiRaft` lock first, then the routing guard. +/// Never call into `MultiRaft` while holding a routing guard. +/// That bans `group_statuses`, `raft_status_fn`, and any `multi_raft.lock()` path. +/// `MultiRaft` reads routing under its own lock, so the nested call re-reads routing. +/// A queued writer blocks that nested read, and the node deadlocks. +/// Take the Raft status first, then the routing guard. #[derive( Debug, Clone, @@ -36,6 +46,13 @@ pub struct RoutingTable { vshard_to_group: Vec, /// raft_group_id → (leader_node, [replica_nodes]). group_members: HashMap, + /// vshard_id → (raft_group_id → epoch at which the vShard moved to that + /// group). A committed `ReassignVShard` metadata entry sets the epoch to + /// its own metadata log index, so every node derives the same value and a + /// replay reproduces it. It is the only event that sets an epoch. A vShard + /// in its initial group has epoch `0`. + #[serde(default)] + vshard_epochs: HashMap>, } #[derive( @@ -50,6 +67,15 @@ pub struct RoutingTable { pub struct GroupInfo { /// Current leader node ID (0 = no leader known). pub leader: u64, + /// The Raft term `leader` is known at: from this node's own Raft, or + /// from a leader redirect that named the leader with its term. `0` when + /// the hint came from a source with no term (the metadata log, + /// placement). A termed hint at a higher term always replaces the hint. + /// A term-less hint never replaces a termed one. A clear keeps the term. + /// Only a confirmation (see [`RoutingTable::confirm_leader`]) fills a + /// cleared hint at that term. + #[serde(default)] + pub leader_term: u64, /// All voting members (including leader). pub members: Vec, /// Non-voting learner peers catching up to this group. @@ -102,6 +128,7 @@ impl RoutingTable { group_id, GroupInfo { leader, + leader_term: 0, members, learners: Vec::new(), placement: None, @@ -116,6 +143,7 @@ impl RoutingTable { 0, GroupInfo { leader: meta_leader, + leader_term: 0, members: meta_members, learners: Vec::new(), placement: None, @@ -125,6 +153,7 @@ impl RoutingTable { Self { vshard_to_group, group_members, + vshard_epochs: HashMap::new(), } } @@ -146,26 +175,137 @@ impl RoutingTable { Ok(info.leader) } + /// The leader hint of the group that owns `vshard_id`, with the term the + /// hint is known at: `(leader, leader_term)`. + pub fn leader_at_term_for_vshard(&self, vshard_id: u32) -> Result<(u64, u64)> { + let group_id = self.group_for_vshard(vshard_id)?; + let info = self + .group_members + .get(&group_id) + .ok_or(ClusterError::GroupNotFound { group_id })?; + Ok((info.leader, info.leader_term)) + } + /// Get group info. pub fn group_info(&self, group_id: u64) -> Option<&GroupInfo> { self.group_members.get(&group_id) } - /// Update the leader for a Raft group. - pub fn set_leader(&mut self, group_id: u64, leader: u64) { + /// Set the leader hint of a Raft group from a source that names a + /// leader without a term. It applies only while the hint holds no term, + /// so it never replaces a leader some node's Raft observed. Returns + /// whether the hint changed. + /// + /// The callers with no term: + /// - the metadata log's `RoutingChange` entries, which name a planned + /// leaseholder or transfer target, not an elected leader; + /// - the migration executor's cut-over when no metadata proposer is + /// wired, which names the transfer target; + /// - a migration compensation that restores a hint it replaced; + /// - routing tables built by tests. + /// + /// Every source that knows the leader's term calls + /// [`Self::observe_leader`] instead. + pub fn set_leader(&mut self, group_id: u64, leader: u64) -> bool { + match self.group_members.get_mut(&group_id) { + Some(info) if info.leader_term == 0 && info.leader != leader => { + info.leader = leader; + true + } + _ => false, + } + } + + /// Forget the leader of a Raft group, as when its leaseholder is + /// suspected dead. The term stays, so an observation of the same leader + /// at the same term does not restore it. A new election's higher term + /// does. + pub fn clear_leader(&mut self, group_id: u64) { if let Some(info) = self.group_members.get_mut(&group_id) { - info.leader = leader; + info.leader = 0; } } - /// Atomically reassign a vShard to a different Raft group. - /// Used during Phase 3 (atomic cut-over) of shard migration. - pub fn reassign_vshard(&mut self, vshard_id: u32, new_group_id: u64) { + /// Whether [`Self::observe_leader`] would change the hint. + pub fn leader_observation_is_new(&self, group_id: u64, leader: u64, term: u64) -> bool { + self.group_members + .get(&group_id) + .is_some_and(|info| leader != 0 && term > info.leader_term) + } + + /// Whether [`Self::confirm_leader`] would change the hint. + pub fn leader_confirmation_is_new(&self, group_id: u64, leader: u64, term: u64) -> bool { + self.group_members.get(&group_id).is_some_and(|info| { + leader != 0 + && (term > info.leader_term || (term == info.leader_term && info.leader == 0)) + }) + } + + /// Record `leader` of `group_id` at `term` from a source that shows the + /// leader serves that term now: + /// - this node's Raft, while it leads or the leader's contact is fresh; + /// - a node that answered as the leader, or named the leader it follows; + /// - SWIM, when a node it suspected answers again. + /// + /// It applies above the hint's term, as [`Self::observe_leader`] does. + /// It also fills a hint cleared at the same term: a clear is a suspicion, + /// and this source shows the leader still serves. Returns whether the + /// hint changed. + pub fn confirm_leader(&mut self, group_id: u64, leader: u64, term: u64) -> bool { + if !self.leader_confirmation_is_new(group_id, leader, term) { + return false; + } + match self.group_members.get_mut(&group_id) { + Some(info) => { + info.leader = leader; + info.leader_term = term; + true + } + None => false, + } + } + + /// Record `leader` of `group_id` as known at `term`: observed by this + /// node's Raft, or named by a leader redirect with the redirecting + /// node's term. Applies only above the hint's term: Raft elects at most + /// one leader per term, so a higher term is newer. Returns whether the + /// hint changed. + pub fn observe_leader(&mut self, group_id: u64, leader: u64, term: u64) -> bool { + if !self.leader_observation_is_new(group_id, leader, term) { + return false; + } + match self.group_members.get_mut(&group_id) { + Some(info) => { + info.leader = leader; + info.leader_term = term; + true + } + None => false, + } + } + + /// Atomically reassign a vShard to a different Raft group, at `epoch`, + /// the metadata log index of the reassignment. + pub fn reassign_vshard(&mut self, vshard_id: u32, new_group_id: u64, epoch: u64) { if (vshard_id as usize) < self.vshard_to_group.len() { self.vshard_to_group[vshard_id as usize] = new_group_id; + let epochs = self.vshard_epochs.entry(vshard_id).or_default(); + let current = epochs.entry(new_group_id).or_insert(epoch); + *current = (*current).max(epoch); } } + /// The epoch at which `vshard_id` moved to `group_id`: `0` while the + /// vShard stays in its initial group. An entry the group applies for the + /// vShard is positioned in this epoch, however late it applies. + pub fn vshard_epoch(&self, vshard_id: u32, group_id: u64) -> u64 { + self.vshard_epochs + .get(&vshard_id) + .and_then(|epochs| epochs.get(&group_id)) + .copied() + .unwrap_or(0) + } + /// All vShards assigned to a given group. pub fn vshards_for_group(&self, group_id: u64) -> Vec { self.vshard_to_group @@ -294,6 +434,7 @@ impl RoutingTable { Self { vshard_to_group, group_members, + vshard_epochs: HashMap::new(), } } } @@ -337,6 +478,22 @@ pub fn partition_hash(placement_hash_id: crate::catalog::PlacementHashId, key: & mod tests { use super::*; + #[test] + fn a_table_with_vshard_epochs_round_trips_through_msgpack() { + let mut rt = RoutingTable::uniform(4, &[1, 2, 3], 3); + rt.reassign_vshard(7, 3, 41); + rt.reassign_vshard(7, 2, 58); + rt.reassign_vshard(9, 4, 60); + let bytes = zerompk::to_msgpack_vec(&rt).expect("encode routing table"); + let decoded: RoutingTable = zerompk::from_msgpack(&bytes).expect("decode routing table"); + assert_eq!(decoded.vshard_epoch(7, 3), 41); + assert_eq!(decoded.vshard_epoch(7, 2), 58); + assert_eq!(decoded.vshard_epoch(9, 4), 60); + assert_eq!(decoded.vshard_epoch(1, 1), 0); + assert_eq!(decoded.group_for_vshard(7).ok(), Some(2)); + assert_eq!(decoded.vshard_epochs, rt.vshard_epochs); + } + #[test] fn uniform_distribution() { // 16 data groups → groups 1..=16 for vShards, plus metadata group 0. @@ -369,10 +526,25 @@ mod tests { let old_group = rt.group_for_vshard(0).unwrap(); // old_group is 1 (first data group); reassign to data group 2. let new_group = if old_group < 4 { old_group + 1 } else { 1 }; - rt.reassign_vshard(0, new_group); + rt.reassign_vshard(0, new_group, 17); assert_eq!(rt.group_for_vshard(0).unwrap(), new_group); } + #[test] + fn a_reassignment_raises_the_vshard_epoch_and_a_replay_keeps_it() { + let mut rt = RoutingTable::uniform(4, &[1, 2, 3], 3); + let initial = rt.group_for_vshard(0).unwrap(); + assert_eq!(rt.vshard_epoch(0, initial), 0); + let moved = if initial < 4 { initial + 1 } else { 1 }; + rt.reassign_vshard(0, moved, 40); + assert_eq!(rt.vshard_epoch(0, moved), 40); + // The old group keeps its epoch, so its late entries stay below. + assert_eq!(rt.vshard_epoch(0, initial), 0); + // A metadata replay applies the same index again. + rt.reassign_vshard(0, moved, 40); + assert_eq!(rt.vshard_epoch(0, moved), 40); + } + #[test] fn set_leader() { let mut rt = RoutingTable::uniform(2, &[1, 2, 3], 3); @@ -381,6 +553,63 @@ mod tests { assert_eq!(rt.leader_for_vshard(0).unwrap(), 99); } + #[test] + fn a_raft_observation_at_a_higher_term_wins_over_every_other_writer() { + let mut rt = RoutingTable::uniform(2, &[1, 2, 3], 3); + // A term-less hint is replaced by the first observation. `uniform` + // seeds group 1's hint with node 1, so the term-less write names + // another node to change it. + assert!(rt.set_leader(1, 3)); + assert_eq!(rt.leader_at_term_for_vshard(0).unwrap(), (3, 0)); + assert!(rt.observe_leader(1, 2, 5)); + assert_eq!(rt.leader_for_vshard(0).unwrap(), 2); + + // SWIM clears the suspected leader. The same leader at the same term + // is not restored, and an unknown leader is never recorded. + rt.clear_leader(1); + assert!(!rt.observe_leader(1, 2, 5)); + assert!(!rt.observe_leader(1, 0, 6)); + assert_eq!(rt.leader_for_vshard(0).unwrap(), 0); + + // The next election's leader is recorded, and an older term never + // replaces it. + assert!(rt.observe_leader(1, 3, 6)); + assert!(!rt.observe_leader(1, 2, 5)); + assert_eq!(rt.leader_for_vshard(0).unwrap(), 3); + + // A term-less hint never replaces the observed leader. + assert!(!rt.set_leader(1, 1)); + assert_eq!(rt.leader_at_term_for_vshard(0).unwrap(), (3, 6)); + + // Nor does it fill a cleared hint that holds a term. + rt.clear_leader(1); + assert!(!rt.set_leader(1, 1)); + assert_eq!(rt.leader_at_term_for_vshard(0).unwrap(), (0, 6)); + } + + #[test] + fn a_confirmation_fills_a_hint_cleared_at_its_term_and_nothing_older() { + let mut rt = RoutingTable::uniform(2, &[1, 2, 3], 3); + assert!(rt.observe_leader(1, 2, 5)); + rt.clear_leader(1); + + // The leader still serves term 5: the confirmation fills the clear. + assert!(!rt.leader_observation_is_new(1, 2, 5)); + assert!(rt.confirm_leader(1, 2, 5)); + assert_eq!(rt.leader_at_term_for_vshard(0).unwrap(), (2, 5)); + + // It never replaces a live hint at the same term, or any older term. + assert!(!rt.confirm_leader(1, 3, 5)); + assert!(!rt.confirm_leader(1, 3, 4)); + rt.clear_leader(1); + assert!(!rt.confirm_leader(1, 3, 4)); + assert!(!rt.confirm_leader(1, 0, 5)); + + // A newer term applies as an observation does. + assert!(rt.confirm_leader(1, 3, 6)); + assert_eq!(rt.leader_at_term_for_vshard(0).unwrap(), (3, 6)); + } + #[test] fn remove_group_member_strips_voter_and_clears_leader() { let mut rt = RoutingTable::uniform(2, &[1, 2, 3], 3); diff --git a/nodedb-cluster/src/routing_liveness.rs b/nodedb-cluster/src/routing_liveness.rs index d11162f18..e3d1d3883 100644 --- a/nodedb-cluster/src/routing_liveness.rs +++ b/nodedb-cluster/src/routing_liveness.rs @@ -3,14 +3,20 @@ //! Liveness-driven routing invalidation. //! //! [`RoutingLivenessHook`] is a [`MembershipSubscriber`] that clears -//! the leader hint for every Raft group whose leaseholder has just -//! been marked `Suspect`, `Dead`, or `Left` by the SWIM failure -//! detector. After the hook fires, the next query that consults the +//! the leader hint for every Raft group this node replicates whose +//! leaseholder has just been marked `Suspect`, `Dead`, or `Left` by the +//! SWIM failure detector. After the hook fires, the next query that consults the //! routing table observes `leader == 0` (the "no leader known" //! sentinel) and falls through to a fresh leader discovery via the -//! existing `NotLeader`-triggered election path. Clients see at most -//! one retry: the stale hint, the failed dispatch, and a refreshed -//! leader lookup. +//! existing `NotLeader`-triggered election path. +//! +//! A clear keeps the hint's term, so only a new election or a +//! confirmation fills it (see [`RoutingTable::confirm_leader`]). A +//! suspicion SWIM refutes is such a confirmation: when a node it marked +//! `Suspect` or `Dead` answers as `Alive` again, the hook fills each +//! hint it cleared for that node back, where the hint is still cleared +//! at the same term. Without that, a leader wrongly suspected under +//! load stays unknown on this node until the group's next election. //! //! The hook is storage-agnostic: it holds `Arc>` //! and a resolver closure that maps the string-keyed SWIM `NodeId` @@ -18,12 +24,12 @@ //! crate. Wiring layers (start_cluster, tests) supply the resolver //! appropriate to their topology source. //! -//! The hook is intentionally sync and cheap — a single `RwLock::write`, -//! a linear scan over group_members, and `set_leader(gid, 0)` for -//! each affected group. No I/O, no spawning. That keeps it safe to -//! call directly from the detector run loop. +//! The hook is sync and cheap: one `RwLock::write` and a linear scan +//! over group_members. No I/O, no spawning. That keeps it safe to call +//! directly from the detector run loop. -use std::sync::{Arc, RwLock}; +use std::collections::HashMap; +use std::sync::{Arc, Mutex, RwLock}; use nodedb_types::NodeId; use tracing::debug; @@ -39,76 +45,152 @@ use crate::swim::subscriber::MembershipSubscriber; /// probe, transient learners, etc.). Those are silently ignored. pub type NodeIdResolver = Arc Option + Send + Sync>; -/// Clears the leader hint for every group led by a node that SWIM -/// has marked Suspect/Dead/Left. +/// Clears the leader hint for every group this node replicates that is +/// led by a node SWIM has marked Suspect/Dead/Left, and fills it back when +/// SWIM sees the node alive again. +/// +/// The hint of a group this node does not replicate is left as it is. This +/// node's Raft never observes such a group, so a cleared hint there names +/// nobody to ask. A hint that names a dead node fails its request instead, +/// and the gateway retry and the leader probe move it on. pub struct RoutingLivenessHook { routing: Arc>, resolver: NodeIdResolver, + /// This node's routing-table id. + local_node_id: u64, + /// Per suspected node: the `(group_id, term)` hints the hook cleared + /// for it. Bounded by the groups the node led when it was suspected. + cleared: Mutex>>, } impl RoutingLivenessHook { - pub fn new(routing: Arc>, resolver: NodeIdResolver) -> Self { - Self { routing, resolver } + pub fn new( + routing: Arc>, + resolver: NodeIdResolver, + local_node_id: u64, + ) -> Self { + Self { + routing, + resolver, + local_node_id, + cleared: Mutex::new(HashMap::new()), + } } -} -impl MembershipSubscriber for RoutingLivenessHook { - fn on_state_change(&self, node_id: &NodeId, _old: Option, new: MemberState) { - // Alive transitions are a no-op: the next query will refresh - // the leader hint naturally on NotLeader. We only invalidate - // when a leader has observably stopped being reachable. - if !matches!( - new, - MemberState::Suspect | MemberState::Dead | MemberState::Left - ) { + /// Clear every hint of a replicated group that names `numeric_id`, and + /// remember each one. + fn clear_led_by(&self, node_id: &NodeId, numeric_id: u64, new: MemberState) { + let local = self.local_node_id; + let mut rt = self.routing.write().unwrap_or_else(|p| p.into_inner()); + let affected: Vec<(u64, u64)> = rt + .group_members() + .iter() + .filter(|(_, info)| { + info.leader == numeric_id + && (info.members.contains(&local) || info.learners.contains(&local)) + }) + .map(|(gid, info)| (*gid, info.leader_term)) + .collect(); + for (gid, _) in &affected { + rt.clear_leader(*gid); + } + drop(rt); + if affected.is_empty() { return; } + debug!( + ?node_id, + ?new, + numeric_id, + groups_invalidated = affected.len(), + "routing liveness hook cleared leader hints" + ); + let mut cleared = self.cleared.lock().unwrap_or_else(|p| p.into_inner()); + let entries = cleared.entry(numeric_id).or_default(); + for (gid, term) in affected { + entries.retain(|(g, _)| *g != gid); + entries.push((gid, term)); + } + } - let Some(numeric_id) = (self.resolver)(node_id) else { - // SWIM knows about a node the routing table doesn't — a - // seed placeholder, a learner mid-join, or a node that - // was never registered. Nothing to invalidate. + /// Fill back every hint cleared for `numeric_id` that is still cleared + /// at the term it was cleared at. + fn restore_led_by(&self, node_id: &NodeId, numeric_id: u64) { + let Some(entries) = self + .cleared + .lock() + .unwrap_or_else(|p| p.into_inner()) + .remove(&numeric_id) + else { return; }; - let mut rt = self.routing.write().unwrap_or_else(|p| p.into_inner()); - let affected: Vec = rt - .group_members() - .iter() - .filter(|(_, info)| info.leader == numeric_id) - .map(|(gid, _)| *gid) - .collect(); - for gid in &affected { - rt.set_leader(*gid, 0); - } - if !affected.is_empty() { + let restored = entries + .into_iter() + .filter(|&(gid, term)| rt.confirm_leader(gid, numeric_id, term)) + .count(); + if restored > 0 { debug!( ?node_id, - ?new, numeric_id, - groups_invalidated = affected.len(), - "routing liveness hook cleared leader hints" + groups_restored = restored, + "routing liveness hook restored leader hints of a node alive again" ); } } } +impl MembershipSubscriber for RoutingLivenessHook { + fn on_state_change(&self, node_id: &NodeId, old: Option, new: MemberState) { + let Some(numeric_id) = (self.resolver)(node_id) else { + // SWIM knows about a node the routing table doesn't — a + // seed placeholder, a learner mid-join, or a node that + // was never registered. Nothing to invalidate. + return; + }; + match new { + MemberState::Suspect | MemberState::Dead | MemberState::Left => { + self.clear_led_by(node_id, numeric_id, new); + } + // A node SWIM suspected answers again: the suspicion was wrong. + MemberState::Alive if matches!(old, Some(MemberState::Suspect | MemberState::Dead)) => { + self.restore_led_by(node_id, numeric_id); + } + MemberState::Alive => {} + } + } +} + #[cfg(test)] mod tests { use super::*; - fn rt_with_leaders(pairs: &[(u64, u64)], rf: usize) -> Arc> { + /// The node the hook runs on. It replicates every group of + /// [`rt_with_leaders`]. + const LOCAL: u64 = 100; + + fn rt_with_leaders(pairs: &[(u64, u64)]) -> Arc> { // Build a routing table with `pairs.len()` groups where group - // `gid` has leader `leader`. Uses the uniform constructor to - // pick a membership, then overrides the leader. - let nodes: Vec = pairs.iter().map(|(_, l)| *l).collect(); - let mut rt = RoutingTable::uniform(pairs.len() as u64, &nodes, rf); + // `gid` has leader `leader`. Every node, `LOCAL` included, is a + // voter of every group. The leader is then overridden. + let mut nodes: Vec = pairs.iter().map(|(_, l)| *l).collect(); + nodes.push(LOCAL); + nodes.sort_unstable(); + nodes.dedup(); + let mut rt = RoutingTable::uniform(pairs.len() as u64, &nodes, nodes.len()); for (gid, leader) in pairs { rt.set_leader(*gid, *leader); } Arc::new(RwLock::new(rt)) } + fn hook( + rt: &Arc>, + map: &'static [(&'static str, u64)], + ) -> RoutingLivenessHook { + RoutingLivenessHook::new(rt.clone(), resolver_for(map), LOCAL) + } + fn resolver_for(map: &'static [(&'static str, u64)]) -> NodeIdResolver { Arc::new(move |nid: &NodeId| { map.iter() @@ -119,9 +201,8 @@ mod tests { #[test] fn dead_transition_clears_leader_for_owned_groups() { - let rt = rt_with_leaders(&[(0, 1), (1, 2), (2, 1), (3, 3)], 1); - let hook = - RoutingLivenessHook::new(rt.clone(), resolver_for(&[("a", 1), ("b", 2), ("c", 3)])); + let rt = rt_with_leaders(&[(0, 1), (1, 2), (2, 1), (3, 3)]); + let hook = hook(&rt, &[("a", 1), ("b", 2), ("c", 3)]); hook.on_state_change( &NodeId::try_new("a").expect("test fixture"), @@ -138,8 +219,8 @@ mod tests { #[test] fn suspect_transition_also_invalidates() { - let rt = rt_with_leaders(&[(0, 7)], 1); - let hook = RoutingLivenessHook::new(rt.clone(), resolver_for(&[("x", 7)])); + let rt = rt_with_leaders(&[(0, 7)]); + let hook = hook(&rt, &[("x", 7)]); hook.on_state_change( &NodeId::try_new("x").expect("test fixture"), Some(MemberState::Alive), @@ -150,8 +231,8 @@ mod tests { #[test] fn alive_transition_is_noop() { - let rt = rt_with_leaders(&[(0, 5)], 1); - let hook = RoutingLivenessHook::new(rt.clone(), resolver_for(&[("q", 5)])); + let rt = rt_with_leaders(&[(0, 5)]); + let hook = hook(&rt, &[("q", 5)]); hook.on_state_change( &NodeId::try_new("q").expect("test fixture"), None, @@ -162,8 +243,8 @@ mod tests { #[test] fn unresolved_node_id_is_ignored() { - let rt = rt_with_leaders(&[(0, 1)], 1); - let hook = RoutingLivenessHook::new(rt.clone(), resolver_for(&[("a", 1)])); + let rt = rt_with_leaders(&[(0, 1)]); + let hook = hook(&rt, &[("a", 1)]); // NodeId "seed:127.0.0.1:9000" is not in the resolver map. hook.on_state_change( &NodeId::try_new("seed:127.0.0.1:9000").expect("test fixture"), @@ -176,8 +257,8 @@ mod tests { #[test] fn left_is_also_invalidating() { - let rt = rt_with_leaders(&[(0, 2)], 1); - let hook = RoutingLivenessHook::new(rt.clone(), resolver_for(&[("b", 2)])); + let rt = rt_with_leaders(&[(0, 2)]); + let hook = hook(&rt, &[("b", 2)]); hook.on_state_change( &NodeId::try_new("b").expect("test fixture"), Some(MemberState::Alive), @@ -185,4 +266,50 @@ mod tests { ); assert_eq!(rt.read().unwrap().group_info(0).unwrap().leader, 0); } + + /// A refuted suspicion fills the cleared hint back, unless a newer + /// election already moved it. + #[test] + fn a_node_alive_again_gets_back_the_hints_its_suspicion_cleared() { + let rt = rt_with_leaders(&[(0, 1), (1, 1)]); + { + let mut table = rt.write().unwrap(); + assert!(table.observe_leader(0, 1, 4)); + assert!(table.observe_leader(1, 1, 4)); + } + let hook = hook(&rt, &[("a", 1)]); + let a = NodeId::try_new("a").expect("test fixture"); + hook.on_state_change(&a, Some(MemberState::Alive), MemberState::Suspect); + assert_eq!(rt.read().unwrap().group_info(0).unwrap().leader, 0); + + // Group 1 elects node 2 at term 5 while node 1 is suspected. + assert!(rt.write().unwrap().observe_leader(1, 2, 5)); + + hook.on_state_change(&a, Some(MemberState::Suspect), MemberState::Alive); + let table = rt.read().unwrap(); + let info0 = table.group_info(0).unwrap(); + assert_eq!((info0.leader, info0.leader_term), (1, 4)); + let info1 = table.group_info(1).unwrap(); + assert_eq!((info1.leader, info1.leader_term), (2, 5)); + } + + /// A group this node does not replicate keeps its hint: nothing else + /// here could name a node to ask. + #[test] + fn a_group_this_node_does_not_replicate_keeps_its_hint() { + let rt = rt_with_leaders(&[(0, 1), (1, 1)]); + { + let mut table = rt.write().unwrap(); + table.set_group_members(1, vec![1, 2]); + } + let hook = hook(&rt, &[("a", 1)]); + hook.on_state_change( + &NodeId::try_new("a").expect("test fixture"), + Some(MemberState::Alive), + MemberState::Dead, + ); + let table = rt.read().unwrap(); + assert_eq!(table.group_info(0).unwrap().leader, 0); + assert_eq!(table.group_info(1).unwrap().leader, 1); + } } diff --git a/nodedb-cluster/src/routing_membership.rs b/nodedb-cluster/src/routing_membership.rs new file mode 100644 index 000000000..f27d4cc83 --- /dev/null +++ b/nodedb-cluster/src/routing_membership.rs @@ -0,0 +1,106 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A group's membership as its leader answers it, adopted into this node's +//! routing view. +//! +//! A node's routing view of a group's voters and learners moves when the node +//! applies the group's conf changes. Two kinds of node never apply them: +//! - a node that hosts no replica of the group; +//! - a node removed from the group before it learned of its removal. +//! +//! Such a node asks a placement node for the group's leader status (see +//! [`crate::rpc_codec::LeaderStatusResponse`]). A leader answers its Raft +//! voters and learners. A leader changes them only by applying a committed +//! conf change, so the answer holds only committed changes. +//! +//! The same view answers whether a node is a replica of a group. A leader +//! hint that names a node the view does not list is stale. + +use crate::routing::RoutingTable; + +impl RoutingTable { + /// Whether this routing view lists `node_id` as a voter or learner of the + /// group `vshard_id` maps to. + /// + /// A leader hint that names `node_id` is stale when this is false: the + /// node left the group, so it cannot serve it. + pub fn is_replica_of_vshard(&self, vshard_id: u32, node_id: u64) -> bool { + self.group_for_vshard(vshard_id) + .ok() + .and_then(|group_id| self.group_info(group_id)) + .is_some_and(|info| info.members.contains(&node_id) || info.learners.contains(&node_id)) + } + + /// Set `group_id`'s voters and learners to the ones `leader` answered at + /// `term`. Returns whether the view changed. + /// + /// Applies only while the leader hint names `leader` at `term`, so the + /// caller confirms the leader first (see [`RoutingTable::confirm_leader`]). + /// An answer from a leader of an older term never passes this check once + /// the hint moved to a newer term. + pub fn adopt_leader_membership( + &mut self, + group_id: u64, + leader: u64, + term: u64, + voters: &[u64], + learners: &[u64], + ) -> bool { + let Some(info) = self.group_info(group_id) else { + return false; + }; + if leader == 0 || term == 0 || info.leader != leader || info.leader_term != term { + return false; + } + let mut voters = voters.to_vec(); + voters.sort_unstable(); + voters.dedup(); + let mut learners = learners.to_vec(); + learners.sort_unstable(); + learners.dedup(); + let mut members_now = info.members.clone(); + members_now.sort_unstable(); + let mut learners_now = info.learners.clone(); + learners_now.sort_unstable(); + if members_now == voters && learners_now == learners { + return false; + } + self.set_group_members(group_id, voters); + self.set_group_learners(group_id, learners); + true + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_confirmed_leaders_membership_replaces_the_view() { + let mut rt = RoutingTable::uniform(1, &[1, 2, 3], 3); + assert!(rt.confirm_leader(1, 2, 5)); + assert!(rt.adopt_leader_membership(1, 2, 5, &[2, 1], &[4])); + let info = rt.group_info(1).expect("group 1"); + assert_eq!(info.members, vec![1, 2]); + assert_eq!(info.learners, vec![4]); + // The same answer again changes nothing. + assert!(!rt.adopt_leader_membership(1, 2, 5, &[1, 2], &[4])); + } + + #[test] + fn an_answer_the_hint_does_not_name_is_ignored() { + let mut rt = RoutingTable::uniform(1, &[1, 2, 3], 3); + assert!(rt.confirm_leader(1, 2, 5)); + // Another leader, or the same leader at another term. + assert!(!rt.adopt_leader_membership(1, 3, 5, &[3], &[])); + assert!(!rt.adopt_leader_membership(1, 2, 4, &[2], &[])); + // A deposed leader's answer after the hint moved on. + assert!(rt.confirm_leader(1, 3, 6)); + assert!(!rt.adopt_leader_membership(1, 2, 5, &[2], &[])); + let mut members = rt.group_info(1).expect("group 1").members.clone(); + members.sort_unstable(); + assert_eq!(members, vec![1, 2, 3]); + // An unknown group. + assert!(!rt.adopt_leader_membership(9, 2, 5, &[2], &[])); + } +} diff --git a/nodedb-cluster/src/rpc_codec/auth_lease.rs b/nodedb-cluster/src/rpc_codec/auth_lease.rs index 219f8f9c0..7e7a45697 100644 --- a/nodedb-cluster/src/rpc_codec/auth_lease.rs +++ b/nodedb-cluster/src/rpc_codec/auth_lease.rs @@ -42,8 +42,9 @@ pub enum AuthLeaseRenewOutcome { /// The report does not cover every acknowledged change. The sender's /// lease is not extended. Withheld, - /// The receiver does not lead the metadata group. - NotLeader { leader_hint: Option }, + /// The receiver does not lead the metadata group. `leader_hint` is the + /// leader the receiver knows at `term`, its current term. + NotLeader { leader_hint: Option, term: u64 }, } /// Response to an [`AuthLeaseRenewRequest`]. @@ -66,8 +67,9 @@ pub struct AuthBarrierRequest { pub enum AuthBarrierOutcome { /// No node can plan against state older than the targets. Released, - /// The receiver does not lead the metadata group. - NotLeader { leader_hint: Option }, + /// The receiver does not lead the metadata group. `leader_hint` is the + /// leader the receiver knows at `term`, its current term. + NotLeader { leader_hint: Option, term: u64 }, /// The barrier did not release in time. Timeout { waited_ms: u64 }, } @@ -182,6 +184,7 @@ mod tests { AuthLeaseRenewOutcome::Withheld, AuthLeaseRenewOutcome::NotLeader { leader_hint: Some(1), + term: 3, }, ] { match roundtrip(RaftRpc::AuthLeaseRenewResponse(AuthLeaseRenewResponse { @@ -207,7 +210,10 @@ mod tests { } for outcome in [ AuthBarrierOutcome::Released, - AuthBarrierOutcome::NotLeader { leader_hint: None }, + AuthBarrierOutcome::NotLeader { + leader_hint: None, + term: 0, + }, AuthBarrierOutcome::Timeout { waited_ms: 5000 }, ] { match roundtrip(RaftRpc::AuthBarrierResponse(AuthBarrierResponse { diff --git a/nodedb-cluster/src/rpc_codec/calvin_parts.rs b/nodedb-cluster/src/rpc_codec/calvin_parts.rs new file mode 100644 index 000000000..695728c30 --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/calvin_parts.rs @@ -0,0 +1,125 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Streamed parts of a multi-part Calvin transaction, coordinator to +//! sequencer leader. +//! +//! A coordinator submits a multi-part transaction's header first. It then +//! sends its parts in order, a bounded batch per [`CalvinPartsRequest`], to +//! the leader that took the header. The leader queues them and answers with +//! one [`CalvinPartsResponse`]: how far the stream got, and whether to go on, +//! wait for room, or stop. One request/response per batch. +//! +//! Discriminants 54/55 are permanently assigned to these variants. + +use super::discriminants::*; +use super::header::write_frame; +use super::raft_rpc::RaftRpc; +use crate::error::{ClusterError, Result}; + +/// Plan bytes one request carries at most: a quarter of the 64 MiB RPC +/// payload limit, so a batch never nears it whatever its framing costs. +pub const MAX_PARTS_BATCH_BYTES: usize = 16 << 20; + +/// One batch of streamed parts. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct CalvinPartsRequest { + /// The stream's coordinator node. + pub stream_node: u64, + /// The stream's sequence on its coordinator. + pub stream_seq: u64, + /// The parts, as a msgpack-encoded `Vec` in index order. + pub parts_bytes: Vec, +} + +/// The leader's answer to a [`CalvinPartsRequest`]. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct CalvinPartsResponse { + /// A `PartsOfferStatus` wire code. + pub status: u8, + /// The index of the next part the stream owes. + pub next_index: u32, + /// Why the leader rejected a part or could not take the batch. + pub detail: Option, +} + +macro_rules! to_bytes { + ($msg:expr) => { + rkyv::to_bytes::($msg) + .map(|b| b.to_vec()) + .map_err(|e| ClusterError::Codec { + detail: format!("rkyv serialize: {e}"), + }) + }; +} + +macro_rules! from_bytes { + ($payload:expr, $T:ty, $name:expr) => {{ + let mut aligned = rkyv::util::AlignedVec::<16>::with_capacity($payload.len()); + aligned.extend_from_slice($payload); + rkyv::from_bytes::<$T, rkyv::rancor::Error>(&aligned).map_err(|e| ClusterError::Codec { + detail: format!("rkyv deserialize {}: {e}", $name), + }) + }}; +} + +pub(super) fn encode_calvin_parts_req(msg: &CalvinPartsRequest, out: &mut Vec) -> Result<()> { + write_frame(RPC_CALVIN_PARTS_REQ, &to_bytes!(msg)?, out) +} + +pub(super) fn encode_calvin_parts_resp(msg: &CalvinPartsResponse, out: &mut Vec) -> Result<()> { + write_frame(RPC_CALVIN_PARTS_RESP, &to_bytes!(msg)?, out) +} + +pub(super) fn decode_calvin_parts_req(payload: &[u8]) -> Result { + Ok(RaftRpc::CalvinPartsRequest(from_bytes!( + payload, + CalvinPartsRequest, + "CalvinPartsRequest" + )?)) +} + +pub(super) fn decode_calvin_parts_resp(payload: &[u8]) -> Result { + Ok(RaftRpc::CalvinPartsResponse(from_bytes!( + payload, + CalvinPartsResponse, + "CalvinPartsResponse" + )?)) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn roundtrip(rpc: RaftRpc) -> RaftRpc { + let epoch = crate::cluster_epoch::ClusterEpochState::default(); + let encoded = super::super::encode(&rpc, &epoch).expect("encode"); + super::super::decode(&encoded, &epoch).expect("decode") + } + + #[test] + fn a_parts_request_and_response_round_trip() { + let RaftRpc::CalvinPartsRequest(req) = + roundtrip(RaftRpc::CalvinPartsRequest(CalvinPartsRequest { + stream_node: 3, + stream_seq: 77, + parts_bytes: vec![1, 2, 3], + })) + else { + panic!("expected a parts request"); + }; + assert_eq!((req.stream_node, req.stream_seq), (3, 77)); + assert_eq!(req.parts_bytes, vec![1, 2, 3]); + + let RaftRpc::CalvinPartsResponse(resp) = + roundtrip(RaftRpc::CalvinPartsResponse(CalvinPartsResponse { + status: 1, + next_index: 12, + detail: Some("full".into()), + })) + else { + panic!("expected a parts response"); + }; + assert_eq!((resp.status, resp.next_index), (1, 12)); + assert_eq!(resp.detail.as_deref(), Some("full")); + } +} diff --git a/nodedb-cluster/src/rpc_codec/cluster_mgmt.rs b/nodedb-cluster/src/rpc_codec/cluster_mgmt.rs index d4e6b3420..2f2fe4071 100644 --- a/nodedb-cluster/src/rpc_codec/cluster_mgmt.rs +++ b/nodedb-cluster/src/rpc_codec/cluster_mgmt.rs @@ -17,11 +17,16 @@ pub struct JoinRequest { pub node_id: u64, pub listen_addr: String, pub wire_version: u16, + /// Joiner's `nodedb_types::wire_version::WIRE_BUILD_ID`. Compared for + /// exact equality against this node's own build — see `handle_join_request`. + pub build_id: String, /// SPIFFE URI SAN from the joiner's mTLS leaf certificate, if present. pub spiffe_id: Option, /// SHA-256 SPKI fingerprint of the joiner's mTLS leaf certificate. /// Stored as `Vec` for rkyv compatibility; always 32 bytes when present. pub spki_pin: Option>, + /// Bound UDP address of the joiner's SWIM failure detector. + pub swim_addr: Option, } /// Response to a join request — carries full cluster state. @@ -48,10 +53,12 @@ pub struct JoinNodeInfo { /// SHA-256 SPKI fingerprint for this node. /// Stored as `Vec` for rkyv compatibility; always 32 bytes when present. pub spki_pin: Option>, + /// Bound UDP address of this node's SWIM failure detector, if it runs one. + pub swim_addr: Option, } /// Raft group membership in the join response wire format. -#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] pub struct JoinGroupInfo { pub group_id: u64, pub leader: u64, @@ -190,8 +197,10 @@ mod tests { node_id: 42, listen_addr: "10.0.0.5:9400".into(), wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), spiffe_id: Some("spiffe://cluster.local/node/42".into()), spki_pin: Some(vec![0xabu8; 32]), + swim_addr: Some("10.0.0.5:9401".into()), }; match roundtrip(RaftRpc::JoinRequest(req)) { RaftRpc::JoinRequest(d) => { @@ -202,6 +211,8 @@ mod tests { Some("spiffe://cluster.local/node/42") ); assert_eq!(d.spki_pin.as_deref(), Some([0xabu8; 32].as_ref())); + assert_eq!(d.build_id, nodedb_types::wire_version::WIRE_BUILD_ID); + assert_eq!(d.swim_addr.as_deref(), Some("10.0.0.5:9401")); } other => panic!("expected JoinRequest, got {other:?}"), } @@ -221,6 +232,7 @@ mod tests { wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, spiffe_id: None, spki_pin: None, + swim_addr: Some("10.0.0.1:9401".into()), }], vshard_to_group: (0..1024u64).map(|i| i % 4).collect(), groups: vec![JoinGroupInfo { @@ -235,6 +247,7 @@ mod tests { assert!(d.success); assert_eq!(d.nodes.len(), 1); assert_eq!(d.vshard_to_group.len(), 1024); + assert_eq!(d.nodes[0].swim_addr.as_deref(), Some("10.0.0.1:9401")); } other => panic!("expected JoinResponse, got {other:?}"), } diff --git a/nodedb-cluster/src/rpc_codec/data_plane_error.rs b/nodedb-cluster/src/rpc_codec/data_plane_error.rs index 506911ffc..5a99333e8 100644 --- a/nodedb-cluster/src/rpc_codec/data_plane_error.rs +++ b/nodedb-cluster/src/rpc_codec/data_plane_error.rs @@ -34,7 +34,6 @@ pub enum DataPlaneErrorCode { expected: [u8; 32], actual: [u8; 32], }, - FanOutExceeded, ResourcesExhausted, RejectedDanglingEdge { missing_node: String, diff --git a/nodedb-cluster/src/rpc_codec/data_propose.rs b/nodedb-cluster/src/rpc_codec/data_propose.rs index f1fcb4831..dd2244da1 100644 --- a/nodedb-cluster/src/rpc_codec/data_propose.rs +++ b/nodedb-cluster/src/rpc_codec/data_propose.rs @@ -35,7 +35,7 @@ pub struct DataProposeRequest { #[derive(Debug, Clone, Copy, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] pub enum ForwardedProposeRefusal { /// The node does not lead the target group. `leader_hint` names the - /// leader it knows, if any. + /// leader it knows, if any, at `leader_term`. NotLeader, /// A leadership transfer of the target group is in flight. LeadershipTransferInProgress, @@ -50,6 +50,9 @@ pub struct DataProposeResponse { pub group_id: u64, pub log_index: u64, pub leader_hint: Option, + /// The refusing node's term, which `leader_hint` is known at. `0` on + /// success and on a refusal that is not `NotLeader`. + pub leader_term: u64, /// The typed reason of a refusal. `None` on success. pub refusal: Option, pub error_message: String, @@ -62,6 +65,7 @@ impl DataProposeResponse { group_id, log_index, leader_hint: None, + leader_term: 0, refusal: None, error_message: String::new(), } @@ -69,13 +73,15 @@ impl DataProposeResponse { /// The response for a proposal the leader refused with `error`. pub fn refused(error: &ClusterError) -> Self { - let (refusal, leader_hint) = match error { - ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint }) => { - (ForwardedProposeRefusal::NotLeader, *leader_hint) - } - ClusterError::Raft(nodedb_raft::RaftError::LeadershipTransferInProgress) => { - (ForwardedProposeRefusal::LeadershipTransferInProgress, None) + let (refusal, leader_hint, leader_term) = match error { + ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint, term }) => { + (ForwardedProposeRefusal::NotLeader, *leader_hint, *term) } + ClusterError::Raft(nodedb_raft::RaftError::LeadershipTransferInProgress) => ( + ForwardedProposeRefusal::LeadershipTransferInProgress, + None, + 0, + ), // A refusal with no retry contract. The forwarding node reads it // as a transport error carrying the leader's message. ClusterError::Raft( @@ -125,14 +131,16 @@ impl DataProposeResponse { | ClusterError::SpatialGather(_) | ClusterError::Bm25Gather(_) | ClusterError::TsGather(_) + | ClusterError::ShufflePush(_) | ClusterError::RemoteUntyped { .. } - | ClusterError::ShardExecution { .. } => (ForwardedProposeRefusal::Failed, None), + | ClusterError::ShardExecution { .. } => (ForwardedProposeRefusal::Failed, None, 0), }; Self { success: false, group_id: 0, log_index: 0, leader_hint, + leader_term, refusal: Some(refusal), error_message: error.to_string(), } @@ -146,6 +154,7 @@ impl DataProposeResponse { Some(ForwardedProposeRefusal::NotLeader) => { ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint: self.leader_hint, + term: self.leader_term, }) } Some(ForwardedProposeRefusal::LeadershipTransferInProgress) => { @@ -261,11 +270,13 @@ mod tests { let error = refusal_across_the_wire(ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint: Some(3), + term: 9, })); assert!(matches!( error, ClusterError::Raft(nodedb_raft::RaftError::NotLeader { - leader_hint: Some(3) + leader_hint: Some(3), + term: 9, }) )); } diff --git a/nodedb-cluster/src/rpc_codec/discriminants.rs b/nodedb-cluster/src/rpc_codec/discriminants.rs index 1682fcff9..4fcf1567a 100644 --- a/nodedb-cluster/src/rpc_codec/discriminants.rs +++ b/nodedb-cluster/src/rpc_codec/discriminants.rs @@ -18,13 +18,9 @@ pub const RPC_PING: u8 = 9; pub const RPC_PONG: u8 = 10; pub const RPC_TOPOLOGY_UPDATE: u8 = 11; pub const RPC_TOPOLOGY_ACK: u8 = 12; -/// Retired in Phase C-δ.6: reserved, do not reuse — was ForwardRequest/Response -/// (SQL-string forwarding path replaced by gateway.execute / ExecuteRequest). -#[allow(dead_code)] +/// Reserved, do not reuse. Number 13 belongs to a retired SQL-string forwarding request. pub const RPC_FORWARD_REQ: u8 = 13; -/// Retired in Phase C-δ.6: reserved, do not reuse — was ForwardRequest/Response -/// (SQL-string forwarding path replaced by gateway.execute / ExecuteRequest). -#[allow(dead_code)] +/// Reserved, do not reuse. Number 14 belongs to a retired SQL-string forwarding response. pub const RPC_FORWARD_RESP: u8 = 14; pub const RPC_VSHARD_ENVELOPE: u8 = 15; pub const RPC_METADATA_PROPOSE_REQ: u8 = 16; @@ -41,7 +37,7 @@ pub const RPC_DATA_PROPOSE_RESP: u8 = 21; pub const RPC_EXECUTE_STREAM_REQ: u8 = 22; pub const RPC_EXECUTE_STREAM_CHUNK: u8 = 23; pub const RPC_EXECUTE_STREAM_END: u8 = 24; -/// Cross-node streaming shuffle (E1). A producer opens one bidi stream per +/// Cross-node streaming shuffle. A producer opens one bidi stream per /// target partition and writes a `RPC_SHUFFLE_PUSH_REQ` envelope, then a /// sequence of `RPC_SHUFFLE_PUSH_CHUNK` envelopes terminated by exactly one /// `RPC_SHUFFLE_PUSH_END` envelope on the same QUIC stream. Direction is @@ -49,7 +45,7 @@ pub const RPC_EXECUTE_STREAM_END: u8 = 24; pub const RPC_SHUFFLE_PUSH_REQ: u8 = 25; pub const RPC_SHUFFLE_PUSH_CHUNK: u8 = 26; pub const RPC_SHUFFLE_PUSH_END: u8 = 27; -/// Cross-node shuffle PRODUCER trigger (E4a). A coordinator sends a +/// Cross-node shuffle PRODUCER trigger. A coordinator sends a /// `RPC_SHUFFLE_PRODUCE_REQ` to a producer node; that node executes a local /// scan fragment, hash-partitions each output row, and fans the rows out to the /// per-part owners as `RPC_SHUFFLE_PUSH_*` streams (looping back into its own @@ -58,7 +54,7 @@ pub const RPC_SHUFFLE_PUSH_END: u8 = 27; /// does NOT stream the scanned rows back to the coordinator. pub const RPC_SHUFFLE_PRODUCE_REQ: u8 = 28; pub const RPC_SHUFFLE_PRODUCE_RESP: u8 = 29; -/// Cross-node shuffle CONSUMER trigger (E4b). A coordinator sends a +/// Cross-node shuffle CONSUMER trigger. A coordinator sends a /// `RPC_SHUFFLE_CONSUME_REQ` to a part-owner node; that node waits for both /// staged sides of its `(shuffle_id, part)` to finalize, runs the node-local /// grace-hash join over them, and replies with exactly one @@ -66,7 +62,7 @@ pub const RPC_SHUFFLE_PRODUCE_RESP: u8 = 29; /// One-shot request/response — no streaming. pub const RPC_SHUFFLE_CONSUME_REQ: u8 = 30; pub const RPC_SHUFFLE_CONSUME_RESP: u8 = 31; -/// Cross-node distributed GROUP BY shuffle CONSUMER trigger (E5b). A coordinator +/// Cross-node distributed GROUP BY shuffle CONSUMER trigger. A coordinator /// sends a `RPC_SHUFFLE_AGG_CONSUME_REQ` to a part-owner node; that node waits /// for its part's single staged producer side (side 0) to finalize, merges the /// staged partial `GroupState`s, finalizes / HAVING-filters / sorts / LIMITs, and @@ -154,6 +150,22 @@ pub const RPC_AUTH_BARRIER_RESP: u8 = 52; /// Answer to an `RPC_VSHARD_ENVELOPE` request whose handler failed. It /// carries the handler's typed error in place of a response envelope. pub const RPC_VSHARD_REFUSAL: u8 = 53; +/// Streamed parts of a multi-part Calvin transaction: a coordinator sends a +/// batch in `RPC_CALVIN_PARTS_REQ` to the sequencer leader and receives one +/// `RPC_CALVIN_PARTS_RESP` naming how far the stream got. +pub const RPC_CALVIN_PARTS_REQ: u8 = 54; +pub const RPC_CALVIN_PARTS_RESP: u8 = 55; +/// Leader status: a node asks another which leader it knows for a Raft +/// group in `RPC_LEADER_STATUS_REQ`, and the receiver answers from its own +/// Raft state, with no quorum round, in one `RPC_LEADER_STATUS_RESP`. +pub const RPC_LEADER_STATUS_REQ: u8 = 56; +pub const RPC_LEADER_STATUS_RESP: u8 = 57; +/// The answer to a request frame the receiver's replay window refused. The +/// sender retries the request under a fresh sequence number. +pub const RPC_FRAME_REFUSAL: u8 = 58; +/// The answer to a request whose handler failed. It carries the typed +/// reason, so the sender learns the peer is up and refused this request. +pub const RPC_REQUEST_REFUSAL: u8 = 59; // VShardMessageType discriminants for distributed array ops (u16, range 80-89). // These mirror `crate::wire::VShardMessageType` repr values and are declared diff --git a/nodedb-cluster/src/rpc_codec/execute/codec.rs b/nodedb-cluster/src/rpc_codec/execute/codec.rs index 3356bc15c..c156002f1 100644 --- a/nodedb-cluster/src/rpc_codec/execute/codec.rs +++ b/nodedb-cluster/src/rpc_codec/execute/codec.rs @@ -165,6 +165,8 @@ mod tests { }, ], txn_id: None, + vshard_id: None, + read_groups: Vec::new(), }; let decoded = roundtrip_req(req.clone()); assert_eq!(decoded.plan_bytes, req.plan_bytes); @@ -189,11 +191,35 @@ mod tests { trace_id: [0u8; 16], descriptor_versions: vec![], txn_id: None, + vshard_id: None, + read_groups: Vec::new(), }; let decoded = roundtrip_req(req); assert!(decoded.descriptor_versions.is_empty()); } + #[test] + fn roundtrip_execute_request_carries_the_scoped_vshard() { + let req = ExecuteRequest { + plan_bytes: vec![0x01], + tenant_id: 1, + database_id: 0, + deadline_remaining_ms: 1000, + trace_id: [0u8; 16], + descriptor_versions: vec![], + txn_id: Some(nodedb_types::id::TxnId::new(9)), + vshard_id: Some(nodedb_types::id::VShardId::new(513)), + read_groups: vec![7, 9], + }; + let decoded = roundtrip_req(req); + assert_eq!( + decoded.vshard_id, + Some(nodedb_types::id::VShardId::new(513)) + ); + assert_eq!(decoded.txn_id, Some(nodedb_types::id::TxnId::new(9))); + assert_eq!(decoded.read_groups, vec![7, 9]); + } + #[test] fn roundtrip_execute_response_success() { let resp = ExecuteResponse::ok( @@ -381,6 +407,8 @@ mod tests { version: 3, }], txn_id: None, + vshard_id: None, + read_groups: Vec::new(), }; let rpc = RaftRpc::ExecuteStreamRequest(req.clone()); let encoded = diff --git a/nodedb-cluster/src/rpc_codec/execute/types.rs b/nodedb-cluster/src/rpc_codec/execute/types.rs index 66727367f..ce85b3aaa 100644 --- a/nodedb-cluster/src/rpc_codec/execute/types.rs +++ b/nodedb-cluster/src/rpc_codec/execute/types.rs @@ -4,7 +4,7 @@ //! //! Field order and enum variant order are the wire ABI: append only. -use nodedb_types::id::TxnId; +use nodedb_types::id::{TxnId, VShardId}; use crate::rpc_codec::data_plane_error::DataPlaneErrorCode; @@ -40,6 +40,16 @@ pub struct ExecuteRequest { /// non-transactional dispatch. Lets the receiver resolve the per-transaction /// staging overlay for the id on the remote node. pub txn_id: Option, + /// The vShard whose owning core runs a vShard-scoped plan (a transaction + /// meta-op such as `StageWrite` or `ResolveTxn`). The receiver sends such a + /// plan to that one core. `None` for every other plan, which fans across + /// all local cores. Every vShard id is valid, so absence is `None`, never + /// a sentinel id. + pub vshard_id: Option, + /// Raft groups a linearizable read leg observes. The receiver takes a read + /// index for each and applies through it before it reads. Empty for a + /// write or a read that accepts this replica as it is. + pub read_groups: Vec, } /// Response to an `ExecuteRequest`. @@ -99,6 +109,13 @@ pub enum TypedClusterError { constraint: String, detail: String, }, + /// A Calvin transaction the sequencer aborted, with the verdict's reason. + /// The coordinator rebuilds the error a local submit returns, so a routed + /// abort keeps its SQLSTATE and message. Appended last: variant order is + /// the wire ABI. + CalvinAborted { + reason: crate::calvin::AbortReason, + }, } /// One streamed chunk of an `ExecuteStreamRequest` result. diff --git a/nodedb-cluster/src/rpc_codec/frame_refusal.rs b/nodedb-cluster/src/rpc_codec/frame_refusal.rs new file mode 100644 index 000000000..ba6e679e6 --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/frame_refusal.rs @@ -0,0 +1,63 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! FrameRefusal wire type and codec: the answer to a request frame the +//! receiver's replay window refused. +//! +//! The frame's MAC verified, so the sender is who it says. Its sequence +//! number was below the receiver's window or seen before. The receiver +//! answers on the same stream instead of dropping it: the sender learns the +//! frame was refused, not that the link failed. It retries under a fresh +//! sequence number, and its circuit breaker does not count the refusal +//! against the peer's health. + +use super::discriminants::RPC_FRAME_REFUSAL; +use super::header::write_frame; +use super::raft_rpc::RaftRpc; +use crate::error::{ClusterError, Result}; + +/// A request frame the receiver refused before it read the request. +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct FrameRefusal { + /// Why the receiver refused the frame, as its replay window put it. + pub detail: String, +} + +pub(super) fn encode_frame_refusal(msg: &FrameRefusal, out: &mut Vec) -> Result<()> { + let bytes = rkyv::to_bytes::(msg) + .map(|b| b.to_vec()) + .map_err(|e| ClusterError::Codec { + detail: format!("rkyv serialize: {e}"), + })?; + write_frame(RPC_FRAME_REFUSAL, &bytes, out) +} + +pub(super) fn decode_frame_refusal(payload: &[u8]) -> Result { + let mut aligned = rkyv::util::AlignedVec::<16>::with_capacity(payload.len()); + aligned.extend_from_slice(payload); + let refusal = rkyv::from_bytes::(&aligned).map_err(|e| { + ClusterError::Codec { + detail: format!("rkyv deserialize FrameRefusal: {e}"), + } + })?; + Ok(RaftRpc::FrameRefused(refusal)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster_epoch::ClusterEpochState; + use crate::rpc_codec::{decode, encode}; + + #[test] + fn a_refusal_survives_the_wire() { + let epoch = ClusterEpochState::default(); + let refusal = FrameRefusal { + detail: "peer 1 sent stale sequence 9, window high is 5000".into(), + }; + let bytes = encode(&RaftRpc::FrameRefused(refusal.clone()), &epoch).expect("encode"); + match decode(&bytes, &epoch).expect("decode") { + RaftRpc::FrameRefused(back) => assert_eq!(back, refusal), + other => panic!("decoded the wrong variant: {other:?}"), + } + } +} diff --git a/nodedb-cluster/src/rpc_codec/leader_status.rs b/nodedb-cluster/src/rpc_codec/leader_status.rs new file mode 100644 index 000000000..d4e070882 --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/leader_status.rs @@ -0,0 +1,136 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! LeaderStatusRequest / LeaderStatusResponse wire types and codecs. +//! +//! A node asks another node which leader it knows for a Raft group. The +//! receiver answers from its own Raft state, with no quorum round: the leader +//! it knows and the term that leader leads. A node that leads answers with +//! itself. A routing hint needs no linearizable confirmation: the routing +//! table's term rules keep a stale answer from moving a hint backwards, and +//! a request sent to a node that no longer leads is redirected. +//! +//! A node that leads also answers the group's membership: its Raft voters +//! and learners. A leader changes its membership only by applying a +//! committed conf change, so its membership holds only committed changes. +//! A node that does not lead answers no membership. + +use super::discriminants::*; +use super::header::write_frame; +use super::raft_rpc::RaftRpc; +use crate::error::{ClusterError, Result}; + +/// Ask a node which leader it knows for `group_id`. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct LeaderStatusRequest { + pub group_id: u64, +} + +/// The receiver's own view of `group_id`'s leader. +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct LeaderStatusResponse { + /// The leader the receiver knows, `0` when it knows none or hosts no + /// replica of the group. + pub leader: u64, + /// The term `leader` leads. `0` when `leader` is `0`. + pub term: u64, + /// The group's membership, when the receiver leads it. `None` from a + /// receiver that does not lead. + pub membership: Option, +} + +/// A leader's voters and learners of a group, sorted ascending. +#[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct LeaderMembership { + /// Voting members, the leader included. + pub voters: Vec, + /// Non-voting learners. + pub learners: Vec, +} + +macro_rules! to_bytes { + ($msg:expr) => { + rkyv::to_bytes::($msg) + .map(|b| b.to_vec()) + .map_err(|e| ClusterError::Codec { + detail: format!("rkyv serialize: {e}"), + }) + }; +} + +macro_rules! from_bytes { + ($payload:expr, $T:ty, $name:expr) => {{ + let mut aligned = rkyv::util::AlignedVec::<16>::with_capacity($payload.len()); + aligned.extend_from_slice($payload); + rkyv::from_bytes::<$T, rkyv::rancor::Error>(&aligned).map_err(|e| ClusterError::Codec { + detail: format!("rkyv deserialize {}: {e}", $name), + }) + }}; +} + +pub(super) fn encode_leader_status_req(msg: &LeaderStatusRequest, out: &mut Vec) -> Result<()> { + write_frame(RPC_LEADER_STATUS_REQ, &to_bytes!(msg)?, out) +} +pub(super) fn encode_leader_status_resp( + msg: &LeaderStatusResponse, + out: &mut Vec, +) -> Result<()> { + write_frame(RPC_LEADER_STATUS_RESP, &to_bytes!(msg)?, out) +} + +pub(super) fn decode_leader_status_req(payload: &[u8]) -> Result { + Ok(RaftRpc::LeaderStatusRequest(from_bytes!( + payload, + LeaderStatusRequest, + "LeaderStatusRequest" + )?)) +} +pub(super) fn decode_leader_status_resp(payload: &[u8]) -> Result { + Ok(RaftRpc::LeaderStatusResponse(from_bytes!( + payload, + LeaderStatusResponse, + "LeaderStatusResponse" + )?)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::cluster_epoch::ClusterEpochState; + use crate::rpc_codec::{decode, encode}; + + fn roundtrip(rpc: RaftRpc) -> RaftRpc { + let epoch = ClusterEpochState::default(); + let encoded = encode(&rpc, &epoch).expect("encode"); + decode(&encoded, &epoch).expect("decode") + } + + #[test] + fn a_request_and_a_response_survive_the_wire() { + match roundtrip(RaftRpc::LeaderStatusRequest(LeaderStatusRequest { + group_id: 7, + })) { + RaftRpc::LeaderStatusRequest(req) => assert_eq!(req.group_id, 7), + other => panic!("decoded the wrong variant: {other:?}"), + } + for status in [ + LeaderStatusResponse { + leader: 3, + term: 9, + membership: None, + }, + LeaderStatusResponse { + leader: 3, + term: 9, + membership: Some(LeaderMembership { + voters: vec![1, 3], + learners: vec![4], + }), + }, + ] { + match roundtrip(RaftRpc::LeaderStatusResponse(status.clone())) { + RaftRpc::LeaderStatusResponse(resp) => assert_eq!(resp, status), + other => panic!("decoded the wrong variant: {other:?}"), + } + } + } +} diff --git a/nodedb-cluster/src/rpc_codec/metadata.rs b/nodedb-cluster/src/rpc_codec/metadata.rs index 34886d868..a0246e374 100644 --- a/nodedb-cluster/src/rpc_codec/metadata.rs +++ b/nodedb-cluster/src/rpc_codec/metadata.rs @@ -11,6 +11,8 @@ use crate::error::{ClusterError, Result}; #[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] pub struct MetadataProposeRequest { pub bytes: Vec, + /// Whether the leader stamps `bytes` as it appends them. + pub stamp: bool, } /// Response to a forwarded metadata-group proposal. @@ -19,6 +21,9 @@ pub struct MetadataProposeResponse { pub success: bool, pub log_index: u64, pub leader_hint: Option, + /// The refusing node's term, which `leader_hint` is known at. `0` on + /// success and on a refusal that is not `NotLeader`. + pub leader_term: u64, pub error_message: String, } @@ -28,15 +33,19 @@ impl MetadataProposeResponse { success: true, log_index, leader_hint: None, + leader_term: 0, error_message: String::new(), } } - pub fn err(message: impl Into, leader_hint: Option) -> Self { + /// A refusal. `leader_hint` is the leader the refusing node knows at + /// `leader_term`. + pub fn err(message: impl Into, leader_hint: Option, leader_term: u64) -> Self { Self { success: false, log_index: 0, leader_hint, + leader_term, error_message: message.into(), } } diff --git a/nodedb-cluster/src/rpc_codec/mod.rs b/nodedb-cluster/src/rpc_codec/mod.rs index b38419f8c..9e72c7647 100644 --- a/nodedb-cluster/src/rpc_codec/mod.rs +++ b/nodedb-cluster/src/rpc_codec/mod.rs @@ -10,19 +10,23 @@ pub mod auth_envelope; pub mod auth_lease; +pub mod calvin_parts; pub mod calvin_submit; pub mod cluster_mgmt; pub mod data_plane_error; pub mod data_propose; pub mod discriminants; pub mod execute; +pub mod frame_refusal; pub mod header; +pub mod leader_status; pub mod mac; pub mod metadata; pub mod peer_seq; pub mod raft_msgs; pub mod raft_rpc; pub mod read_index; +pub mod request_refusal; pub mod reservation; pub mod shard_error; pub mod shuffle; @@ -36,6 +40,7 @@ pub use auth_lease::{ AuthBarrierOutcome, AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewOutcome, AuthLeaseRenewRequest, AuthLeaseRenewResponse, GroupCoverage, }; +pub use calvin_parts::{CalvinPartsRequest, CalvinPartsResponse, MAX_PARTS_BATCH_BYTES}; pub use calvin_submit::{ SubmitCalvinInboxRequest, SubmitCalvinInboxResponse, SubmitCalvinTxnRequest, SubmitCalvinTxnResponse, @@ -52,12 +57,15 @@ pub use execute::{ DescriptorVersionEntry, ExecuteRequest, ExecuteResponse, ExecuteStreamChunk, ExecuteStreamEnd, PLAN_DECODE_FAILED, TypedClusterError, }; +pub use frame_refusal::FrameRefusal; pub use header::{HEADER_SIZE, MAX_RPC_PAYLOAD_SIZE}; +pub use leader_status::{LeaderMembership, LeaderStatusRequest, LeaderStatusResponse}; pub use mac::{MAC_LEN, MacKey}; pub use metadata::{MetadataProposeRequest, MetadataProposeResponse}; -pub use peer_seq::{PeerSeqSender, PeerSeqWindow, REPLAY_WINDOW}; +pub use peer_seq::{BOOT_EPOCH_SHIFT, PeerSeqSender, PeerSeqWindow, REPLAY_WINDOW}; pub use raft_rpc::{RaftRpc, decode, encode, frame_size}; pub use read_index::{ReadIndexOutcome, ReadIndexRequest, ReadIndexResponse}; +pub use request_refusal::{RefusalReason, RequestRefusal}; pub use reservation::{ ReleaseReservationRequest, ReleaseReservationResponse, ReserveReadRequest, ReserveReadResponse, }; diff --git a/nodedb-cluster/src/rpc_codec/peer_seq.rs b/nodedb-cluster/src/rpc_codec/peer_seq.rs index 84d399827..90dc143b0 100644 --- a/nodedb-cluster/src/rpc_codec/peer_seq.rs +++ b/nodedb-cluster/src/rpc_codec/peer_seq.rs @@ -1,29 +1,33 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Per-peer monotonic sequence counters and a 64-entry sliding-window -//! replay detector. +//! Per-peer monotonic sequence counters and a [`REPLAY_WINDOW`]-entry +//! sliding-window replay detector. //! //! # Outbound: [`PeerSeqSender`] //! //! Each local-node-to-peer direction has a distinct counter. Sent frames -//! carry strictly-increasing sequence numbers starting from 1. `0` is -//! reserved as a sentinel meaning "never sent". +//! carry strictly-increasing sequence numbers. Each boot of a node sends in +//! its own range, above its durable boot epoch shifted by +//! [`BOOT_EPOCH_SHIFT`], so a restarted node sends above every number an +//! earlier boot sent. `0` is reserved as a sentinel meaning "never sent". //! //! # Inbound: [`PeerSeqWindow`] //! -//! A 64-bit bitmap anchored at `last_accepted_seq`. Frame with sequence -//! `n` is: +//! A [`REPLAY_WINDOW`]-bit bitmap anchored at `last_accepted_seq`. Frame +//! with sequence `n` is: //! - accepted and window advanced if `n > last_accepted_seq` -//! - accepted and bit set if `last_accepted_seq - 63 <= n < last_accepted_seq` -//! and the bit was previously unset +//! - accepted and bit set if `last_accepted_seq - (REPLAY_WINDOW - 1) <= n < +//! last_accepted_seq` and the bit was previously unset //! - rejected as replay if the bit was already set, or if `n` is older //! than the window //! -//! The window model is identical to IPsec AH/ESP (RFC 4303 §3.4.3). -//! 64 entries is the standard default — large enough that legitimate -//! reordering in-flight (rare over one QUIC stream but possible across -//! QUIC streams) is tolerated, and small enough to fit in a single -//! `u64`. +//! The window model is IPsec AH/ESP's (RFC 4303 §3.4.3), with a wider +//! window. A sender numbers every frame from one counter, to every peer, +//! and runs many RPCs at once, each on its own QUIC stream. Frames reach a +//! peer out of order by up to the number the sender numbered while they +//! were in flight, across all its peers. A 64-entry window rejects a burst +//! of concurrent RPCs as stale; [`REPLAY_WINDOW`] covers every frame the +//! sender's concurrent streams hold at once. use std::collections::HashMap; use std::sync::RwLock; @@ -31,8 +35,19 @@ use std::sync::atomic::{AtomicU64, Ordering}; use crate::error::{ClusterError, Result}; -/// Size of the inbound replay-detection window. -pub const REPLAY_WINDOW: u64 = 64; +/// Size of the inbound replay-detection window, in sequence numbers. It is +/// 16 times a connection's default limit of concurrent streams, so frames a +/// sender numbered for other peers in the same moment fit as well. +pub const REPLAY_WINDOW: u64 = 4096; + +/// 64-bit words in one peer's window bitmap. +const WINDOW_WORDS: usize = (REPLAY_WINDOW / 64) as usize; + +/// Bits of a sequence number below the boot epoch. A boot owns `2^44` +/// (about 1.8 × 10^13) sequence numbers: at 100,000 frames a second that +/// lasts over five years of one boot's sending. Above the shift, `2^20` +/// (about a million) boots fit. +pub const BOOT_EPOCH_SHIFT: u32 = 44; /// Outbound monotonic counter for this `AuthContext`. One counter total /// — not one per target — because the receiver's replay window is keyed @@ -41,18 +56,48 @@ pub const REPLAY_WINDOW: u64 = 64; /// any node that receives from both: seq=1 from target=A and seq=1 from /// target=B collide in the receiver's `window[sender_id]`. A single /// counter makes every outbound seq globally unique per sender. -#[derive(Default, Debug)] +#[derive(Debug)] pub struct PeerSeqSender { counter: AtomicU64, } +impl Default for PeerSeqSender { + fn default() -> Self { + Self::new() + } +} + impl PeerSeqSender { + /// A counter whose first sequence number is 1. A node moves it into its + /// boot's own range with [`Self::enter_boot_epoch`] before it sends. pub fn new() -> Self { - Self::default() + Self::starting_after(0) } - /// Reserve and return the next outbound sequence number. Starts at 1 - /// and is strictly increasing across all targets for this sender. + /// A counter whose first sequence number is `floor + 1`. + pub fn starting_after(floor: u64) -> Self { + Self { + counter: AtomicU64::new(floor), + } + } + + /// Move the counter into boot `epoch`'s range: its next sequence number + /// is above `epoch << BOOT_EPOCH_SHIFT`. Never moves it back. + /// + /// A peer keeps its window for this node across this node's restart. A + /// restarted node that counted from 1 again would send only sequence + /// numbers the peer rejects as stale. The boot epoch rises durably at + /// every boot (see `ClusterCatalog::advance_boot_epoch`), so each boot's + /// range starts above every number an earlier boot sent: a boot sends + /// fewer than `1 << BOOT_EPOCH_SHIFT` frames. No clock is read. + pub fn enter_boot_epoch(&self, epoch: u64) { + let floor = epoch.saturating_mul(1u64 << BOOT_EPOCH_SHIFT); + self.counter.fetch_max(floor, Ordering::AcqRel); + } + + /// Reserve and return the next outbound sequence number. Starts one + /// above the counter's floor and is strictly increasing across all + /// targets for this sender. pub fn next(&self) -> u64 { self.counter.fetch_add(1, Ordering::Relaxed) + 1 } @@ -72,13 +117,59 @@ pub struct PeerSeqWindow { } /// Sliding-window state for one peer. -#[derive(Default, Debug, Clone, Copy)] +#[derive(Debug, Clone)] struct WindowState { /// Highest accepted sequence seen from this peer. 0 if none yet. high: u64, - /// Bitmap of accepted sequences in `[high - 63, high]`. Bit 0 is - /// `high`, bit 63 is `high - 63`. - mask: u64, + /// Ring bitmap of accepted sequences in `[high - (REPLAY_WINDOW - 1), + /// high]`. Sequence `s` sits at bit `s % REPLAY_WINDOW`. + seen: [u64; WINDOW_WORDS], +} + +impl Default for WindowState { + fn default() -> Self { + Self { + high: 0, + seen: [0; WINDOW_WORDS], + } + } +} + +impl WindowState { + /// The word and bit of sequence `seq` in the ring. + fn slot(seq: u64) -> (usize, u64) { + let pos = seq % REPLAY_WINDOW; + ((pos / 64) as usize, 1u64 << (pos % 64)) + } + + fn is_set(&self, seq: u64) -> bool { + let (word, bit) = Self::slot(seq); + self.seen[word] & bit != 0 + } + + fn set(&mut self, seq: u64) { + let (word, bit) = Self::slot(seq); + self.seen[word] |= bit; + } + + fn clear(&mut self, seq: u64) { + let (word, bit) = Self::slot(seq); + self.seen[word] &= !bit; + } + + /// Move the window's top to `seq`, above `high`: every slot the window + /// gains is cleared, as no sequence there was accepted yet. + fn advance(&mut self, seq: u64) { + let delta = seq - self.high; + if delta >= REPLAY_WINDOW { + self.seen = [0; WINDOW_WORDS]; + } else { + for gained in self.high + 1..=seq { + self.clear(gained); + } + } + self.high = seq; + } } impl PeerSeqWindow { @@ -102,14 +193,9 @@ impl PeerSeqWindow { let state = guard.entry(peer_id).or_default(); if seq > state.high { - // Frame advances the window. Shift by the delta and set bit 0. - let delta = seq - state.high; - state.mask = if delta >= REPLAY_WINDOW { - 1 - } else { - (state.mask << delta) | 1 - }; - state.high = seq; + // Frame advances the window. + state.advance(seq); + state.set(seq); return Ok(()); } @@ -123,8 +209,7 @@ impl PeerSeqWindow { ), }); } - let bit = 1u64 << offset; - if state.mask & bit != 0 { + if state.is_set(seq) { return Err(ClusterError::Codec { detail: format!( "peer {peer_id} replayed sequence {seq} (window high {})", @@ -132,7 +217,7 @@ impl PeerSeqWindow { ), }); } - state.mask |= bit; + state.set(seq); Ok(()) } @@ -148,13 +233,57 @@ mod tests { use super::*; #[test] - fn outbound_counter_starts_at_one() { - let s = PeerSeqSender::new(); + fn outbound_counter_starts_above_its_floor() { + let s = PeerSeqSender::starting_after(0); assert_eq!(s.next(), 1); assert_eq!(s.next(), 2); assert_eq!(s.next(), 3); } + /// A restarted sender enters the next boot epoch and starts above + /// everything the previous boot sent, so the peer's window accepts its + /// first frame. + #[test] + fn a_restarted_sender_is_accepted_by_the_peers_window() { + let window = PeerSeqWindow::new(); + let before = PeerSeqSender::new(); + before.enter_boot_epoch(1); + for _ in 0..100 { + window.accept(9, before.next()).unwrap(); + } + let after = PeerSeqSender::new(); + after.enter_boot_epoch(2); + window.accept(9, after.next()).unwrap(); + } + + /// The start depends on the boot epoch alone. A restart whose wall clock + /// reads earlier than the previous boot's still starts above it: no + /// clock goes into the sequence number. + #[test] + fn a_restart_with_an_earlier_clock_is_still_accepted() { + let window = PeerSeqWindow::new(); + // The previous boot ran far into its range, as a long uptime does. + let before = PeerSeqSender::starting_after((7u64 << BOOT_EPOCH_SHIFT) + 1_000_000); + for _ in 0..10 { + window.accept(9, before.next()).unwrap(); + } + // The next boot comes up with its clock set back. Its boot epoch + // is the durable counter plus one. + let after = PeerSeqSender::new(); + after.enter_boot_epoch(8); + window.accept(9, after.next()).unwrap(); + } + + #[test] + fn entering_a_boot_epoch_never_moves_the_counter_back() { + let s = PeerSeqSender::new(); + s.enter_boot_epoch(3); + let first = s.next(); + assert_eq!(first, (3u64 << BOOT_EPOCH_SHIFT) + 1); + s.enter_boot_epoch(2); + assert_eq!(s.next(), first + 1); + } + #[test] fn outbound_counter_is_single_across_all_targets() { // The outbound counter is intentionally shared across targets: the @@ -163,7 +292,7 @@ mod tests { // single monotonic counter guarantees every emitted seq is unique // from the receiver's point of view regardless of which target // the sender was aiming at. - let s = PeerSeqSender::new(); + let s = PeerSeqSender::starting_after(0); assert_eq!(s.next(), 1); assert_eq!(s.next(), 2); assert_eq!(s.next(), 3); @@ -218,12 +347,39 @@ mod tests { #[test] fn window_rejects_stale_outside_window() { let w = PeerSeqWindow::new(); - w.accept(1, 100).unwrap(); - // Window is [37, 100]. seq=36 is stale. - let err = w.accept(1, 36).unwrap_err(); - assert!(err.to_string().contains("stale sequence 36")); - // seq=37 is inside the window edge and acceptable. - w.accept(1, 37).unwrap(); + let high = 10_000; + w.accept(1, high).unwrap(); + // The window is [high - REPLAY_WINDOW + 1, high]. + let stale = high - REPLAY_WINDOW; + let err = w.accept(1, stale).unwrap_err(); + assert!(err.to_string().contains(&format!("stale sequence {stale}"))); + // The window's lowest sequence is acceptable. + w.accept(1, stale + 1).unwrap(); + } + + /// A burst of concurrent RPCs arrives out of order by more than 64 + /// frames. Every frame is accepted once. + #[test] + fn a_reordered_burst_wider_than_64_frames_is_accepted() { + let w = PeerSeqWindow::new(); + let burst: Vec = (1..=600).collect(); + for &seq in burst.iter().rev() { + w.accept(3, seq).unwrap(); + } + for &seq in &burst { + assert!(w.accept(3, seq).is_err(), "sequence {seq} replays"); + } + } + + /// A slot the window moves past is free again for the sequence that + /// lands on it one window later. + #[test] + fn a_ring_slot_is_reused_after_the_window_moves_past_it() { + let w = PeerSeqWindow::new(); + w.accept(1, 5).unwrap(); + w.accept(1, 5 + REPLAY_WINDOW).unwrap(); + w.accept(1, 4 + REPLAY_WINDOW).unwrap(); + assert!(w.accept(1, 5 + REPLAY_WINDOW).is_err()); } #[test] @@ -231,7 +387,7 @@ mod tests { let w = PeerSeqWindow::new(); w.accept(1, 1).unwrap(); w.accept(1, 2).unwrap(); - w.accept(1, 100).unwrap(); + w.accept(1, 100 + REPLAY_WINDOW).unwrap(); // Sequences 1, 2 are now outside the window anchored at 100 and // must be rejected on replay (not accepted as fresh within mask). let err = w.accept(1, 1).unwrap_err(); diff --git a/nodedb-cluster/src/rpc_codec/raft_msgs.rs b/nodedb-cluster/src/rpc_codec/raft_msgs.rs index 6e60c4fed..4518f7c98 100644 --- a/nodedb-cluster/src/rpc_codec/raft_msgs.rs +++ b/nodedb-cluster/src/rpc_codec/raft_msgs.rs @@ -175,10 +175,14 @@ mod tests { ], leader_commit: 98, group_id: 7, + round: 11, + replicated_floor: 7, }; match roundtrip(RaftRpc::AppendEntriesRequest(req)) { RaftRpc::AppendEntriesRequest(d) => { assert_eq!(d.term, 5); + assert_eq!(d.round, 11); + assert_eq!(d.replicated_floor, 7); assert_eq!(d.entries.len(), 2); assert_eq!(d.entries[0].data, b"put x=1"); } @@ -196,6 +200,8 @@ mod tests { entries: vec![], leader_commit: 8, group_id: 0, + round: 2, + replicated_floor: 0, }; match roundtrip(RaftRpc::AppendEntriesRequest(req)) { RaftRpc::AppendEntriesRequest(d) => { @@ -212,11 +218,15 @@ mod tests { term: 5, success: true, last_log_index: 100, + round: 11, + needs_snapshot: true, }; match roundtrip(RaftRpc::AppendEntriesResponse(resp)) { RaftRpc::AppendEntriesResponse(d) => { assert_eq!(d.term, 5); + assert_eq!(d.round, 11); assert!(d.success); + assert!(d.needs_snapshot); } other => panic!("expected AppendEntriesResponse, got {other:?}"), } @@ -230,10 +240,12 @@ mod tests { last_log_index: 200, last_log_term: 9, group_id: 42, + transfer: true, }; match roundtrip(RaftRpc::RequestVoteRequest(req)) { RaftRpc::RequestVoteRequest(d) => { assert_eq!(d.term, 10); + assert!(d.transfer); assert_eq!(d.group_id, 42); } other => panic!("expected RequestVoteRequest, got {other:?}"), @@ -306,6 +318,8 @@ mod tests { done: false, group_id: 3, total_size: 0, + voters: Vec::new(), + learners: Vec::new(), }; match roundtrip(RaftRpc::InstallSnapshotRequest(req)) { RaftRpc::InstallSnapshotRequest(d) => { @@ -329,10 +343,14 @@ mod tests { done: true, group_id: 3, total_size: 0, + voters: vec![1, 2], + learners: vec![4], }; match roundtrip(RaftRpc::InstallSnapshotRequest(req)) { RaftRpc::InstallSnapshotRequest(d) => { assert!(d.done); + assert_eq!(d.voters, vec![1, 2]); + assert_eq!(d.learners, vec![4]); assert_eq!(d.offset, 4096); } other => panic!("expected InstallSnapshotRequest, got {other:?}"), @@ -378,6 +396,8 @@ mod tests { done: false, group_id: 0, total_size: 0, + voters: Vec::new(), + learners: Vec::new(), }; match roundtrip(RaftRpc::InstallSnapshotRequest(req)) { RaftRpc::InstallSnapshotRequest(d) => { diff --git a/nodedb-cluster/src/rpc_codec/raft_rpc.rs b/nodedb-cluster/src/rpc_codec/raft_rpc.rs index b01149021..5e6ebb452 100644 --- a/nodedb-cluster/src/rpc_codec/raft_rpc.rs +++ b/nodedb-cluster/src/rpc_codec/raft_rpc.rs @@ -10,6 +10,7 @@ use nodedb_raft::message::{ use super::auth_lease::{ AuthBarrierRequest, AuthBarrierResponse, AuthLeaseRenewRequest, AuthLeaseRenewResponse, }; +use super::calvin_parts::{CalvinPartsRequest, CalvinPartsResponse}; use super::calvin_submit::{ SubmitCalvinInboxRequest, SubmitCalvinInboxResponse, SubmitCalvinTxnRequest, SubmitCalvinTxnResponse, @@ -20,9 +21,12 @@ use super::cluster_mgmt::{ use super::data_propose::{DataProposeRequest, DataProposeResponse}; use super::discriminants::*; use super::execute::{ExecuteRequest, ExecuteResponse, ExecuteStreamChunk, ExecuteStreamEnd}; +use super::frame_refusal::FrameRefusal; use super::header::HEADER_SIZE; +use super::leader_status::{LeaderStatusRequest, LeaderStatusResponse}; use super::metadata::{MetadataProposeRequest, MetadataProposeResponse}; use super::read_index::{ReadIndexRequest, ReadIndexResponse}; +use super::request_refusal::RequestRefusal; use super::reservation::{ ReleaseReservationRequest, ReleaseReservationResponse, ReserveReadRequest, ReserveReadResponse, }; @@ -34,8 +38,9 @@ use super::shuffle::{ use super::surrogate::{AssignSurrogateRequest, AssignSurrogateResponse}; use super::vshard::VShardRefusal; use super::{ - auth_lease, calvin_submit, cluster_mgmt, data_propose, execute, metadata, raft_msgs, - read_index, reservation, shuffle, surrogate, vshard, + auth_lease, calvin_parts, calvin_submit, cluster_mgmt, data_propose, execute, frame_refusal, + leader_status, metadata, raft_msgs, read_index, request_refusal, reservation, shuffle, + surrogate, vshard, }; use crate::error::{ClusterError, Result}; use crate::wire_version::{unwrap_bytes_versioned, wrap_bytes_versioned}; @@ -66,7 +71,7 @@ pub enum RaftRpc { // Topology broadcast TopologyUpdate(TopologyUpdate), TopologyAck(TopologyAck), - // Discriminants 13/14 (ForwardRequest/ForwardResponse) retired in C-δ.6. + // Discriminants 13/14 (ForwardRequest/ForwardResponse) are retired. // VShardEnvelope VShardEnvelope(Vec), // Metadata-group proposal forwarding (group 0) @@ -130,6 +135,11 @@ pub enum RaftRpc { // error. SubmitCalvinInboxRequest(SubmitCalvinInboxRequest), SubmitCalvinInboxResponse(SubmitCalvinInboxResponse), + // Streamed parts of a multi-part Calvin transaction. A coordinator sends + // one bounded batch of parts to the sequencer leader, which queues them + // and answers how far the stream got. + CalvinPartsRequest(CalvinPartsRequest), + CalvinPartsResponse(CalvinPartsResponse), // Routed reserve-read (Calvin OLLP). A coordinator sends a // `ReserveReadRequest` carrying a msgpack-encoded `LockKeyWire` to the // sequencer-group leader; the leader assign-only reserves the read lock @@ -152,6 +162,10 @@ pub enum RaftRpc { // for a read index confirmed against a quorum. ReadIndexRequest(ReadIndexRequest), ReadIndexResponse(ReadIndexResponse), + // Leader status. A node asks another which leader it knows for a group; + // the receiver answers from its own Raft state, with no quorum round. + LeaderStatusRequest(LeaderStatusRequest), + LeaderStatusResponse(LeaderStatusResponse), // Authorization lease renewal and the writer-side barrier, both answered // by the metadata group leader. AuthLeaseRenewRequest(AuthLeaseRenewRequest), @@ -161,6 +175,10 @@ pub enum RaftRpc { // Answer to a `VShardEnvelope` request whose handler failed. It carries // the handler's typed error. VShardRefusal(VShardRefusal), + // Answer to a request frame the receiver's replay window refused. + FrameRefused(FrameRefusal), + // Answer to a request whose handler failed. It carries the typed reason. + RequestRefused(RequestRefusal), } /// Encode a [`RaftRpc`] into a framed binary message stamped with `epoch`. @@ -220,6 +238,8 @@ pub fn encode(rpc: &RaftRpc, epoch: &crate::cluster_epoch::ClusterEpochState) -> RaftRpc::SubmitCalvinInboxResponse(m) => { calvin_submit::encode_submit_calvin_inbox_resp(m, &mut out) } + RaftRpc::CalvinPartsRequest(m) => calvin_parts::encode_calvin_parts_req(m, &mut out), + RaftRpc::CalvinPartsResponse(m) => calvin_parts::encode_calvin_parts_resp(m, &mut out), RaftRpc::ReserveReadRequest(m) => reservation::encode_reserve_read_req(m, &mut out), RaftRpc::ReserveReadResponse(m) => reservation::encode_reserve_read_resp(m, &mut out), RaftRpc::ReleaseReservationRequest(m) => { @@ -232,11 +252,15 @@ pub fn encode(rpc: &RaftRpc, epoch: &crate::cluster_epoch::ClusterEpochState) -> RaftRpc::DataProposeResponse(m) => data_propose::encode_data_propose_resp(m, &mut out), RaftRpc::ReadIndexRequest(m) => read_index::encode_read_index_req(m, &mut out), RaftRpc::ReadIndexResponse(m) => read_index::encode_read_index_resp(m, &mut out), + RaftRpc::LeaderStatusRequest(m) => leader_status::encode_leader_status_req(m, &mut out), + RaftRpc::LeaderStatusResponse(m) => leader_status::encode_leader_status_resp(m, &mut out), RaftRpc::AuthLeaseRenewRequest(m) => auth_lease::encode_renew_req(m, &mut out), RaftRpc::AuthLeaseRenewResponse(m) => auth_lease::encode_renew_resp(m, &mut out), RaftRpc::AuthBarrierRequest(m) => auth_lease::encode_barrier_req(m, &mut out), RaftRpc::AuthBarrierResponse(m) => auth_lease::encode_barrier_resp(m, &mut out), RaftRpc::VShardRefusal(m) => vshard::encode_vshard_refusal(m, &mut out), + RaftRpc::FrameRefused(m) => frame_refusal::encode_frame_refusal(m, &mut out), + RaftRpc::RequestRefused(m) => request_refusal::encode_request_refusal(m, &mut out), }?; super::header::stamp_epoch(&mut out, epoch)?; Ok(out) @@ -300,7 +324,7 @@ pub fn decode(data: &[u8], epoch: &crate::cluster_epoch::ClusterEpochState) -> R RPC_FORWARD_REQ | RPC_FORWARD_RESP => Err(ClusterError::Codec { detail: format!( "rpc_type {rpc_type} is a retired wire variant (ForwardRequest/ForwardResponse, \ - retired in C-δ.6); upgrade all cluster nodes to remove this peer" + retired); upgrade all cluster nodes to remove this peer" ), }), RPC_VSHARD_ENVELOPE => vshard::decode_vshard_envelope(payload), @@ -326,6 +350,8 @@ pub fn decode(data: &[u8], epoch: &crate::cluster_epoch::ClusterEpochState) -> R RPC_SUBMIT_CALVIN_TXN_RESP => calvin_submit::decode_submit_calvin_txn_resp(payload), RPC_SUBMIT_CALVIN_INBOX_REQ => calvin_submit::decode_submit_calvin_inbox_req(payload), RPC_SUBMIT_CALVIN_INBOX_RESP => calvin_submit::decode_submit_calvin_inbox_resp(payload), + RPC_CALVIN_PARTS_REQ => calvin_parts::decode_calvin_parts_req(payload), + RPC_CALVIN_PARTS_RESP => calvin_parts::decode_calvin_parts_resp(payload), RPC_RESERVE_READ_REQ => reservation::decode_reserve_read_req(payload), RPC_RESERVE_READ_RESP => reservation::decode_reserve_read_resp(payload), RPC_RELEASE_RESERVATION_REQ => reservation::decode_release_reservation_req(payload), @@ -334,11 +360,15 @@ pub fn decode(data: &[u8], epoch: &crate::cluster_epoch::ClusterEpochState) -> R RPC_DATA_PROPOSE_RESP => data_propose::decode_data_propose_resp(payload), RPC_READ_INDEX_REQ => read_index::decode_read_index_req(payload), RPC_READ_INDEX_RESP => read_index::decode_read_index_resp(payload), + RPC_LEADER_STATUS_REQ => leader_status::decode_leader_status_req(payload), + RPC_LEADER_STATUS_RESP => leader_status::decode_leader_status_resp(payload), RPC_AUTH_LEASE_RENEW_REQ => auth_lease::decode_renew_req(payload), RPC_AUTH_LEASE_RENEW_RESP => auth_lease::decode_renew_resp(payload), RPC_AUTH_BARRIER_REQ => auth_lease::decode_barrier_req(payload), RPC_AUTH_BARRIER_RESP => auth_lease::decode_barrier_resp(payload), RPC_VSHARD_REFUSAL => vshard::decode_vshard_refusal(payload), + RPC_FRAME_REFUSAL => frame_refusal::decode_frame_refusal(payload), + RPC_REQUEST_REFUSAL => request_refusal::decode_request_refusal(payload), _ => Err(ClusterError::Codec { detail: format!("unknown rpc_type: {rpc_type}"), }), @@ -441,6 +471,8 @@ mod tests { term: 1, success: true, last_log_index: 5, + round: 3, + needs_snapshot: false, }); let encoded = encode(&rpc, &crate::cluster_epoch::ClusterEpochState::default()).unwrap(); let header: [u8; HEADER_SIZE] = encoded[..HEADER_SIZE].try_into().unwrap(); diff --git a/nodedb-cluster/src/rpc_codec/read_index.rs b/nodedb-cluster/src/rpc_codec/read_index.rs index 420fbff29..5b2f8a50f 100644 --- a/nodedb-cluster/src/rpc_codec/read_index.rs +++ b/nodedb-cluster/src/rpc_codec/read_index.rs @@ -24,11 +24,12 @@ pub struct ReadIndexRequest { /// The leader's answer to a [`ReadIndexRequest`]. #[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] pub enum ReadIndexOutcome { - /// A quorum confirmed the leader. Reads may be served at `read_index`. - Confirmed { read_index: u64 }, + /// A quorum confirmed the leader at `term`. Reads may be served at + /// `read_index`. + Confirmed { read_index: u64, term: u64 }, /// The receiver does not lead the group. `leader_hint` names the leader - /// it knows of. - NotLeader { leader_hint: Option }, + /// it knows of at `term`, its own term for the group. + NotLeader { leader_hint: Option, term: u64 }, /// The receiver leads the group, but no quorum answered in time. Timeout { waited_ms: u64 }, } @@ -111,11 +112,18 @@ mod tests { #[test] fn every_outcome_survives_the_wire() { for outcome in [ - ReadIndexOutcome::Confirmed { read_index: 42 }, + ReadIndexOutcome::Confirmed { + read_index: 42, + term: 3, + }, ReadIndexOutcome::NotLeader { leader_hint: Some(3), + term: 6, + }, + ReadIndexOutcome::NotLeader { + leader_hint: None, + term: 6, }, - ReadIndexOutcome::NotLeader { leader_hint: None }, ReadIndexOutcome::Timeout { waited_ms: 750 }, ] { let rpc = roundtrip(RaftRpc::ReadIndexResponse(ReadIndexResponse { diff --git a/nodedb-cluster/src/rpc_codec/request_refusal.rs b/nodedb-cluster/src/rpc_codec/request_refusal.rs new file mode 100644 index 000000000..ff9e37a36 --- /dev/null +++ b/nodedb-cluster/src/rpc_codec/request_refusal.rs @@ -0,0 +1,159 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! RequestRefusal wire type and codec: the answer to a request whose +//! handler failed. +//! +//! The request's MAC verified, so the sender is who it says and its link is +//! up. The receiver answers on the same stream instead of dropping it. The +//! sender learns that the peer refused this one request, not that the link +//! failed. Its circuit breaker counts the answer as a success, and it does +//! not resend the request. + +use super::discriminants::RPC_REQUEST_REFUSAL; +use super::header::write_frame; +use super::raft_rpc::RaftRpc; +use super::shard_error::ShardErrorWire; +use crate::error::{ClusterError, Result}; + +/// Why the receiver's handler refused a request. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub enum RefusalReason { + /// The request names a Raft group this node does not host. + GroupNotHosted { group_id: u64 }, + /// The handler failed with another error, in its typed wire form. + /// + /// The error is never a link failure. A handler error that is one, from + /// the receiver's own outbound call, crosses as `Untyped`. + Handler { error: ShardErrorWire }, +} + +/// A request the receiver's handler refused. +#[derive(Debug, Clone, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] +pub struct RequestRefusal { + pub reason: RefusalReason, +} + +impl From for RequestRefusal { + fn from(error: ClusterError) -> Self { + let reason = match error { + ClusterError::GroupNotFound { group_id } + | ClusterError::Raft(nodedb_raft::RaftError::GroupNotFound { group_id }) => { + RefusalReason::GroupNotHosted { group_id } + } + // The receiver's own link to a third node failed. To the sender + // that is an answer, so it must not read as its own link failing. + link if link.is_link_failure() => RefusalReason::Handler { + error: ShardErrorWire::Untyped { + detail: link.to_string(), + }, + }, + other => RefusalReason::Handler { + error: other.into(), + }, + }; + Self { reason } + } +} + +impl RequestRefusal { + /// The typed error the sender returns to its caller. + /// + /// `GroupNotHosted` becomes `GroupNotFound`. Neither result is a link + /// failure, and neither is retryable. + pub fn into_error(self) -> ClusterError { + match self.reason { + RefusalReason::GroupNotHosted { group_id } => ClusterError::GroupNotFound { group_id }, + RefusalReason::Handler { error } => ClusterError::from(error), + } + } +} + +pub(super) fn encode_request_refusal(msg: &RequestRefusal, out: &mut Vec) -> Result<()> { + let bytes = rkyv::to_bytes::(msg).map_err(|e| ClusterError::Codec { + detail: format!("rkyv serialize RequestRefusal: {e}"), + })?; + write_frame(RPC_REQUEST_REFUSAL, &bytes, out) +} + +pub(super) fn decode_request_refusal(payload: &[u8]) -> Result { + let mut aligned = rkyv::util::AlignedVec::<16>::with_capacity(payload.len()); + aligned.extend_from_slice(payload); + let refusal = + rkyv::from_bytes::(&aligned).map_err(|e| { + ClusterError::Codec { + detail: format!("rkyv deserialize RequestRefusal: {e}"), + } + })?; + Ok(RaftRpc::RequestRefused(refusal)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::circuit_breaker::RetryPolicy; + use crate::cluster_epoch::ClusterEpochState; + use crate::rpc_codec::{decode, encode}; + + /// Refuse with `error`, cross the wire, and rebuild the sender's error. + fn round_trip(error: ClusterError) -> ClusterError { + let epoch = ClusterEpochState::default(); + let rpc = RaftRpc::RequestRefused(RequestRefusal::from(error)); + let bytes = encode(&rpc, &epoch).expect("encode"); + match decode(&bytes, &epoch).expect("decode") { + RaftRpc::RequestRefused(refusal) => refusal.into_error(), + other => panic!("decoded the wrong variant: {other:?}"), + } + } + + #[test] + fn an_unhosted_group_survives_the_wire() { + let rebuilt = round_trip(ClusterError::GroupNotFound { group_id: 4 }); + assert!(matches!( + rebuilt, + ClusterError::GroupNotFound { group_id: 4 } + )); + assert!(!rebuilt.is_link_failure()); + assert!(!RetryPolicy::is_retryable(&rebuilt)); + } + + #[test] + fn a_raft_unhosted_group_is_the_same_refusal() { + let error = ClusterError::Raft(nodedb_raft::RaftError::GroupNotFound { group_id: 9 }); + match RequestRefusal::from(error).reason { + RefusalReason::GroupNotHosted { group_id } => assert_eq!(group_id, 9), + other => panic!("expected GroupNotHosted, got {other:?}"), + } + } + + #[test] + fn a_handler_link_error_never_reads_as_the_senders_link() { + for error in [ + ClusterError::Transport { + detail: "node 3 reset the stream".into(), + }, + ClusterError::CircuitOpen { + node_id: 3, + failures: 5, + }, + ClusterError::NodeUnreachable { node_id: 3 }, + ] { + let message = error.to_string(); + let rebuilt = round_trip(error); + assert!(!rebuilt.is_link_failure(), "{rebuilt:?}"); + assert!(!RetryPolicy::is_retryable(&rebuilt), "{rebuilt:?}"); + match rebuilt { + ClusterError::RemoteUntyped { detail } => assert_eq!(detail, message), + other => panic!("expected RemoteUntyped, got {other:?}"), + } + } + } + + #[test] + fn a_typed_handler_error_keeps_its_type() { + let rebuilt = round_trip(ClusterError::ReadIndexNotLeader { group_id: 2 }); + assert!(matches!( + rebuilt, + ClusterError::ReadIndexNotLeader { group_id: 2 } + )); + } +} diff --git a/nodedb-cluster/src/rpc_codec/shard_error/convert.rs b/nodedb-cluster/src/rpc_codec/shard_error/convert.rs index eda394f3b..a0b1f4c9f 100644 --- a/nodedb-cluster/src/rpc_codec/shard_error/convert.rs +++ b/nodedb-cluster/src/rpc_codec/shard_error/convert.rs @@ -135,7 +135,8 @@ impl From for ShardErrorWire { | ClusterError::VectorGather(_) | ClusterError::SpatialGather(_) | ClusterError::Bm25Gather(_) - | ClusterError::TsGather(_)) => Self::Untyped { + | ClusterError::TsGather(_) + | ClusterError::ShufflePush(_)) => Self::Untyped { detail: other.to_string(), }, } diff --git a/nodedb-cluster/src/rpc_codec/shard_error/raft.rs b/nodedb-cluster/src/rpc_codec/shard_error/raft.rs index f4d4512d6..26e6ba772 100644 --- a/nodedb-cluster/src/rpc_codec/shard_error/raft.rs +++ b/nodedb-cluster/src/rpc_codec/shard_error/raft.rs @@ -6,11 +6,13 @@ use nodedb_raft::RaftError; /// A `RaftError` carried across a node hop. `NotLeader` keeps its leader -/// hint, so the caller can chase the redirect. +/// hint and the term the responder knew it at, so the caller can chase the +/// redirect and rank it against the hint it holds. #[derive(Debug, Clone, PartialEq, Eq, rkyv::Archive, rkyv::Serialize, rkyv::Deserialize)] pub enum RaftErrorWire { NotLeader { leader_hint: Option, + term: u64, }, LogCompacted { requested: u64, @@ -48,7 +50,7 @@ pub enum RaftErrorWire { impl From for RaftErrorWire { fn from(error: RaftError) -> Self { match error { - RaftError::NotLeader { leader_hint } => Self::NotLeader { leader_hint }, + RaftError::NotLeader { leader_hint, term } => Self::NotLeader { leader_hint, term }, RaftError::LogCompacted { requested, first_available, @@ -79,7 +81,7 @@ impl From for RaftErrorWire { impl From for RaftError { fn from(wire: RaftErrorWire) -> Self { match wire { - RaftErrorWire::NotLeader { leader_hint } => Self::NotLeader { leader_hint }, + RaftErrorWire::NotLeader { leader_hint, term } => Self::NotLeader { leader_hint, term }, RaftErrorWire::LogCompacted { requested, first_available, diff --git a/nodedb-cluster/src/rpc_codec/shuffle.rs b/nodedb-cluster/src/rpc_codec/shuffle.rs index 6d41baf7c..379046d5d 100644 --- a/nodedb-cluster/src/rpc_codec/shuffle.rs +++ b/nodedb-cluster/src/rpc_codec/shuffle.rs @@ -109,6 +109,9 @@ pub struct ShuffleProduceRequest { pub deadline_remaining_ms: u64, pub trace_id: [u8; 16], pub descriptor_versions: Vec, + /// Raft groups the producer confirms before its scan, as a linearizable + /// read leg (see `ExecuteRequest::read_groups`). Empty otherwise. + pub read_groups: Vec, } /// Terminal reply to a [`ShuffleProduceRequest`]. @@ -589,6 +592,7 @@ mod tests { collection: "orders".into(), version: 11, }], + read_groups: vec![3, 8], }; let decoded = roundtrip_produce_req(req.clone()); assert_eq!(decoded.shuffle_id, req.shuffle_id); @@ -607,6 +611,7 @@ mod tests { assert_eq!(decoded.descriptor_versions.len(), 1); assert_eq!(decoded.descriptor_versions[0].collection, "orders"); assert_eq!(decoded.descriptor_versions[0].version, 11); + assert_eq!(decoded.read_groups, vec![3, 8]); } #[test] @@ -624,6 +629,7 @@ mod tests { deadline_remaining_ms: 1000, trace_id: [0u8; 16], descriptor_versions: vec![], + read_groups: vec![], }; let decoded = roundtrip_produce_req(req); assert!(decoded.keys.is_empty()); diff --git a/nodedb-cluster/src/rpc_codec/vshard.rs b/nodedb-cluster/src/rpc_codec/vshard.rs index 45c4bd98b..eb7dd85fe 100644 --- a/nodedb-cluster/src/rpc_codec/vshard.rs +++ b/nodedb-cluster/src/rpc_codec/vshard.rs @@ -108,11 +108,13 @@ mod tests { fn a_raft_redirect_keeps_its_leader_hint() { let error = ClusterError::Raft(nodedb_raft::RaftError::NotLeader { leader_hint: Some(5), + term: 4, }); assert!(matches!( round_trip(error), ClusterError::Raft(nodedb_raft::RaftError::NotLeader { - leader_hint: Some(5) + leader_hint: Some(5), + term: 4, }) )); } diff --git a/nodedb-cluster/src/subsystem/impls/mod.rs b/nodedb-cluster/src/subsystem/impls/mod.rs index cdf562b5f..d3e687cd6 100644 --- a/nodedb-cluster/src/subsystem/impls/mod.rs +++ b/nodedb-cluster/src/subsystem/impls/mod.rs @@ -9,4 +9,4 @@ pub mod swim_subsystem; pub use decommission_subsystem::DecommissionSubsystem; pub use reachability_subsystem::ReachabilitySubsystem; pub use rebalancer_subsystem::RebalancerSubsystem; -pub use swim_subsystem::{SwimSubsystem, SwimSubsystemConfig}; +pub use swim_subsystem::{SwimSubsystem, SwimSubsystemConfig, SwimWiring}; diff --git a/nodedb-cluster/src/subsystem/impls/swim_subsystem.rs b/nodedb-cluster/src/subsystem/impls/swim_subsystem.rs index 5f71d3220..514791b8b 100644 --- a/nodedb-cluster/src/subsystem/impls/swim_subsystem.rs +++ b/nodedb-cluster/src/subsystem/impls/swim_subsystem.rs @@ -5,10 +5,14 @@ //! This is the root subsystem (no dependencies) that spawns the SWIM //! run loop with all membership subscribers attached **before** the //! UDP socket starts exchanging probes. This eliminates the first-rumour -//! race: `spawn_with_subscribers` adds subscribers to the +//! race: `spawn_with_members` adds subscribers to the //! `FailureDetector` before `detector.run()` is called, so the very //! first `on_state_change` callback fires on the first probe round. //! +//! The UDP socket is bound before cluster startup so bootstrap, join, and +//! restart can advertise its address. The detector is seeded from every +//! peer's advertised SWIM address in topology, with real node ids. +//! //! Subscribers attached here: //! - [`RoutingLivenessHook`] — invalidates routing leader hints when //! SWIM marks a peer Suspect / Dead / Left. @@ -27,12 +31,12 @@ use tokio::sync::watch; use crate::routing::RoutingTable; use crate::routing_liveness::{NodeIdResolver, RoutingLivenessHook}; -use crate::swim::bootstrap::{SwimHandle, spawn_with_subscribers}; +use crate::swim::bootstrap::{SwimHandle, spawn_with_members}; use crate::swim::config::SwimConfig; -use crate::swim::detector::UdpTransport; +use crate::swim::detector::{Transport, UdpTransport}; use crate::swim::incarnation_store::IncarnationStore; use crate::swim::subscriber::MembershipSubscriber; -use crate::topology::ClusterTopology; +use crate::topology::{ClusterTopology, NodeState}; use super::super::context::BootstrapCtx; use super::super::errors::{BootstrapError, ShutdownError}; @@ -46,15 +50,18 @@ pub struct SwimSubsystemConfig { pub swim: SwimConfig, /// This node's stable identity (used in member records). pub local_id: NodeId, - /// The address the SWIM UDP socket will bind to. - pub swim_addr: SocketAddr, - /// Initial seed peers for the membership list. - pub seeds: Vec, /// Persists self-refutation incarnation bumps across restarts. /// Production passes a catalog-backed store. pub incarnation_store: Option>, } +/// What the host supplies to run SWIM: the socket it bound before cluster +/// startup, and extra subscribers to attach before the first probe. +pub struct SwimWiring { + pub transport: Arc, + pub subscribers: Vec>, +} + /// Owns the SWIM failure detector lifetime. /// /// `RoutingLivenessHook` is attached as a subscriber before the run @@ -64,9 +71,13 @@ pub struct SwimSubsystem { cfg: SwimSubsystemConfig, routing: Arc>, topology: Arc>, + /// UDP socket bound before cluster startup, so its address could be + /// advertised in this node's topology entry. + transport: Arc, extra_subscribers: Vec>, - /// Stashed handle after `start()` so `shutdown()` can stop it. - handle: Mutex>, + /// Running detector after `start()`. Shared with the subsystem handle's + /// task; whichever of it and `shutdown()` runs first stops the detector. + handle: Arc>>, } impl SwimSubsystem { @@ -74,14 +85,16 @@ impl SwimSubsystem { cfg: SwimSubsystemConfig, routing: Arc>, topology: Arc>, + transport: Arc, extra_subscribers: Vec>, ) -> Self { Self { cfg, routing, topology, + transport, extra_subscribers, - handle: Mutex::new(None), + handle: Arc::new(Mutex::new(None)), } } } @@ -113,28 +126,34 @@ impl ClusterSubsystem for SwimSubsystem { .filter(|&id| topo.get_node(id).is_some()) }); + // The local id is the node's numeric id as a decimal string. + let local_node_id = self.cfg.local_id.as_str().parse::().map_err(|e| { + BootstrapError::SubsystemStart { + name: "swim", + cause: Box::new(e), + } + })?; let routing_hook = Arc::new(RoutingLivenessHook::new( Arc::clone(&self.routing), resolver, + local_node_id, )); let mut subscribers: Vec> = vec![routing_hook]; subscribers.extend(self.extra_subscribers.iter().cloned()); - let mac_key = _ctx.transport.mac_key(); - let transport = UdpTransport::bind(self.cfg.swim_addr, mac_key) - .await - .map_err(|e| BootstrapError::SubsystemStart { - name: "swim", - cause: Box::new(e), - })?; + let local_addr = self.transport.local_addr(); + let peers = topology_swim_peers( + &self.topology.read().unwrap_or_else(|p| p.into_inner()), + &self.cfg.local_id, + ); - let swim_handle = spawn_with_subscribers( + let swim_handle = spawn_with_members( self.cfg.swim.clone(), self.cfg.local_id.clone(), - self.cfg.swim_addr, - self.cfg.seeds.clone(), - Arc::new(transport), + local_addr, + peers, + Arc::clone(&self.transport) as Arc, subscribers, self.cfg.incarnation_store.clone(), ) @@ -144,27 +163,26 @@ impl ClusterSubsystem for SwimSubsystem { cause: Box::new(e), })?; - // Extract the shutdown channel from the SWIM handle to build a - // `SubsystemHandle`. We store the `SwimHandle` so `shutdown()` - // can call `swim_handle.shutdown()` for graceful drain. - // - // The `SubsystemHandle` owns a dummy task that finishes immediately - // — the real work is in the SWIM detector task that `SwimHandle` - // holds. The shutdown watch is shared: when the registry sends - // `true` on `shutdown_tx`, the SWIM detector's recv loop sees it - // via the same receiver that `SwimHandle` already subscribes to. - let (dummy_tx, _dummy_rx) = watch::channel(false); - // We share the _real_ shutdown sender via the stored handle. - // The subsystem handle's watch is used only for signalling; the - // actual wait is done in `shutdown()` which calls `SwimHandle::shutdown`. - let dummy_join = tokio::spawn(async {}); - let subsystem_handle = SubsystemHandle::new("swim", dummy_join, dummy_tx); - { let mut guard = self.handle.lock().unwrap_or_else(|p| p.into_inner()); *guard = Some(swim_handle); } + // The subsystem handle's task owns the detector's lifetime: a shutdown + // signal, or the handle being dropped, stops the detector so its + // socket stops answering probes. + let (shutdown_tx, mut shutdown_rx) = watch::channel(false); + let slot = Arc::clone(&self.handle); + let join = tokio::spawn(async move { + // A closed channel means the handle was dropped: stop as well. + let _ = shutdown_rx.wait_for(|stop| *stop).await; + let taken = slot.lock().unwrap_or_else(|p| p.into_inner()).take(); + if let Some(swim_handle) = taken { + swim_handle.shutdown().await; + } + }); + let subsystem_handle = SubsystemHandle::new("swim", join, shutdown_tx); + Ok(subsystem_handle) } @@ -194,13 +212,31 @@ impl ClusterSubsystem for SwimSubsystem { } } +/// SWIM seeds for every other node in `topology` that advertises a SWIM +/// address. The QUIC address is never used: it is a different socket. +fn topology_swim_peers(topology: &ClusterTopology, local_id: &NodeId) -> Vec<(NodeId, SocketAddr)> { + topology + .all_nodes() + .filter(|n| n.state != NodeState::Decommissioned) + .filter_map(|n| { + // A decimal u64 is always a valid id: non-empty, short, no NUL. + let id = NodeId::from_validated(n.node_id.to_string()); + if &id == local_id { + return None; + } + n.swim_socket_addr().map(|addr| (id, addr)) + }) + .collect() +} + #[cfg(test)] mod tests { use super::*; + use crate::rpc_codec::MacKey; + use crate::topology::NodeInfo; - #[test] - fn swim_subsystem_name_and_deps() { - let dummy_cfg = SwimSubsystemConfig { + fn dummy_cfg() -> SwimSubsystemConfig { + SwimSubsystemConfig { swim: crate::swim::config::SwimConfig { probe_interval: std::time::Duration::from_millis(100), probe_timeout: std::time::Duration::from_millis(40), @@ -212,38 +248,67 @@ mod tests { fanout_lambda: 3, }, local_id: NodeId::try_new("1").expect("test fixture"), - swim_addr: "127.0.0.1:0".parse().unwrap(), - seeds: vec![], incarnation_store: None, - }; + } + } + + async fn subsystem() -> SwimSubsystem { + let transport = UdpTransport::bind("127.0.0.1:0".parse().unwrap(), MacKey::zero()) + .await + .expect("bind a free port"); let routing = Arc::new(RwLock::new(RoutingTable::uniform(1, &[1], 1))); let topology = Arc::new(RwLock::new(ClusterTopology::new())); - let s = SwimSubsystem::new(dummy_cfg, routing, topology, vec![]); + SwimSubsystem::new(dummy_cfg(), routing, topology, Arc::new(transport), vec![]) + } + + #[tokio::test] + async fn swim_subsystem_name_and_deps() { + let s = subsystem().await; assert_eq!(s.name(), "swim"); assert!(s.dependencies().is_empty()); } - #[test] - fn health_is_stopped_before_start() { - let dummy_cfg = SwimSubsystemConfig { - swim: crate::swim::config::SwimConfig { - probe_interval: std::time::Duration::from_millis(100), - probe_timeout: std::time::Duration::from_millis(40), - indirect_probes: 2, - suspicion_mult: 4, - min_suspicion: std::time::Duration::from_millis(500), - initial_incarnation: crate::swim::incarnation::Incarnation::ZERO, - max_piggyback: 6, - fanout_lambda: 3, - }, - local_id: NodeId::try_new("1").expect("test fixture"), - swim_addr: "127.0.0.1:0".parse().unwrap(), - seeds: vec![], - incarnation_store: None, - }; - let routing = Arc::new(RwLock::new(RoutingTable::uniform(1, &[1], 1))); - let topology = Arc::new(RwLock::new(ClusterTopology::new())); - let s = SwimSubsystem::new(dummy_cfg, routing, topology, vec![]); + #[tokio::test] + async fn health_is_stopped_before_start() { + let s = subsystem().await; assert_eq!(s.health(), SubsystemHealth::Stopped); } + + /// Seeds come from each peer's advertised SWIM address, never its QUIC + /// address, and skip this node, decommissioned nodes, and nodes with none. + #[test] + fn seeds_are_the_advertised_swim_addresses() { + let mut topo = ClusterTopology::new(); + topo.add_node( + NodeInfo::new(1, "10.0.0.1:9400".parse().unwrap(), NodeState::Active) + .with_swim_addr("10.0.0.1:9401".parse().ok()), + ); + topo.add_node( + NodeInfo::new(2, "10.0.0.2:9400".parse().unwrap(), NodeState::Active) + .with_swim_addr("10.0.0.2:9401".parse().ok()), + ); + topo.add_node(NodeInfo::new( + 3, + "10.0.0.3:9400".parse().unwrap(), + NodeState::Active, + )); + topo.add_node( + NodeInfo::new( + 4, + "10.0.0.4:9400".parse().unwrap(), + NodeState::Decommissioned, + ) + .with_swim_addr("10.0.0.4:9401".parse().ok()), + ); + + let local = NodeId::try_new("1").expect("test fixture"); + let peers = topology_swim_peers(&topo, &local); + assert_eq!( + peers, + vec![( + NodeId::try_new("2").expect("test fixture"), + "10.0.0.2:9401".parse::().unwrap() + )] + ); + } } diff --git a/nodedb-cluster/src/subsystem/mod.rs b/nodedb-cluster/src/subsystem/mod.rs index ef6830c13..021034ea9 100644 --- a/nodedb-cluster/src/subsystem/mod.rs +++ b/nodedb-cluster/src/subsystem/mod.rs @@ -13,7 +13,7 @@ pub use errors::{BootstrapError, ShutdownError, TopoError}; pub use health::{ClusterHealth, SubsystemHealth}; pub use impls::{ DecommissionSubsystem, ReachabilitySubsystem, RebalancerSubsystem, SwimSubsystem, - SwimSubsystemConfig, + SwimSubsystemConfig, SwimWiring, }; pub use registry::{RunningCluster, SubsystemRegistry}; pub use topo_sort::topo_sort; diff --git a/nodedb-cluster/src/swim/bootstrap.rs b/nodedb-cluster/src/swim/bootstrap.rs index b608962a4..8ec79e73a 100644 --- a/nodedb-cluster/src/swim/bootstrap.rs +++ b/nodedb-cluster/src/swim/bootstrap.rs @@ -7,10 +7,11 @@ //! //! 1. Constructs a [`MembershipList`] containing the local node at //! incarnation 0. -//! 2. Seeds the list with an `Alive` entry for every address in -//! `seeds`, using a synthetic `NodeId` of the form `"seed:"`. -//! The first successful probe replaces the placeholder with the -//! peer's real node id via the normal merge path. +//! 2. Seeds the list with an `Alive` entry per peer. [`spawn_with_members`] +//! takes real node ids (cluster startup reads them from topology). +//! [`spawn`] and [`spawn_with_subscribers`] take bare addresses and use +//! a synthetic `NodeId` of the form `"seed:"`, which the first +//! successful probe replaces with the peer's real id. //! 3. Validates [`SwimConfig`] and constructs a [`FailureDetector`]. //! 4. Spawns the detector's run loop on a fresh tokio task. //! 5. Returns a [`SwimHandle`] the caller can use to read membership, @@ -119,6 +120,37 @@ pub async fn spawn_with_subscribers( transport: Arc, subscribers: Vec>, incarnation_store: Option>, +) -> Result { + // Address-only seeds get placeholder ids, replaced on the first ack. + // Callers without a topology to read real ids from seed this way. + let peers = seeds + .into_iter() + // SocketAddr display always produces a valid ID: non-empty, well under cap, no NUL. + .map(|addr| (NodeId::from_validated(format!("seed:{addr}")), addr)) + .collect(); + spawn_with_members( + cfg, + local_id, + local_addr, + peers, + transport, + subscribers, + incarnation_store, + ) + .await +} + +/// Same as [`spawn_with_subscribers`], seeded with peers whose real ids are +/// already known, such as every node in the cluster topology. Entries at +/// `local_addr` are skipped. +pub async fn spawn_with_members( + cfg: SwimConfig, + local_id: NodeId, + local_addr: SocketAddr, + peers: Vec<(NodeId, SocketAddr)>, + transport: Arc, + subscribers: Vec>, + incarnation_store: Option>, ) -> Result { cfg.validate()?; @@ -128,16 +160,14 @@ pub async fn spawn_with_subscribers( cfg.initial_incarnation, )); - // Seed the membership table so the first probe round has somewhere - // to go. Placeholder ids are replaced on the first ack. - for seed_addr in &seeds { - if *seed_addr == local_addr { + // Seed the membership table so the first probe round has somewhere to go. + for (node_id, addr) in peers { + if addr == local_addr || node_id == local_id { continue; } membership.apply(&MemberUpdate { - // SocketAddr display always produces a valid ID: non-empty, well under cap, no NUL. - node_id: NodeId::from_validated(format!("seed:{seed_addr}")), - addr: seed_addr.to_string(), + node_id, + addr: addr.to_string(), state: MemberState::Alive, incarnation: Incarnation::ZERO, }); diff --git a/nodedb-cluster/src/swim/detector/mod.rs b/nodedb-cluster/src/swim/detector/mod.rs index de1cf78e6..d7b53e3c1 100644 --- a/nodedb-cluster/src/swim/detector/mod.rs +++ b/nodedb-cluster/src/swim/detector/mod.rs @@ -10,6 +10,7 @@ //! detector without touching its logic. pub mod probe_round; +mod round_slot; pub mod runner; pub mod scheduler; pub mod suspicion; diff --git a/nodedb-cluster/src/swim/detector/round_slot.rs b/nodedb-cluster/src/swim/detector/round_slot.rs new file mode 100644 index 000000000..d84652eea --- /dev/null +++ b/nodedb-cluster/src/swim/detector/round_slot.rs @@ -0,0 +1,74 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The probe round in flight, polled beside the detector's inbound arm. +//! +//! A probe round waits for acks, and only the run loop's inbound arm reads +//! them off the transport. The loop therefore never awaits a round inline. +//! It parks the round here and polls it as one `select!` arm, so inbound +//! datagrams keep flowing while the round waits. + +use std::future::Future; +use std::pin::Pin; + +type RoundFuture<'a> = Pin + Send + 'a>>; + +/// At most one probe round in flight. +#[derive(Default)] +pub(super) struct RoundSlot<'a> { + round: Option>, +} + +impl<'a> RoundSlot<'a> { + /// Whether no round is in flight. + pub(super) fn is_idle(&self) -> bool { + self.round.is_none() + } + + /// Park `round` as the round in flight. The caller starts a round only + /// while the slot is idle. + pub(super) fn start(&mut self, round: impl Future + Send + 'a) { + self.round = Some(Box::pin(round)); + } + + /// Resolves once the round in flight completes, and leaves the slot idle. + /// Pends forever while no round is in flight. + /// + /// Cancel-safe: dropping this future leaves the round parked, and the + /// next call resumes it where it stopped. + pub(super) async fn finished(&mut self) { + match self.round.as_mut() { + Some(round) => { + round.as_mut().await; + self.round = None; + } + None => std::future::pending().await, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[tokio::test] + async fn a_parked_round_resumes_after_its_poll_is_dropped() { + let (tx, rx) = tokio::sync::oneshot::channel::<()>(); + let mut slot = RoundSlot::default(); + slot.start(async move { + let _ = rx.await; + }); + assert!(!slot.is_idle()); + + // The first poll is dropped while the round still waits. + tokio::select! { + biased; + () = slot.finished() => panic!("the round cannot finish before its signal"), + () = std::future::ready(()) => {} + } + assert!(!slot.is_idle(), "a dropped poll keeps the round parked"); + + let _ = tx.send(()); + slot.finished().await; + assert!(slot.is_idle()); + } +} diff --git a/nodedb-cluster/src/swim/detector/runner.rs b/nodedb-cluster/src/swim/detector/runner.rs index 3e12cb41b..f9ddbb785 100644 --- a/nodedb-cluster/src/swim/detector/runner.rs +++ b/nodedb-cluster/src/swim/detector/runner.rs @@ -5,7 +5,8 @@ //! One instance per node. Owns the membership list (shared via `Arc`), //! the probe scheduler, the suspicion timer, the inflight-probe registry, //! and the async transport. Drives a `tokio::select!` loop over four -//! arms: probe tick, inbound datagram, suspicion expiry, shutdown. +//! arms: shutdown, the probe round in flight, probe tick, and inbound +//! datagram. Suspicion expiry runs at the start of each probe round. use std::net::SocketAddr; use std::sync::Arc; @@ -27,6 +28,7 @@ use crate::swim::subscriber::MembershipSubscriber; use crate::swim::wire::{Ack, Ping, PingReq, ProbeId, SwimMessage}; use super::probe_round::{InflightProbes, ProbeOutcome, ProbeRound}; +use super::round_slot::RoundSlot; use super::scheduler::ProbeScheduler; use super::suspicion::SuspicionTimer; use super::transport::Transport; @@ -210,23 +212,31 @@ impl FailureDetector { ProbeId::new(self.probe_counter.fetch_add(1, Ordering::Relaxed)) } - /// Main loop. Returns when `shutdown` receives `true`. + /// Main loop. Returns when `shutdown` receives `true` or its sender is + /// gone. + /// + /// A probe round waits for acks that only the inbound arm reads. The + /// round is therefore polled as its own arm, never awaited inline, and + /// the inbound arm runs while it waits. A tick starts a round only when + /// none is in flight. pub async fn run(self: Arc, mut shutdown: watch::Receiver) { let mut tick = interval(self.cfg.probe_interval); tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); // Consume the first immediate tick so the first probe aligns // with a full interval from start. tick.tick().await; + let mut round = RoundSlot::default(); loop { tokio::select! { biased; changed = shutdown.changed() => { - if changed.is_ok() && *shutdown.borrow() { + if changed.is_err() || *shutdown.borrow() { break; } } - _ = tick.tick() => { - self.on_tick().await; + () = round.finished() => {} + _ = tick.tick(), if round.is_idle() => { + round.start(self.on_tick()); } recv = self.transport.recv() => { match recv { @@ -602,6 +612,92 @@ mod tests { let _ = tokio::time::timeout(Duration::from_millis(200), h_c).await; } + /// Every state change one detector reports. + #[derive(Default)] + struct Verdicts(std::sync::Mutex>); + + impl MembershipSubscriber for Verdicts { + fn on_state_change(&self, node_id: &NodeId, _old: Option, new: MemberState) { + self.0 + .lock() + .unwrap_or_else(|p| p.into_inner()) + .push((node_id.clone(), new)); + } + } + + /// Live peers ack every probe within the probe timeout, so no detector + /// ever suspects one. An ack reaches the round that waits for it while + /// the round still waits. No false verdict means no refutation, so no + /// local incarnation moves. + /// + /// `sleep` rather than `advance`: the paused clock moves only once every + /// task is idle, so each ping and its ack are delivered before any probe + /// timeout can fire. + #[tokio::test(start_paused = true)] + async fn live_peers_that_ack_in_time_are_never_suspected() { + let fab = TransportFabric::new(); + let nodes = [("a", 7200u16), ("b", 7201), ("c", 7202)]; + let mut running = Vec::new(); + for (id, port) in nodes { + let transport: Arc = Arc::new(fab.bind(addr(port)).await); + let list = Arc::new(MembershipList::new_local( + NodeId::try_new(id).expect("test fixture"), + addr(port), + Incarnation::ZERO, + )); + for (peer_id, peer_port) in nodes.iter().filter(|(peer, _)| *peer != id) { + list.apply(&MemberUpdate { + node_id: NodeId::try_new(*peer_id).expect("test fixture"), + addr: addr(*peer_port).to_string(), + state: MemberState::Alive, + incarnation: Incarnation::ZERO, + }); + } + let verdicts = Arc::new(Verdicts::default()); + let detector = Arc::new(FailureDetector::with_subscribers( + cfg(), + list, + transport, + ProbeScheduler::with_seed(u64::from(port)), + vec![Arc::clone(&verdicts) as Arc], + )); + let (tx, rx) = watch::channel(false); + let handle = tokio::spawn({ + let detector = Arc::clone(&detector); + async move { detector.run(rx).await } + }); + running.push((id, detector, verdicts, tx, handle)); + } + + // Thirty probe intervals: every node probes each peer many times. + tokio::time::sleep(cfg().probe_interval * 30).await; + + for (id, detector, verdicts, _, _) in &running { + let seen = verdicts.0.lock().unwrap_or_else(|p| p.into_inner()).clone(); + assert!( + seen.iter().all(|(_, state)| *state == MemberState::Alive), + "{id} reported a false verdict on a live peer: {seen:?}" + ); + for (peer, _) in nodes.iter().filter(|(peer, _)| peer != id) { + let member = detector + .membership + .get(&NodeId::try_new(*peer).expect("test fixture")) + .expect("peer in list"); + assert_eq!(member.state, MemberState::Alive, "{id} sees {peer}"); + } + assert_eq!( + *detector.local_incarnation.lock().await, + Incarnation::ZERO, + "{id} refuted a suspicion no live peer should have raised" + ); + } + + for (_, _, _, tx, handle) in running { + let _ = tx.send(true); + let _ = tokio::time::timeout(Duration::from_millis(200), handle).await; + } + } + #[tokio::test(start_paused = true)] async fn ping_triggers_ack_reply() { let fab = TransportFabric::new(); diff --git a/nodedb-cluster/src/swim/detector/transport/udp.rs b/nodedb-cluster/src/swim/detector/transport/udp.rs index 18fba6880..12c27739c 100644 --- a/nodedb-cluster/src/swim/detector/transport/udp.rs +++ b/nodedb-cluster/src/swim/detector/transport/udp.rs @@ -56,11 +56,13 @@ impl UdpTransport { /// mode (mirrors the Raft transport's `TransportCredentials::Insecure` /// escape hatch — only safe on isolated networks). pub async fn bind(addr: SocketAddr, mac_key: MacKey) -> Result { - let socket = UdpSocket::bind(addr).await.map_err(|e| SwimError::Encode { - detail: format!("udp bind {addr}: {e}"), + let socket = UdpSocket::bind(addr).await.map_err(|e| SwimError::Bind { + addr, + detail: e.to_string(), })?; - let local_addr = socket.local_addr().map_err(|e| SwimError::Encode { - detail: format!("udp local_addr: {e}"), + let local_addr = socket.local_addr().map_err(|e| SwimError::Bind { + addr, + detail: format!("local_addr: {e}"), })?; Ok(Self { socket: Arc::new(socket), @@ -69,6 +71,13 @@ impl UdpTransport { recv_buf: Mutex::new(vec![0u8; RECV_BUF_BYTES]), }) } + + /// Send every later datagram in boot `epoch`'s sequence range. A peer + /// keeps its replay window for this address across this node's restart, + /// so the node calls it once per boot, before the detector sends. + pub fn enter_boot_epoch(&self, epoch: u64) { + self.auth.enter_boot_epoch(epoch); + } } #[async_trait] diff --git a/nodedb-cluster/src/swim/error.rs b/nodedb-cluster/src/swim/error.rs index 9d5de6dd6..878b78963 100644 --- a/nodedb-cluster/src/swim/error.rs +++ b/nodedb-cluster/src/swim/error.rs @@ -61,6 +61,19 @@ pub enum SwimError { #[error("swim: decode failure: {detail}")] Decode { detail: String }, + /// The SWIM UDP socket could not bind `addr`. Startup fails rather than + /// falling back to another port peers do not know. + #[error("swim: cannot bind UDP listener on {addr}: {detail}")] + Bind { + addr: std::net::SocketAddr, + detail: String, + }, + + /// The default SWIM address cannot be derived from the QUIC listen + /// address: its port is the highest one, so `port + 1` does not exist. + #[error("swim: no default listen address for QUIC listener {quic_listen}; set it explicitly")] + NoDefaultAddr { quic_listen: std::net::SocketAddr }, + /// Transport backend has been closed; no further I/O is possible. /// Returned by [`super::detector::Transport::recv`] on shutdown. #[error("swim: transport closed")] diff --git a/nodedb-cluster/src/swim/listen_addr.rs b/nodedb-cluster/src/swim/listen_addr.rs new file mode 100644 index 000000000..42f08ce76 --- /dev/null +++ b/nodedb-cluster/src/swim/listen_addr.rs @@ -0,0 +1,102 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The UDP address a node's SWIM detector binds and advertises. +//! +//! The default is the QUIC listen IP, one port above the QUIC port. Peers +//! never derive it themselves: every node advertises its bound address in +//! its topology entry, so an override on one node needs no change elsewhere. + +use std::net::SocketAddr; + +use crate::rpc_codec::MacKey; + +use super::detector::UdpTransport; +use super::error::SwimError; + +/// The SWIM address used when none is configured: the QUIC listen IP with +/// port `quic_port + 1`. +/// +/// A QUIC port of `0` asks the OS for a free port, so SWIM does the same. +/// The bound port is what gets advertised. +pub fn default_swim_addr(quic_listen: SocketAddr) -> Result { + if quic_listen.port() == 0 { + return Ok(quic_listen); + } + let port = quic_listen + .port() + .checked_add(1) + .ok_or(SwimError::NoDefaultAddr { quic_listen })?; + Ok(SocketAddr::new(quic_listen.ip(), port)) +} + +/// Bind the SWIM UDP socket at `configured`, or at [`default_swim_addr`] of +/// `quic_listen` when unset. A bind error is returned as [`SwimError::Bind`], +/// never retried on another port. +pub async fn bind_swim_listener( + configured: Option, + quic_listen: SocketAddr, + mac_key: MacKey, +) -> Result { + let addr = match configured { + Some(addr) => addr, + None => default_swim_addr(quic_listen)?, + }; + UdpTransport::bind(addr, mac_key).await +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::swim::detector::Transport; + + #[test] + fn default_is_the_quic_ip_one_port_up() { + let quic: SocketAddr = "10.1.2.3:9400".parse().expect("literal address"); + assert_eq!( + default_swim_addr(quic).expect("derivable"), + "10.1.2.3:9401" + .parse::() + .expect("literal address") + ); + let quic_v6: SocketAddr = "[fd00::7]:7000".parse().expect("literal address"); + assert_eq!( + default_swim_addr(quic_v6).expect("derivable"), + "[fd00::7]:7001" + .parse::() + .expect("literal address") + ); + } + + #[test] + fn the_highest_quic_port_has_no_default() { + let quic: SocketAddr = "10.1.2.3:65535".parse().expect("literal address"); + assert!(matches!( + default_swim_addr(quic), + Err(SwimError::NoDefaultAddr { quic_listen }) if quic_listen == quic + )); + } + + #[test] + fn an_os_assigned_quic_port_gives_an_os_assigned_swim_port() { + let quic: SocketAddr = "127.0.0.1:0".parse().expect("literal address"); + assert_eq!(default_swim_addr(quic).expect("derivable"), quic); + } + + #[tokio::test] + async fn a_taken_port_fails_with_the_address() { + let first = bind_swim_listener( + Some("127.0.0.1:0".parse().expect("literal address")), + "127.0.0.1:0".parse().expect("literal address"), + MacKey::zero(), + ) + .await + .expect("bind a free port"); + let taken = first.local_addr(); + + match bind_swim_listener(Some(taken), taken, MacKey::zero()).await { + Err(SwimError::Bind { addr, .. }) => assert_eq!(addr, taken), + Err(other) => panic!("expected a bind error naming {taken}, got {other}"), + Ok(_) => panic!("binding a taken port must fail"), + } + } +} diff --git a/nodedb-cluster/src/swim/mod.rs b/nodedb-cluster/src/swim/mod.rs index ee1db512f..81a893d90 100644 --- a/nodedb-cluster/src/swim/mod.rs +++ b/nodedb-cluster/src/swim/mod.rs @@ -26,6 +26,7 @@ pub mod dissemination; pub mod error; pub mod incarnation; pub mod incarnation_store; +pub mod listen_addr; pub mod member; pub mod membership; pub mod subscriber; @@ -39,6 +40,7 @@ pub use detector::{ pub use dissemination::{DisseminationQueue, PendingUpdate, apply_and_disseminate}; pub use error::SwimError; pub use incarnation::Incarnation; +pub use listen_addr::{bind_swim_listener, default_swim_addr}; pub use member::{Member, MemberState}; pub use membership::{MembershipList, MembershipSnapshot, merge_update}; pub use subscriber::MembershipSubscriber; diff --git a/nodedb-cluster/src/swim/wire/authenticated.rs b/nodedb-cluster/src/swim/wire/authenticated.rs index f12136199..b2265f9e4 100644 --- a/nodedb-cluster/src/swim/wire/authenticated.rs +++ b/nodedb-cluster/src/swim/wire/authenticated.rs @@ -87,6 +87,12 @@ impl SwimAuth { } } + /// Move the outbound counter into boot `epoch`'s sequence range (see + /// [`PeerSeqSender::enter_boot_epoch`]). + pub fn enter_boot_epoch(&self, epoch: u64) { + self.seq_out.enter_boot_epoch(epoch); + } + /// Hash of the local bound address — used as the envelope's /// `from_node_id` on every outbound datagram. pub fn local_addr_hash(&self) -> u64 { diff --git a/nodedb-cluster/src/topology.rs b/nodedb-cluster/src/topology.rs index 54b51e67f..da34a28fa 100644 --- a/nodedb-cluster/src/topology.rs +++ b/nodedb-cluster/src/topology.rs @@ -5,6 +5,8 @@ use std::collections::HashMap; use std::net::SocketAddr; +use crate::rpc_codec::JoinNodeInfo; + /// Wire format version carried on every `NodeInfo`. Re-exported from /// `nodedb_types::wire_version`, which is the single source of truth /// shared with `nodedb::version` and any other crate that needs to @@ -128,6 +130,10 @@ pub struct NodeInfo { /// transmitted its identity fields. #[serde(default = "default_spki_pin")] pub spki_pin: Option<[u8; 32]>, + /// Bound UDP address of this node's SWIM failure detector. Peers seed + /// SWIM from it. `None` for a node that runs no SWIM detector. + #[serde(default)] + pub swim_addr: Option, } impl NodeInfo { @@ -143,6 +149,7 @@ impl NodeInfo { wire_version: CLUSTER_WIRE_FORMAT_VERSION, spiffe_id: None, spki_pin: None, + swim_addr: None, } } @@ -166,9 +173,58 @@ impl NodeInfo { self } + /// Set the SWIM address this node advertises. Builder-style. + pub fn with_swim_addr(mut self, swim_addr: Option) -> Self { + self.swim_addr = swim_addr.map(|a| a.to_string()); + self + } + pub fn socket_addr(&self) -> Option { self.addr.parse().ok() } + + /// The advertised SWIM address, if present and parseable. + pub fn swim_socket_addr(&self) -> Option { + self.swim_addr.as_deref().and_then(|a| a.parse().ok()) + } + + /// Wire form carried in join responses and topology updates. + pub fn to_wire(&self) -> JoinNodeInfo { + JoinNodeInfo { + node_id: self.node_id, + addr: self.addr.clone(), + state: self.state.as_u8(), + raft_groups: self.raft_groups.clone(), + wire_version: self.wire_version, + spiffe_id: self.spiffe_id.clone(), + spki_pin: self.spki_pin.map(|arr| arr.to_vec()), + swim_addr: self.swim_addr.clone(), + } + } + + /// Rebuild a `NodeInfo` from its wire form. + /// + /// An unknown state reads as `Active`. An unparseable address keeps the + /// unspecified address, so the entry stays visible but unroutable. A pin + /// that is not 32 bytes is dropped. + pub fn from_wire(node: &JoinNodeInfo) -> Self { + let state = NodeState::from_u8(node.state).unwrap_or(NodeState::Active); + let spki_pin: Option<[u8; 32]> = node + .spki_pin + .as_deref() + .and_then(|b| <[u8; 32]>::try_from(b).ok()); + let addr = node + .addr + .parse() + .unwrap_or_else(|_| SocketAddr::from(([0, 0, 0, 0], 0))); + let mut info = NodeInfo::new(node.node_id, addr, state) + .with_wire_version(node.wire_version) + .with_spiffe_id(node.spiffe_id.clone()) + .with_spki_pin(spki_pin); + info.raft_groups = node.raft_groups.clone(); + info.swim_addr = node.swim_addr.clone(); + info + } } /// Cluster topology — authoritative registry of all nodes. @@ -252,6 +308,12 @@ impl ClusterTopology { self.version } + /// Take the version of a topology adopted wholesale from a peer, so the + /// next version comparison sees the two as equal. + pub(crate) fn adopt_version(&mut self, version: u64) { + self.version = version; + } + pub fn contains(&self, node_id: u64) -> bool { self.nodes.contains_key(&node_id) } diff --git a/nodedb-cluster/src/transport/client/attempt.rs b/nodedb-cluster/src/transport/client/attempt.rs new file mode 100644 index 000000000..888e3eadb --- /dev/null +++ b/nodedb-cluster/src/transport/client/attempt.rs @@ -0,0 +1,404 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Outbound RPC to a known peer: one attempt, its circuit-breaker record, +//! and the retry loop around it. +//! +//! Only a link failure counts against the peer's circuit breaker. That is a +//! failed connect, handshake, stream open or write, a read timeout, or a lost +//! or reset connection. Every answer the peer sends counts as a success, a +//! typed refusal included. A refusal goes back to the caller as its typed +//! error and is not resent. +//! +//! A link failure drops the pooled connection only when the connection +//! itself failed. Other sends share that connection, so one stream's +//! failure must not cut them off. The connection fails when: +//! - it is closed, lost or reset; +//! - it reached a node other than the target; +//! - a read timed out and the peer sent nothing on it during the wait. +//! +//! A live peer acknowledges the request within its ACK delay. A read that +//! times out on a connection that received datagrams meanwhile is a slow +//! handler, and the connection stays pooled. + +use std::time::Duration; + +use tracing::debug; + +use crate::circuit_breaker::{Admission, RetryPolicy}; +use crate::error::{ClusterError, Result}; +use crate::rpc_codec::{self, RaftRpc}; +use crate::transport::frame_io::read_envelope_or_finish; + +use super::transport::NexarTransport; + +/// How one attempt ended. +#[derive(Debug)] +enum Attempt { + /// The peer answered. The result is final: a response, or the typed + /// error the peer refused the request with. + Answered(Result), + /// The peer's replay window refused the frame. The peer is up, and a + /// retry goes out under a fresh sequence number. + FrameRefused(ClusterError), + /// The link to the peer failed. + LinkFailed(LinkFailure), +} + +/// A link failure, and the pooled connection it condemns. +#[derive(Debug)] +struct LinkFailure { + error: ClusterError, + /// Stable id of the connection to drop from the pool. `None` when no + /// connection was obtained or the connection still works. + condemned: Option, +} + +impl LinkFailure { + /// A failure that leaves the pool as it is. + fn keep_connection(error: ClusterError) -> Self { + Self { + error, + condemned: None, + } + } +} + +/// Sort the outcome of one send into an [`Attempt`]. +/// +/// The outer error is a link failure. The inner result is what the peer +/// sent back, or why its reply is unreadable. +fn classify(target: u64, sent: std::result::Result, LinkFailure>) -> Attempt { + match sent { + Err(link) => Attempt::LinkFailed(link), + Ok(Ok(RaftRpc::FrameRefused(refusal))) => Attempt::FrameRefused(ClusterError::Transport { + detail: format!("node {target} refused the frame: {}", refusal.detail), + }), + Ok(Ok(RaftRpc::RequestRefused(refusal))) => Attempt::Answered(Err(refusal.into_error())), + Ok(answer) => Attempt::Answered(answer), + } +} + +impl NexarTransport { + /// Send an RPC to a peer with retry and circuit breaker. + /// + /// The response read is bounded by the transport's default `rpc_timeout`. + /// For RPCs whose handler legitimately blocks far longer than a normal + /// request/response round-trip (e.g. a routed Calvin submit-and-await, which + /// the leader-side handler holds open until the transaction is sequenced AND + /// completion-acked), use [`send_rpc_with_read_timeout`](Self::send_rpc_with_read_timeout) + /// so the generic short timeout does not abort the call while the remote + /// handler is still legitimately working. + pub async fn send_rpc(&self, target: u64, rpc: RaftRpc) -> Result { + self.send_rpc_with_read_timeout(target, rpc, self.rpc_timeout) + .await + } + + /// [`send_rpc`](Self::send_rpc) with an explicit response-read timeout. + /// + /// `read_timeout` bounds the wait for the response envelope on each attempt + /// (the connect / handshake / write phases still use the transport's pooled + /// connection). Callers pass a value derived from the remote handler's own + /// deadline (plus a margin) so a long-running handler is not aborted early. + pub async fn send_rpc_with_read_timeout( + &self, + target: u64, + rpc: RaftRpc, + read_timeout: Duration, + ) -> Result { + self.check_not_severed(target)?; + // Encode the inner RPC once (codec errors are not retryable). + // Each retry wraps it in a fresh envelope so the seq advances + // per attempt — a retry is a new frame, not a replayed frame. + let inner = rpc_codec::encode(&rpc, &self.auth.epoch)?; + + let mut last_err = None; + for attempt in 0..self.retry_policy.max_attempts { + if attempt > 0 { + let delay = self.retry_policy.delay_for_attempt(attempt - 1); + debug!(target, attempt, ?delay, "retrying RPC"); + tokio::time::sleep(delay).await; + } + // The breaker admits each attempt right before it goes out. + let admission = self.circuit_breaker.check(target)?; + + match self.attempt(target, &inner, read_timeout, admission).await { + Attempt::Answered(answer) => return answer, + Attempt::FrameRefused(e) => last_err = Some(e), + Attempt::LinkFailed(failure) if RetryPolicy::is_retryable(&failure.error) => { + last_err = Some(failure.error) + } + Attempt::LinkFailed(failure) => return Err(failure.error), + } + } + + Err(last_err.unwrap_or_else(|| ClusterError::Transport { + detail: format!("send_rpc to node {target}: all attempts exhausted"), + })) + } + + /// Send a recovery probe: one attempt that an open circuit never refuses. + /// + /// The probe's outcome decides the circuit. An answer closes it, and a + /// link failure reopens it. The health monitor and the reachability + /// driver probe peers this way, so a peer whose circuit is open is still + /// probed and can recover. + pub async fn send_probe_rpc(&self, target: u64, rpc: RaftRpc) -> Result { + self.check_not_severed(target)?; + let inner = rpc_codec::encode(&rpc, &self.auth.epoch)?; + let admission = self.circuit_breaker.admit_probe(target); + match self + .attempt(target, &inner, self.rpc_timeout, admission) + .await + { + Attempt::Answered(answer) => answer, + Attempt::FrameRefused(e) => Err(e), + Attempt::LinkFailed(failure) => Err(failure.error), + } + } + + /// One attempt, recorded on the circuit breaker under `admission`. + /// + /// A link failure that condemns the pooled connection also evicts it, so + /// the next attempt dials a fresh one. + async fn attempt( + &self, + target: u64, + inner: &[u8], + read_timeout: Duration, + admission: Admission, + ) -> Attempt { + let outcome = classify( + target, + self.try_send_once(target, inner, read_timeout).await, + ); + match &outcome { + Attempt::LinkFailed(failure) => { + self.circuit_breaker.record_failure(target, admission); + if let Some(stable_id) = failure.condemned { + self.evict_connection(target, stable_id); + } + } + Attempt::Answered(_) | Attempt::FrameRefused(_) => { + self.circuit_breaker.record_success(target, admission); + } + } + outcome + } + + /// Single-attempt RPC send (no retry, no circuit breaker). `inner` is the + /// encoded RPC. It is wrapped in a fresh envelope once the stream is + /// open. + /// + /// The outer error is a link failure. The inner result is the peer's + /// reply. A stream the peer finished without a reply is an inner error: + /// the peer refused the request before its handler, and a resend gets + /// the same answer. + async fn try_send_once( + &self, + target: u64, + inner: &[u8], + read_timeout: Duration, + ) -> std::result::Result, LinkFailure> { + let conn = self + .get_or_connect(target) + .await + .map_err(LinkFailure::keep_connection)?; + let stable_id = conn.stable_id(); + if let Err(error) = self.verify_connection_target(&conn, target) { + return Err(LinkFailure { + error, + condemned: Some(stable_id), + }); + } + let received_before = received_datagrams(&conn); + self.exchange(&conn, target, inner, read_timeout) + .await + .map_err(|failure| { + let (error, timed_out) = match failure { + StreamFailure::ReadTimeout(error) => (error, true), + StreamFailure::Other(error) => (error, false), + }; + let heard = received_datagrams(&conn) > received_before; + let broken = condemns_connection(timed_out, conn.close_reason().is_some(), heard); + LinkFailure { + error, + condemned: broken.then_some(stable_id), + } + }) + } + + /// Send one request on its own stream of `conn` and read the reply. + async fn exchange( + &self, + conn: &quinn::Connection, + target: u64, + inner: &[u8], + read_timeout: Duration, + ) -> std::result::Result, StreamFailure> { + let (mut send, mut recv) = conn.open_bi().await.map_err(|e| { + StreamFailure::Other(ClusterError::Transport { + detail: format!("open_bi to node {target}: {e}"), + }) + })?; + + let envelope = self.wrap_inner(inner).map_err(StreamFailure::Other)?; + send.write_all(&envelope).await.map_err(|e| { + StreamFailure::Other(ClusterError::Transport { + detail: format!("write to node {target}: {e}"), + }) + })?; + send.finish().map_err(|e| { + StreamFailure::Other(ClusterError::Transport { + detail: format!("finish send to node {target}: {e}"), + }) + })?; + + let response_envelope = + tokio::time::timeout(read_timeout, read_envelope_or_finish(&mut recv)) + .await + .map_err(|_| { + StreamFailure::ReadTimeout(ClusterError::Transport { + detail: format!( + "RPC timeout ({}ms) to node {target}", + read_timeout.as_millis() + ), + }) + })? + .map_err(StreamFailure::Other)?; + let Some(response_envelope) = response_envelope else { + return Ok(Err(ClusterError::RemoteUntyped { + detail: format!("node {target} finished the stream without a reply"), + })); + }; + + // Envelope / MAC / replay-window / codec errors are not transport + // errors — return them wrapped in Ok so retry logic doesn't retry + // a failed MAC as if it were a flaky network. + Ok(self.parse_inbound(&response_envelope, Some(target))) + } +} + +/// How one stream on a pooled connection failed. +#[derive(Debug)] +enum StreamFailure { + /// The reply did not arrive within the read timeout. + ReadTimeout(ClusterError), + /// Opening, writing or reading the stream failed. + Other(ClusterError), +} + +/// Whether a stream failure condemns its connection. +/// +/// `closed` holds when the connection is closed, lost or reset. `heard` +/// holds when the connection received a datagram since the stream opened. +/// Any other stream failure leaves the connection to the sends sharing it. +fn condemns_connection(timed_out: bool, closed: bool, heard: bool) -> bool { + closed || (timed_out && !heard) +} + +/// UDP datagrams `conn` received so far. +fn received_datagrams(conn: &quinn::Connection) -> u64 { + conn.stats().udp_rx.datagrams +} + +#[cfg(test)] +mod tests { + use nodedb_raft::message::RequestVoteRequest; + + use super::super::transport::tests::{STALL, STALLED_GROUP, serve_echo}; + use super::*; + use crate::rpc_codec::{FrameRefusal, PongResponse, RequestRefusal}; + + #[test] + fn only_a_closed_or_silent_connection_is_condemned() { + // A closed connection is condemned whatever failed on it. + assert!(condemns_connection(false, true, true)); + assert!(condemns_connection(true, true, true)); + // A read timeout with no datagram during the wait: the peer is gone. + assert!(condemns_connection(true, false, false)); + // A read timeout while the peer kept acknowledging: a slow handler. + assert!(!condemns_connection(true, false, true)); + // A stream reset or refused write on an open connection. + assert!(!condemns_connection(false, false, false)); + assert!(!condemns_connection(false, false, true)); + } + + /// A handler slower than the caller's read timeout counts against the + /// breaker, but the connection other sends share stays pooled. + #[tokio::test] + async fn a_read_timeout_on_a_live_connection_keeps_it_pooled() { + let (_server, client, _shutdown) = serve_echo().await; + client.warm_peer(1).await.expect("warm"); + let pooled = client.peer_connection_stable_id(1).expect("pooled"); + let vote = RaftRpc::RequestVoteRequest(RequestVoteRequest { + term: 1, + candidate_id: 2, + last_log_index: 0, + last_log_term: 0, + group_id: STALLED_GROUP, + transfer: false, + }); + let inner = rpc_codec::encode(&vote, &client.auth.epoch).expect("encode"); + let read_timeout = STALL / 4; + + let outcome = client + .attempt(1, &inner, read_timeout, Admission::Normal) + .await; + + match outcome { + Attempt::LinkFailed(LinkFailure { + condemned: None, .. + }) => {} + other => panic!("expected a kept connection, got {other:?}"), + } + assert_eq!(client.peer_connection_stable_id(1), Some(pooled)); + assert_eq!(client.circuit_breaker().failure_count(1), 1); + } + + #[test] + fn a_typed_refusal_is_a_final_answer() { + let refusal = RaftRpc::RequestRefused(RequestRefusal::from(ClusterError::GroupNotFound { + group_id: 4, + })); + match classify(1, Ok(Ok(refusal))) { + Attempt::Answered(Err(ClusterError::GroupNotFound { group_id: 4 })) => {} + other => panic!("expected a final GroupNotFound, got {other:?}"), + } + } + + #[test] + fn a_frame_refusal_is_resent() { + let refusal = RaftRpc::FrameRefused(FrameRefusal { + detail: "stale sequence".into(), + }); + assert!(matches!( + classify(1, Ok(Ok(refusal))), + Attempt::FrameRefused(_) + )); + } + + #[test] + fn only_the_outer_error_is_a_link_failure() { + let link = LinkFailure::keep_connection(ClusterError::Transport { + detail: "connection lost".into(), + }); + assert!(matches!(classify(1, Err(link)), Attempt::LinkFailed(_))); + + let unreadable = ClusterError::Codec { + detail: "bad crc".into(), + }; + assert!(matches!( + classify(1, Ok(Err(unreadable))), + Attempt::Answered(Err(ClusterError::Codec { .. })) + )); + + let pong = RaftRpc::Pong(PongResponse { + responder_id: 1, + topology_version: 2, + }); + assert!(matches!( + classify(1, Ok(Ok(pong))), + Attempt::Answered(Ok(RaftRpc::Pong(_))) + )); + } +} diff --git a/nodedb-cluster/src/transport/client/mod.rs b/nodedb-cluster/src/transport/client/mod.rs index 2b2a55b5c..675a9337e 100644 --- a/nodedb-cluster/src/transport/client/mod.rs +++ b/nodedb-cluster/src/transport/client/mod.rs @@ -9,6 +9,7 @@ //! //! [`RaftTransport`]: nodedb_raft::transport::RaftTransport +mod attempt; pub mod close; pub mod pool; pub mod raft_impl; diff --git a/nodedb-cluster/src/transport/client/pool.rs b/nodedb-cluster/src/transport/client/pool.rs index 2bd2052f7..bbb97c8c9 100644 --- a/nodedb-cluster/src/transport/client/pool.rs +++ b/nodedb-cluster/src/transport/client/pool.rs @@ -2,7 +2,10 @@ //! Per-peer QUIC connection pool: registration, dialling, eviction, warm-up. +use std::collections::HashMap; use std::net::SocketAddr; +use std::sync::{Arc, Mutex}; +use std::time::Instant; use tracing::debug; @@ -51,19 +54,26 @@ impl NexarTransport { }) } + /// The pooled connection to `target`, if it is open. + fn pooled_connection(&self, target: u64) -> Option { + let peers = self.peers.read().unwrap_or_else(|p| p.into_inner()); + peers + .get(&target) + .filter(|conn| conn.close_reason().is_none()) + .cloned() + } + /// Get an existing connection to a peer, or establish a new one. + /// + /// Dials to one peer are single-flight. A caller that finds no open + /// connection waits for the peer's dial gate. It then reuses the + /// connection an earlier dial pooled meanwhile, or shares the failure of + /// a dial that ended while it waited. Only a caller with neither dials, + /// so a burst of sends to one peer opens one connection. pub(super) async fn get_or_connect(&self, target: u64) -> Result { - // Check cache — fast path. - { - let peers = self.peers.read().unwrap_or_else(|p| p.into_inner()); - if let Some(conn) = peers.get(&target) - && conn.close_reason().is_none() - { - return Ok(conn.clone()); - } + if let Some(conn) = self.pooled_connection(target) { + return Ok(conn); } - - // Resolve address. let addr = { let addrs = self.peer_addrs.read().unwrap_or_else(|p| p.into_inner()); addrs @@ -72,6 +82,40 @@ impl NexarTransport { .ok_or(ClusterError::NodeUnreachable { node_id: target })? }; + let waiting_since = Instant::now(); + let gate = self.dial_gates.gate(target); + // Held across this peer's dial only. Eviction takes no gate. + let mut last_dial = gate.lock().await; + if let Some(conn) = self.pooled_connection(target) { + return Ok(conn); + } + if let Some(failure) = last_dial.failed_since(waiting_since) { + return Err(ClusterError::Transport { + detail: format!( + "dial to node {target} at {addr} failed while this send waited: {failure}" + ), + }); + } + self.drop_closed_connection(target); + match self.dial(target, addr).await { + Ok(conn) => { + last_dial.failure = None; + let mut peers = self.peers.write().unwrap_or_else(|p| p.into_inner()); + peers.insert(target, conn.clone()); + Ok(conn) + } + Err(error) => { + last_dial.failure = Some(DialFailure { + at: Instant::now(), + detail: error.to_string(), + }); + Err(error) + } + } + } + + /// Connect to `target` at `addr` and negotiate the wire version. + async fn dial(&self, target: u64, addr: SocketAddr) -> Result { // Connect — bounded by rpc_timeout so a hung QUIC handshake // (peer not yet serving) doesn't block for the full 30s idle timeout. let connecting = self @@ -130,24 +174,149 @@ impl NexarTransport { // Cache the agreed version keyed on the QUIC connection's stable id. self.store_agreed_version(conn.stable_id(), agreed); - - // Cache (harmless race: last writer wins, both connections valid). - let mut peers = self.peers.write().unwrap_or_else(|p| p.into_inner()); - peers.insert(target, conn.clone()); Ok(conn) } - /// Remove a cached connection (forces reconnect on next use). - pub(super) fn evict_peer(&self, target: u64) { - let stable_id = { + /// Drop the pooled connection to `target` if it is closed. + fn drop_closed_connection(&self, target: u64) { + let closed = { let peers = self.peers.read().unwrap_or_else(|p| p.into_inner()); - peers.get(&target).map(|c| c.stable_id()) + peers + .get(&target) + .filter(|conn| conn.close_reason().is_some()) + .map(quinn::Connection::stable_id) }; - let mut peers = self.peers.write().unwrap_or_else(|p| p.into_inner()); - peers.remove(&target); - drop(peers); - if let Some(id) = stable_id { - self.evict_agreed_version(id); + if let Some(stable_id) = closed { + self.evict_connection(target, stable_id); + } + } + + /// Drop the connection `stable_id` from the pool, so the next send to + /// `target` dials a fresh one. A newer connection pooled for `target` + /// stays. + pub(super) fn evict_connection(&self, target: u64, stable_id: usize) { + { + let mut peers = self.peers.write().unwrap_or_else(|p| p.into_inner()); + if peers + .get(&target) + .is_some_and(|conn| conn.stable_id() == stable_id) + { + peers.remove(&target); + } } + self.evict_agreed_version(stable_id); + } +} + +/// The last failed dial to one peer. +#[derive(Debug)] +struct DialFailure { + /// When the dial ended. + at: Instant, + detail: String, +} + +/// Outcome of the last dial to one peer, behind the peer's dial gate. +#[derive(Debug, Default)] +struct LastDial { + failure: Option, +} + +impl LastDial { + /// The failure of a dial that ended at or after `since`. + fn failed_since(&self, since: Instant) -> Option<&str> { + self.failure + .as_ref() + .filter(|failure| failure.at >= since) + .map(|failure| failure.detail.as_str()) + } +} + +/// One dial gate per peer. A gate is held across a dial to its peer only. +#[derive(Debug, Default)] +pub(super) struct DialGates { + gates: Mutex>>>, +} + +impl DialGates { + /// The dial gate of `target`. The map holds one gate per peer ever + /// dialled, so it stays bounded by the cluster's nodes. + fn gate(&self, target: u64) -> Arc> { + let mut gates = self.gates.lock().unwrap_or_else(|p| p.into_inner()); + Arc::clone(gates.entry(target).or_default()) + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use super::super::transport::tests::serve_echo; + use super::*; + use crate::transport::credentials::TransportCredentials; + + /// A burst of sends to one peer opens one connection. + #[tokio::test] + async fn concurrent_callers_share_one_dial() { + let (_server, client, _shutdown) = serve_echo().await; + let callers = (0..16).map(|_| { + let client = Arc::clone(&client); + async move { client.get_or_connect(1).await.map(|conn| conn.stable_id()) } + }); + + let ids = futures::future::try_join_all(callers).await.expect("dial"); + + assert!(ids.windows(2).all(|pair| pair[0] == pair[1]), "{ids:?}"); + assert_eq!(client.peer_connection_stable_id(1), ids.first().copied()); + } + + /// Callers that waited behind a failed dial share its failure. None of + /// them dials again, so a peer that is down costs one dial timeout. + #[tokio::test] + async fn waiters_share_a_failed_dial() { + // Bound but never answering: every dial runs into the handshake timeout. + let silent = std::net::UdpSocket::bind("127.0.0.1:0").expect("bind"); + let client = Arc::new( + NexarTransport::with_timeout( + 2, + "127.0.0.1:0".parse().expect("addr"), + Duration::from_millis(300), + TransportCredentials::Insecure, + ) + .expect("transport"), + ); + client.register_peer(1, silent.local_addr().expect("addr")); + let callers = (0..8).map(|_| { + let client = Arc::clone(&client); + async move { client.get_or_connect(1).await } + }); + + let outcomes = futures::future::join_all(callers).await; + + let shared = outcomes + .iter() + .filter(|outcome| { + outcome + .as_ref() + .is_err_and(|e| e.to_string().contains("while this send waited")) + }) + .count(); + assert!(outcomes.iter().all(Result::is_err)); + assert_eq!(shared, outcomes.len() - 1, "one dial, its failure shared"); + } + + /// Eviction removes only the connection that failed. A newer connection + /// pooled for the same peer stays. + #[tokio::test] + async fn eviction_spares_a_newer_connection() { + let (_server, client, _shutdown) = serve_echo().await; + let conn = client.get_or_connect(1).await.expect("dial"); + let stale = conn.stable_id().wrapping_add(1); + + client.evict_connection(1, stale); + assert_eq!(client.peer_connection_stable_id(1), Some(conn.stable_id())); + + client.evict_connection(1, conn.stable_id()); + assert_eq!(client.peer_connection_stable_id(1), None); } } diff --git a/nodedb-cluster/src/transport/client/raft_impl.rs b/nodedb-cluster/src/transport/client/raft_impl.rs index f2b52ba76..4ea0c4651 100644 --- a/nodedb-cluster/src/transport/client/raft_impl.rs +++ b/nodedb-cluster/src/transport/client/raft_impl.rs @@ -16,9 +16,57 @@ use crate::rpc_codec::RaftRpc; use super::transport::NexarTransport; +/// Map a send error onto the Raft transport's error type. +/// +/// A group the peer does not host stays typed. Every other error crosses as +/// a transport error with its message. fn to_raft_err(e: ClusterError) -> nodedb_raft::RaftError { - nodedb_raft::RaftError::Transport { - detail: e.to_string(), + match e { + ClusterError::Raft(e) => e, + ClusterError::GroupNotFound { group_id } => { + nodedb_raft::RaftError::GroupNotFound { group_id } + } + other => nodedb_raft::RaftError::Transport { + detail: other.to_string(), + }, + } +} + +/// The error for a reply of the wrong type. +fn unexpected_reply(expected: &str, reply: RaftRpc) -> ClusterError { + ClusterError::Codec { + detail: format!("expected {expected}, got {reply:?}"), + } +} + +impl NexarTransport { + /// Send AppendEntries and keep the typed cluster error. + /// + /// The tick loop tells a link failure from a refusal by this error. + pub async fn send_append_entries( + &self, + target: u64, + req: AppendEntriesRequest, + ) -> crate::error::Result { + match self + .send_rpc(target, RaftRpc::AppendEntriesRequest(req)) + .await? + { + RaftRpc::AppendEntriesResponse(r) => Ok(r), + other => Err(unexpected_reply("AppendEntriesResponse", other)), + } + } + + /// Send a PreVote probe and keep the typed cluster error. + pub async fn send_pre_vote( + &self, + target: u64, + req: PreVoteRequest, + ) -> crate::error::Result { + match self.send_rpc(target, RaftRpc::PreVoteRequest(req)).await? { + RaftRpc::PreVoteResponse(r) => Ok(r), + other => Err(unexpected_reply("PreVoteResponse", other)), + } } } @@ -28,16 +76,9 @@ impl RaftTransport for NexarTransport { target: u64, req: AppendEntriesRequest, ) -> nodedb_raft::Result { - let resp = self - .send_rpc(target, RaftRpc::AppendEntriesRequest(req)) + self.send_append_entries(target, req) .await - .map_err(to_raft_err)?; - match resp { - RaftRpc::AppendEntriesResponse(r) => Ok(r), - other => Err(nodedb_raft::RaftError::Transport { - detail: format!("expected AppendEntriesResponse, got {other:?}"), - }), - } + .map_err(to_raft_err) } async fn request_vote( @@ -62,16 +103,7 @@ impl RaftTransport for NexarTransport { target: u64, req: PreVoteRequest, ) -> nodedb_raft::Result { - let resp = self - .send_rpc(target, RaftRpc::PreVoteRequest(req)) - .await - .map_err(to_raft_err)?; - match resp { - RaftRpc::PreVoteResponse(r) => Ok(r), - other => Err(nodedb_raft::RaftError::Transport { - detail: format!("expected PreVoteResponse, got {other:?}"), - }), - } + self.send_pre_vote(target, req).await.map_err(to_raft_err) } async fn install_snapshot( diff --git a/nodedb-cluster/src/transport/client/send.rs b/nodedb-cluster/src/transport/client/send.rs index f662269a8..fc6942836 100644 --- a/nodedb-cluster/src/transport/client/send.rs +++ b/nodedb-cluster/src/transport/client/send.rs @@ -1,7 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 //! Outbound RPC: encode, wrap in authenticated envelope, dial/reuse, -//! send-and-receive, retry + circuit breaker. +//! send-and-receive. The retry loop and the circuit breaker live in +//! [`super::attempt`]. //! //! Every outbound RPC is wrapped in an //! [`auth_envelope`](crate::rpc_codec::auth_envelope)-framed wire message @@ -10,20 +11,17 @@ //! run through the per-peer replay window before being decoded. use std::net::SocketAddr; -use std::time::Duration; use futures::Stream; use rustls::pki_types::CertificateDer; -use tracing::debug; -use crate::circuit_breaker::RetryPolicy; use crate::error::{ClusterError, Result}; use crate::rpc_codec::{self, RaftRpc, TypedClusterError, auth_envelope}; use crate::transport::config::SNI_HOSTNAME; +use crate::transport::frame_io::read_envelope; use crate::transport::peer_identity_verifier::{ VerifyOutcome, spki_pin_from_cert_der, verify_peer_identity, }; -use crate::transport::server; use crate::wire_version::handshake_io::perform_version_handshake_client; use super::transport::NexarTransport; @@ -72,8 +70,6 @@ impl NexarTransport { rpc: RaftRpc, enforce_bootstrap_pin: bool, ) -> Result { - let envelope = self.wrap_outbound(&rpc)?; - let conn = self .listener .endpoint() @@ -120,6 +116,8 @@ impl NexarTransport { detail: format!("open_bi to {addr}: {e}"), })?; + // Numbered once the stream is open (see `wrap_inner`). + let envelope = self.wrap_outbound(&rpc)?; send.write_all(&envelope) .await .map_err(|e| ClusterError::Transport { @@ -129,76 +127,14 @@ impl NexarTransport { detail: format!("finish send to {addr}: {e}"), })?; - let response_envelope = server::read_envelope(&mut recv).await?; - self.parse_inbound(&response_envelope, None) - } - - /// Send an RPC to a peer with retry and circuit breaker. - /// - /// The response read is bounded by the transport's default `rpc_timeout`. - /// For RPCs whose handler legitimately blocks far longer than a normal - /// request/response round-trip (e.g. a routed Calvin submit-and-await, which - /// the leader-side handler holds open until the transaction is sequenced AND - /// completion-acked), use [`send_rpc_with_read_timeout`](Self::send_rpc_with_read_timeout) - /// so the generic short timeout does not abort the call while the remote - /// handler is still legitimately working. - pub async fn send_rpc(&self, target: u64, rpc: RaftRpc) -> Result { - self.send_rpc_with_read_timeout(target, rpc, self.rpc_timeout) - .await - } - - /// [`send_rpc`](Self::send_rpc) with an explicit response-read timeout. - /// - /// `read_timeout` bounds the wait for the response envelope on each attempt - /// (the connect / handshake / write phases still use the transport's pooled - /// connection). Callers pass a value derived from the remote handler's own - /// deadline (plus a margin) so a long-running handler is not aborted early. - pub async fn send_rpc_with_read_timeout( - &self, - target: u64, - rpc: RaftRpc, - read_timeout: Duration, - ) -> Result { - self.check_not_severed(target)?; - // Encode the inner RPC once (codec errors are not retryable). - // Each retry wraps it in a fresh envelope so the seq advances - // per attempt — a retry is a new frame, not a replayed frame. - let inner = rpc_codec::encode(&rpc, &self.auth.epoch)?; - - let mut last_err = None; - for attempt in 0..self.retry_policy.max_attempts { - // Check circuit breaker before each attempt. - self.circuit_breaker.check(target)?; - - if attempt > 0 { - let delay = self.retry_policy.delay_for_attempt(attempt - 1); - debug!(target, attempt, ?delay, "retrying RPC"); - tokio::time::sleep(delay).await; - } - - let envelope = self.wrap_inner(&inner)?; - match self.try_send_once(target, &envelope, read_timeout).await { - Ok(resp) => { - self.circuit_breaker.record_success(target); - return resp; - } - Err(e) if RetryPolicy::is_retryable(&e) => { - self.circuit_breaker.record_failure(target); - // Evict stale connection so retry gets a fresh one. - self.evict_peer(target); - last_err = Some(e); - } - Err(e) => { - // Non-retryable error — fail immediately. - self.circuit_breaker.record_failure(target); - return Err(e); - } - } + let response_envelope = read_envelope(&mut recv).await?; + match self.parse_inbound(&response_envelope, None)? { + RaftRpc::FrameRefused(refusal) => Err(ClusterError::Transport { + detail: format!("{addr} refused the frame: {}", refusal.detail), + }), + RaftRpc::RequestRefused(refusal) => Err(refusal.into_error()), + response => Ok(response), } - - Err(last_err.unwrap_or_else(|| ClusterError::Transport { - detail: format!("send_rpc to node {target}: all attempts exhausted"), - })) } /// Fire-and-forget one-way RPC: encode, send, and return without reading a @@ -206,12 +142,13 @@ impl NexarTransport { /// expects no response and tolerates loss — no retry or circuit-breaker, so /// a dropped frame is the caller's concern to recover from. pub async fn send_rpc_oneway(&self, target: u64, rpc: RaftRpc) -> Result<()> { - let envelope = self.wrap_outbound(&rpc)?; + let inner = rpc_codec::encode(&rpc, &self.auth.epoch)?; let conn = self.get_or_connect(target).await?; self.verify_connection_target(&conn, target)?; let (mut send, _recv) = conn.open_bi().await.map_err(|e| ClusterError::Transport { detail: format!("oneway open_bi to node {target}: {e}"), })?; + let envelope = self.wrap_inner(&inner)?; send.write_all(&envelope) .await .map_err(|e| ClusterError::Transport { @@ -225,11 +162,11 @@ impl NexarTransport { /// Open a streaming RPC to `target` and return a stream of result chunks. /// - /// Sibling of [`try_send_once`](Self::try_send_once) for multi-frame + /// Sibling of the one-shot send in [`super::attempt`] for multi-frame /// responses: opens a bidi stream on the pooled connection, writes the /// request envelope (typically a `RaftRpc::ExecuteStreamRequest`), finishes /// the send side, and returns a [`Stream`] that loops - /// [`server::read_envelope`] until the terminal `RPC_EXECUTE_STREAM_END` + /// [`read_envelope`] until the terminal `RPC_EXECUTE_STREAM_END` /// frame: /// /// - `RPC_EXECUTE_STREAM_CHUNK` → yields `Ok((payload, watermark_lsn))`. @@ -246,7 +183,7 @@ impl NexarTransport { target: u64, rpc: RaftRpc, ) -> Result, u64)>> + Send + use<>> { - let envelope = self.wrap_outbound(&rpc)?; + let inner = rpc_codec::encode(&rpc, &self.auth.epoch)?; let conn = self.get_or_connect(target).await?; self.verify_connection_target(&conn, target)?; @@ -254,6 +191,7 @@ impl NexarTransport { detail: format!("open_bi (stream) to node {target}: {e}"), })?; + let envelope = self.wrap_inner(&inner)?; send.write_all(&envelope) .await .map_err(|e| ClusterError::Transport { @@ -273,7 +211,7 @@ impl NexarTransport { // server-side `accept_bi` stream is not reset early. let _send = send; loop { - let envelope = server::read_envelope(&mut recv).await?; + let envelope = read_envelope(&mut recv).await?; let (fields, inner_frame) = auth_envelope::parse_envelope(&envelope, &auth.mac_key)?; if fields.from_node_id != target { @@ -302,6 +240,15 @@ impl NexarTransport { } } } + RaftRpc::FrameRefused(refusal) => { + Err(ClusterError::Transport { + detail: format!( + "node {target} refused the stream request: {}", + refusal.detail + ), + })?; + return; + } other => { Err(ClusterError::Transport { detail: format!( @@ -315,45 +262,6 @@ impl NexarTransport { }) } - /// Single-attempt RPC send (no retry, no circuit breaker). - async fn try_send_once( - &self, - target: u64, - envelope: &[u8], - read_timeout: Duration, - ) -> std::result::Result, ClusterError> { - let conn = self.get_or_connect(target).await?; - self.verify_connection_target(&conn, target)?; - - let (mut send, mut recv) = conn.open_bi().await.map_err(|e| ClusterError::Transport { - detail: format!("open_bi to node {target}: {e}"), - })?; - - send.write_all(envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write to node {target}: {e}"), - })?; - send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish send to node {target}: {e}"), - })?; - - let response_envelope = - tokio::time::timeout(read_timeout, server::read_envelope(&mut recv)) - .await - .map_err(|_| ClusterError::Transport { - detail: format!( - "RPC timeout ({}ms) to node {target}", - read_timeout.as_millis() - ), - })??; - - // Envelope / MAC / replay-window / codec errors are not transport - // errors — return them wrapped in Ok so retry logic doesn't retry - // a failed MAC as if it were a flaky network. - Ok(self.parse_inbound(&response_envelope, Some(target))) - } - /// Encode and wrap an RPC in an authenticated envelope. fn wrap_outbound(&self, rpc: &RaftRpc) -> Result> { let inner = rpc_codec::encode(rpc, &self.auth.epoch)?; @@ -361,7 +269,12 @@ impl NexarTransport { } /// Wrap an already-encoded inner frame in an authenticated envelope. - fn wrap_inner(&self, inner: &[u8]) -> Result> { + /// + /// Callers wrap once the frame's stream is open, right before the write. + /// A number taken earlier would wait out the connect and the stream + /// credit while later frames go ahead of it, and reach the peer further + /// out of order than its replay window allows. + pub(super) fn wrap_inner(&self, inner: &[u8]) -> Result> { let seq = self.auth.peer_seq_out.next(); let mut out = Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + inner.len()); auth_envelope::write_envelope( @@ -374,17 +287,7 @@ impl NexarTransport { Ok(out) } - /// Parse an inbound envelope: verify MAC, check replay window, decode - /// inner RPC. - /// - /// Self-addressed frames skip the replay-window check. In a single-node - /// test (or when a node genuinely dispatches an RPC to itself over the - /// transport) the client and server share one `AuthContext`, which - /// means one `peer_seq_in` window is updated by *both* the server-side - /// request-accept and the client-side response-accept. Without this - /// guard the second accept trips on its own first — the envelope - /// was never replayed, the same window simply saw traffic from both - /// directions for `peer_id == local_node_id`. + /// Check that `conn` reaches the TLS identity pinned for `target`. pub(super) fn verify_connection_target( &self, conn: &quinn::Connection, @@ -422,7 +325,22 @@ impl NexarTransport { } } - fn parse_inbound(&self, envelope: &[u8], expected_node_id: Option) -> Result { + /// Parse an inbound envelope: verify MAC, check replay window, decode + /// inner RPC. + /// + /// Self-addressed frames skip the replay-window check. In a single-node + /// test (or when a node genuinely dispatches an RPC to itself over the + /// transport) the client and server share one `AuthContext`, which + /// means one `peer_seq_in` window is updated by *both* the server-side + /// request-accept and the client-side response-accept. Without this + /// guard the second accept trips on its own first — the envelope + /// was never replayed, the same window simply saw traffic from both + /// directions for `peer_id == local_node_id`. + pub(super) fn parse_inbound( + &self, + envelope: &[u8], + expected_node_id: Option, + ) -> Result { let (fields, inner_frame) = auth_envelope::parse_envelope(envelope, &self.auth.mac_key)?; if let Some(expected) = expected_node_id && fields.from_node_id != expected diff --git a/nodedb-cluster/src/transport/client/shuffle_push.rs b/nodedb-cluster/src/transport/client/shuffle_push.rs index 413f57baa..45afa0aff 100644 --- a/nodedb-cluster/src/transport/client/shuffle_push.rs +++ b/nodedb-cluster/src/transport/client/shuffle_push.rs @@ -8,10 +8,10 @@ use std::sync::Arc; use crate::error::{ClusterError, Result}; use crate::rpc_codec::{ - self, RaftRpc, ShufflePushChunk, ShufflePushEnd, ShufflePushRequest, TypedClusterError, - auth_envelope, + RaftRpc, ShufflePushChunk, ShufflePushEnd, ShufflePushRequest, TypedClusterError, }; use crate::transport::auth_context::AuthContext; +use crate::transport::frame_io::encode_rpc_frame; use super::transport::NexarTransport; @@ -96,7 +96,7 @@ impl ShufflePushStream { })?; let auth = Arc::clone(transport.auth()); - let req_envelope = wrap_with_auth(&auth, &RaftRpc::ShufflePushRequest(req))?; + let req_envelope = encode_rpc_frame(&RaftRpc::ShufflePushRequest(req), &auth)?; send.write_all(&req_envelope) .await .map_err(|e| ClusterError::Transport { @@ -108,9 +108,9 @@ impl ShufflePushStream { /// Write one [`ShufflePushChunk`] envelope (a standalone msgpack row array). pub async fn push_chunk(&mut self, payload: Vec) -> Result<()> { - let chunk_envelope = wrap_with_auth( - &self.auth, + let chunk_envelope = encode_rpc_frame( &RaftRpc::ShufflePushChunk(ShufflePushChunk { payload }), + &self.auth, )?; self.send .write_all(&chunk_envelope) @@ -123,9 +123,9 @@ impl ShufflePushStream { /// Write the terminal [`ShufflePushEnd`] envelope (`error: None` for a clean /// EOF, `Some(e)` to fail the receiver fast) and finish the send half. pub async fn finish(mut self, error: Option) -> Result<()> { - let end_envelope = wrap_with_auth( - &self.auth, + let end_envelope = encode_rpc_frame( &RaftRpc::ShufflePushEnd(ShufflePushEnd { error }), + &self.auth, )?; self.send .write_all(&end_envelope) @@ -139,14 +139,3 @@ impl ShufflePushStream { Ok(()) } } - -/// Encode `rpc` and wrap it in an authenticated envelope with a fresh outbound -/// `seq` — the standalone form of [`NexarTransport::wrap_outbound`] for the -/// owned-[`AuthContext`] [`ShufflePushStream`]. -fn wrap_with_auth(auth: &AuthContext, rpc: &RaftRpc) -> Result> { - let inner = rpc_codec::encode(rpc, &auth.epoch)?; - let seq = auth.peer_seq_out.next(); - let mut out = Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + inner.len()); - auth_envelope::write_envelope(auth.local_node_id, seq, &inner, &auth.mac_key, &mut out)?; - Ok(out) -} diff --git a/nodedb-cluster/src/transport/client/transport.rs b/nodedb-cluster/src/transport/client/transport.rs index 3b5829b86..742e31357 100644 --- a/nodedb-cluster/src/transport/client/transport.rs +++ b/nodedb-cluster/src/transport/client/transport.rs @@ -26,7 +26,8 @@ use crate::transport::topology_identity_store::TopologyIdentityStore; /// Resilience features: /// - **Retry**: Transient transport failures are retried with exponential backoff. /// - **Circuit breaker**: Peers with consecutive failures are fast-failed until cooldown. -/// - **Connection eviction**: Stale connections are evicted on failure and re-established on retry. +/// - **Connection eviction**: A connection that failed is evicted and re-established on retry. +/// - **Single-flight dial**: Concurrent sends to one peer share one dial. /// /// [`RaftTransport`]: nodedb_raft::transport::RaftTransport /// [`serve`]: Self::serve @@ -38,6 +39,8 @@ pub struct NexarTransport { pub(super) peers: RwLock>, /// Known peer addresses for connection establishment. pub(super) peer_addrs: RwLock>, + /// Per-peer gates that make dials single-flight. See [`super::pool`]. + pub(super) dial_gates: super::pool::DialGates, pub(super) rpc_timeout: Duration, pub(super) circuit_breaker: Arc, pub(super) retry_policy: RetryPolicy, @@ -220,6 +223,7 @@ impl NexarTransport { client_config, peers: RwLock::new(HashMap::new()), peer_addrs: RwLock::new(HashMap::new()), + dial_gates: super::pool::DialGates::default(), rpc_timeout, circuit_breaker: Arc::new(CircuitBreaker::new(CircuitBreakerConfig::default())), retry_policy: RetryPolicy::default(), @@ -290,6 +294,14 @@ impl NexarTransport { std::sync::Arc::clone(&self.auth.epoch) } + /// Send every later frame in boot `epoch`'s sequence range (see + /// [`crate::rpc_codec::PeerSeqSender::enter_boot_epoch`]). The node calls + /// it once per boot, with the epoch its catalog just raised, before the + /// transport sends anything. + pub fn enter_boot_epoch(&self, epoch: u64) { + self.auth.peer_seq_out.enter_boot_epoch(epoch); + } + /// The cluster MAC key carried by this transport. SWIM subsystem /// uses it to authenticate UDP datagrams on the same key material. pub fn mac_key(&self) -> crate::rpc_codec::MacKey { @@ -407,8 +419,9 @@ pub struct TransportPeerSnapshot { /// Unit tests for [`NexarTransport`]: end-to-end RPC roundtrips, concurrent /// fan-out, connection reuse, and unreachable-peer errors. #[cfg(test)] -mod tests { +pub(super) mod tests { use std::sync::Arc; + use std::sync::atomic::{AtomicU32, Ordering}; use std::time::Duration; use nodedb_raft::message::{ @@ -418,6 +431,7 @@ mod tests { }; use nodedb_raft::transport::RaftTransport; + use crate::circuit_breaker::{Admission, CircuitBreakerConfig, CircuitState}; use crate::error::{ClusterError, Result}; use crate::rpc_codec::RaftRpc; use crate::transport::credentials::TransportCredentials; @@ -425,17 +439,45 @@ mod tests { use super::NexarTransport; + /// The group `EchoHandler` does not host. + const UNHOSTED_GROUP: u64 = 404; + + /// The group whose vote requests `EchoHandler` answers after [`STALL`]. + pub(crate) const STALLED_GROUP: u64 = 503; + + /// How long `EchoHandler` holds a vote request of [`STALLED_GROUP`]. + pub(crate) const STALL: Duration = Duration::from_secs(2); + + /// AppendEntries requests `EchoHandler` refused for `UNHOSTED_GROUP`. + static UNHOSTED_APPENDS: AtomicU32 = AtomicU32::new(0); + /// Mock handler that returns fixed responses for testing. struct EchoHandler; impl RaftRpcHandler for EchoHandler { async fn handle_rpc(&self, rpc: RaftRpc) -> Result { match rpc { + RaftRpc::AppendEntriesRequest(req) if req.group_id == UNHOSTED_GROUP => { + UNHOSTED_APPENDS.fetch_add(1, Ordering::SeqCst); + Err(ClusterError::GroupNotFound { + group_id: req.group_id, + }) + } + RaftRpc::Ping(req) => Ok(crate::health::handle_ping(1, 0, &req)), RaftRpc::AppendEntriesRequest(req) => { Ok(RaftRpc::AppendEntriesResponse(AppendEntriesResponse { term: req.term, success: true, last_log_index: req.prev_log_index + req.entries.len() as u64, + round: req.round, + needs_snapshot: false, + })) + } + RaftRpc::RequestVoteRequest(req) if req.group_id == STALLED_GROUP => { + tokio::time::sleep(STALL).await; + Ok(RaftRpc::RequestVoteResponse(RequestVoteResponse { + term: req.term, + vote_granted: true, })) } RaftRpc::RequestVoteRequest(req) => { @@ -594,7 +636,7 @@ mod tests { ); } - fn make_transport(node_id: u64) -> NexarTransport { + pub(crate) fn make_transport(node_id: u64) -> NexarTransport { NexarTransport::new( node_id, "127.0.0.1:0".parse().unwrap(), @@ -639,6 +681,8 @@ mod tests { ], leader_commit: 10, group_id: 7, + round: 1, + replicated_floor: 0, }; let resp = client.append_entries(1, req).await.unwrap(); @@ -677,6 +721,8 @@ mod tests { trace_id: [0u8; 16], descriptor_versions: vec![], txn_id: None, + vshard_id: None, + read_groups: Vec::new(), }); let stream = client.send_rpc_stream(1, req).await.unwrap(); @@ -721,6 +767,7 @@ mod tests { last_log_index: 100, last_log_term: 9, group_id: 3, + transfer: false, }; let resp = client.request_vote(1, req).await.unwrap(); @@ -755,6 +802,8 @@ mod tests { done: true, group_id: 0, total_size: 0, + voters: Vec::new(), + learners: Vec::new(), }; let resp = client.install_snapshot(1, req).await.unwrap(); @@ -790,6 +839,8 @@ mod tests { entries: vec![], leader_commit: i * 10, group_id: 0, + round: i, + replicated_floor: 0, }; let resp = c.append_entries(1, req).await.unwrap(); assert_eq!(resp.term, i); @@ -826,6 +877,7 @@ mod tests { last_log_index: 0, last_log_term: 0, group_id: 0, + transfer: false, }; client.request_vote(1, req).await.unwrap(); } @@ -846,6 +898,8 @@ mod tests { entries: vec![], leader_commit: 0, group_id: 0, + round: 1, + replicated_floor: 0, }; let err = client.append_entries(99, req).await.unwrap_err(); @@ -880,6 +934,8 @@ mod tests { entries: vec![], leader_commit: 50, group_id: 0, + round: 1, + replicated_floor: 0, }; let resp = client.append_entries(1, req).await.unwrap(); @@ -887,4 +943,95 @@ mod tests { assert!(resp.success); assert_eq!(resp.last_log_index, 50); } + + /// Start `EchoHandler` on node 1 and a client on node 2 that knows it. + pub(crate) async fn serve_echo() -> ( + Arc, + Arc, + tokio::sync::watch::Sender, + ) { + let server = Arc::new(make_transport(1)); + let client = Arc::new(make_transport(2)); + client.register_peer(1, server.local_addr()); + let (shutdown_tx, shutdown_rx) = tokio::sync::watch::channel(false); + let srv = server.clone(); + tokio::spawn(async move { + srv.serve(Arc::new(EchoHandler), shutdown_rx).await.unwrap(); + }); + tokio::time::sleep(Duration::from_millis(20)).await; + (server, client, shutdown_tx) + } + + #[tokio::test] + async fn a_handler_refusal_is_answered_and_not_resent() { + let (_server, client, _shutdown) = serve_echo().await; + let req = AppendEntriesRequest { + term: 3, + leader_id: 2, + prev_log_index: 0, + prev_log_term: 0, + entries: vec![], + leader_commit: 0, + group_id: UNHOSTED_GROUP, + round: 1, + replicated_floor: 0, + }; + let before = UNHOSTED_APPENDS.load(Ordering::SeqCst); + + let err = client.send_append_entries(1, req).await.unwrap_err(); + + assert!( + matches!(err, ClusterError::GroupNotFound { group_id } if group_id == UNHOSTED_GROUP), + "{err:?}" + ); + assert!(!err.is_link_failure()); + assert_eq!( + UNHOSTED_APPENDS.load(Ordering::SeqCst) - before, + 1, + "a refusal is not resent" + ); + let breaker = client.circuit_breaker(); + assert_eq!(breaker.state(1), CircuitState::Closed); + assert_eq!(breaker.failure_count(1), 0, "a refusal is an answer"); + assert!( + client.peers.read().unwrap().contains_key(&1), + "a refusal keeps the connection" + ); + } + + #[tokio::test] + async fn a_handler_error_never_reads_as_a_link_failure() { + let (_server, client, _shutdown) = serve_echo().await; + // `EchoHandler` fails a topology update with a transport error. + let update = RaftRpc::TopologyUpdate(crate::rpc_codec::TopologyUpdate { + version: 1, + nodes: vec![], + }); + + let err = client.send_rpc(1, update).await.unwrap_err(); + + assert!(matches!(err, ClusterError::RemoteUntyped { .. }), "{err:?}"); + assert_eq!(client.circuit_breaker().failure_count(1), 0); + } + + #[tokio::test] + async fn a_probe_pong_closes_an_open_breaker() { + let (_server, client, _shutdown) = serve_echo().await; + let breaker = client.circuit_breaker(); + for _ in 0..CircuitBreakerConfig::default().failure_threshold { + breaker.record_failure(1, Admission::Normal); + } + assert_eq!(breaker.state(1), CircuitState::Open); + assert!(breaker.check(1).is_err(), "the cooldown still runs"); + + let ping = RaftRpc::Ping(crate::rpc_codec::PingRequest { + sender_id: 2, + topology_version: 0, + }); + let reply = client.send_probe_rpc(1, ping).await.unwrap(); + + assert!(matches!(reply, RaftRpc::Pong(_)), "{reply:?}"); + assert_eq!(breaker.state(1), CircuitState::Closed); + assert_eq!(breaker.failure_count(1), 0); + } } diff --git a/nodedb-cluster/src/transport/config.rs b/nodedb-cluster/src/transport/config.rs index 1ae0020ad..36ac0042d 100644 --- a/nodedb-cluster/src/transport/config.rs +++ b/nodedb-cluster/src/transport/config.rs @@ -412,12 +412,7 @@ pub fn ca_fingerprint(cert: &rustls::pki_types::CertificateDer<'_>) -> [u8; 32] /// Format a CA fingerprint as a short lowercase hex string (8 bytes). /// Used as the filename stem under `data_dir/tls/ca.d/.crt`. pub fn ca_fingerprint_hex(fp: &[u8; 32]) -> String { - let mut out = String::with_capacity(16); - for b in &fp[..8] { - use std::fmt::Write as _; - let _ = write!(out, "{b:02x}"); - } - out + hex::encode(&fp[..8]) } /// Load CRLs from a PEM file. diff --git a/nodedb-cluster/src/transport/frame_io.rs b/nodedb-cluster/src/transport/frame_io.rs new file mode 100644 index 000000000..cffe23a44 --- /dev/null +++ b/nodedb-cluster/src/transport/frame_io.rs @@ -0,0 +1,176 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Authenticated frame I/O on QUIC streams. +//! +//! One outbound RPC frame is `rpc_codec::encode`, then a fresh outbound +//! `seq`, then `auth_envelope::write_envelope`, then one `write_all`. +//! [`write_rpc_frame`] does that sequence. [`read_envelope`] and +//! [`read_envelope_or_finish`] read one envelope back. + +use crate::error::{ClusterError, Result}; +use crate::rpc_codec::{self, MAX_RPC_PAYLOAD_SIZE, RaftRpc, auth_envelope}; +use crate::transport::auth_context::AuthContext; + +/// Envelope pre-header: version(1) + from_node_id(8) + seq(8) + inner_len(4). +const ENV_HDR_LEN: usize = 21; + +/// Encode `rpc` into one authenticated envelope from the local node. +/// +/// Each call takes a fresh outbound `seq`. +pub(super) fn encode_rpc_frame(rpc: &RaftRpc, auth: &AuthContext) -> Result> { + let inner = rpc_codec::encode(rpc, &auth.epoch)?; + let seq = auth.peer_seq_out.next(); + let mut envelope = Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + inner.len()); + auth_envelope::write_envelope( + auth.local_node_id, + seq, + &inner, + &auth.mac_key, + &mut envelope, + )?; + Ok(envelope) +} + +/// Encode `rpc` into one authenticated envelope and write it on `send`. +/// +/// The write is awaited inline, so QUIC flow control throttles the caller. +/// `what` names the frame in the error detail. +pub(super) async fn write_rpc_frame( + send: &mut quinn::SendStream, + auth: &AuthContext, + rpc: &RaftRpc, + what: &str, +) -> Result<()> { + let envelope = encode_rpc_frame(rpc, auth)?; + send.write_all(&envelope) + .await + .map_err(|e| ClusterError::Transport { + detail: format!("write {what}: {e}"), + }) +} + +/// Finish the send half of a response stream. +/// +/// `what` names the response in the error detail. +pub(super) fn finish_stream(send: &mut quinn::SendStream, what: &str) -> Result<()> { + send.finish().map_err(|e| ClusterError::Transport { + detail: format!("finish {what}: {e}"), + }) +} + +/// Write `rpc` as the single response frame on `send`, then finish `send`. +pub(super) async fn reply_and_finish( + send: &mut quinn::SendStream, + auth: &AuthContext, + rpc: &RaftRpc, + what: &str, +) -> Result<()> { + write_rpc_frame(send, auth, rpc, what).await?; + finish_stream(send, what) +} + +/// Read a complete authenticated envelope from a QUIC receive stream. +/// +/// Reads the fixed envelope pre-header, then the inner frame, then the MAC +/// tag. Returns the full envelope bytes for caller-side parsing. A stream +/// that finishes before the envelope is an error. +pub(crate) async fn read_envelope(recv: &mut quinn::RecvStream) -> Result> { + read_envelope_or_finish(recv) + .await? + .ok_or_else(|| ClusterError::Transport { + detail: "read envelope header: stream finished before an envelope".into(), + }) +} + +/// Read one authenticated envelope, or `None` on a clean finish. +/// +/// A clean finish is the peer finishing its send half with 0 bytes of a +/// new envelope read. A finish part way through an envelope, a reset, or a +/// lost connection is an error. +pub(crate) async fn read_envelope_or_finish( + recv: &mut quinn::RecvStream, +) -> Result>> { + let mut hdr = [0u8; ENV_HDR_LEN]; + if !header_read_outcome(recv.read_exact(&mut hdr).await)? { + return Ok(None); + } + + let inner_len = u32::from_le_bytes([hdr[17], hdr[18], hdr[19], hdr[20]]); + if inner_len > MAX_RPC_PAYLOAD_SIZE { + return Err(ClusterError::Codec { + detail: format!( + "envelope inner length {inner_len} exceeds maximum {MAX_RPC_PAYLOAD_SIZE}" + ), + }); + } + + let total = ENV_HDR_LEN + inner_len as usize + rpc_codec::MAC_LEN; + let mut buf = vec![0u8; total]; + buf[..ENV_HDR_LEN].copy_from_slice(&hdr); + if total > ENV_HDR_LEN { + recv.read_exact(&mut buf[ENV_HDR_LEN..]) + .await + .map_err(|e| ClusterError::Transport { + detail: format!("read envelope payload+mac: {e}"), + })?; + } + + Ok(Some(buf)) +} + +/// Classify the result of reading an envelope header. +/// +/// `Ok(true)` is a full header. `Ok(false)` is a clean finish with 0 bytes +/// read. Every other outcome is a transport error. +fn header_read_outcome(read: std::result::Result<(), quinn::ReadExactError>) -> Result { + match read { + Ok(()) => Ok(true), + Err(quinn::ReadExactError::FinishedEarly(0)) => Ok(false), + Err(e) => Err(ClusterError::Transport { + detail: format!("read envelope header: {e}"), + }), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::rpc_codec::{ExecuteStreamEnd, parse_envelope}; + use crate::transport::credentials::TransportCredentials; + + #[test] + fn only_a_zero_byte_finish_is_a_clean_end() { + assert!(matches!(header_read_outcome(Ok(())), Ok(true))); + assert!(matches!( + header_read_outcome(Err(quinn::ReadExactError::FinishedEarly(0))), + Ok(false) + )); + assert!(matches!( + header_read_outcome(Err(quinn::ReadExactError::FinishedEarly(5))), + Err(ClusterError::Transport { .. }) + )); + assert!(matches!( + header_read_outcome(Err(quinn::ReadExactError::ReadError( + quinn::ReadError::ClosedStream + ))), + Err(ClusterError::Transport { .. }) + )); + } + + #[test] + fn encoded_frame_parses_back_with_fresh_seqs() { + let auth = AuthContext::from_credentials(7, &TransportCredentials::Insecure); + let rpc = RaftRpc::ExecuteStreamEnd(ExecuteStreamEnd { error: None }); + let first = encode_rpc_frame(&rpc, &auth).expect("encode first"); + let second = encode_rpc_frame(&rpc, &auth).expect("encode second"); + + let (f1, inner1) = parse_envelope(&first, &auth.mac_key).expect("parse first"); + let (f2, _) = parse_envelope(&second, &auth.mac_key).expect("parse second"); + assert_eq!(f1.from_node_id, 7); + assert!(f2.seq > f1.seq, "each frame takes a fresh seq"); + assert!(matches!( + rpc_codec::decode(inner1, &auth.epoch).expect("decode"), + RaftRpc::ExecuteStreamEnd(ExecuteStreamEnd { error: None }) + )); + } +} diff --git a/nodedb-cluster/src/transport/identity_admission.rs b/nodedb-cluster/src/transport/identity_admission.rs index 09e933b60..93a13b63b 100644 --- a/nodedb-cluster/src/transport/identity_admission.rs +++ b/nodedb-cluster/src/transport/identity_admission.rs @@ -90,8 +90,10 @@ mod tests { node_id, listen_addr: "127.0.0.1:9400".into(), wire_version: 1, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), spiffe_id: None, spki_pin: Some(pin.to_vec()), + swim_addr: None, }) } diff --git a/nodedb-cluster/src/transport/mod.rs b/nodedb-cluster/src/transport/mod.rs index 64d854956..65a244415 100644 --- a/nodedb-cluster/src/transport/mod.rs +++ b/nodedb-cluster/src/transport/mod.rs @@ -4,13 +4,16 @@ pub mod auth_context; pub mod client; pub mod config; pub mod credentials; +mod frame_io; mod identity_admission; pub mod peer_identity_store; pub mod peer_identity_verifier; pub mod pinned_verifier; pub mod rpc_handler; pub mod server; +mod shuffle_drain; mod stream_dispatch; +mod stream_identity; mod topology_identity_store; pub use auth_context::AuthContext; @@ -41,4 +44,5 @@ pub use peer_identity_verifier::{ IDENTITY_MISMATCH_QUIC_ERROR, VerifyMethod, VerifyOutcome, spki_pin_from_cert_der, }; pub use rpc_handler::RaftRpcHandler; +pub use shuffle_drain::ShufflePushError; pub use topology_identity_store::TopologyIdentityStore; diff --git a/nodedb-cluster/src/transport/server.rs b/nodedb-cluster/src/transport/server.rs index 14fbb746c..ec2cd16d6 100644 --- a/nodedb-cluster/src/transport/server.rs +++ b/nodedb-cluster/src/transport/server.rs @@ -19,7 +19,8 @@ //! 4. decodes the inner frame and dispatches to the handler, //! 5. wraps the handler's response in its own authenticated envelope //! with `from_node_id = local_node_id` and a fresh outbound seq for -//! the caller's id. +//! the caller's id. A handler error is answered with a typed +//! `RequestRefused` frame in place of the response. //! //! # Cooperative shutdown //! @@ -40,18 +41,18 @@ use tracing::{debug, warn}; use crate::error::{ClusterError, Result}; use crate::forward::ChunkSink; use crate::rpc_codec::{ - self, ExecuteStreamChunk, ExecuteStreamEnd, MAX_RPC_PAYLOAD_SIZE, RaftRpc, auth_envelope, + self, ExecuteStreamChunk, ExecuteStreamEnd, FrameRefusal, RaftRpc, RequestRefusal, + auth_envelope, }; use crate::transport::auth_context::AuthContext; -use crate::transport::identity_admission::enrollment_matches; use crate::transport::peer_identity_store::PeerIdentityStore; -use crate::transport::peer_identity_verifier::{ - IDENTITY_MISMATCH_QUIC_ERROR, VerifyOutcome, verify_peer_identity, -}; use crate::transport::rpc_handler::RaftRpcHandler; use crate::wire_version::handshake_io::{local_version_range, perform_version_handshake_server}; +use super::frame_io::{finish_stream, read_envelope, reply_and_finish, write_rpc_frame}; +use super::shuffle_drain::drain_shuffle_push; use super::stream_dispatch; +use super::stream_identity::{reject_peer_identity, verify_stream_identity}; /// Transport-local [`ChunkSink`] that writes one `RPC_EXECUTE_STREAM_CHUNK` /// envelope per chunk to a QUIC send stream. @@ -79,22 +80,7 @@ impl ChunkSink for QuicChunkSink<'_> { payload, watermark_lsn, }); - let inner = rpc_codec::encode(&rpc, &self.auth.epoch)?; - let seq = self.auth.peer_seq_out.next(); - let mut envelope = Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + inner.len()); - auth_envelope::write_envelope( - self.auth.local_node_id, - seq, - &inner, - &self.auth.mac_key, - &mut envelope, - )?; - self.send - .write_all(&envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write stream chunk: {e}"), - }) + write_rpc_frame(self.send, self.auth, &rpc, "stream chunk").await } } @@ -228,14 +214,6 @@ struct StreamContext { shutdown: watch::Receiver, } -fn reject_peer_identity(conn: &quinn::Connection, node_id: u64) -> Result<()> { - warn!(node_id, "peer identity mismatch — closing connection"); - conn.close(IDENTITY_MISMATCH_QUIC_ERROR, b"peer identity mismatch"); - Err(ClusterError::Transport { - detail: format!("peer identity mismatch for node {node_id}"), - }) -} - /// Handle a single bidi stream: read request → dispatch → write response. /// /// Every long-lived await is racing a shutdown signal — see the @@ -267,8 +245,24 @@ async fn handle_stream( // keeps the two flows from tripping on each other's entries — // a self-addressed frame can't have been replayed by an // external attacker by definition. - if fields.from_node_id != auth.local_node_id { - auth.peer_seq_in.accept(fields.from_node_id, fields.seq)?; + // A refused frame is answered with a typed `FrameRefused`, never a + // dropped stream: the MAC verified, so the sender is genuine and + // its link is up. It retries under a fresh sequence number, and a + // dropped stream would count against this node's health instead. + if fields.from_node_id != auth.local_node_id + && let Err(e) = auth.peer_seq_in.accept(fields.from_node_id, fields.seq) + { + debug!( + from_node_id = fields.from_node_id, + error = %e, + "raft RPC frame refused by the replay window" + ); + let refusal = RaftRpc::FrameRefused(FrameRefusal { + detail: e.to_string(), + }); + write_rpc_frame(&mut send, &auth, &refusal, "frame refusal").await?; + finish_stream(&mut send, "frame refusal")?; + return Ok::<(), ClusterError>(()); } // 3. Decode before the identity decision so an unknown, CA-verified @@ -277,51 +271,19 @@ async fn handle_stream( validate_join_sender(&request, fields.from_node_id)?; // 3b. Bind the MAC-authenticated node id to the mTLS leaf identity. - // Unknown identities may submit only a JoinRequest whose node id and - // advertised pins exactly match that leaf. Every other RPC fails - // closed until the successful join is visible in topology. - if fields.from_node_id != auth.local_node_id && identity_store.enforces_peer_identity() { - let cert_der = peer_cert_der - .as_deref() - .ok_or_else(|| ClusterError::Transport { - detail: "authenticated cluster peer did not present a leaf certificate".into(), - })?; - match identity_store.get_node_info(fields.from_node_id) { - Some(ref info) => match verify_peer_identity(info, cert_der) { - VerifyOutcome::Accepted { method } => { - debug!( - node_id = fields.from_node_id, - ?method, - "peer identity verified" - ); - } - VerifyOutcome::Rejected => { - reject_peer_identity(&conn, fields.from_node_id)?; - } - }, - None if enrollment_matches( - &request, - fields.from_node_id, - cert_der, - &*identity_store, - ) => - { - debug!( - node_id = fields.from_node_id, - "accepted identity-bound cluster join enrollment" - ); - } - None => { - reject_peer_identity(&conn, fields.from_node_id)?; - } - } - } + verify_stream_identity( + &conn, + &*identity_store, + peer_cert_der.as_deref(), + &auth, + fields.from_node_id, + &request, + )?; // 4b. Streaming path: an `ExecuteStreamRequest` produces a multi-frame // response — N `RPC_EXECUTE_STREAM_CHUNK` envelopes (each written // inline so QUIC flow control throttles the producer) followed by // exactly one `RPC_EXECUTE_STREAM_END` envelope, then `finish()`. - // The non-streaming path below is unchanged: one response envelope. if let RaftRpc::ExecuteStreamRequest(req) = request { let terminal = { let sink = QuicChunkSink { @@ -332,86 +294,24 @@ async fn handle_stream( }; let end_rpc = RaftRpc::ExecuteStreamEnd(ExecuteStreamEnd { error: terminal }); - let end_inner = rpc_codec::encode(&end_rpc, &auth.epoch)?; - let end_seq = auth.peer_seq_out.next(); - let mut end_envelope = - Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + end_inner.len()); - auth_envelope::write_envelope( - auth.local_node_id, - end_seq, - &end_inner, - &auth.mac_key, - &mut end_envelope, - )?; - send.write_all(&end_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write stream end: {e}"), - })?; - send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish stream response: {e}"), - })?; + write_rpc_frame(&mut send, &auth, &end_rpc, "stream end").await?; + finish_stream(&mut send, "stream response")?; return Ok::<(), ClusterError>(()); } - // 4c. Cross-node streaming shuffle (E1): a `ShufflePushRequest` is the - // opening frame of a producer → receiver stream. The producer keeps - // writing `ShufflePushChunk` envelopes on the SAME bidi stream - // (this read half), terminated by exactly one `ShufflePushEnd`. - // The server reads inbound frames, deposits them via the handler, - // and writes NO reply — the producer fire-and-finishes (mirroring - // the response-direction `send_rpc_stream`, which finishes its send - // half before reading). The loop exits on the `End` frame or on a - // clean stream close (`read_envelope` surfacing a transport error - // after the producer's `finish()`). + // 4c. Cross-node streaming shuffle: a `ShufflePushRequest` opens a + // producer → receiver stream on this read half. The server + // deposits each inbound frame and writes no reply. if let RaftRpc::ShufflePushRequest(req) = request { - let shuffle_id = req.shuffle_id; - let part = req.part; - let side = req.side; - handler.on_shuffle_request(req).await; - - loop { - let frame_envelope = match read_envelope(&mut recv).await { - Ok(e) => e, - // Producer closed the stream without (or after) an End. - // A graceful finish surfaces here as a transport read - // error; treat it as end-of-stream rather than propagating. - Err(_) => return Ok::<(), ClusterError>(()), - }; - let (frame_fields, frame_inner) = - auth_envelope::parse_envelope(&frame_envelope, &auth.mac_key)?; - if frame_fields.from_node_id != fields.from_node_id { - reject_peer_identity(&conn, frame_fields.from_node_id)?; - } - if frame_fields.from_node_id != auth.local_node_id { - auth.peer_seq_in - .accept(frame_fields.from_node_id, frame_fields.seq)?; - } - match rpc_codec::decode(frame_inner, &auth.epoch)? { - RaftRpc::ShufflePushChunk(chunk) => { - handler - .on_shuffle_chunk(shuffle_id, part, side, chunk.payload) - .await?; - } - RaftRpc::ShufflePushEnd(end) => { - handler - .on_shuffle_end(shuffle_id, part, side, end.error) - .await; - return Ok::<(), ClusterError>(()); - } - other => { - return Err(ClusterError::Transport { - detail: format!("unexpected frame in shuffle push stream: {other:?}"), - }); - } - } - } + let opener = fields.from_node_id; + return drain_shuffle_push(&*handler, &auth, &mut recv, req, opener, |node| { + reject_peer_identity(&conn, node) + }) + .await; } - // 4d/4e/4f. One-shot shuffle RPCs (ShuffleProduce / ShuffleConsume / - // ShuffleAggregateConsume). Each is a single request/response — no - // additional frames on `recv`. Handled in a shared helper to keep - // this function under the file-size limit. + // 4d onward. One-shot RPCs: one request, one response, no further + // frames on `recv`. let request = match stream_dispatch::try_handle_oneshot_rpc(&*handler, request, &mut send, &auth) .await? @@ -420,31 +320,22 @@ async fn handle_stream( Some(req) => req, }; - let response = handler.handle_rpc(request).await?; + // A handler error is answered with a typed `RequestRefused`, never a + // dropped stream. The MAC verified, so the sender is genuine and its + // link is up. The sender reads a dropped stream as a link failure. + let outcome = handler.handle_rpc(request).await; + if let Err(e) = &outcome { + debug!( + from_node_id = fields.from_node_id, + error = %e, + "raft RPC refused by the handler" + ); + } + let response = reply_for(outcome); // 5. Wrap the response in its own envelope. `from = local_node_id`, // `seq = next outbound seq scoped to the caller`. - let response_inner = rpc_codec::encode(&response, &auth.epoch)?; - let response_seq = auth.peer_seq_out.next(); - let mut response_envelope = - Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + response_inner.len()); - auth_envelope::write_envelope( - auth.local_node_id, - response_seq, - &response_inner, - &auth.mac_key, - &mut response_envelope, - )?; - - send.write_all(&response_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write response: {e}"), - })?; - send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish response: {e}"), - })?; - Ok::<(), ClusterError>(()) + reply_and_finish(&mut send, &auth, &response, "response").await }; tokio::select! { @@ -454,6 +345,15 @@ async fn handle_stream( } } +/// The frame that answers a one-shot request: the handler's response, or a +/// typed refusal carrying its error. +fn reply_for(outcome: Result) -> RaftRpc { + match outcome { + Ok(response) => response, + Err(error) => RaftRpc::RequestRefused(RequestRefusal::from(error)), + } +} + fn validate_join_sender(request: &RaftRpc, authenticated_node_id: u64) -> Result<()> { if let RaftRpc::JoinRequest(join) = request && join.node_id != authenticated_node_id @@ -468,49 +368,31 @@ fn validate_join_sender(request: &RaftRpc, authenticated_node_id: u64) -> Result Ok(()) } -/// Read a complete authenticated envelope from a QUIC receive stream. -/// -/// Reads the fixed envelope pre-header (version + from_node_id + seq + -/// inner_len), then the inner frame, then the MAC tag. Returns the full -/// envelope bytes for caller-side parsing. -pub(crate) async fn read_envelope(recv: &mut quinn::RecvStream) -> Result> { - // Envelope header is version(1) + from_node_id(8) + seq(8) + inner_len(4). - const ENV_HDR_LEN: usize = 21; - - let mut hdr = [0u8; ENV_HDR_LEN]; - recv.read_exact(&mut hdr) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("read envelope header: {e}"), - })?; - - let inner_len = u32::from_le_bytes([hdr[17], hdr[18], hdr[19], hdr[20]]); - if inner_len > MAX_RPC_PAYLOAD_SIZE { - return Err(ClusterError::Codec { - detail: format!( - "envelope inner length {inner_len} exceeds maximum {MAX_RPC_PAYLOAD_SIZE}" - ), - }); - } - - let total = ENV_HDR_LEN + inner_len as usize + rpc_codec::MAC_LEN; - let mut buf = vec![0u8; total]; - buf[..ENV_HDR_LEN].copy_from_slice(&hdr); - if total > ENV_HDR_LEN { - recv.read_exact(&mut buf[ENV_HDR_LEN..]) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("read envelope payload+mac: {e}"), - })?; - } - - Ok(buf) -} - #[cfg(test)] mod tests { use super::*; - use crate::rpc_codec::JoinRequest; + use crate::rpc_codec::{JoinRequest, RefusalReason}; + + #[test] + fn a_handler_error_is_answered_with_a_typed_refusal() { + let reply = reply_for(Err(ClusterError::GroupNotFound { group_id: 4 })); + match reply { + RaftRpc::RequestRefused(refusal) => assert!(matches!( + refusal.reason, + RefusalReason::GroupNotHosted { group_id: 4 } + )), + other => panic!("expected a refusal, got {other:?}"), + } + } + + #[test] + fn a_handler_response_is_answered_as_is() { + let pong = RaftRpc::Pong(crate::rpc_codec::PongResponse { + responder_id: 1, + topology_version: 3, + }); + assert!(matches!(reply_for(Ok(pong)), RaftRpc::Pong(_))); + } #[test] fn join_request_must_match_authenticated_sender() { @@ -518,8 +400,10 @@ mod tests { node_id: 9, listen_addr: "127.0.0.1:9400".into(), wire_version: crate::topology::CLUSTER_WIRE_FORMAT_VERSION, + build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), spiffe_id: None, spki_pin: Some(vec![1; 32]), + swim_addr: None, }); assert!(validate_join_sender(&request, 9).is_ok()); let error = validate_join_sender(&request, 8).unwrap_err(); diff --git a/nodedb-cluster/src/transport/shuffle_drain.rs b/nodedb-cluster/src/transport/shuffle_drain.rs new file mode 100644 index 000000000..2ef71004a --- /dev/null +++ b/nodedb-cluster/src/transport/shuffle_drain.rs @@ -0,0 +1,418 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Receiver side of a cross-node streaming shuffle push. +//! +//! A `ShufflePushRequest` opens a producer → receiver stream. The producer +//! writes `ShufflePushChunk` envelopes on the same bidi stream, then exactly +//! one `ShufflePushEnd`. The receiver deposits each chunk and writes no +//! reply. + +use std::future::Future; + +use crate::error::{ClusterError, Result}; +use crate::rpc_codec::{self, RaftRpc, ShufflePushRequest, TypedClusterError, auth_envelope}; +use crate::transport::auth_context::AuthContext; +use crate::transport::rpc_handler::RaftRpcHandler; + +use super::frame_io::read_envelope_or_finish; + +/// The `(shuffle_id, part, side)` a push stream feeds. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct PushKey { + pub shuffle_id: u64, + pub part: u32, + pub side: u8, +} + +/// A shuffle push stream broke its frame protocol. +#[derive(Debug, thiserror::Error, Clone, PartialEq, Eq)] +pub enum ShufflePushError { + /// The producer finished its send stream before the `ShufflePushEnd` + /// frame. The receiver holds a partial partition. + #[error("shuffle push stream ({shuffle_id},{part},{side}) ended before its end frame")] + Truncated { + shuffle_id: u64, + part: u32, + side: u8, + }, +} + +/// A stream of inbound envelopes. `None` is a clean finish. +pub(super) trait EnvelopeSource { + fn next_envelope(&mut self) -> impl Future>>> + Send; +} + +impl EnvelopeSource for quinn::RecvStream { + fn next_envelope(&mut self) -> impl Future>>> + Send { + read_envelope_or_finish(self) + } +} + +/// The receiver inbox a push stream feeds. +pub(super) trait ShuffleInbox: Sync { + fn open(&self, req: ShufflePushRequest) -> impl Future + Send; + fn deposit(&self, key: PushKey, payload: Vec) -> impl Future> + Send; + fn end( + &self, + key: PushKey, + error: Option, + ) -> impl Future + Send; +} + +impl ShuffleInbox for H { + fn open(&self, req: ShufflePushRequest) -> impl Future + Send { + self.on_shuffle_request(req) + } + + fn deposit(&self, key: PushKey, payload: Vec) -> impl Future> + Send { + self.on_shuffle_chunk(key.shuffle_id, key.part, key.side, payload) + } + + fn end( + &self, + key: PushKey, + error: Option, + ) -> impl Future + Send { + self.on_shuffle_end(key.shuffle_id, key.part, key.side, error) + } +} + +/// Open the inbox for `req`, then deposit every inbound frame until the +/// `ShufflePushEnd` frame. +/// +/// A clean finish before the `ShufflePushEnd` frame is a truncated stream +/// and fails with [`ShufflePushError::Truncated`]. Every frame must come from `opener_node_id`. A frame from another node +/// calls `reject_identity`. Any error ends the inbox with that error, so +/// the consumer never waits on a producer that is gone, and the error +/// returns to the caller. +pub(super) async fn drain_shuffle_push( + inbox: &I, + auth: &AuthContext, + source: &mut S, + req: ShufflePushRequest, + opener_node_id: u64, + reject_identity: impl Fn(u64) -> Result<()> + Send, +) -> Result<()> +where + I: ShuffleInbox + ?Sized, + S: EnvelopeSource + Send, +{ + let key = PushKey { + shuffle_id: req.shuffle_id, + part: req.part, + side: req.side, + }; + inbox.open(req).await; + let result = drain_frames(inbox, auth, source, key, opener_node_id, reject_identity).await; + if let Err(err) = &result { + inbox.end(key, Some(stream_failure(key, err))).await; + } + result +} + +/// Deposit frames until the `End` frame. A clean finish before it is +/// [`ShufflePushError::Truncated`]. +async fn drain_frames( + inbox: &I, + auth: &AuthContext, + source: &mut S, + key: PushKey, + opener_node_id: u64, + reject_identity: impl Fn(u64) -> Result<()> + Send, +) -> Result<()> +where + I: ShuffleInbox + ?Sized, + S: EnvelopeSource + Send, +{ + loop { + let Some(envelope) = source.next_envelope().await? else { + return Err(ShufflePushError::Truncated { + shuffle_id: key.shuffle_id, + part: key.part, + side: key.side, + } + .into()); + }; + let (fields, inner) = auth_envelope::parse_envelope(&envelope, &auth.mac_key)?; + if fields.from_node_id != opener_node_id { + reject_identity(fields.from_node_id)?; + } + if fields.from_node_id != auth.local_node_id { + auth.peer_seq_in.accept(fields.from_node_id, fields.seq)?; + } + match rpc_codec::decode(inner, &auth.epoch)? { + RaftRpc::ShufflePushChunk(chunk) => inbox.deposit(key, chunk.payload).await?, + RaftRpc::ShufflePushEnd(end) => { + inbox.end(key, end.error).await; + return Ok(()); + } + other => { + return Err(ClusterError::Transport { + detail: format!("unexpected frame in shuffle push stream: {other:?}"), + }); + } + } + } +} + +/// The inbox error for a push stream that failed with `err`. +fn stream_failure(key: PushKey, err: &ClusterError) -> TypedClusterError { + TypedClusterError::Internal { + code: 0, + message: format!( + "shuffle push stream ({},{},{}) failed: {err}", + key.shuffle_id, key.part, key.side + ), + } +} + +#[cfg(test)] +mod tests { + use std::collections::VecDeque; + use std::sync::Mutex; + + use super::*; + use crate::rpc_codec::{ShufflePushChunk, ShufflePushEnd}; + use crate::transport::credentials::TransportCredentials; + use crate::transport::frame_io::encode_rpc_frame; + + const PRODUCER: u64 = 2; + const RECEIVER: u64 = 1; + + #[derive(Debug, PartialEq)] + enum Event { + Open, + Deposit(Vec), + End { failed: bool }, + } + + #[derive(Default)] + struct Recorder(Mutex>); + + impl Recorder { + fn push(&self, event: Event) { + self.0.lock().expect("recorder lock").push(event); + } + + fn events(&self) -> Vec { + std::mem::take(&mut *self.0.lock().expect("recorder lock")) + } + } + + impl ShuffleInbox for Recorder { + fn open(&self, _req: ShufflePushRequest) -> impl Future + Send { + self.push(Event::Open); + std::future::ready(()) + } + + fn deposit( + &self, + _key: PushKey, + payload: Vec, + ) -> impl Future> + Send { + self.push(Event::Deposit(payload)); + std::future::ready(Ok(())) + } + + fn end( + &self, + _key: PushKey, + error: Option, + ) -> impl Future + Send { + self.push(Event::End { + failed: error.is_some(), + }); + std::future::ready(()) + } + } + + struct Scripted(VecDeque>>>); + + impl EnvelopeSource for Scripted { + fn next_envelope(&mut self) -> impl Future>>> + Send { + std::future::ready(self.0.pop_front().unwrap_or(Ok(None))) + } + } + + fn request() -> ShufflePushRequest { + ShufflePushRequest { + shuffle_id: 9, + part: 3, + side: 1, + num_parts: 4, + producer_count: 1, + } + } + + fn auths() -> (AuthContext, AuthContext) { + ( + AuthContext::from_credentials(RECEIVER, &TransportCredentials::Insecure), + AuthContext::from_credentials(PRODUCER, &TransportCredentials::Insecure), + ) + } + + fn chunk(producer: &AuthContext, payload: &[u8]) -> Result>> { + let rpc = RaftRpc::ShufflePushChunk(ShufflePushChunk { + payload: payload.to_vec(), + }); + encode_rpc_frame(&rpc, producer).map(Some) + } + + fn no_reject(node: u64) -> Result<()> { + Err(ClusterError::Transport { + detail: format!("unexpected identity check for node {node}"), + }) + } + + /// A read error mid-stream ends the inbox with an error and returns the + /// error. It is never a clean end of stream. + #[tokio::test] + async fn read_error_ends_the_inbox_with_the_error() { + let (receiver, producer) = auths(); + let mut source = Scripted(VecDeque::from([ + chunk(&producer, b"a"), + Err(ClusterError::Transport { + detail: "read envelope header: connection lost".into(), + }), + ])); + let inbox = Recorder::default(); + let result = drain_shuffle_push( + &inbox, + &receiver, + &mut source, + request(), + PRODUCER, + no_reject, + ) + .await; + assert!(matches!(result, Err(ClusterError::Transport { .. }))); + assert_eq!( + inbox.events(), + vec![ + Event::Open, + Event::Deposit(b"a".to_vec()), + Event::End { failed: true } + ] + ); + } + + /// The `End` frame ends the inbox once, with the producer's outcome. + #[tokio::test] + async fn end_frame_ends_the_inbox_once() { + let (receiver, producer) = auths(); + let end = encode_rpc_frame( + &RaftRpc::ShufflePushEnd(ShufflePushEnd { error: None }), + &producer, + ) + .map(Some); + let mut source = Scripted(VecDeque::from([chunk(&producer, b"a"), end])); + let inbox = Recorder::default(); + drain_shuffle_push( + &inbox, + &receiver, + &mut source, + request(), + PRODUCER, + no_reject, + ) + .await + .expect("clean drain"); + assert_eq!( + inbox.events(), + vec![ + Event::Open, + Event::Deposit(b"a".to_vec()), + Event::End { failed: false } + ] + ); + } + + /// A clean finish after the `End` frame is the end of the stream. A + /// clean finish before it is a truncated stream: the drain fails and + /// ends the inbox with the error. + #[tokio::test] + async fn clean_finish_is_end_of_stream() { + let (receiver, producer) = auths(); + let end = encode_rpc_frame( + &RaftRpc::ShufflePushEnd(ShufflePushEnd { error: None }), + &producer, + ) + .map(Some); + let mut source = Scripted(VecDeque::from([chunk(&producer, b"a"), end, Ok(None)])); + let inbox = Recorder::default(); + drain_shuffle_push( + &inbox, + &receiver, + &mut source, + request(), + PRODUCER, + no_reject, + ) + .await + .expect("clean finish after the end frame"); + assert_eq!( + inbox.events(), + vec![ + Event::Open, + Event::Deposit(b"a".to_vec()), + Event::End { failed: false } + ] + ); + + let mut source = Scripted(VecDeque::from([chunk(&producer, b"a"), Ok(None)])); + let inbox = Recorder::default(); + let result = drain_shuffle_push( + &inbox, + &receiver, + &mut source, + request(), + PRODUCER, + no_reject, + ) + .await; + assert!(matches!( + result, + Err(ClusterError::ShufflePush(ShufflePushError::Truncated { + shuffle_id: 9, + part: 3, + side: 1, + })) + )); + assert_eq!( + inbox.events(), + vec![ + Event::Open, + Event::Deposit(b"a".to_vec()), + Event::End { failed: true } + ] + ); + } + + /// A frame from a node other than the opener fails the drain and ends + /// the inbox with the error. + #[tokio::test] + async fn foreign_frame_fails_the_drain() { + let (receiver, _) = auths(); + let intruder = AuthContext::from_credentials(5, &TransportCredentials::Insecure); + let mut source = Scripted(VecDeque::from([chunk(&intruder, b"x")])); + let inbox = Recorder::default(); + let result = drain_shuffle_push( + &inbox, + &receiver, + &mut source, + request(), + PRODUCER, + |node| { + Err(ClusterError::Transport { + detail: format!("peer identity mismatch for node {node}"), + }) + }, + ) + .await; + assert!(matches!(result, Err(ClusterError::Transport { .. }))); + assert_eq!( + inbox.events(), + vec![Event::Open, Event::End { failed: true }] + ); + } +} diff --git a/nodedb-cluster/src/transport/stream_dispatch.rs b/nodedb-cluster/src/transport/stream_dispatch.rs index 0032948a8..d6173614b 100644 --- a/nodedb-cluster/src/transport/stream_dispatch.rs +++ b/nodedb-cluster/src/transport/stream_dispatch.rs @@ -8,19 +8,21 @@ //! already-decoded [`RaftRpc`] value, calls the handler, encodes the response, //! and finishes the send stream. //! -//! Arms that read additional frames from `recv` (the `ExecuteStreamRequest` -//! and `ShufflePushRequest` streaming arms) remain in `server.rs` because -//! their borrow shape differs. +//! Arms that read additional frames from `recv` live elsewhere: the +//! `ExecuteStreamRequest` arm in `server.rs` and the `ShufflePushRequest` +//! arm in `shuffle_drain.rs`. -use crate::error::{ClusterError, Result}; +use crate::error::Result; use crate::rpc_codec::{ - self, AssignSurrogateResponse, RaftRpc, ReleaseReservationResponse, ReserveReadResponse, + AssignSurrogateResponse, RaftRpc, ReleaseReservationResponse, ReserveReadResponse, ShuffleAggregateConsumeResponse, ShuffleConsumeResponse, SubmitCalvinInboxResponse, - SubmitCalvinTxnResponse, auth_envelope, + SubmitCalvinTxnResponse, }; use crate::transport::auth_context::AuthContext; use crate::transport::rpc_handler::RaftRpcHandler; +use super::frame_io::{finish_stream, reply_and_finish}; + /// Attempt to handle a one-shot (single-request / single-response) RPC. /// /// Checks whether `request` is one of the known one-shot shuffle variants @@ -38,7 +40,7 @@ pub(super) async fn try_handle_oneshot_rpc( send: &mut quinn::SendStream, auth: &AuthContext, ) -> Result> { - // 4d. Cross-node shuffle PRODUCER trigger (E4a): a `ShuffleProduceRequest` + // 4d. Cross-node shuffle PRODUCER trigger: a `ShuffleProduceRequest` // is a ONE-SHOT request/response (NOT a stream from the coordinator). // The producer runs a local scan, fans the hash-partitioned rows out // to the part-owners on its OWN outbound `ShufflePush` streams, then @@ -49,29 +51,11 @@ pub(super) async fn try_handle_oneshot_rpc( if let RaftRpc::ShuffleProduceRequest(req) = request { let resp = handler.on_shuffle_produce(req).await; let resp_rpc = RaftRpc::ShuffleProduceResponse(resp); - let resp_inner = rpc_codec::encode(&resp_rpc, &auth.epoch)?; - let resp_seq = auth.peer_seq_out.next(); - let mut resp_envelope = - Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + resp_inner.len()); - auth_envelope::write_envelope( - auth.local_node_id, - resp_seq, - &resp_inner, - &auth.mac_key, - &mut resp_envelope, - )?; - send.write_all(&resp_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write shuffle produce response: {e}"), - })?; - send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish shuffle produce response: {e}"), - })?; + reply_and_finish(send, auth, &resp_rpc, "shuffle produce response").await?; return Ok(None); } - // 4e. Cross-node shuffle CONSUMER trigger (E4b): a `ShuffleConsumeRequest` + // 4e. Cross-node shuffle CONSUMER trigger: a `ShuffleConsumeRequest` // is a ONE-SHOT request/response. The part-owner waits for both staged // sides of its part to finalize, runs the node-local grace join, and // replies with exactly one `ShuffleConsumeResponse` carrying the join @@ -81,29 +65,11 @@ pub(super) async fn try_handle_oneshot_rpc( if let RaftRpc::ShuffleConsumeRequest(req) = request { let resp: ShuffleConsumeResponse = handler.on_shuffle_consume(req).await; let resp_rpc = RaftRpc::ShuffleConsumeResponse(resp); - let resp_inner = rpc_codec::encode(&resp_rpc, &auth.epoch)?; - let resp_seq = auth.peer_seq_out.next(); - let mut resp_envelope = - Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + resp_inner.len()); - auth_envelope::write_envelope( - auth.local_node_id, - resp_seq, - &resp_inner, - &auth.mac_key, - &mut resp_envelope, - )?; - send.write_all(&resp_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write shuffle consume response: {e}"), - })?; - send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish shuffle consume response: {e}"), - })?; + reply_and_finish(send, auth, &resp_rpc, "shuffle consume response").await?; return Ok(None); } - // 4f. Cross-node distributed GROUP BY shuffle CONSUMER trigger (E5b): a + // 4f. Cross-node distributed GROUP BY shuffle CONSUMER trigger: a // `ShuffleAggregateConsumeRequest` is a ONE-SHOT request/response and // the single-sided aggregate sibling of the consume arm above. The // part-owner waits for its part's single staged producer side to @@ -115,29 +81,11 @@ pub(super) async fn try_handle_oneshot_rpc( if let RaftRpc::ShuffleAggregateConsumeRequest(req) = request { let resp: ShuffleAggregateConsumeResponse = handler.on_shuffle_aggregate(req).await; let resp_rpc = RaftRpc::ShuffleAggregateConsumeResponse(resp); - let resp_inner = rpc_codec::encode(&resp_rpc, &auth.epoch)?; - let resp_seq = auth.peer_seq_out.next(); - let mut resp_envelope = - Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + resp_inner.len()); - auth_envelope::write_envelope( - auth.local_node_id, - resp_seq, - &resp_inner, - &auth.mac_key, - &mut resp_envelope, - )?; - send.write_all(&resp_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write shuffle aggregate consume response: {e}"), - })?; - send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish shuffle aggregate consume response: {e}"), - })?; + reply_and_finish(send, auth, &resp_rpc, "shuffle aggregate consume response").await?; return Ok(None); } - // 4g. Routed-surrogate-exchange (F1b): an `AssignSurrogateRequest` is a + // 4g. Routed-surrogate-exchange: an `AssignSurrogateRequest` is a // ONE-SHOT request/response. This node is the home vShard's leader; it // assign-or-returns the authoritative surrogate for the `(collection, // pk)` endpoint key and replies with exactly one @@ -147,29 +95,11 @@ pub(super) async fn try_handle_oneshot_rpc( if let RaftRpc::AssignSurrogateRequest(req) = request { let resp: AssignSurrogateResponse = handler.on_assign_surrogate(req).await; let resp_rpc = RaftRpc::AssignSurrogateResponse(resp); - let resp_inner = rpc_codec::encode(&resp_rpc, &auth.epoch)?; - let resp_seq = auth.peer_seq_out.next(); - let mut resp_envelope = - Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + resp_inner.len()); - auth_envelope::write_envelope( - auth.local_node_id, - resp_seq, - &resp_inner, - &auth.mac_key, - &mut resp_envelope, - )?; - send.write_all(&resp_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write assign surrogate response: {e}"), - })?; - send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish assign surrogate response: {e}"), - })?; + reply_and_finish(send, auth, &resp_rpc, "assign surrogate response").await?; return Ok(None); } - // 4h. Routed Calvin-submit (Cv1): a `SubmitCalvinTxnRequest` is a ONE-SHOT + // 4h. Routed Calvin-submit: a `SubmitCalvinTxnRequest` is a ONE-SHOT // request/response. This node is the SEQUENCER-GROUP leader; it submits // the carried `TxClass` to its local Calvin sequencer inbox, awaits // assignment + completion, and replies with exactly one @@ -179,29 +109,11 @@ pub(super) async fn try_handle_oneshot_rpc( if let RaftRpc::SubmitCalvinTxnRequest(req) = request { let resp: SubmitCalvinTxnResponse = handler.on_submit_calvin_txn(req).await; let resp_rpc = RaftRpc::SubmitCalvinTxnResponse(resp); - let resp_inner = rpc_codec::encode(&resp_rpc, &auth.epoch)?; - let resp_seq = auth.peer_seq_out.next(); - let mut resp_envelope = - Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + resp_inner.len()); - auth_envelope::write_envelope( - auth.local_node_id, - resp_seq, - &resp_inner, - &auth.mac_key, - &mut resp_envelope, - )?; - send.write_all(&resp_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write submit calvin txn response: {e}"), - })?; - send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish submit calvin txn response: {e}"), - })?; + reply_and_finish(send, auth, &resp_rpc, "submit calvin txn response").await?; return Ok(None); } - // 4i. Routed Calvin-INBOX submit (Cv1): a `SubmitCalvinInboxRequest` is + // 4i. Routed Calvin-INBOX submit:a `SubmitCalvinInboxRequest` is // a ONE-SHOT request/response and the OLLP dependent sibling of the // submit-calvin-txn arm above. This node is the SEQUENCER-GROUP leader; it // submits the carried `TxClass` to its local Calvin sequencer inbox, @@ -212,25 +124,7 @@ pub(super) async fn try_handle_oneshot_rpc( if let RaftRpc::SubmitCalvinInboxRequest(req) = request { let resp: SubmitCalvinInboxResponse = handler.on_submit_calvin_inbox(req).await; let resp_rpc = RaftRpc::SubmitCalvinInboxResponse(resp); - let resp_inner = rpc_codec::encode(&resp_rpc, &auth.epoch)?; - let resp_seq = auth.peer_seq_out.next(); - let mut resp_envelope = - Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + resp_inner.len()); - auth_envelope::write_envelope( - auth.local_node_id, - resp_seq, - &resp_inner, - &auth.mac_key, - &mut resp_envelope, - )?; - send.write_all(&resp_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write submit calvin inbox response: {e}"), - })?; - send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish submit calvin inbox response: {e}"), - })?; + reply_and_finish(send, auth, &resp_rpc, "submit calvin inbox response").await?; return Ok(None); } @@ -244,25 +138,7 @@ pub(super) async fn try_handle_oneshot_rpc( if let RaftRpc::ReserveReadRequest(req) = request { let resp: ReserveReadResponse = handler.on_reserve_read(req).await; let resp_rpc = RaftRpc::ReserveReadResponse(resp); - let resp_inner = rpc_codec::encode(&resp_rpc, &auth.epoch)?; - let resp_seq = auth.peer_seq_out.next(); - let mut resp_envelope = - Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + resp_inner.len()); - auth_envelope::write_envelope( - auth.local_node_id, - resp_seq, - &resp_inner, - &auth.mac_key, - &mut resp_envelope, - )?; - send.write_all(&resp_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write reserve read response: {e}"), - })?; - send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish reserve read response: {e}"), - })?; + reply_and_finish(send, auth, &resp_rpc, "reserve read response").await?; return Ok(None); } @@ -277,25 +153,7 @@ pub(super) async fn try_handle_oneshot_rpc( if let RaftRpc::ReleaseReservationRequest(req) = request { let resp: ReleaseReservationResponse = handler.on_release_reservation(req).await; let resp_rpc = RaftRpc::ReleaseReservationResponse(resp); - let resp_inner = rpc_codec::encode(&resp_rpc, &auth.epoch)?; - let resp_seq = auth.peer_seq_out.next(); - let mut resp_envelope = - Vec::with_capacity(auth_envelope::ENVELOPE_OVERHEAD + resp_inner.len()); - auth_envelope::write_envelope( - auth.local_node_id, - resp_seq, - &resp_inner, - &auth.mac_key, - &mut resp_envelope, - )?; - send.write_all(&resp_envelope) - .await - .map_err(|e| ClusterError::Transport { - detail: format!("write release reservation response: {e}"), - })?; - send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish release reservation response: {e}"), - })?; + reply_and_finish(send, auth, &resp_rpc, "release reservation response").await?; return Ok(None); } @@ -304,9 +162,7 @@ pub(super) async fn try_handle_oneshot_rpc( // response frame. The sender does not await a reply (it discards recv). if let RaftRpc::TimeoutNowRequest(req) = request { handler.on_timeout_now(req).await; - send.finish().map_err(|e| ClusterError::Transport { - detail: format!("finish timeout_now (no-response): {e}"), - })?; + finish_stream(send, "timeout_now (no-response)")?; return Ok(None); } diff --git a/nodedb-cluster/src/transport/stream_identity.rs b/nodedb-cluster/src/transport/stream_identity.rs new file mode 100644 index 000000000..23e3a118b --- /dev/null +++ b/nodedb-cluster/src/transport/stream_identity.rs @@ -0,0 +1,63 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Bind the MAC-authenticated sender of an inbound stream to its mTLS leaf +//! identity. + +use tracing::{debug, warn}; + +use crate::error::{ClusterError, Result}; +use crate::rpc_codec::RaftRpc; +use crate::transport::auth_context::AuthContext; +use crate::transport::identity_admission::enrollment_matches; +use crate::transport::peer_identity_store::PeerIdentityStore; +use crate::transport::peer_identity_verifier::{ + IDENTITY_MISMATCH_QUIC_ERROR, VerifyOutcome, verify_peer_identity, +}; + +/// Close `conn` for a peer identity mismatch and return the matching error. +pub(super) fn reject_peer_identity(conn: &quinn::Connection, node_id: u64) -> Result<()> { + warn!(node_id, "peer identity mismatch — closing connection"); + conn.close(IDENTITY_MISMATCH_QUIC_ERROR, b"peer identity mismatch"); + Err(ClusterError::Transport { + detail: format!("peer identity mismatch for node {node_id}"), + }) +} + +/// Check that `from_node_id` matches the leaf certificate `peer_cert_der`. +/// +/// Self-addressed frames and stores that do not enforce peer identity pass. +/// An unknown identity may submit only a `JoinRequest` whose node id and +/// advertised pins match that leaf. Every other RPC fails closed until the +/// join is visible in topology. A mismatch closes `conn`. +pub(super) fn verify_stream_identity( + conn: &quinn::Connection, + identity_store: &S, + peer_cert_der: Option<&[u8]>, + auth: &AuthContext, + from_node_id: u64, + request: &RaftRpc, +) -> Result<()> { + if from_node_id == auth.local_node_id || !identity_store.enforces_peer_identity() { + return Ok(()); + } + let cert_der = peer_cert_der.ok_or_else(|| ClusterError::Transport { + detail: "authenticated cluster peer did not present a leaf certificate".into(), + })?; + match identity_store.get_node_info(from_node_id) { + Some(ref info) => match verify_peer_identity(info, cert_der) { + VerifyOutcome::Accepted { method } => { + debug!(node_id = from_node_id, ?method, "peer identity verified"); + Ok(()) + } + VerifyOutcome::Rejected => reject_peer_identity(conn, from_node_id), + }, + None if enrollment_matches(request, from_node_id, cert_der, identity_store) => { + debug!( + node_id = from_node_id, + "accepted identity-bound cluster join enrollment" + ); + Ok(()) + } + None => reject_peer_identity(conn, from_node_id), + } +} diff --git a/nodedb-cluster/src/wire.rs b/nodedb-cluster/src/wire.rs index 2aa4a973e..90a9f7e3a 100644 --- a/nodedb-cluster/src/wire.rs +++ b/nodedb-cluster/src/wire.rs @@ -202,7 +202,24 @@ impl VShardEnvelope { } let payload = buf[28..28 + payload_len].to_vec(); - let msg_type = match msg_type_raw { + let msg_type = VShardMessageType::from_raw(msg_type_raw)?; + + Some(Self { + version, + msg_type, + source_node, + target_node, + vshard_id, + payload, + }) + } +} + +impl VShardMessageType { + /// The variant whose discriminant is `raw`, or `None` for a discriminant + /// no variant carries. The one opcode table of the vShard wire format. + pub fn from_raw(raw: u16) -> Option { + let msg_type = match raw { 1 => VShardMessageType::SegmentChunk, 2 => VShardMessageType::SegmentComplete, 3 => VShardMessageType::WalTail, @@ -248,15 +265,7 @@ impl VShardEnvelope { 89 => VShardMessageType::ArrayShardSurrogateBitmapResp, _ => return None, }; - - Some(Self { - version, - msg_type, - source_node, - target_node, - vshard_id, - payload, - }) + Some(msg_type) } } diff --git a/nodedb-cluster/src/wire_version/error.rs b/nodedb-cluster/src/wire_version/error.rs index afaa09a6c..245c91cdd 100644 --- a/nodedb-cluster/src/wire_version/error.rs +++ b/nodedb-cluster/src/wire_version/error.rs @@ -39,6 +39,19 @@ pub enum WireVersionError { remote_min: WireVersion, remote_max: WireVersion, }, + + /// The peer negotiated a compatible wire version but is running a + /// different build. Refused unconditionally before 1.0 — see + /// `nodedb_types::wire_version` for why build identity, not wire + /// version, is the invariant that actually protects mixed builds. + #[error( + "build identity mismatch: local build {local_build_id}, peer build {peer_build_id} — \ + all nodes must run one build before 1.0; restart every node on the same build" + )] + BuildIdMismatch { + local_build_id: String, + peer_build_id: String, + }, } impl From for crate::error::ClusterError { diff --git a/nodedb-cluster/src/wire_version/handshake_io.rs b/nodedb-cluster/src/wire_version/handshake_io.rs index 5655d61ba..e377e71ac 100644 --- a/nodedb-cluster/src/wire_version/handshake_io.rs +++ b/nodedb-cluster/src/wire_version/handshake_io.rs @@ -31,12 +31,30 @@ use crate::wire::WIRE_VERSION; /// length prefixes before we allocate a receive buffer). const MAX_HANDSHAKE_BYTES: u32 = 4 * 1024; // 4 KiB — far more than needed -/// The local version range derived from the compile-time constants in -/// `rpc_codec::header`. Single source of truth for both sides. +/// The local version range derived from the compile-time constant in +/// `crate::wire::WIRE_VERSION`. Single source of truth for both sides. +/// +/// Floor == ceiling: this build advertises only its own exact frame +/// version, never a range of supported older versions — there is no +/// rolling-upgrade window pre-1.0 (see `nodedb_types::wire_version`). pub fn local_version_range() -> VersionRange { - // Supported range: [1, WIRE_VERSION]. Min is 1 (oldest supported); - // max is the current build's wire version. - VersionRange::new(WireVersion(1), WireVersion(WIRE_VERSION)) + VersionRange::new(WireVersion(WIRE_VERSION), WireVersion(WIRE_VERSION)) +} + +/// This build's exact identity, compared for equality in every handshake. +pub fn local_build_id() -> &'static str { + nodedb_types::wire_version::WIRE_BUILD_ID +} + +/// Compare two build identities for exact equality. +fn check_build_id(local: &str, remote: &str) -> std::result::Result<(), WireVersionError> { + if local != remote { + return Err(WireVersionError::BuildIdMismatch { + local_build_id: local.to_owned(), + peer_build_id: remote.to_owned(), + }); + } + Ok(()) } /// Write a length-prefixed zerompk message to `send`. @@ -119,7 +137,19 @@ pub async fn perform_version_handshake_server( } })?; - let ack = VersionHandshakeAck::new(agreed); + if let Err(e) = check_build_id(local_build_id(), &client_hs.build_id) { + let reason = e.to_string(); + let reason_bytes = reason.as_bytes(); + conn.close( + quinn::VarInt::from_u32(0x01), + &reason_bytes[..reason_bytes.len().min(100)], + ); + return Err(ClusterError::Transport { + detail: format!("wire version handshake failed (server): {e}"), + }); + } + + let ack = VersionHandshakeAck::new(agreed, local_build_id().to_owned()); write_framed(send, &ack).await?; Ok(agreed) @@ -135,7 +165,7 @@ pub async fn perform_version_handshake_client( recv: &mut quinn::RecvStream, ) -> Result { let local = local_version_range(); - let hs = VersionHandshake::from_range(local); + let hs = VersionHandshake::from_range(local, local_build_id().to_owned()); write_framed(send, &hs).await?; let ack: VersionHandshakeAck = read_framed(recv).await?; @@ -153,6 +183,13 @@ pub async fn perform_version_handshake_client( }); } + // Belt-and-suspenders: the server already refuses a build mismatch + // before sending an ack, but a misbehaving server must not be trusted + // silently either. + check_build_id(local_build_id(), &ack.build_id).map_err(|e| ClusterError::Transport { + detail: format!("wire version handshake failed (client): {e}"), + })?; + Ok(agreed) } @@ -196,7 +233,7 @@ mod tests { r.min, r.max ); - assert_eq!(r.min, v(1)); + assert_eq!(r.min, v(WIRE_VERSION)); assert_eq!(r.max, v(WIRE_VERSION)); } @@ -235,6 +272,7 @@ mod tests { let hs = VersionHandshake { range: (1, 3), capabilities: caps, + build_id: "test-build".to_owned(), }; let bytes = zerompk::to_msgpack_vec(&hs).unwrap(); let decoded: VersionHandshake = zerompk::from_msgpack(&bytes).unwrap(); @@ -249,10 +287,28 @@ mod tests { let ack = VersionHandshakeAck { agreed: 2, capabilities: caps, + build_id: "test-build".to_owned(), }; let bytes = zerompk::to_msgpack_vec(&ack).unwrap(); let decoded: VersionHandshakeAck = zerompk::from_msgpack(&bytes).unwrap(); assert_eq!(decoded.agreed_version(), v(2)); assert_eq!(decoded.capabilities, caps); } + + /// A build-id mismatch is refused with the operator-facing message. + #[test] + fn build_id_mismatch_is_refused_with_message() { + let err = check_build_id("build-a", "build-b").unwrap_err(); + assert!(matches!(err, WireVersionError::BuildIdMismatch { .. })); + let msg = err.to_string(); + assert!(msg.contains("all nodes must run one build before 1.0")); + assert!(msg.contains("build-a")); + assert!(msg.contains("build-b")); + } + + /// Matching build identities are accepted. + #[test] + fn matching_build_id_is_accepted() { + assert!(check_build_id("same-build", "same-build").is_ok()); + } } diff --git a/nodedb-cluster/src/wire_version/negotiation.rs b/nodedb-cluster/src/wire_version/negotiation.rs index 1e802fca0..7b0738c6e 100644 --- a/nodedb-cluster/src/wire_version/negotiation.rs +++ b/nodedb-cluster/src/wire_version/negotiation.rs @@ -88,6 +88,9 @@ pub struct VersionHandshake { /// capabilities) so older peers that do not set this field remain compatible. #[serde(default)] pub capabilities: u64, + /// Sender's `nodedb_types::wire_version::WIRE_BUILD_ID`. Compared for + /// exact equality by the receiver — see `handshake_io::perform_version_handshake_server`. + pub build_id: String, } /// Server-side acknowledgement returned after negotiation succeeds. @@ -107,14 +110,18 @@ pub struct VersionHandshakeAck { /// Unknown bits are ignored by the receiver. Defaults to `0`. #[serde(default)] pub capabilities: u64, + /// Server's `nodedb_types::wire_version::WIRE_BUILD_ID`, echoed so the + /// client can also refuse a build mismatch the server failed to catch. + pub build_id: String, } impl VersionHandshake { - /// Build a handshake from a [`VersionRange`]. - pub fn from_range(range: VersionRange) -> Self { + /// Build a handshake from a [`VersionRange`] and this build's identity. + pub fn from_range(range: VersionRange, build_id: String) -> Self { Self { range: (range.min.0, range.max.0), capabilities: 0, + build_id, } } @@ -125,11 +132,12 @@ impl VersionHandshake { } impl VersionHandshakeAck { - /// Construct an ack for the given agreed wire version. - pub fn new(agreed: WireVersion) -> Self { + /// Construct an ack for the given agreed wire version and build identity. + pub fn new(agreed: WireVersion, build_id: String) -> Self { Self { agreed: agreed.0, capabilities: 0, + build_id, } } @@ -183,9 +191,10 @@ mod tests { #[test] fn handshake_roundtrip() { let r = range(1, 2); - let hs = VersionHandshake::from_range(r); + let hs = VersionHandshake::from_range(r, "test-build".to_owned()); let bytes = zerompk::to_msgpack_vec(&hs).unwrap(); let decoded: VersionHandshake = zerompk::from_msgpack(&bytes).unwrap(); assert_eq!(decoded.to_range(), r); + assert_eq!(decoded.build_id, "test-build"); } } diff --git a/nodedb-crdt/src/state/history.rs b/nodedb-crdt/src/state/history.rs index 685d17134..238a8b2fc 100644 --- a/nodedb-crdt/src/state/history.rs +++ b/nodedb-crdt/src/state/history.rs @@ -90,8 +90,19 @@ impl CrdtState { /// Discards oplog entries before the target version. Current state and /// every version at or after the target, including the target itself, /// stays readable. + /// + /// A target at or below the current compaction boundary is a no-op. A + /// compaction is re-driven after a restart or a re-delivery, and it or a + /// newer compaction may already have moved the boundary there. pub fn compact_at_version(&mut self, version: &loro::VersionVector) -> Result<()> { - self.compact_to_frontiers(&self.doc.vv_to_frontiers(version)) + let frontiers = self.doc.vv_to_frontiers(version); + if self.doc.is_shallow() + && (self.doc.shallow_since_vv().to_vv().includes_vv(version) + || self.doc.shallow_since_frontiers() == frontiers) + { + return Ok(()); + } + self.compact_to_frontiers(&frontiers) } /// Generate a forward restore delta without changing authoritative state. @@ -355,6 +366,49 @@ mod tests { ); } + /// Compacting to a target the document already discarded changes nothing + /// and succeeds, so a re-driven older compaction cannot fail forever. + #[test] + fn compaction_below_the_boundary_is_a_no_op() { + let mut history = three_versions(); + history + .state + .compact_at_version(&history.after_v2) + .expect("compact"); + history + .state + .compact_at_version(&history.after_v1) + .expect("an older target is already compacted"); + + let row = history + .state + .read_at_version("docs", "doc-1", &history.after_v2) + .expect("the newer target still reads") + .expect("row exists"); + assert_eq!(title_of(&row), Some(LoroValue::String("v2".into()))); + } + + /// Re-running the same compaction succeeds and keeps the target readable. + #[test] + fn repeating_a_compaction_succeeds() { + let mut history = three_versions(); + history + .state + .compact_at_version(&history.after_v2) + .expect("compact"); + history + .state + .compact_at_version(&history.after_v2) + .expect("a repeated compaction"); + + let row = history + .state + .read_at_version("docs", "doc-1", &history.after_v2) + .expect("the target still reads") + .expect("row exists"); + assert_eq!(title_of(&row), Some(LoroValue::String("v2".into()))); + } + /// The compaction target itself stays readable. It names the shallow /// root, which the document keeps, so the guard must admit it. #[test] diff --git a/nodedb-graph/src/csr/index/label_filter.rs b/nodedb-graph/src/csr/index/label_filter.rs new file mode 100644 index 000000000..2b29cf757 --- /dev/null +++ b/nodedb-graph/src/csr/index/label_filter.rs @@ -0,0 +1,80 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! An edge-label filter resolved against one CSR partition. +//! +//! A filter names a label as a string, and the CSR stores labels as dense +//! ids. A label this partition has never interned carries no edge here, so +//! the filter keeps no durable edge. Falling back to "no filter" instead +//! would return every other label's edges under the caller's label. In a +//! cluster that happens whenever the label lives only on other nodes. + +use super::types::CsrIndex; + +/// Which durable edges a label filter keeps in one partition. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum LabelFilter { + /// No filter: every edge. + Any, + /// Edges with this dense label id. + Only(u32), + /// A label this partition has never seen: no edge. + Unknown, +} + +impl LabelFilter { + /// Whether an edge with dense label id `lid` passes the filter. + #[inline] + pub fn keeps(self, lid: u32) -> bool { + match self { + LabelFilter::Any => true, + LabelFilter::Only(id) => id == lid, + LabelFilter::Unknown => false, + } + } +} + +impl CsrIndex { + /// Resolve `filter` against this partition's interned labels. + pub fn label_filter(&self, filter: Option<&str>) -> LabelFilter { + match filter { + None => LabelFilter::Any, + Some(label) => match self.label_to_id.get(label) { + Some(&id) => LabelFilter::Only(id), + None => LabelFilter::Unknown, + }, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::csr::index::Direction; + use crate::test_support::test_memory; + + fn csr() -> CsrIndex { + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge_in_collection("a", "knows", "b", "people") + .unwrap_or_else(|e| panic!("seed edge: {e}")); + csr + } + + #[test] + fn an_unknown_label_keeps_no_edge() { + let csr = csr(); + assert_eq!(csr.label_filter(Some("likes")), LabelFilter::Unknown); + assert!(!csr.label_filter(Some("likes")).keeps(0)); + assert!(csr.neighbors("a", Some("likes"), Direction::Out).is_empty()); + assert!( + csr.neighbors_in_collection("a", Some("likes"), Direction::Out, "people") + .is_empty() + ); + } + + #[test] + fn no_filter_and_a_known_label_keep_their_edges() { + let csr = csr(); + assert_eq!(csr.neighbors("a", None, Direction::Out).len(), 1); + assert_eq!(csr.neighbors("a", Some("knows"), Direction::Out).len(), 1); + } +} diff --git a/nodedb-graph/src/csr/index/lookup.rs b/nodedb-graph/src/csr/index/lookup.rs index 80cf8de13..cce17b63e 100644 --- a/nodedb-graph/src/csr/index/lookup.rs +++ b/nodedb-graph/src/csr/index/lookup.rs @@ -51,13 +51,13 @@ impl CsrIndex { return Vec::new(); }; self.record_access(node_id); - let label_id = label_filter.and_then(|l| self.label_to_id.get(l).copied()); + let labels = self.label_filter(label_filter); let mut result = Vec::new(); if matches!(direction, Direction::Out | Direction::Both) { for (lid, dst) in self.dense_iter_out(node_id) { - if label_id.is_none_or(|f| f == lid) { + if labels.keeps(lid) { result.push(( self.id_to_label[lid as usize].clone(), self.id_to_node[dst as usize].clone(), @@ -67,7 +67,7 @@ impl CsrIndex { } if matches!(direction, Direction::In | Direction::Both) { for (lid, src) in self.dense_iter_in(node_id) { - if label_id.is_none_or(|f| f == lid) { + if labels.keeps(lid) { result.push(( self.id_to_label[lid as usize].clone(), self.id_to_node[src as usize].clone(), @@ -94,7 +94,9 @@ impl CsrIndex { .iter() .filter_map(|l| self.label_to_id.get(*l).copied()) .collect(); - let match_label = |lid: u32| label_ids.is_empty() || label_ids.contains(&lid); + // Filters this partition has never seen match no edge here; they must + // not widen the filter to every edge. + let match_label = |lid: u32| label_filters.is_empty() || label_ids.contains(&lid); let mut result = Vec::new(); diff --git a/nodedb-graph/src/csr/index/mod.rs b/nodedb-graph/src/csr/index/mod.rs index 5afdfebfb..cd5b2769b 100644 --- a/nodedb-graph/src/csr/index/mod.rs +++ b/nodedb-graph/src/csr/index/mod.rs @@ -8,15 +8,18 @@ //! - `mutation` — `add_edge`, `remove_edge`, `remove_node_edges` //! - `lookup` — neighbor queries, accessors, degree, iterators //! - `scoped` — collection-scoped read paths (MATCH / RAG) +//! - `label_filter` — an edge-label filter resolved against the partition //! - `restore` — exact edge writes and the reversals a rollback uses pub mod interning; +pub mod label_filter; pub mod lookup; pub mod mutation; pub mod restore; pub mod scoped; pub mod types; +pub use label_filter::LabelFilter; pub use types::CsrIndex; // Re-export shared Direction from nodedb-types via the types submodule. pub use types::Direction; diff --git a/nodedb-graph/src/csr/index/scoped.rs b/nodedb-graph/src/csr/index/scoped.rs index 322ffd5e9..8e19fa604 100644 --- a/nodedb-graph/src/csr/index/scoped.rs +++ b/nodedb-graph/src/csr/index/scoped.rs @@ -32,8 +32,8 @@ impl CsrIndex { return Vec::new(); }; self.record_access(node_id); - let label_id = label_filter.and_then(|l| self.label_to_id.get(l).copied()); - let keep = |lid: u32| label_id.is_none_or(|f| f == lid); + let labels = self.label_filter(label_filter); + let keep = |lid: u32| labels.keeps(lid); let mut result = Vec::new(); if matches!(direction, Direction::Out | Direction::Both) { diff --git a/nodedb-graph/src/lib.rs b/nodedb-graph/src/lib.rs index eeac64985..310356069 100644 --- a/nodedb-graph/src/lib.rs +++ b/nodedb-graph/src/lib.rs @@ -34,5 +34,5 @@ pub use overlay_delta::GraphOverlayDelta; pub use params::{AlgoColumnType, AlgoParams, GraphAlgorithm}; pub use path_params::ShortestPathParams; pub use sharded::ShardedCsrIndex; -pub use traversal_options::{GraphResponseMeta, GraphTraversalOptions, MAX_GRAPH_TRAVERSAL_DEPTH}; +pub use traversal_options::{GraphTraversalOptions, MAX_GRAPH_TRAVERSAL_DEPTH}; pub use traversal_surrogate::{SurrogateBfsParams, SurrogateHops}; diff --git a/nodedb-graph/src/path_overlay.rs b/nodedb-graph/src/path_overlay.rs index 1c4164478..fd9731ce3 100644 --- a/nodedb-graph/src/path_overlay.rs +++ b/nodedb-graph/src/path_overlay.rs @@ -59,50 +59,57 @@ impl CsrIndex { let mut fwd_frontier: Vec = vec![src.to_string()]; let mut bwd_frontier: Vec = vec![dst.to_string()]; - let label_id = label_filter.and_then(|l| self.label_id(l)); + let labels = self.label_filter(label_filter); for _depth in 0..max_depth { if fwd_parent.len() + bwd_parent.len() >= max_visited { break; } - let mut next_fwd = Vec::new(); + // Each level's edges are relaxed in (neighbour, frontier node) + // name order, as the durable search relaxes them. + let mut candidates: Vec<(String, String)> = Vec::new(); for node in std::mem::take(&mut fwd_frontier) { - let neighbors = - self.forward_neighbors(&node, label_id, label_filter, frontier_bitmap, overlay); - for neighbor in neighbors { - if let Some(meeting) = relax( - &neighbor, - &node, - &mut fwd_parent, - &bwd_parent, - &mut next_fwd, - ) { - return Some(reconstruct(&meeting, &fwd_parent, &bwd_parent)); - } + for neighbor in + self.forward_neighbors(&node, labels, label_filter, frontier_bitmap, overlay) + { + candidates.push((neighbor, node.clone())); + } + } + candidates.sort(); + let mut next_fwd = Vec::new(); + for (neighbor, node) in candidates { + if let Some(meeting) = relax( + &neighbor, + &node, + &mut fwd_parent, + &bwd_parent, + &mut next_fwd, + ) { + return Some(reconstruct(&meeting, &fwd_parent, &bwd_parent)); } } fwd_frontier = next_fwd; - let mut next_bwd = Vec::new(); + let mut candidates: Vec<(String, String)> = Vec::new(); for node in std::mem::take(&mut bwd_frontier) { - let neighbors = self.backward_neighbors( + for neighbor in + self.backward_neighbors(&node, labels, label_filter, frontier_bitmap, overlay) + { + candidates.push((neighbor, node.clone())); + } + } + candidates.sort(); + let mut next_bwd = Vec::new(); + for (neighbor, node) in candidates { + if let Some(meeting) = relax( + &neighbor, &node, - label_id, - label_filter, - frontier_bitmap, - overlay, - ); - for neighbor in neighbors { - if let Some(meeting) = relax( - &neighbor, - &node, - &mut bwd_parent, - &fwd_parent, - &mut next_bwd, - ) { - return Some(reconstruct(&meeting, &fwd_parent, &bwd_parent)); - } + &mut bwd_parent, + &fwd_parent, + &mut next_bwd, + ) { + return Some(reconstruct(&meeting, &fwd_parent, &bwd_parent)); } } bwd_frontier = next_bwd; @@ -120,7 +127,7 @@ impl CsrIndex { fn forward_neighbors( &self, node: &str, - label_id: Option, + labels: crate::csr::index::LabelFilter, label_filter: Option<&str>, frontier_bitmap: Option<&nodedb_types::SurrogateBitmap>, overlay: &GraphOverlayDelta, @@ -129,7 +136,7 @@ impl CsrIndex { if let Some(&node_id) = self.node_to_id.get(node) { self.record_access(node_id); for (lid, dst) in self.dense_iter_out(node_id) { - if label_id.is_some_and(|f| f != lid) { + if !labels.keeps(lid) { continue; } let dst_name = &self.id_to_node[dst as usize]; @@ -156,7 +163,7 @@ impl CsrIndex { fn backward_neighbors( &self, node: &str, - label_id: Option, + labels: crate::csr::index::LabelFilter, label_filter: Option<&str>, frontier_bitmap: Option<&nodedb_types::SurrogateBitmap>, overlay: &GraphOverlayDelta, @@ -165,7 +172,7 @@ impl CsrIndex { if let Some(&node_id) = self.node_to_id.get(node) { self.record_access(node_id); for (lid, src) in self.dense_iter_in(node_id) { - if label_id.is_some_and(|f| f != lid) { + if !labels.keeps(lid) { continue; } let src_name = &self.id_to_node[src as usize]; diff --git a/nodedb-graph/src/traversal.rs b/nodedb-graph/src/traversal.rs index 05c9509e6..5e5393738 100644 --- a/nodedb-graph/src/traversal.rs +++ b/nodedb-graph/src/traversal.rs @@ -44,8 +44,13 @@ impl CsrIndex { } } - /// Durable-only BFS over the dense u32 CSR ids. Behavior and performance - /// are identical to the pre-overlay traversal. + /// Durable-only BFS over the dense u32 CSR ids. + /// + /// The walk runs level by level. Each level's new nodes are admitted in + /// node-name order until `max_visited` nodes are visited, so a capped walk + /// admits the same nodes however the edges are stored. A cluster + /// coordinator walking the same edges across partitions admits the same + /// nodes. fn traverse_bfs_dense(&self, params: BfsParams<'_>) -> Vec { let BfsParams { start_nodes, @@ -55,54 +60,46 @@ impl CsrIndex { max_visited, frontier_bitmap, } = params; - let label_id = label_filter.and_then(|l| self.label_id(l)); + let labels = self.label_filter(label_filter); + let in_bitmap = |id: u32| { + frontier_bitmap.is_none_or(|bm| { + bm.contains(nodedb_types::Surrogate::new(self.node_surrogate_raw(id))) + }) + }; let mut visited: HashSet = HashSet::new(); - let mut queue: VecDeque<(u32, usize)> = VecDeque::new(); - + let mut frontier: Vec = Vec::new(); for &node in start_nodes { if let Some(&id) = self.node_to_id.get(node) && visited.insert(id) { - queue.push_back((id, 0)); + frontier.push(id); } } - while let Some((node_id, depth)) = queue.pop_front() { - if depth >= max_depth || visited.len() >= max_visited { - continue; + for _depth in 0..max_depth { + if frontier.is_empty() || visited.len() >= max_visited { + break; } - - // Track access for hot/cold partition decisions. - self.record_access(node_id); - - if matches!(direction, Direction::Out | Direction::Both) { - for (lid, dst) in self.dense_iter_out(node_id) { - if label_id.is_none_or(|f| f == lid) - && visited.len() < max_visited - && frontier_bitmap.is_none_or(|bm| { - bm.contains(nodedb_types::Surrogate::new(self.node_surrogate_raw(dst))) - }) - && visited.insert(dst) - { - self.prefetch_node(dst); - queue.push_back((dst, depth + 1)); + let mut candidates: Vec = Vec::new(); + for &node_id in &frontier { + // Track access for hot/cold partition decisions. + self.record_access(node_id); + if matches!(direction, Direction::Out | Direction::Both) { + for (lid, dst) in self.dense_iter_out(node_id) { + if labels.keeps(lid) && !visited.contains(&dst) && in_bitmap(dst) { + candidates.push(dst); + } } } - } - if matches!(direction, Direction::In | Direction::Both) { - for (lid, src) in self.dense_iter_in(node_id) { - if label_id.is_none_or(|f| f == lid) - && visited.len() < max_visited - && frontier_bitmap.is_none_or(|bm| { - bm.contains(nodedb_types::Surrogate::new(self.node_surrogate_raw(src))) - }) - && visited.insert(src) - { - self.prefetch_node(src); - queue.push_back((src, depth + 1)); + if matches!(direction, Direction::In | Direction::Both) { + for (lid, src) in self.dense_iter_in(node_id) { + if labels.keeps(lid) && !visited.contains(&src) && in_bitmap(src) { + candidates.push(src); + } } } } + frontier = self.admit_by_name(candidates, &mut visited, max_visited); } visited @@ -111,6 +108,29 @@ impl CsrIndex { .collect() } + /// Admit `candidates` into `visited` in node-name order, until `visited` + /// holds `max_visited` nodes. Returns the nodes admitted, in name order. + pub(crate) fn admit_by_name( + &self, + mut candidates: Vec, + visited: &mut HashSet, + max_visited: usize, + ) -> Vec { + self.sort_by_name(&mut candidates); + candidates.dedup(); + let mut admitted = Vec::with_capacity(candidates.len()); + for id in candidates { + if visited.len() >= max_visited { + break; + } + if visited.insert(id) { + self.prefetch_node(id); + admitted.push(id); + } + } + admitted + } + /// BFS traversal returning nodes with depth information. /// /// `max_visited` caps the number of nodes visited to prevent supernode fan-out @@ -143,7 +163,9 @@ impl CsrIndex { .iter() .filter_map(|l| self.label_id(l)) .collect(); - let match_label = |lid: u32| label_ids.is_empty() || label_ids.contains(&lid); + // Filters this partition has never seen match no edge here; they must + // not widen the filter to every edge. + let match_label = |lid: u32| label_filters.is_empty() || label_ids.contains(&lid); let mut visited: HashMap = HashMap::new(); let mut queue: VecDeque<(u32, u8)> = VecDeque::new(); @@ -215,8 +237,12 @@ impl CsrIndex { } } - /// Durable-only bidirectional BFS over the dense u32 CSR ids. Behavior and - /// performance are identical to the pre-overlay shortest path. + /// Durable-only bidirectional BFS over the dense u32 CSR ids. + /// + /// Each step expands one forward level, then one backward level. A level + /// relaxes its edges in `(neighbour, frontier node)` name order, and the + /// search stops at the first node both sides reached. The cap is checked + /// before each step. fn shortest_path_dense(&self, params: ShortestPathParams<'_>) -> Option> { let ShortestPathParams { src, @@ -232,7 +258,12 @@ impl CsrIndex { return Some(vec![src.to_string()]); } - let label_id = label_filter.and_then(|l| self.label_id(l)); + let labels = self.label_filter(label_filter); + let in_bitmap = |id: u32| { + frontier_bitmap.is_none_or(|bm| { + bm.contains(nodedb_types::Surrogate::new(self.node_surrogate_raw(id))) + }) + }; let mut fwd_parent: HashMap = HashMap::new(); let mut bwd_parent: HashMap = HashMap::new(); fwd_parent.insert(src_id, src_id); @@ -246,50 +277,52 @@ impl CsrIndex { break; } - let mut next_fwd = Vec::new(); + // Each level's edges are relaxed in (neighbour, frontier node) + // name order, so the parent a node gets, and the meeting point, + // do not depend on how the edges are stored. A cluster coordinator + // relaxes cross-shard hops in the same order. + let mut candidates: Vec<(u32, u32)> = Vec::new(); for &node in &fwd_frontier { self.record_access(node); for (lid, neighbor) in self.dense_iter_out(node) { - if label_id.is_none_or(|f| f == lid) - && frontier_bitmap.is_none_or(|bm| { - bm.contains(nodedb_types::Surrogate::new( - self.node_surrogate_raw(neighbor), - )) - }) - { - if let Entry::Vacant(e) = fwd_parent.entry(neighbor) { - e.insert(node); - next_fwd.push(neighbor); - } - if bwd_parent.contains_key(&neighbor) { - return Some(self.reconstruct_path(neighbor, &fwd_parent, &bwd_parent)); - } + if labels.keeps(lid) && in_bitmap(neighbor) { + candidates.push((neighbor, node)); } } } + self.sort_edges_by_name(&mut candidates); + let mut next_fwd = Vec::new(); + for (neighbor, node) in candidates { + if let Entry::Vacant(e) = fwd_parent.entry(neighbor) { + e.insert(node); + next_fwd.push(neighbor); + } + if bwd_parent.contains_key(&neighbor) { + return Some(self.reconstruct_path(neighbor, &fwd_parent, &bwd_parent)); + } + } fwd_frontier = next_fwd; - let mut next_bwd = Vec::new(); + let mut candidates: Vec<(u32, u32)> = Vec::new(); for &node in &bwd_frontier { self.record_access(node); for (lid, neighbor) in self.dense_iter_in(node) { - if label_id.is_none_or(|f| f == lid) - && frontier_bitmap.is_none_or(|bm| { - bm.contains(nodedb_types::Surrogate::new( - self.node_surrogate_raw(neighbor), - )) - }) - { - if let Entry::Vacant(e) = bwd_parent.entry(neighbor) { - e.insert(node); - next_bwd.push(neighbor); - } - if fwd_parent.contains_key(&neighbor) { - return Some(self.reconstruct_path(neighbor, &fwd_parent, &bwd_parent)); - } + if labels.keeps(lid) && in_bitmap(neighbor) { + candidates.push((neighbor, node)); } } } + self.sort_edges_by_name(&mut candidates); + let mut next_bwd = Vec::new(); + for (neighbor, node) in candidates { + if let Entry::Vacant(e) = bwd_parent.entry(neighbor) { + e.insert(node); + next_bwd.push(neighbor); + } + if fwd_parent.contains_key(&neighbor) { + return Some(self.reconstruct_path(neighbor, &fwd_parent, &bwd_parent)); + } + } bwd_frontier = next_bwd; if fwd_frontier.is_empty() && bwd_frontier.is_empty() { @@ -299,6 +332,14 @@ impl CsrIndex { None } + /// Order `(neighbour, frontier node)` edges by the two node names. + fn sort_edges_by_name(&self, edges: &mut [(u32, u32)]) { + edges.sort_by(|a, b| { + let name = |id: u32| self.node_name_checked(id); + (name(a.0), name(a.1)).cmp(&(name(b.0), name(b.1))) + }); + } + fn reconstruct_path( &self, meeting: u32, @@ -376,52 +417,58 @@ impl CsrIndex { max_depth: usize, max_visited: usize, ) -> Vec<(String, String, String)> { - let label_id = label_filter.and_then(|l| self.label_id(l)); + let labels = self.label_filter(label_filter); let mut visited: HashSet = HashSet::new(); - let mut queue: VecDeque<(u32, usize)> = VecDeque::new(); + let mut frontier: Vec = Vec::new(); let mut edges = Vec::new(); for &node in start_nodes { if let Some(&id) = self.node_to_id.get(node) && visited.insert(id) { - queue.push_back((id, 0)); + frontier.push(id); } } - while let Some((node_id, depth)) = queue.pop_front() { - if depth >= max_depth || visited.len() >= max_visited { - continue; + // Level by level, as `traverse_bfs_dense`: every frontier node's edges + // are recorded, then the level's new nodes are admitted in name order. + for _depth in 0..max_depth { + if frontier.is_empty() || visited.len() >= max_visited { + break; } - self.record_access(node_id); - if matches!(direction, Direction::Out | Direction::Both) { - for (lid, dst) in self.dense_iter_out(node_id) { - if label_id.is_none_or(|f| f == lid) { - edges.push(( - self.id_to_node[node_id as usize].clone(), - self.label_name(lid).to_string(), - self.id_to_node[dst as usize].clone(), - )); - if visited.len() < max_visited && visited.insert(dst) { - queue.push_back((dst, depth + 1)); + let mut candidates: Vec = Vec::new(); + for &node_id in &frontier { + self.record_access(node_id); + if matches!(direction, Direction::Out | Direction::Both) { + for (lid, dst) in self.dense_iter_out(node_id) { + if labels.keeps(lid) { + edges.push(( + self.id_to_node[node_id as usize].clone(), + self.label_name(lid).to_string(), + self.id_to_node[dst as usize].clone(), + )); + if !visited.contains(&dst) { + candidates.push(dst); + } } } } - } - if matches!(direction, Direction::In | Direction::Both) { - for (lid, src) in self.dense_iter_in(node_id) { - if label_id.is_none_or(|f| f == lid) { - edges.push(( - self.id_to_node[src as usize].clone(), - self.label_name(lid).to_string(), - self.id_to_node[node_id as usize].clone(), - )); - if visited.len() < max_visited && visited.insert(src) { - queue.push_back((src, depth + 1)); + if matches!(direction, Direction::In | Direction::Both) { + for (lid, src) in self.dense_iter_in(node_id) { + if labels.keeps(lid) { + edges.push(( + self.id_to_node[src as usize].clone(), + self.label_name(lid).to_string(), + self.id_to_node[node_id as usize].clone(), + )); + if !visited.contains(&src) { + candidates.push(src); + } } } } } + frontier = self.admit_by_name(candidates, &mut visited, max_visited); } edges @@ -478,6 +525,49 @@ mod tests { assert_eq!(result, vec!["a", "b", "e"]); } + /// `a` points at `z`, `m` and `b`, stored in that order. A cap of 3 leaves + /// room for two of them, and name order picks `b` and `m`. + fn fan_out_csr() -> CsrIndex { + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge("a", "L", "z").unwrap(); + csr.add_edge("a", "L", "m").unwrap(); + csr.add_edge("a", "L", "b").unwrap(); + csr + } + + #[test] + fn a_capped_bfs_admits_each_level_in_name_order() { + let csr = fan_out_csr(); + let mut result = csr.traverse_bfs( + BfsParams { + start_nodes: &["a"], + label_filter: None, + direction: Direction::Out, + max_depth: 2, + max_visited: 3, + frontier_bitmap: None, + }, + None, + ); + result.sort(); + assert_eq!(result, vec!["a", "b", "m"]); + } + + #[test] + fn a_capped_subgraph_expands_only_admitted_levels() { + let mut csr = fan_out_csr(); + csr.add_edge("z", "L", "y").unwrap(); + csr.add_edge("b", "L", "c").unwrap(); + let mut edges = csr.subgraph(&["a"], None, Direction::Out, 3, 3, None); + edges.sort(); + // Level 1 fills the cap, so no level-1 node expands. + let expected: Vec<(String, String, String)> = ["b", "m", "z"] + .iter() + .map(|dst| ("a".to_string(), "L".to_string(), dst.to_string())) + .collect(); + assert_eq!(edges, expected); + } + #[test] fn bfs_cycle() { let mut csr = CsrIndex::new(test_memory()); @@ -542,6 +632,20 @@ mod tests { assert_eq!(path, vec!["a", "b", "c"]); } + #[test] + fn shortest_path_takes_the_smallest_named_tie() { + // Two paths of equal length, the `z` one stored first. + let mut csr = CsrIndex::new(test_memory()); + csr.add_edge("a", "L", "z").unwrap(); + csr.add_edge("z", "L", "d").unwrap(); + csr.add_edge("a", "L", "b").unwrap(); + csr.add_edge("b", "L", "d").unwrap(); + let path = csr + .shortest_path(path_params("a", "d", None, 5, None), None) + .unwrap(); + assert_eq!(path, vec!["a", "b", "d"]); + } + #[test] fn shortest_path_same_node() { let csr = make_csr(); diff --git a/nodedb-graph/src/traversal_options.rs b/nodedb-graph/src/traversal_options.rs index f1f7be0b2..59d40492f 100644 --- a/nodedb-graph/src/traversal_options.rs +++ b/nodedb-graph/src/traversal_options.rs @@ -1,9 +1,6 @@ // SPDX-License-Identifier: Apache-2.0 //! Per-query graph traversal configuration. -//! -//! Adaptive fan-out uses a two-tier limit with optional graceful -//! degradation instead of a hard kill. /// Largest accepted value for any graph-DSL depth parameter /// (`DEPTH`, `MAX_DEPTH`, `EXPANSION_DEPTH`). @@ -17,9 +14,6 @@ pub const MAX_GRAPH_TRAVERSAL_DEPTH: usize = 64; use serde::{Deserialize, Serialize}; /// Per-query graph traversal configuration. -/// -/// Controls fan-out limits, partial result handling, and visited node caps -/// for scatter-gather graph queries across shards. #[derive( Debug, Clone, @@ -31,27 +25,6 @@ use serde::{Deserialize, Serialize}; zerompk::FromMessagePack, )] pub struct GraphTraversalOptions { - /// Soft warning threshold (shards per hop). - /// - /// When the number of shards reached in a single hop exceeds this value, - /// a fan-out warning is emitted but execution continues. - /// Default: 12 - pub fan_out_soft: u16, - - /// Hard limit (shards per hop). - /// - /// Maximum number of shards that can be queried in a single hop. - /// If exceeded and `fan_out_partial` is false, returns FAN_OUT_EXCEEDED error. - /// Default: 16 - pub fan_out_hard: u16, - - /// If true, return partial results instead of FAN_OUT_EXCEEDED error. - /// - /// When the hard limit is exceeded, instead of failing with FAN_OUT_EXCEEDED, - /// this flag allows the response to be marked as truncated with partial results. - /// Default: false - pub fan_out_partial: bool, - /// Cap on total visited nodes across all shards. /// /// Once this limit is reached, no further node exploration occurs. @@ -62,9 +35,6 @@ pub struct GraphTraversalOptions { impl Default for GraphTraversalOptions { fn default() -> Self { Self { - fan_out_soft: 12, - fan_out_hard: 16, - fan_out_partial: false, max_visited: 100_000, } } @@ -77,72 +47,6 @@ impl GraphTraversalOptions { } } -/// Response metadata for scatter-gather graph query results. -/// -/// Tracks how many shards were reached, skipped, and whether results are -/// complete or truncated due to adaptive fan-out limits. -#[derive(Debug, Clone, Default, Serialize, Deserialize)] -pub struct GraphResponseMeta { - /// Number of shards that were queried and returned results. - pub shards_reached: u16, - - /// Number of shards that were skipped due to fan-out limits. - pub shards_skipped: u16, - - /// Whether results are incomplete (true) or complete (false). - pub truncated: bool, - - /// Fan-out warning message if soft limit was exceeded. - /// - /// Format: "X/Y" where X is shards_reached and Y is fan_out_hard. - /// None if no warning. - pub fan_out_warning: Option, - - /// Whether results gathered beyond the soft limit are approximate. - /// - /// Set to true when shards_reached > fan_out_soft. - pub approximate: bool, -} - -impl GraphResponseMeta { - /// Check if this response has no warnings or truncation. - /// - /// Returns true if: - /// - No fan-out warning - /// - Not truncated - /// - Not approximate - pub fn is_clean(&self) -> bool { - self.fan_out_warning.is_none() && !self.truncated && !self.approximate - } - - /// Create response metadata with a fan-out warning. - /// - /// Indicates that the soft limit was exceeded but execution continued. - /// Creates a warning string like "12/16" showing reached vs hard limit. - pub fn with_warning(shards_reached: u16, shards_skipped: u16, fan_out_hard: u16) -> Self { - Self { - shards_reached, - shards_skipped, - truncated: false, - fan_out_warning: Some(format!("{}/{}", shards_reached, fan_out_hard)), - approximate: true, - } - } - - /// Create response metadata for truncated results. - /// - /// Indicates that results were incomplete due to fan-out limits. - pub fn with_truncation(shards_reached: u16, shards_skipped: u16) -> Self { - Self { - shards_reached, - shards_skipped, - truncated: true, - fan_out_warning: None, - approximate: true, - } - } -} - #[cfg(test)] mod tests { use super::*; @@ -151,9 +55,6 @@ mod tests { #[test] fn default_options_have_expected_values() { let opts = GraphTraversalOptions::default(); - assert_eq!(opts.fan_out_soft, 12); - assert_eq!(opts.fan_out_hard, 16); - assert!(!opts.fan_out_partial); assert_eq!(opts.max_visited, 100_000); } @@ -163,74 +64,13 @@ mod tests { assert_eq!(opts, GraphTraversalOptions::default()); } - #[test] - fn default_meta_is_clean() { - let meta = GraphResponseMeta::default(); - assert!(meta.is_clean()); - assert_eq!(meta.shards_reached, 0); - assert_eq!(meta.shards_skipped, 0); - assert!(!meta.truncated); - assert!(meta.fan_out_warning.is_none()); - assert!(!meta.approximate); - } - - #[test] - fn with_warning_generates_correct_string() { - let meta = GraphResponseMeta::with_warning(12, 4, 16); - assert_eq!(meta.shards_reached, 12); - assert_eq!(meta.shards_skipped, 4); - assert!(!meta.truncated); - assert_eq!(meta.fan_out_warning, Some("12/16".to_string())); - assert!(meta.approximate); - } - - #[test] - fn with_truncation_sets_flags() { - let meta = GraphResponseMeta::with_truncation(10, 6); - assert_eq!(meta.shards_reached, 10); - assert_eq!(meta.shards_skipped, 6); - assert!(meta.truncated); - assert!(meta.fan_out_warning.is_none()); - assert!(meta.approximate); - } - - #[test] - fn with_warning_is_not_clean() { - let meta = GraphResponseMeta::with_warning(12, 4, 16); - assert!(!meta.is_clean()); - } - - #[test] - fn with_truncation_is_not_clean() { - let meta = GraphResponseMeta::with_truncation(10, 6); - assert!(!meta.is_clean()); - } - #[test] fn serialization_roundtrip() { let opts = GraphTraversalOptions { - fan_out_soft: 8, - fan_out_hard: 12, - fan_out_partial: true, max_visited: 50_000, }; let json = sonic_rs::to_string(&opts).unwrap(); let deserialized: GraphTraversalOptions = sonic_rs::from_str(&json).unwrap(); - assert_eq!(opts.fan_out_soft, deserialized.fan_out_soft); - assert_eq!(opts.fan_out_hard, deserialized.fan_out_hard); - assert_eq!(opts.fan_out_partial, deserialized.fan_out_partial); - assert_eq!(opts.max_visited, deserialized.max_visited); - } - - #[test] - fn meta_serialization_roundtrip() { - let meta = GraphResponseMeta::with_warning(15, 1, 16); - let json = sonic_rs::to_string(&meta).unwrap(); - let deserialized: GraphResponseMeta = sonic_rs::from_str(&json).unwrap(); - assert_eq!(meta.shards_reached, deserialized.shards_reached); - assert_eq!(meta.shards_skipped, deserialized.shards_skipped); - assert_eq!(meta.truncated, deserialized.truncated); - assert_eq!(meta.fan_out_warning, deserialized.fan_out_warning); - assert_eq!(meta.approximate, deserialized.approximate); + assert_eq!(opts, deserialized); } } diff --git a/nodedb-graph/src/traversal_overlay.rs b/nodedb-graph/src/traversal_overlay.rs index efb50f778..640f3fdea 100644 --- a/nodedb-graph/src/traversal_overlay.rs +++ b/nodedb-graph/src/traversal_overlay.rs @@ -9,8 +9,11 @@ //! still be followed at the next hop. Durable nodes still resolve through //! `node_to_id` for the CSR expansion, so the dense adjacency is used wherever //! it exists; the string key is what lets staged-only nodes participate. +//! +//! Both walks run level by level, as the durable paths do: each level's new +//! nodes are admitted in node-name order under `max_visited`. -use std::collections::{HashSet, VecDeque}; +use std::collections::HashSet; use crate::bfs_params::BfsParams; use crate::csr::{CsrIndex, Direction}; @@ -31,97 +34,80 @@ impl CsrIndex { max_visited, frontier_bitmap, } = params; - let label_id = label_filter.and_then(|l| self.label_id(l)); + let labels = self.label_filter(label_filter); + let in_bitmap = |id: u32| { + frontier_bitmap.is_none_or(|bm| { + bm.contains(nodedb_types::Surrogate::new(self.node_surrogate_raw(id))) + }) + }; let mut visited: HashSet = HashSet::new(); - let mut queue: VecDeque<(String, usize)> = VecDeque::new(); - + let mut frontier: Vec = Vec::new(); for &node in start_nodes { if visited.insert(node.to_string()) { - queue.push_back((node.to_string(), 0)); + frontier.push(node.to_string()); } } let want_out = matches!(direction, Direction::Out | Direction::Both); let want_in = matches!(direction, Direction::In | Direction::Both); - while let Some((node, depth)) = queue.pop_front() { - if depth >= max_depth || visited.len() >= max_visited { - continue; + for _depth in 0..max_depth { + if frontier.is_empty() || visited.len() >= max_visited { + break; } - let next_depth = depth + 1; - - // Durable CSR expansion for nodes that carry a surrogate. - if let Some(&node_id) = self.node_to_id.get(node.as_str()) { - self.record_access(node_id); - if want_out { - for (lid, dst) in self.dense_iter_out(node_id) { - if label_id.is_some_and(|f| f != lid) { - continue; - } - let dst_name = &self.id_to_node[dst as usize]; - if overlay.is_tombstoned(&node, self.label_name(lid), dst_name) { - continue; - } - if !frontier_bitmap.is_none_or(|bm| { - bm.contains(nodedb_types::Surrogate::new(self.node_surrogate_raw(dst))) - }) { - continue; + let mut candidates: Vec = Vec::new(); + for node in &frontier { + // Durable CSR expansion for nodes that carry a surrogate. + if let Some(&node_id) = self.node_to_id.get(node.as_str()) { + self.record_access(node_id); + if want_out { + for (lid, dst) in self.dense_iter_out(node_id) { + let dst_name = &self.id_to_node[dst as usize]; + if labels.keeps(lid) + && !overlay.is_tombstoned(node, self.label_name(lid), dst_name) + && in_bitmap(dst) + && !visited.contains(dst_name) + { + candidates.push(dst_name.clone()); + } } - self.enqueue_str( - dst_name, - next_depth, - max_visited, - &mut visited, - &mut queue, - ); } - } - if want_in { - for (lid, src) in self.dense_iter_in(node_id) { - if label_id.is_some_and(|f| f != lid) { - continue; - } - let src_name = &self.id_to_node[src as usize]; - if overlay.is_tombstoned(src_name, self.label_name(lid), &node) { - continue; + if want_in { + for (lid, src) in self.dense_iter_in(node_id) { + let src_name = &self.id_to_node[src as usize]; + if labels.keeps(lid) + && !overlay.is_tombstoned(src_name, self.label_name(lid), node) + && in_bitmap(src) + && !visited.contains(src_name) + { + candidates.push(src_name.clone()); + } } - if !frontier_bitmap.is_none_or(|bm| { - bm.contains(nodedb_types::Surrogate::new(self.node_surrogate_raw(src))) - }) { - continue; - } - self.enqueue_str( - src_name, - next_depth, - max_visited, - &mut visited, - &mut queue, - ); } } - } - // Staged edges — followed for durable and staged-only nodes alike. - // Staged edges are the transaction's own writes, so bitmap gating - // (which needs a durable surrogate) does not apply. - if want_out { - let staged: Vec = overlay - .out_neighbors(&node, label_filter) - .map(|(_, dst)| dst.to_string()) - .collect(); - for dst in staged { - self.enqueue_str(&dst, next_depth, max_visited, &mut visited, &mut queue); + // Staged edges — followed for durable and staged-only nodes + // alike. Staged edges are the transaction's own writes, so + // bitmap gating (which needs a durable surrogate) does not + // apply. + if want_out { + candidates.extend( + overlay + .out_neighbors(node, label_filter) + .map(|(_, dst)| dst.to_string()) + .filter(|dst| !visited.contains(dst)), + ); } - } - if want_in { - let staged: Vec = overlay - .in_neighbors(&node, label_filter) - .map(|(_, src)| src.to_string()) - .collect(); - for src in staged { - self.enqueue_str(&src, next_depth, max_visited, &mut visited, &mut queue); + if want_in { + candidates.extend( + overlay + .in_neighbors(node, label_filter) + .map(|(_, src)| src.to_string()) + .filter(|src| !visited.contains(src)), + ); } } + frontier = admit_names(candidates, &mut visited, max_visited); } visited.into_iter().collect() @@ -137,109 +123,105 @@ impl CsrIndex { max_visited: usize, overlay: &GraphOverlayDelta, ) -> Vec<(String, String, String)> { - let label_id = label_filter.and_then(|l| self.label_id(l)); + let labels = self.label_filter(label_filter); let mut visited: HashSet = HashSet::new(); - let mut queue: VecDeque<(String, usize)> = VecDeque::new(); + let mut frontier: Vec = Vec::new(); let mut edges: Vec<(String, String, String)> = Vec::new(); for &node in start_nodes { if visited.insert(node.to_string()) { - queue.push_back((node.to_string(), 0)); + frontier.push(node.to_string()); } } let want_out = matches!(direction, Direction::Out | Direction::Both); let want_in = matches!(direction, Direction::In | Direction::Both); - while let Some((node, depth)) = queue.pop_front() { - if depth >= max_depth || visited.len() >= max_visited { - continue; + for _depth in 0..max_depth { + if frontier.is_empty() || visited.len() >= max_visited { + break; } - let next_depth = depth + 1; + let mut candidates: Vec = Vec::new(); + for node in &frontier { + if let Some(&node_id) = self.node_to_id.get(node.as_str()) { + self.record_access(node_id); + if want_out { + for (lid, dst) in self.dense_iter_out(node_id) { + if !labels.keeps(lid) { + continue; + } + let label = self.label_name(lid); + let dst_name = &self.id_to_node[dst as usize]; + if overlay.is_tombstoned(node, label, dst_name) { + continue; + } + edges.push((node.clone(), label.to_string(), dst_name.clone())); + if !visited.contains(dst_name) { + candidates.push(dst_name.clone()); + } + } + } + if want_in { + for (lid, src) in self.dense_iter_in(node_id) { + if !labels.keeps(lid) { + continue; + } + let label = self.label_name(lid); + let src_name = &self.id_to_node[src as usize]; + if overlay.is_tombstoned(src_name, label, node) { + continue; + } + edges.push((src_name.clone(), label.to_string(), node.clone())); + if !visited.contains(src_name) { + candidates.push(src_name.clone()); + } + } + } + } - if let Some(&node_id) = self.node_to_id.get(node.as_str()) { - self.record_access(node_id); if want_out { - for (lid, dst) in self.dense_iter_out(node_id) { - if label_id.is_some_and(|f| f != lid) { - continue; - } - let label = self.label_name(lid); - let dst_name = &self.id_to_node[dst as usize]; - if overlay.is_tombstoned(&node, label, dst_name) { - continue; + for (label, dst) in overlay.out_neighbors(node, label_filter) { + edges.push((node.clone(), label.to_string(), dst.to_string())); + if !visited.contains(dst) { + candidates.push(dst.to_string()); } - edges.push((node.clone(), label.to_string(), dst_name.clone())); - self.enqueue_str( - dst_name, - next_depth, - max_visited, - &mut visited, - &mut queue, - ); } } if want_in { - for (lid, src) in self.dense_iter_in(node_id) { - if label_id.is_some_and(|f| f != lid) { - continue; - } - let label = self.label_name(lid); - let src_name = &self.id_to_node[src as usize]; - if overlay.is_tombstoned(src_name, label, &node) { - continue; + for (label, src) in overlay.in_neighbors(node, label_filter) { + edges.push((src.to_string(), label.to_string(), node.clone())); + if !visited.contains(src) { + candidates.push(src.to_string()); } - edges.push((src_name.clone(), label.to_string(), node.clone())); - self.enqueue_str( - src_name, - next_depth, - max_visited, - &mut visited, - &mut queue, - ); } } } - - if want_out { - let staged: Vec<(String, String)> = overlay - .out_neighbors(&node, label_filter) - .map(|(l, d)| (l.to_string(), d.to_string())) - .collect(); - for (label, dst) in staged { - edges.push((node.clone(), label, dst.clone())); - self.enqueue_str(&dst, next_depth, max_visited, &mut visited, &mut queue); - } - } - if want_in { - let staged: Vec<(String, String)> = overlay - .in_neighbors(&node, label_filter) - .map(|(l, s)| (l.to_string(), s.to_string())) - .collect(); - for (label, src) in staged { - edges.push((src.clone(), label, node.clone())); - self.enqueue_str(&src, next_depth, max_visited, &mut visited, &mut queue); - } - } + frontier = admit_names(candidates, &mut visited, max_visited); } edges } +} - /// Enqueue `name` at `next_depth` if it is newly visited and the visited - /// cap has not been reached. - fn enqueue_str( - &self, - name: &str, - next_depth: usize, - max_visited: usize, - visited: &mut HashSet, - queue: &mut VecDeque<(String, usize)>, - ) { - if visited.len() < max_visited && visited.insert(name.to_string()) { - queue.push_back((name.to_string(), next_depth)); +/// Admit `candidates` into `visited` in name order, until `visited` holds +/// `max_visited` nodes. Returns the nodes admitted, in name order. +fn admit_names( + mut candidates: Vec, + visited: &mut HashSet, + max_visited: usize, +) -> Vec { + candidates.sort(); + candidates.dedup(); + let mut admitted = Vec::with_capacity(candidates.len()); + for name in candidates { + if visited.len() >= max_visited { + break; + } + if visited.insert(name.clone()) { + admitted.push(name); } } + admitted } #[cfg(test)] @@ -280,6 +262,29 @@ mod tests { assert_eq!(r, vec!["a", "b", "x", "y"]); } + #[test] + fn a_capped_overlay_bfs_admits_in_name_order() { + // Durable a->b, staged a->x and a->c: a cap of 3 admits b and c. + let csr = base(); + let mut ov = GraphOverlayDelta::new(); + ov.stage_edge("a", "KNOWS", "x"); + ov.stage_edge("a", "KNOWS", "c"); + + let mut r = csr.traverse_bfs( + BfsParams { + start_nodes: &["a"], + label_filter: Some("KNOWS"), + direction: Direction::Out, + max_depth: 2, + max_visited: 3, + frontier_bitmap: None, + }, + Some(&ov), + ); + r.sort(); + assert_eq!(r, vec!["a", "b", "c"]); + } + #[test] fn tombstone_skips_durable_edge() { let csr = base(); diff --git a/nodedb-graph/src/traversal_surrogate.rs b/nodedb-graph/src/traversal_surrogate.rs index e87526737..82c835039 100644 --- a/nodedb-graph/src/traversal_surrogate.rs +++ b/nodedb-graph/src/traversal_surrogate.rs @@ -15,7 +15,7 @@ //! directly with any other engine's. Node names are resolved once, by the //! caller, and only for the rows that survive fusion. -use std::collections::{HashSet, VecDeque}; +use std::collections::HashSet; use nodedb_types::{Surrogate, SurrogateBitmap}; @@ -104,6 +104,12 @@ impl CsrIndex { /// string-keyed path; the difference is that nothing is ever converted to a /// name. Nodes without a surrogate stay traversable — they just cannot be /// reported (see [`SurrogateHops::unaddressable`]). + /// + /// The walk admits nodes in `(depth, node name)` order: each level's new + /// nodes in name order. When `max_visited` cuts the walk, the admitted set + /// is the same however the edges are stored, which lets a cluster + /// coordinator walking the same edges across partitions admit the same + /// nodes. pub fn traverse_surrogates_in_collection( &self, params: SurrogateBfsParams<'_>, @@ -131,56 +137,71 @@ impl CsrIndex { let label_id = label_filter.and_then(|l| self.label_id(l)); let mut visited: HashSet = HashSet::with_capacity(max_visited.min(1024)); - let mut queue: VecDeque<(u32, usize)> = VecDeque::new(); + let mut frontier: Vec = Vec::new(); for &local in seeds { - if !self.is_local_node(local) { - continue; - } - if visited.insert(local) { - hops.record(self, local, 0); - queue.push_back((local, 0)); + if self.is_local_node(local) && visited.insert(local) { + frontier.push(local); } } + self.sort_by_name(&mut frontier); + for &local in &frontier { + hops.record(self, local, 0); + } let want_out = matches!(direction, Direction::Out | Direction::Both); let want_in = matches!(direction, Direction::In | Direction::Both); - while let Some((node, depth)) = queue.pop_front() { - if depth >= max_depth { - continue; - } - self.record_access(node); - let next_depth = depth + 1; - - let mut neighbors: Vec<(u32, u32)> = Vec::new(); - if want_out { - neighbors.extend(self.iter_out_edges_raw_in(node, collection_id)); + // Level by level. Each level's new nodes are admitted in node-name + // order, so under the visit cap the admitted set is the same whatever + // order the edges are stored in, here or split across partitions. + for depth in 0..max_depth { + if frontier.is_empty() { + break; } - if want_in { - neighbors.extend(self.iter_in_edges_raw_in(node, collection_id)); - } - - for (lid, other) in neighbors { - if label_id.is_some_and(|f| f != lid) { - continue; + let mut candidates: Vec = Vec::new(); + let mut offered: HashSet = HashSet::new(); + for &node in &frontier { + self.record_access(node); + let mut neighbors: Vec<(u32, u32)> = Vec::new(); + if want_out { + neighbors.extend(self.iter_out_edges_raw_in(node, collection_id)); + } + if want_in { + neighbors.extend(self.iter_in_edges_raw_in(node, collection_id)); } - if visited.contains(&other) { - continue; + for (lid, other) in neighbors { + if label_id.is_some_and(|f| f != lid) + || visited.contains(&other) + || !offered.insert(other) + { + continue; + } + candidates.push(other); } + } + self.sort_by_name(&mut candidates); + let mut next: Vec = Vec::with_capacity(candidates.len()); + for other in candidates { if visited.len() >= max_visited { hops.truncated = true; return hops; } visited.insert(other); - hops.record(self, other, next_depth); + hops.record(self, other, depth + 1); self.prefetch_node(other); - queue.push_back((other, next_depth)); + next.push(other); } + frontier = next; } hops } + /// Order CSR-local ids by node name. + pub(crate) fn sort_by_name(&self, nodes: &mut [u32]) { + nodes.sort_by(|a, b| self.node_name_checked(*a).cmp(&self.node_name_checked(*b))); + } + /// Record the addressable seeds and nothing else. Used when the requested /// edge label does not exist in this partition, so no expansion is possible /// but the seeds themselves are still legitimately reachable at depth 0. @@ -365,6 +386,28 @@ mod tests { assert!(hops.reached.contains(Surrogate::new(10))); } + /// Under the cap, a level admits its nodes in name order, whatever order + /// their edges were inserted in. + #[test] + fn a_capped_level_admits_nodes_in_name_order() { + let mut csr = CsrIndex::new(test_memory()); + for dst in ["z", "m", "b"] { + csr.add_edge_in_collection("a", "knows", dst, "people") + .unwrap_or_else(|e| panic!("seed edge a->{dst}: {e}")); + } + let seeds = [local(&csr, "a")]; + let mut p = params(&seeds, "people"); + p.max_visited = 3; + let hops = csr.traverse_surrogates_in_collection(p); + assert!(hops.truncated); + let admitted: Vec<&str> = hops + .distances + .iter() + .filter_map(|&(l, _)| csr.node_name_checked(l)) + .collect(); + assert_eq!(admitted, vec!["a", "b", "m"]); + } + #[test] fn hitting_max_visited_reports_truncation() { let csr = seeded_csr(); diff --git a/nodedb-physical/src/convert_context.rs b/nodedb-physical/src/convert_context.rs deleted file mode 100644 index 8f1afda29..000000000 --- a/nodedb-physical/src/convert_context.rs +++ /dev/null @@ -1,43 +0,0 @@ -// SPDX-License-Identifier: Apache-2.0 - -//! Deployment-neutral context threaded through the shared `SqlPlan → -//! PhysicalPlan` converter helpers in `crate::convert`. -//! -//! Carries only fields both Origin and Lite can supply. Origin-only state -//! (WAL handle, array catalog, credential store, retention registries) lives -//! on Origin's wrapper context and is consumed by Origin-only converter -//! arms (array DDL/DML, timeseries-retention tier-down) that the shared -//! helpers never touch. - -use std::sync::Arc; - -use nodedb_types::DatabaseId; - -use crate::SurrogateAssigner; - -/// Inputs every shared converter helper needs. -/// -/// Origin and Lite construct this with the same shape; their visitor -/// implementations wrap it (Origin adds catalog/WAL handles, Lite passes -/// it through unchanged). -pub struct SharedConvertContext { - /// Database scope for vShard computation. Every `CollectionKey` the - /// converter builds uses this value, so collections in different - /// databases route to distinct shards. - pub database_id: DatabaseId, - - /// Per-tenant maximum vector dimension (0 = unlimited). Checked during - /// `VectorPrimaryInsert` lowering. - pub max_vector_dim: u32, - - /// `true` when the node is running in cluster mode with a live - /// topology. Origin's array DML/query converters emit `ClusterArray` - /// variants when set; single-node Origin and Lite leave this `false`. - pub cluster_enabled: bool, - - /// CP-side surrogate assigner. Threaded into INSERT/UPSERT/KV-INSERT - /// helpers to bind `(collection, pk_bytes)` → `Surrogate` before the - /// op crosses any plane boundary. `None` only for sub-planners that - /// never lower to surrogate-bearing variants. - pub surrogate_assigner: Option>, -} diff --git a/nodedb-physical/src/error.rs b/nodedb-physical/src/error.rs index 8456027e4..90c1c2925 100644 --- a/nodedb-physical/src/error.rs +++ b/nodedb-physical/src/error.rs @@ -7,8 +7,6 @@ //! Origin maps `ConvertError → nodedb::Error`; Lite will map it to its //! own error type. -use crate::surrogate::SurrogateAssignError; - #[derive(Debug, thiserror::Error)] pub enum ConvertError { /// The plan shape is invalid (unsupported combination, missing field, etc.). @@ -27,10 +25,6 @@ pub enum ConvertError { max: u64, }, - /// Surrogate allocation failed. - #[error(transparent)] - Surrogate(#[from] SurrogateAssignError), - /// Serialization failure (msgpack encoding of filters, projections, etc.). #[error("serialization: {0}")] Serialization(String), diff --git a/nodedb-physical/src/lib.rs b/nodedb-physical/src/lib.rs index 301925bb9..7c9cf79e9 100644 --- a/nodedb-physical/src/lib.rs +++ b/nodedb-physical/src/lib.rs @@ -9,15 +9,11 @@ //! pre-serialisation, cross-plane envelope fields) live in an Origin-side //! wrapper that contains a `PhysicalTask`, not in this crate. -pub mod convert_context; pub mod error; pub mod kv_atomic; pub mod physical_plan; pub mod physical_task; -pub mod surrogate; pub mod visitor; -pub use convert_context::SharedConvertContext; pub use error::ConvertError; -pub use surrogate::{SurrogateAssignError, SurrogateAssigner}; pub use visitor::{PhysicalTaskVisitor, dispatch}; diff --git a/nodedb-physical/src/physical_plan/array.rs b/nodedb-physical/src/physical_plan/array.rs index e32064fa7..b01e0b039 100644 --- a/nodedb-physical/src/physical_plan/array.rs +++ b/nodedb-physical/src/physical_plan/array.rs @@ -108,6 +108,9 @@ pub enum ArrayOp { /// `None` for locally-originated or Raft-replicated writes. #[serde(default)] provenance: Option, + /// The vShard every cell of the op homes to: the vShard its tile + /// hashes to. A Calvin transaction routes and locks the write by it. + vshard_id: u32, }, /// Delete by exact coordinates. `coords_msgpack` is an zerompk @@ -122,6 +125,10 @@ pub enum ArrayOp { /// `None` for locally-originated or Raft-replicated writes. #[serde(default)] provenance: Option, + /// The vShard every coordinate of the op homes to: the vShard its + /// tile hashes to. A Calvin transaction routes and locks the write + /// by it. + vshard_id: u32, }, /// Coord-range slice with optional attribute projection. @@ -251,9 +258,10 @@ pub enum ArrayOp { /// externally planned `DROP ARRAY` operation and is idempotent. DropArray { array_id: ArrayId }, - /// Undo a staged array drop by restoring its deterministic tombstone. - /// This is an internal all-core compensation operation. - RestoreArrayDrop { array_id: ArrayId }, + /// Move a store from `array_id` to `target` on one core: flush, close, + /// and rename its directory. The MOVE TENANT rekey sends it to every + /// core. Idempotent once the store sits under `target`. + RekeyArray { array_id: ArrayId, target: ArrayId }, /// Permanently purge a successfully dropped array's tombstone. This is an /// internal all-core operation; failures must be retried before recreation. @@ -276,7 +284,7 @@ impl ArrayOp { | ArrayOp::Flush { array_id, .. } | ArrayOp::Compact { array_id, .. } | ArrayOp::DropArray { array_id } - | ArrayOp::RestoreArrayDrop { array_id } + | ArrayOp::RekeyArray { array_id, .. } | ArrayOp::PurgeArrayDrop { array_id } => array_id, ArrayOp::Elementwise { left, .. } => left, } diff --git a/nodedb-physical/src/physical_plan/cluster_event.rs b/nodedb-physical/src/physical_plan/cluster_event.rs index 2407d4849..7c38fc69d 100644 --- a/nodedb-physical/src/physical_plan/cluster_event.rs +++ b/nodedb-physical/src/physical_plan/cluster_event.rs @@ -24,31 +24,50 @@ pub const MAX_REMOTE_CDC_COMMITTED_OFFSETS: usize = 4_096; zerompk::FromMessagePack, )] pub enum ClusterEventOp { - /// Consume CDC events from the leader node's local event buffer. + /// Consume CDC events from a replica's local event buffer. /// /// `committed_offsets` belongs to the caller node: each tuple is - /// `(partition_id, lsn, sequence)`. The receiver must use these cursors - /// rather than its own consumer-group offset store. Missing partitions - /// start at the initial `(0, 0)` position. + /// `(partition_id, epoch, index, sequence)`. The receiver must use these + /// cursors rather than its own consumer-group offset store. Missing + /// partitions start at the initial `(0, 0, 0)` position. ConsumeStream { database_id: DatabaseId, stream_name: String, group_name: String, partition: Option, limit: u64, - committed_offsets: Vec<(u32, u64, u64)>, - }, - /// Publish one durable-topic message on the topic's home node. - PublishTopic { - database_id: DatabaseId, - topic_name: String, - payload: String, + committed_offsets: Vec<(u32, u64, u64, u64)>, }, /// Read the receiving node's tenant write marks of `group_ids`, once it /// applied every entry the groups committed before the request. RESTORE's /// staleness guard asks a replica of each group this way when the /// restoring node does not replicate the group. TenantWriteMarks { tenant_id: u64, group_ids: Vec }, + /// Read the receiving node's PK→surrogate binds of `tenant_id`'s + /// `collections` in `database_id` with a home among `vshards`. A backup + /// or MOVE TENANT capture asks the source node of each vShard this way, + /// so every bind comes from a node that holds it. + SurrogateBinds { + tenant_id: u64, + database_id: DatabaseId, + vshards: Vec, + collections: Vec, + }, + /// Answer once the receiving node applied the metadata log through + /// `index`. A restore asks every node this way after it raises the + /// surrogate high-water mark, so no node issues a surrogate the restore + /// binds. + MetadataApplied { index: u64 }, + /// Read which primary key the receiving node binds each + /// `(collection, surrogate)` of `entries` to, in `tenant_id`'s + /// `database_id`. A re-issue asks each carried surrogate's home leader + /// this way before it binds, so a surrogate already bound to another key + /// fails the re-issue. + SurrogateHolders { + tenant_id: u64, + database_id: DatabaseId, + entries: Vec<(String, u32)>, + }, } #[cfg(test)] @@ -64,7 +83,7 @@ mod tests { group_name: "Analytics".into(), partition: Some(7), limit: 128, - committed_offsets: vec![(7, 42, 3)], + committed_offsets: vec![(7, 1, 42, 3)], }); let encoded = wire::encode(&plan).expect("encode typed cluster event"); assert_eq!( diff --git a/nodedb-physical/src/physical_plan/collection.rs b/nodedb-physical/src/physical_plan/collection.rs index b031081d0..4bd095a99 100644 --- a/nodedb-physical/src/physical_plan/collection.rs +++ b/nodedb-physical/src/physical_plan/collection.rs @@ -91,7 +91,7 @@ impl PhysicalPlan { } // Read-only resolve wrapper: it reports the wrapped ingest's // collection, which is what the propose step routes on. - PhysicalPlan::Timeseries(TimeseriesOp::ResolveIngest(inner)) => match inner.as_ref() { + PhysicalPlan::Timeseries(TimeseriesOp::ResolveIngest(inner)) => match &inner.ingest { TimeseriesOp::Scan { collection, .. } | TimeseriesOp::Ingest { collection, .. } | TimeseriesOp::Truncate { collection, .. } => Some(collection.as_str()), @@ -162,14 +162,15 @@ impl PhysicalPlan { /// write several collections, so [`Self::collection`] reports none for /// them. A caller that keys on a collection name uses this instead. pub fn named_collections(&self) -> Vec<&str> { - if let PhysicalPlan::Meta( - MetaOp::ApplyTransactionRedo { collections, .. } - | MetaOp::CalvinFlush { collections, .. }, - ) = self - { - collections.iter().map(String::as_str).collect() - } else { - self.collection().into_iter().collect() + match self { + PhysicalPlan::Meta( + MetaOp::ApplyTransactionRedo { collections, .. } + | MetaOp::CalvinFlush { collections, .. }, + ) => collections.iter().map(String::as_str).collect(), + PhysicalPlan::Meta(MetaOp::RestoreRedo(batch)) => { + batch.collections.iter().map(String::as_str).collect() + } + _ => self.collection().into_iter().collect(), } } } diff --git a/nodedb-physical/src/physical_plan/crdt/collection.rs b/nodedb-physical/src/physical_plan/crdt/collection.rs index fdf4165b4..67ad12a1a 100644 --- a/nodedb-physical/src/physical_plan/crdt/collection.rs +++ b/nodedb-physical/src/physical_plan/crdt/collection.rs @@ -83,7 +83,7 @@ mod tests { delta: Vec::new(), peer_id: 1, mutation_id: 1, - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), provenance: None, constraint_version_required: 0, expected_frontier_digest: None, @@ -94,7 +94,7 @@ mod tests { delta: Vec::new(), peer_id: 1, mutation_id: 1, - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), provenance: provenance(), constraint_version_required: 0, expected_frontier_digest: None, @@ -144,7 +144,7 @@ mod tests { collection: coll("restore_to_version"), document_id: "d".to_string(), target_version_json: "{}".to_string(), - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), }, CrdtOp::CompactAtVersion { collection: coll("compact_at_version"), @@ -156,14 +156,14 @@ mod tests { list_path: "blocks".to_string(), index: 0, fields_json: "{}".to_string(), - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), }, CrdtOp::ListDelete { collection: coll("list_delete"), document_id: "d".to_string(), list_path: "blocks".to_string(), index: 0, - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), }, CrdtOp::ListMove { collection: coll("list_move"), @@ -171,13 +171,13 @@ mod tests { list_path: "blocks".to_string(), from_index: 0, to_index: 1, - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), }, CrdtOp::DocUpsert { collection: coll("doc_upsert"), document_id: "d".to_string(), fields_json: "{}".to_string(), - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), partial: false, verb: crate::physical_plan::CrdtWriteVerb::Insert, returning: None, @@ -186,7 +186,7 @@ mod tests { CrdtOp::DocDelete { collection: coll("doc_delete"), document_id: "d".to_string(), - surrogate: Surrogate::ZERO, + surrogate: Some(Surrogate::new(1)), returning: None, rls_filters: Vec::new(), }, diff --git a/nodedb-physical/src/physical_plan/crdt/op.rs b/nodedb-physical/src/physical_plan/crdt/op.rs index 202c287d9..854b95621 100644 --- a/nodedb-physical/src/physical_plan/crdt/op.rs +++ b/nodedb-physical/src/physical_plan/crdt/op.rs @@ -29,8 +29,8 @@ pub enum CrdtOp { /// /// Binds the user-visible `document_id` to a stable cross-engine /// `Surrogate`. UPSERT-aware: if the document already has a surrogate, - /// the assigner returns the existing one. `Surrogate::ZERO` only - /// appears in test fixtures. + /// the assigner returns the existing one. Never `Surrogate::ZERO`: an + /// apply that carries it is refused. Apply { collection: QualifiedCollection, document_id: String, @@ -247,7 +247,9 @@ pub enum CrdtOp { DocDelete { collection: QualifiedCollection, document_id: String, - surrogate: Surrogate, + /// `None` when the key is unbound in its database: no row matches, + /// and the delete answers a count of 0. + surrogate: Option, /// When `Some`, return the STORED pre-image of the deleted row — /// projected per spec. Carried across replication — see /// `DocUpsert::returning`. diff --git a/nodedb-physical/src/physical_plan/document/op.rs b/nodedb-physical/src/physical_plan/document/op.rs index c59eaa0fe..0a127cd39 100644 --- a/nodedb-physical/src/physical_plan/document/op.rs +++ b/nodedb-physical/src/physical_plan/document/op.rs @@ -27,8 +27,10 @@ pub enum DocumentOp { document_id: String, /// Catalog-bound identity for `(collection, document_id)`. Hex-encoded /// by the handler for the substrate row key — user-PK strings are - /// never used for storage addressing here. - surrogate: Surrogate, + /// never used for storage addressing here. `None` when the key is + /// unbound in this database: the read matches no row here, and a + /// clone reads through to its source. + surrogate: Option, /// Raw primary-key bytes, for follower-side WAL decode to re-derive /// the surrogate via the catalog rev table. pk_bytes: Vec, @@ -87,8 +89,7 @@ pub enum DocumentOp { if_absent: bool, /// Stable cross-engine identity assigned by the CP-side /// `SurrogateAssigner` from `(collection, document_id_bytes)`. - /// `Surrogate::ZERO` is reserved as a sentinel and only appears - /// in test fixtures. + /// Never `Surrogate::ZERO`: a put that carries it is refused. surrogate: Surrogate, /// When `Some`, return the STORED post-image of the inserted row /// projected per spec — see `PointPut::returning`. A conflict skipped @@ -116,8 +117,10 @@ pub enum DocumentOp { collection: QualifiedCollection, document_id: String, /// Catalog-bound identity for `(collection, document_id)`. The - /// handler hex-encodes this for the substrate row key. - surrogate: Surrogate, + /// handler hex-encodes this for the substrate row key. `None` when + /// the key is unbound in this database: the delete matches no row + /// here, and a clone resolves it against its source. + surrogate: Option, /// Raw primary-key bytes for follower WAL decode rebind. pk_bytes: Vec, /// When `Some`, return the pre-deletion document projected per spec. @@ -142,8 +145,10 @@ pub enum DocumentOp { collection: QualifiedCollection, document_id: String, /// Catalog-bound identity for `(collection, document_id)`. The - /// handler hex-encodes this for the substrate row key. - surrogate: Surrogate, + /// handler hex-encodes this for the substrate row key. `None` when + /// the key is unbound in this database: the update matches no row + /// here, and a clone copies the source row up first. + surrogate: Option, /// Raw primary-key bytes for follower WAL decode rebind. pk_bytes: Vec, /// Field name → assignment RHS (literal bytes or row-scope expression). @@ -199,9 +204,9 @@ pub enum DocumentOp { collection: QualifiedCollection, /// (document_id, value_bytes) pairs. documents: Vec<(String, Vec)>, - /// Per-row surrogates (parallel to `documents`). When non-empty and - /// same length as `documents`, the handler uses these for FTS indexing. - /// `Surrogate::ZERO` entries are silently skipped by the FTS path. + /// Per-row surrogates, parallel to `documents`: one bound surrogate + /// per row. A batch with a `Surrogate::ZERO` entry is refused before + /// any row is written. surrogates: Vec, /// When `Some`, return one row per inserted document — the STORED /// post-image of each, in `documents` order — projected per spec. @@ -362,7 +367,8 @@ pub enum DocumentOp { value: Vec, on_conflict_updates: Vec<(String, UpdateValue)>, /// Stable cross-engine identity assigned by the CP-side - /// `SurrogateAssigner`. `Surrogate::ZERO` only in test fixtures. + /// `SurrogateAssigner`. Never `Surrogate::ZERO`: an upsert that + /// carries it is refused. surrogate: Surrogate, /// Write policy gating the persist against whichever body actually /// lands: the insert body when absent, the merged/conflict-updated @@ -547,6 +553,12 @@ pub enum DocumentOp { count: usize, system_as_of_ms: Option, // Point-in-time snapshot: `AllVersions` is rejected upstream. + /// Return each body as stored, transcoded to MessagePack with no + /// `id` added. Every clone materializer copy reads these: a + /// `HASH_CHAIN` link covers the stored contents, and a strict target + /// refuses a field its schema lacks. + #[serde(default)] + raw_bodies: bool, }, /// Add a signed amount to a materialized-sum balance on a TARGET row. diff --git a/nodedb-physical/src/physical_plan/graph/algo_stage.rs b/nodedb-physical/src/physical_plan/graph/algo_stage.rs new file mode 100644 index 000000000..bee765060 --- /dev/null +++ b/nodedb-physical/src/physical_plan/graph/algo_stage.rs @@ -0,0 +1,48 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The input stage of [`super::op::GraphOp::Algo`]. +//! +//! Graph edges are partitioned across cores and nodes by endpoint key. An +//! algorithm that needs the whole graph runs in two stages: every core that +//! holds edges exports them, then one core runs the algorithm over the union. + +/// One weighted edge of a collection's graph. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct AlgoEdge { + pub src: String, + pub label: String, + pub dst: String, + /// The edge's `weight` property, `1.0` when it carries none. + pub weight: f64, +} + +/// Where an algorithm reads its graph from. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum AlgoStage { + /// Run over this core's own edges. + Local, + /// Run nothing. Answer this core's edges of the collection, after the + /// label filter, as a msgpack array of [`AlgoEdge`]. `system_as_of_ms` + /// bounds them to the edges live at that system time; `None` answers the + /// current edges. + ExportEdges { system_as_of_ms: Option }, + /// Run over these edges, gathered from every core that holds the + /// collection. The core reads no edges of its own. + Gathered { edges: Vec }, +} diff --git a/nodedb-physical/src/physical_plan/graph/batch_edge.rs b/nodedb-physical/src/physical_plan/graph/batch_edge.rs index 190e78627..14f2f007d 100644 --- a/nodedb-physical/src/physical_plan/graph/batch_edge.rs +++ b/nodedb-physical/src/physical_plan/graph/batch_edge.rs @@ -8,8 +8,7 @@ use nodedb_types::{QualifiedCollection, Surrogate}; /// /// `src_surrogate` / `dst_surrogate` carry the global row identity for the /// edge endpoints (resolved at construction time via the surrogate assigner). -/// `Surrogate::ZERO` is used in test fixtures and on in-memory paths where -/// no catalog is wired; production paths always populate real surrogates. +/// Neither is ever `Surrogate::ZERO`: an edge that carries it is refused. #[derive( Debug, Clone, diff --git a/nodedb-physical/src/physical_plan/graph/bsp.rs b/nodedb-physical/src/physical_plan/graph/bsp.rs index fbfdb7f7e..58e5e4e2d 100644 --- a/nodedb-physical/src/physical_plan/graph/bsp.rs +++ b/nodedb-physical/src/physical_plan/graph/bsp.rs @@ -19,13 +19,18 @@ use nodedb_graph::{AlgoParams, GraphAlgorithm}; zerompk::FromMessagePack, )] pub struct BspSuperstepPlan { - /// Algorithm selector. Only `PageRank` is supported in Phase A; other - /// variants surface a typed `Unsupported` error from the handler. + /// Algorithm selector. Only `PageRank` has a BSP form. Every other + /// variant surfaces a typed `Unsupported` error from the handler. pub algorithm: GraphAlgorithm, /// Algorithm parameters. Carries the target `collection` (mirroring `Algo`) /// plus `damping`. pub params: AlgoParams, - /// Zero-based superstep index. `0` triggers `1/global_n` initialization. + /// Zero-based superstep index. + /// + /// - `0` sets the initial rank (`1/global_n`, or the seed share under + /// personalization) and scatters it. It applies no contribution. + /// - `>= 1` takes the rank from `rank_seed`, computes the next rank from + /// it and `incoming_contributions`, and scatters the next rank. pub superstep: u32, /// Total OWNED nodes across all shards (Control-Plane computed). Used as the /// PageRank `n` in the teleport / dangling redistribution terms. @@ -36,29 +41,27 @@ pub struct BspSuperstepPlan { /// `vertex_count` into the real `global_n`. On that sentinel the handler /// short-circuits after building the owned-node set and runs NO superstep — /// it returns only `vertex_count` + `node_names`. Every real superstep - /// (`superstep >= 0` of the actual run) passes `global_n > 0`. + /// passes `global_n > 0`. pub global_n: usize, /// The vShards this shard owns (Control-Plane supplied). A destination node /// whose `VShardId::from_key(name)` is not in this set is a ghost /// (cross-shard) edge target and its contribution is emitted in `outbound` /// rather than scattered locally. pub owned_vshards: Vec, - /// Cross-shard contributions routed to THIS shard's owned nodes for this - /// superstep: `(dst_node_name, contribution)`. + /// Cross-shard contributions routed to this shard's owned nodes: + /// `(dst_node_name, contribution)`. Every shard scattered them from the + /// rank in `rank_seed`, in the previous superstep. Empty on superstep 0. + /// A contribution to a node this shard does not own is an error. pub incoming_contributions: Vec<(String, f64)>, - /// Round-tripped per-shard rank seed as `(node_name, rank)` pairs (name-keyed, - /// NOT positional) so the same plan can be fanned across a node's cores and - /// each core self-filters to its owned nodes by name. EMPTY on superstep 0 → - /// the handler initializes every owned node to `1/global_n`. A node absent from - /// the seed also falls back to `1/global_n`. + /// The current rank as `(node_name, rank)` pairs: the rank the previous + /// superstep returned. Name-keyed, so the same plan fans across a node's + /// cores and each core picks its owned nodes by name. Empty on superstep + /// 0. From superstep 1, an owned node absent from it is an error. pub rank_seed: Vec<(String, f64)>, - /// Global dangling-node rank mass aggregated by the coordinator from the - /// PREVIOUS superstep across all shards; used for the teleport base so dangling - /// mass redistributes across the WHOLE graph, not just this shard. - /// - /// `0.0` on superstep 0 and the count phase: no previous local sums exist yet, - /// so the base collapses to the plain teleport `(1−d)/n` — identical to a - /// non-dangling graph and correct for initialization. + /// The dangling-node mass of the rank in `rank_seed`, summed by the + /// coordinator over every shard's previous `dangling_sum`. It feeds the + /// redistributed base, so dangling mass spreads over the whole graph. + /// `0.0` on superstep 0 and the count phase. pub global_dangling: f64, /// Coordinator-computed GLOBAL `Σ max(w, 0.0)` over the Personalized-PageRank /// seed map (`params.personalization_vector`), summed across the WHOLE cluster. @@ -75,17 +78,31 @@ pub struct BspSuperstepPlan { /// uniformly. Normalizing by the cluster-wide sum (never a per-shard sum) is /// what preserves the mass-conservation invariant across shards. pub personalization_sum: f64, + /// The watermark of the Calvin cut marker the run's read cut comes + /// from, `0` when `system_as_of` already holds the cut. + /// + /// Every superstep of a run reads the same graph, on every node and + /// core. The run's first dispatch (the count phase) carries the marker. + /// Each node proposes it, waits until it applied and every Calvin + /// scheduler the node runs passed it, and reads at the marker's epoch + /// instant. Every edge version sequenced before the marker is then + /// installed and at or below that cut. Every version sequenced after it + /// is above the cut. The count phase answers the cut in + /// `BspSuperstepResult::system_as_of`, and every later dispatch carries + /// it as `system_as_of`. + pub read_cut_marker: u64, + /// The system-time ordinal every core reads the graph at. The handler + /// refuses a plan without one. + pub system_as_of: Option, } /// Result of one [`super::op::GraphOp::BspSuperstep`] on a single shard. /// /// `rank_vec` and `node_names` are positionally aligned: `rank_vec[i]` is the /// post-superstep PageRank of the owned node `node_names[i]`. The Control-Plane -/// coordinator (Phase B) round-trips `rank_vec` back into the next superstep's -/// `GraphOp::BspSuperstep::rank_vec` and uses `node_names` to map indices back -/// to node identities for final assembly and for routing `outbound` -/// contributions to the owning shard. `node_names` is returned on every -/// superstep (it is cheap and keeps the op stateless). +/// coordinator zips them into the next superstep's `rank_seed` and uses +/// `node_names` for final assembly. `node_names` is returned on every +/// superstep, which keeps the op stateless. #[derive( Debug, Clone, @@ -98,10 +115,11 @@ pub struct BspSuperstepPlan { )] pub struct BspSuperstepResult { /// Sum of `|rank_old - rank_new|` over this shard's owned nodes — the - /// shard's contribution to the global convergence delta. + /// shard's contribution to the global convergence delta. `0.0` on + /// superstep 0, which changes no rank. pub local_delta: f64, - /// Cross-shard contributions to scatter to other shards next superstep: - /// `(target_vshard, dst_node_name, contribution)`. + /// Cross-shard contributions scattered from the returned `rank_vec`, for + /// the next superstep: `(target_vshard, dst_node_name, contribution)`. pub outbound: Vec<(u32, String, f64)>, /// Post-superstep rank vector over this shard's owned nodes, aligned with /// `node_names`. @@ -110,10 +128,9 @@ pub struct BspSuperstepResult { pub vertex_count: usize, /// Owned-node names, positionally aligned with `rank_vec`. pub node_names: Vec, - /// This shard's dangling-node rank mass this superstep (sum of `rank` for all - /// owned nodes with out-degree 0, computed BEFORE the rank swap). The - /// coordinator sums these across shards into the next superstep's - /// `global_dangling` field so dangling mass redistributes globally. + /// The dangling-node mass of the returned `rank_vec`: the rank sum of + /// every owned node with out-degree 0. The coordinator sums these across + /// shards into the next superstep's `global_dangling`. pub dangling_sum: f64, /// Number of this shard's OWNED nodes that appear as a positively-weighted key /// in the Personalized-PageRank seed map (`params.personalization_vector`), @@ -123,4 +140,6 @@ pub struct BspSuperstepResult { /// (matching single-node `build_personalization` returning `None`). `0` on /// every real superstep (only the count phase populates it). pub seed_hits: usize, + /// The system-time ordinal the node read the graph at. + pub system_as_of: Option, } diff --git a/nodedb-physical/src/physical_plan/graph/mod.rs b/nodedb-physical/src/physical_plan/graph/mod.rs index 9c8ff0c5c..61c3c1a5e 100644 --- a/nodedb-physical/src/physical_plan/graph/mod.rs +++ b/nodedb-physical/src/physical_plan/graph/mod.rs @@ -2,12 +2,16 @@ //! Graph engine operations dispatched to the Data Plane. +pub mod algo_stage; pub mod batch_edge; pub mod bsp; pub mod op; +pub mod rag_stage; pub mod wcc; +pub use algo_stage::{AlgoEdge, AlgoStage}; pub use batch_edge::BatchEdge; pub use bsp::{BspSuperstepPlan, BspSuperstepResult}; pub use op::GraphOp; +pub use rag_stage::{RagBindingRow, RagLegs, RagStage, RagTextHit, RagVectorHit}; pub use wcc::{WccSuperstepPlan, WccSuperstepResult}; diff --git a/nodedb-physical/src/physical_plan/graph/op.rs b/nodedb-physical/src/physical_plan/graph/op.rs index 8b5b969dd..955abf8eb 100644 --- a/nodedb-physical/src/physical_plan/graph/op.rs +++ b/nodedb-physical/src/physical_plan/graph/op.rs @@ -7,8 +7,10 @@ use nodedb_types::{ QualifiedCollection, RlsWriteCheck, Surrogate, SurrogateBitmap, SystemTimeScope, }; +use super::algo_stage::AlgoStage; use super::batch_edge::BatchEdge; use super::bsp::BspSuperstepPlan; +use super::rag_stage::RagStage; use super::wcc::WccSuperstepPlan; /// Graph engine physical operations. @@ -178,12 +180,16 @@ pub enum GraphOp { bm25_query: Option, /// Document field scored by BM25. Required when `bm25_query` is set. bm25_field: Option, + /// Which part of the fusion the core runs. + stage: RagStage, }, /// Graph algorithm execution (PageRank, WCC, SSSP, etc.). Algo { algorithm: GraphAlgorithm, params: AlgoParams, + /// Where the algorithm reads its graph from. + stage: AlgoStage, }, /// Graph pattern matching (MATCH clause execution). @@ -303,4 +309,61 @@ pub enum GraphOp { collection: Option, as_of: Option, }, + + /// Guard a node delete on the node's key home, `from_key(node_id)`, + /// which holds every edge incident on the node. + /// + /// The live edges of `collection` with `node_id` as source or + /// destination must be exactly `expected`. Any other state answers + /// `OllpRetryRequired` and the transaction writes nothing. The planner + /// reads the edges, deletes each one in the same transaction, and this + /// guard proves that no edge appeared or vanished in between. The guard + /// itself writes nothing. + NodeEdgeGuard { + collection: QualifiedCollection, + node_id: String, + expected: Vec, + }, + + /// A CRDT document delete's presence guard on the collection's vShard + /// `vshard`. It runs only inside a Calvin transaction. + /// + /// It holds only when every id in `present` is a stored document of + /// `collection` and no id in `absent` is. Any other state answers + /// `OllpRetryRequired`, and the transaction writes nothing. The planner + /// tombstones the edges of the present documents' nodes only, and this + /// guard proves no document appeared or vanished since + /// [`GraphOp::NodePresenceRead`] read them. + NodePresenceGuard { + collection: QualifiedCollection, + vshard: u32, + present: Vec, + absent: Vec, + }, + + /// Tombstone every live edge of `collection` with a home on `vshard` + /// whose newest version is below the transaction's ordinal. + /// + /// TRUNCATE of an edge-bearing collection is one Calvin transaction: the + /// rows' truncate on the collection's vShard, and one copy of this on + /// every vShard. The copy stages nothing. Its resolve runs at the + /// transaction's turn on `vshard`, reads the collection's live edges + /// there, and tombstones each one at the transaction's ordinal, which + /// every participant shares. So both homes of an edge stamp one key, an + /// edge sequenced before the TRUNCATE is gone, and one sequenced after + /// it stays. + TruncateEdges { + collection: QualifiedCollection, + vshard: u32, + }, + + /// The stored CRDT documents of `collection` among `ids`, read on the + /// collection's vShard `vshard`. The answer is a msgpack array of the + /// stored ids. A CRDT document delete's planner reads it before it + /// builds the [`GraphOp::NodePresenceGuard`]. It writes nothing. + NodePresenceRead { + collection: QualifiedCollection, + vshard: u32, + ids: Vec, + }, } diff --git a/nodedb-physical/src/physical_plan/graph/rag_stage.rs b/nodedb-physical/src/physical_plan/graph/rag_stage.rs new file mode 100644 index 000000000..9f6738ed2 --- /dev/null +++ b/nodedb-physical/src/physical_plan/graph/rag_stage.rs @@ -0,0 +1,143 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The stage of [`super::op::GraphOp::RagFusion`] a core runs. +//! +//! On one node, one core runs the whole fusion over its own partitions. In a +//! cluster the legs live apart: the vector index and the text index sit on +//! the collection's owner, and graph edges sit on their endpoints' key +//! vShards. A coordinator then runs the fusion in stages: +//! +//! 1. The owner exports the raw vector and BM25 hits ([`RagStage::ExportLegs`]). +//! 2. Every graph owner answers which of the hits' surrogates name graph +//! nodes ([`RagStage::Bindings`]). +//! 3. The coordinator walks the graph from those nodes across shards. +//! 4. Every graph owner answers which reached nodes carry a surrogate +//! ([`RagStage::Bindings`] again). +//! 5. The coordinator fuses the three lists as the single-core fusion does. + +/// Which part of a RAG fusion a core runs. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum RagStage { + /// The whole fusion, over this core's own indexes and edges. + Local, + /// The vector leg, and the BM25 leg of a three-source fusion, only. + /// Answered as a one-element msgpack array holding a [`RagLegs`]. + ExportLegs, + /// This core's graph bindings for the given surrogates and names, and + /// whether it holds edges of the fusion's collection. Answered as a + /// msgpack array of [`RagBindingRow`]. + Bindings { + surrogates: Vec, + names: Vec, + }, +} + +/// One vector-leg hit, in rank order. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct RagVectorHit { + /// The row's global surrogate. `None` for an index entry that predates + /// surrogates: it ranks under a key that matches nothing else. + pub surrogate: Option, + /// The HNSW entry id, which keys a hit with no surrogate. + pub entry_id: u32, + pub distance: f32, +} + +/// One BM25-leg hit, in rank order. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct RagTextHit { + pub surrogate: u32, + pub score: f32, +} + +/// The owner's raw legs of a RAG fusion. +#[derive( + Debug, + Clone, + Default, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct RagLegs { + pub vector: Vec, + /// Empty for a two-source fusion. + pub text: Vec, +} + +/// One row of a core's [`RagStage::Bindings`] answer. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum RagBindingRow { + /// Graph node `name` carries `surrogate` in this core's partition. + Bound { name: String, surrogate: u32 }, + /// This core's partition holds edges of the fusion's collection. + KnowsCollection, +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn legs_and_binding_rows_round_trip() { + let legs = RagLegs { + vector: vec![RagVectorHit { + surrogate: Some(7), + entry_id: 3, + distance: 0.25, + }], + text: vec![RagTextHit { + surrogate: 9, + score: 1.5, + }], + }; + let bytes = zerompk::to_msgpack_vec(&vec![legs.clone()]).expect("encode legs"); + let decoded: Vec = zerompk::from_msgpack(&bytes).expect("decode legs"); + assert_eq!(decoded, vec![legs]); + + let rows = vec![ + RagBindingRow::Bound { + name: "alice".into(), + surrogate: 7, + }, + RagBindingRow::KnowsCollection, + ]; + let bytes = zerompk::to_msgpack_vec(&rows).expect("encode rows"); + let decoded: Vec = zerompk::from_msgpack(&bytes).expect("decode rows"); + assert_eq!(decoded, rows); + } +} diff --git a/nodedb-physical/src/physical_plan/graph/wcc.rs b/nodedb-physical/src/physical_plan/graph/wcc.rs index a327e258b..b2ef82a1a 100644 --- a/nodedb-physical/src/physical_plan/graph/wcc.rs +++ b/nodedb-physical/src/physical_plan/graph/wcc.rs @@ -29,6 +29,14 @@ pub struct WccSuperstepPlan { /// (cross-shard) edge target and the edge is recorded as a boundary edge /// rather than unioned locally. pub owned_vshards: Vec, + /// The watermark of the Calvin cut marker the read cut comes from, `0` + /// when `system_as_of` already holds it. Every node resolves the same + /// cut from it (see `BspSuperstepPlan::read_cut_marker`), so every node + /// reads the same graph. + pub read_cut_marker: u64, + /// The system-time ordinal every core reads the graph at. The handler + /// refuses a plan without one. + pub system_as_of: Option, } /// Result of one [`super::op::GraphOp::WccSuperstep`] on a single shard. @@ -59,4 +67,6 @@ pub struct WccSuperstepResult { pub boundary_edges: Vec<(String, String)>, /// Number of owned nodes on this shard (== `node_labels.len()`). pub vertex_count: usize, + /// The system-time ordinal the node read the graph at. + pub system_as_of: Option, } diff --git a/nodedb-physical/src/physical_plan/kv/op.rs b/nodedb-physical/src/physical_plan/kv/op.rs index 9c392caef..9e1366330 100644 --- a/nodedb-physical/src/physical_plan/kv/op.rs +++ b/nodedb-physical/src/physical_plan/kv/op.rs @@ -50,8 +50,8 @@ pub enum KvOp { /// Per-key TTL override in milliseconds. 0 = use collection default. ttl_ms: u64, /// Stable cross-engine identity assigned by the CP-side - /// `SurrogateAssigner` from `(collection, key)`. - /// `Surrogate::ZERO` only appears in test fixtures. + /// `SurrogateAssigner` from `(collection, key)`. Never + /// `Surrogate::ZERO`: a write that carries it is refused. surrogate: Surrogate, /// When `Some`, return the STORED post-image (row as `SELECT` would /// show it, `key` included). Never the caller's submitted body. @@ -75,7 +75,7 @@ pub enum KvOp { key: Vec, value: Vec, ttl_ms: u64, - /// Stable cross-engine identity. `Surrogate::ZERO` only in tests. + /// Stable cross-engine identity. Never `Surrogate::ZERO`. surrogate: Surrogate, /// See `Put::returning`. #[serde(default)] @@ -92,7 +92,7 @@ pub enum KvOp { key: Vec, value: Vec, ttl_ms: u64, - /// Stable cross-engine identity. `Surrogate::ZERO` only in tests. + /// Stable cross-engine identity. Never `Surrogate::ZERO`. surrogate: Surrogate, /// See `Put::returning`. #[serde(default)] @@ -114,7 +114,7 @@ pub enum KvOp { value: Vec, ttl_ms: u64, updates: Vec<(String, crate::physical_plan::document::UpdateValue)>, - /// Stable cross-engine identity. `Surrogate::ZERO` only in tests. + /// Stable cross-engine identity. Never `Surrogate::ZERO`. surrogate: Surrogate, /// Write policy against the body actually persisted — insert branch /// or the conflict-merge, neither of which exists at plan time. @@ -231,7 +231,7 @@ pub enum KvOp { /// Stable cross-engine identity for each entry, same order and /// length as `entries`, assigned by the CP-side `SurrogateAssigner` /// from `(collection, key)` -- the same mechanism `Put`/`Insert` - /// use. `Surrogate::ZERO` only appears in test fixtures. + /// use. Never `Surrogate::ZERO`. #[serde(default)] surrogates: Vec, /// When `Some`, return one row per written entry — the STORED @@ -344,7 +344,7 @@ pub enum KvOp { /// protocol boundary. It is parsed once, where it is added, so no /// digit is lost to an `f64` on the way. delta: String, - /// Stable cross-engine identity. `Surrogate::ZERO` only in tests. + /// Stable cross-engine identity. Never `Surrogate::ZERO`. surrogate: Surrogate, /// Compiled row-level-security WRITE predicate — see `Incr`, whose /// engine-internal compute-and-persist this mirrors. @@ -362,7 +362,7 @@ pub enum KvOp { key: Vec, expected: Vec, new_value: Vec, - /// Stable cross-engine identity. `Surrogate::ZERO` only in tests. + /// Stable cross-engine identity. Never `Surrogate::ZERO`. surrogate: Surrogate, /// Compiled row-level-security WRITE predicate, evaluated against /// `new_value` before the swap is attempted, or the reason no @@ -377,7 +377,7 @@ pub enum KvOp { collection: QualifiedCollection, key: Vec, new_value: Vec, - /// Stable cross-engine identity. `Surrogate::ZERO` only in tests. + /// Stable cross-engine identity. Never `Surrogate::ZERO`. surrogate: Surrogate, /// Row-level-security READ filters applied to the OLD value this op /// hands back. The reply is a row body, so a row the read policy hides diff --git a/nodedb-physical/src/physical_plan/kv/resolved_mutation.rs b/nodedb-physical/src/physical_plan/kv/resolved_mutation.rs index e13408651..e2aec6f5a 100644 --- a/nodedb-physical/src/physical_plan/kv/resolved_mutation.rs +++ b/nodedb-physical/src/physical_plan/kv/resolved_mutation.rs @@ -29,7 +29,8 @@ use nodedb_types::{QualifiedCollection, Surrogate}; zerompk::FromMessagePack, )] pub enum KvResolvedMutation { - /// Write `value` under `key`, replacing whatever is there. + /// Write `value` under `key`, replacing whatever is there. `surrogate` + /// is the row's bound identity and is never `Surrogate::ZERO`. Put { collection: QualifiedCollection, key: Vec, @@ -45,6 +46,20 @@ pub enum KvResolvedMutation { surrogate: Surrogate, precondition: Option>, }, + /// Replace the value of the row `key` already holds, keeping its bound + /// identity. The row must be present and hold exactly `precondition`. A + /// predicate update resolves to this: it rewrites existing rows and + /// allocates no identity. + Rewrite { + collection: QualifiedCollection, + key: Vec, + value: Vec, + /// See `Put::ttl_ms`. + ttl_ms: u64, + /// See `Put::expire_at_ms`. + expire_at_ms: u64, + precondition: Vec, + }, /// Remove `key`. Delete { collection: QualifiedCollection, @@ -75,6 +90,7 @@ impl KvResolvedMutation { pub fn collection(&self) -> &QualifiedCollection { match self { KvResolvedMutation::Put { collection, .. } + | KvResolvedMutation::Rewrite { collection, .. } | KvResolvedMutation::Delete { collection, .. } | KvResolvedMutation::Expire { collection, .. } | KvResolvedMutation::Persist { collection, .. } => collection, @@ -85,6 +101,7 @@ impl KvResolvedMutation { pub fn key(&self) -> &[u8] { match self { KvResolvedMutation::Put { key, .. } + | KvResolvedMutation::Rewrite { key, .. } | KvResolvedMutation::Delete { key, .. } | KvResolvedMutation::Expire { key, .. } | KvResolvedMutation::Persist { key, .. } => key.as_slice(), @@ -98,6 +115,7 @@ impl KvResolvedMutation { | KvResolvedMutation::Delete { precondition, .. } | KvResolvedMutation::Expire { precondition, .. } | KvResolvedMutation::Persist { precondition, .. } => precondition.as_deref(), + KvResolvedMutation::Rewrite { precondition, .. } => Some(precondition.as_slice()), } } } diff --git a/nodedb-physical/src/physical_plan/meta.rs b/nodedb-physical/src/physical_plan/meta.rs index 9a2bec498..727acc5e7 100644 --- a/nodedb-physical/src/physical_plan/meta.rs +++ b/nodedb-physical/src/physical_plan/meta.rs @@ -102,6 +102,16 @@ pub enum MetaOp { /// Data Plane never reads it. #[serde(default)] cut_watermark: Option, + /// `Some` asks the receiving node to take the cut with barriers that + /// capture the request's tenants, and to answer with the captures it + /// parked instead of a snapshot. Requires `cut_watermark`. The Data + /// Plane never reads it. + #[serde(default)] + cut_capture: Option, + /// Export every array cell version. Only a backup and a Raft group + /// snapshot read them, so every other snapshot skips the export. + #[serde(default)] + arrays: bool, }, /// Restore a tenant's data across all engines from a snapshot. @@ -116,17 +126,19 @@ pub enum MetaOp { tenant_id: u64, snapshot: Vec, replace_mode: bool, - /// vShard IDs whose state must be cleared before install (clear-then-install - /// for a lagging follower). Empty = legacy install-over-present behavior. + /// Collections this core clears before it installs, pre-resolved by + /// the Raft snapshot applier from the local catalog. Empty = no clear. #[serde(default)] - clear_vshards: Vec, - /// `(database_id, tenant_id, collection)` triples to clear before - /// install — pre-resolved by the applier from the local catalog for the - /// cleared vShards. `collection` is the name the Data Plane stores the - /// collection under: database-qualified outside the default database. - /// Empty = no clear. + collections_to_clear: Vec, + /// The data group's vShards, whose records the install replaces: the + /// array cells with the snapshot's `arrays`, and every graph edge + /// with an endpoint home among them with the snapshot's edges. With + /// it set, `collections_to_clear` keeps graph edges: an edge lives + /// on its endpoints' vShards, not on its collection's home. Empty = + /// no group install: no array replace, and collection clears take + /// their edges too. #[serde(default)] - collections_to_clear: Vec<(u64, u64, String)>, + group_vshards: Vec, }, /// Purge ALL data for a tenant across every engine and cache. @@ -190,6 +202,11 @@ pub enum MetaOp { /// without waiting for a purge cycle. QueryCollectionSize { tenant_id: u64, name: String }, + /// Walk a `HASH_CHAIN` collection's chain in install order from genesis + /// over its raw stored rows. Read-only. The response payload is a + /// MessagePack `ChainVerdict` naming the first break, if any. + VerifyHashChain { collection: QualifiedCollection }, + /// Enforce retention on a timeseries collection: drop segments older than /// the cutoff. Called by the retention policy enforcement loop. EnforceTimeseriesRetention { @@ -340,6 +357,11 @@ pub enum MetaOp { /// defaults to empty on decode of older entries. #[serde(default)] versioned_reads: Vec, + /// Indexes into `plans` of the plans a trigger body buffered. Their + /// rows commit under `Trigger`, so they fire no trigger. Empty for a + /// transaction no body joined. + #[serde(default)] + body_plans: Vec, }, /// Calvin dependent-read executor: passive participant reads keys and @@ -438,23 +460,6 @@ pub enum MetaOp { /// Called after the Control Plane has already removed it from the catalog. DeleteSynonymGroup { tenant_id: u64, name: String }, - /// Re-key all documents and secondary indexes for a collection from - /// `old_collection` (db-qualified source name) to `new_collection` - /// (db-qualified target name) in the local Data Plane sparse engine. - /// - /// Called after `MoveTenantCutover` applies so that physical data is - /// accessible under the new database context. Both `old_collection` and - /// `new_collection` are the `db_qualified` strings used as the logical - /// collection identifier in the sparse store - /// (e.g. `"2/orders"` for database 2, collection `orders`). - RenameCollection { - tenant_id: u64, - old_database_id: u64, - new_database_id: u64, - old_collection: QualifiedCollection, - new_collection: QualifiedCollection, - }, - /// Execute a point write at STATEMENT time by STAGING it into the /// per-transaction overlay, instead of buffering it for COMMIT. /// @@ -606,4 +611,26 @@ pub enum MetaOp { sum_targets: Vec, origin: super::RedoOrigin, }, + + /// Install one batch of a RESTORE as part of a Calvin transaction, on + /// the vShard the batch names. + /// + /// It stages nothing. The transaction's resolve appends the batch's rows + /// and edge versions to its redo record, each edge version applied at + /// the transaction's ordinal, and its flush installs the record as a + /// RESTORE. Its write keys are the rows and edges the batch writes. + RestoreRedo(Box), + + /// Report the current version of each home in `probes`. + /// + /// A read-only transaction's commit sends this to the leader of the + /// vShards its cross-shard graph reads observed, one request per leader. + /// The leader hands each core the probes whose vShard that core owns. The + /// core answers each probe with the collection's write floor on the core, + /// or with the core watermark for a probe with no collection. The payload + /// is a msgpack array of `HomeVersion`. It reads nothing else and writes + /// nothing. + HomeVersions { + probes: Vec, + }, } diff --git a/nodedb-physical/src/physical_plan/meta_home.rs b/nodedb-physical/src/physical_plan/meta_home.rs new file mode 100644 index 000000000..cec6fb3d0 --- /dev/null +++ b/nodedb-physical/src/physical_plan/meta_home.rs @@ -0,0 +1,94 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The request and answer rows of `MetaOp::HomeVersions`. + +/// One vShard home whose current version a core reports. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + PartialOrd, + Ord, + Hash, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct HomeVersionProbe { + /// The vShard the home read observed. + pub vshard: u32, + /// The database-qualified collection the read scoped, or `None` when the + /// read walked every collection. + pub collection: Option, +} + +/// A node's answer for one [`HomeVersionProbe`]. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct HomeVersion { + pub probe: HomeVersionProbe, + pub answer: HomeAnswer, +} + +/// What a node answers for one home. +#[derive( + Debug, + Clone, + Copy, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum HomeAnswer { + /// The collection's write floor on the probe's core, or that core's + /// watermark for a probe with no collection. + Version(u64), + /// The node does not hold the leader lease of the probe's group, so it + /// cannot answer. `leader_node` is the leader its routing table names at + /// `leader_term`, `0` when it names none. + NotLeader { leader_node: u64, leader_term: u64 }, +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn answers_round_trip_through_msgpack() { + let answers = vec![ + HomeVersion { + probe: HomeVersionProbe { + vshard: 7, + collection: Some("db1.edges".into()), + }, + answer: HomeAnswer::Version(42), + }, + HomeVersion { + probe: HomeVersionProbe { + vshard: 9, + collection: None, + }, + answer: HomeAnswer::NotLeader { + leader_node: 2, + leader_term: 7, + }, + }, + ]; + let bytes = zerompk::to_msgpack_vec(&answers).expect("encode"); + let decoded: Vec = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(decoded, answers); + } +} diff --git a/nodedb-physical/src/physical_plan/meta_restore.rs b/nodedb-physical/src/physical_plan/meta_restore.rs new file mode 100644 index 000000000..c4336c7b7 --- /dev/null +++ b/nodedb-physical/src/physical_plan/meta_restore.rs @@ -0,0 +1,142 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The payload of `MetaOp::RestoreRedo`: one batch of a RESTORE's rows and +//! edge versions, installed as a Calvin transaction. +//! +//! A RESTORE re-issues what a backup captured in the Calvin sequence, so its +//! writes order against every other transaction, a TRUNCATE included. Each +//! participant installs the batch at its transaction's turn. A restored row +//! installs as the backup holds it. A restored edge version keeps its +//! historical `system_from` and is applied at the transaction's ordinal. + +/// One restored edge version. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct RestoredEdgeVersion { + /// The collection, as the Data Plane stores it. + pub collection: String, + pub src_id: String, + pub label: String, + pub dst_id: String, + pub src_surrogate: u32, + pub dst_surrogate: u32, + /// The system time the backup holds the version at. + pub system_from: i64, + /// The version's properties, `None` for a tombstone. + pub properties: Option>, +} + +/// One restored document row: the keys its writers lock. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct RestoredRow { + /// The collection, as the Data Plane stores it. + pub collection: String, + /// The row's key, as its surrogate binds it. + pub document_id: String, + pub surrogate: u32, +} + +/// One `(collection, key) → surrogate` identity every participant binds +/// before the batch installs. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct RestoredIdentity { + /// The bare catalog name. + pub collection: String, + pub pk_bytes: Vec, + pub surrogate: u32, +} + +/// One batch of a RESTORE on one vShard. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +#[msgpack(map)] +pub struct RestoredRedo { + /// The vShard every row and edge version of the batch writes. + pub vshard: u32, + /// The zerompk-encoded redo record of the restored rows: their + /// sub-records and the change events their install publishes. Empty + /// when the batch restores edges only. + pub rows_redo: Vec, + /// Every row `rows_redo` writes. + pub rows: Vec, + /// The restored edge versions, in apply order. + pub edges: Vec, + /// Every collection the batch writes, as the Data Plane stores it. + pub collections: Vec, + /// Every identity the rows and edge endpoints are stored under. + pub identities: Vec, +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_restored_batch_round_trips_through_msgpack() { + let batch = RestoredRedo { + vshard: 7, + rows_redo: vec![1, 2, 3], + rows: vec![RestoredRow { + collection: "people".into(), + document_id: "alice".into(), + surrogate: 4, + }], + edges: vec![RestoredEdgeVersion { + collection: "knows".into(), + src_id: "alice".into(), + label: "KNOWS".into(), + dst_id: "bob".into(), + src_surrogate: 4, + dst_surrogate: 5, + system_from: 100, + properties: None, + }], + collections: vec!["knows".into(), "people".into()], + identities: vec![RestoredIdentity { + collection: "people".into(), + pk_bytes: b"alice".to_vec(), + surrogate: 4, + }], + }; + let bytes = zerompk::to_msgpack_vec(&batch).expect("encode"); + let decoded: RestoredRedo = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(decoded, batch); + } +} diff --git a/nodedb-physical/src/physical_plan/meta_snapshot.rs b/nodedb-physical/src/physical_plan/meta_snapshot.rs new file mode 100644 index 000000000..0af8302af --- /dev/null +++ b/nodedb-physical/src/physical_plan/meta_snapshot.rs @@ -0,0 +1,51 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Types carried by `MetaOp::RestoreTenantSnapshot` and +//! `MetaOp::CreateTenantSnapshot`. + +/// One collection a Raft snapshot install clears on a core before it +/// installs that core's share of the snapshot. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct SnapshotClearTarget { + pub database_id: u64, + pub tenant_id: u64, + /// The name the Data Plane stores the collection under: + /// database-qualified outside the default database. + pub collection: String, + /// Whether this core also removes the collection's shared on-disk L1 + /// files. Those paths are keyed by `(database, tenant, collection)`, not + /// by core, so exactly one core per collection (its home core) sets it. + pub reclaim_l1_files: bool, +} + +/// A database backup's request to capture its tenants at its cut barrier. +/// +/// The barrier the backup proposes into each data group carries it. The +/// group's leader, when it applies the first barrier of `request_id`, +/// snapshots every tenant of `tenants` in `database_id` before it applies the +/// next entry, and parks the capture for the backup to collect. +#[derive( + Debug, + Clone, + PartialEq, + Eq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct CutCaptureRequest { + /// Unique per backup. It keys every parked capture. + pub request_id: u64, + pub database_id: u64, + pub tenants: Vec, +} diff --git a/nodedb-physical/src/physical_plan/mod.rs b/nodedb-physical/src/physical_plan/mod.rs index 2f00b5eb8..787c86e99 100644 --- a/nodedb-physical/src/physical_plan/mod.rs +++ b/nodedb-physical/src/physical_plan/mod.rs @@ -18,6 +18,9 @@ pub mod graph; pub mod kv; pub mod meta; pub mod meta_calvin; +pub mod meta_home; +pub mod meta_restore; +pub mod meta_snapshot; pub mod plan; pub mod query; pub mod redo_origin; @@ -47,12 +50,16 @@ pub use document::{ }; pub use exchange::{ExchangeMode, ExchangeOp}; pub use graph::{ - BatchEdge, BspSuperstepPlan, BspSuperstepResult, GraphOp, WccSuperstepPlan, WccSuperstepResult, + AlgoEdge, AlgoStage, BatchEdge, BspSuperstepPlan, BspSuperstepResult, GraphOp, RagBindingRow, + RagLegs, RagStage, RagTextHit, RagVectorHit, WccSuperstepPlan, WccSuperstepResult, }; pub use kv::{ KvCounterShape, KvOp, KvResolveOutcome, KvResolvedMutation, SortedIndexRead, SortedIndexSpec, }; pub use meta::{MetaOp, SAVEPOINT_MARKER_BYTES}; +pub use meta_home::{HomeAnswer, HomeVersion, HomeVersionProbe}; +pub use meta_restore::{RestoredEdgeVersion, RestoredIdentity, RestoredRedo, RestoredRow}; +pub use meta_snapshot::{CutCaptureRequest, SnapshotClearTarget}; pub use plan::PhysicalPlan; pub use query::{AggregateSpec, GroupKeySpec, JoinProjection, QueryOp}; pub use redo_origin::RedoOrigin; @@ -61,7 +68,7 @@ pub use set_op::SetOpKind; pub use sort_key::SortKeySpec; pub use spatial::{SpatialOp, SpatialPredicate}; pub use text::TextOp; -pub use timeseries::{TimeseriesOp, UNBOUNDED_TIME_RANGE}; +pub use timeseries::{TimeseriesOp, TimeseriesResolve, UNBOUNDED_TIME_RANGE}; pub use vector::{ VectorDirectWriteIntent, VectorOp, VectorResolveOutcome, VectorResolvedMutation, VectorWriteTargets, diff --git a/nodedb-physical/src/physical_plan/routing.rs b/nodedb-physical/src/physical_plan/routing.rs index ba224ae09..e9e9170fe 100644 --- a/nodedb-physical/src/physical_plan/routing.rs +++ b/nodedb-physical/src/physical_plan/routing.rs @@ -107,6 +107,10 @@ pub fn plan_contains_cluster_partitioned_leaf(plan: &PhysicalPlan) -> bool { | PhysicalPlan::Graph(GraphOp::ResolveEdgeDelete(_)) | PhysicalPlan::Graph(GraphOp::SetNodeLabels { .. }) | PhysicalPlan::Graph(GraphOp::RemoveNodeLabels { .. }) + | PhysicalPlan::Graph(GraphOp::NodeEdgeGuard { .. }) + | PhysicalPlan::Graph(GraphOp::NodePresenceGuard { .. }) + | PhysicalPlan::Graph(GraphOp::TruncateEdges { .. }) + | PhysicalPlan::Graph(GraphOp::NodePresenceRead { .. }) | PhysicalPlan::Vector(_) | PhysicalPlan::Document(_) | PhysicalPlan::Kv(_) diff --git a/nodedb-physical/src/physical_plan/spatial.rs b/nodedb-physical/src/physical_plan/spatial.rs index c36b6850a..eca3968d5 100644 --- a/nodedb-physical/src/physical_plan/spatial.rs +++ b/nodedb-physical/src/physical_plan/spatial.rs @@ -63,8 +63,10 @@ pub enum SpatialOp { Delete { collection: QualifiedCollection, field: String, - /// Stable global surrogate for the row. - surrogate: Surrogate, + /// Stable global surrogate for the row. `None` when the key's home + /// binds none: the delete removes nothing and still commits its sync + /// provenance. + surrogate: Option, /// Sync provenance: identifies the originating peer and sequence for idempotency. #[serde(default)] provenance: Option, diff --git a/nodedb-physical/src/physical_plan/text.rs b/nodedb-physical/src/physical_plan/text.rs index 34e87c641..9a8dfc487 100644 --- a/nodedb-physical/src/physical_plan/text.rs +++ b/nodedb-physical/src/physical_plan/text.rs @@ -104,8 +104,10 @@ pub enum TextOp { /// Used by the sync path when a Lite client sends an `FtsDelete` frame. FtsDeleteDoc { collection: QualifiedCollection, - /// Pre-assigned global surrogate for `(collection, doc_id)`. - surrogate: nodedb_types::Surrogate, + /// The global surrogate `(collection, doc_id)` is bound to. `None` + /// when the key's home binds none: the delete removes nothing and + /// still commits its sync provenance. + surrogate: Option, /// Sync provenance: identifies the originating peer and sequence for idempotency. #[serde(default)] provenance: Option, diff --git a/nodedb-physical/src/physical_plan/timeseries.rs b/nodedb-physical/src/physical_plan/timeseries.rs index 70867a666..c7cdbedba 100644 --- a/nodedb-physical/src/physical_plan/timeseries.rs +++ b/nodedb-physical/src/physical_plan/timeseries.rs @@ -100,11 +100,11 @@ pub enum TimeseriesOp { rls_filters: Vec, }, - /// Read-only resolve pass for a governed [`TimeseriesOp::Ingest`]: a - /// follower can't judge a live predicate, so this normalizes the payload - /// into stamped ILP lines (memtable schema is Data-Plane-only, so - /// normalization must happen here) and decides the policy without writing. - ResolveIngest(Box), + /// Read-only resolve pass for a [`TimeseriesOp::Ingest`]: it normalizes + /// the payload into stamped ILP lines (memtable schema is Data-Plane-only, + /// so normalization must happen here), decides the policy, and resolves + /// the lines to the rows they store, without writing. + ResolveIngest(Box), /// `TRUNCATE` of a timeseries collection: the memtable, every on-disk /// partition, the series catalog, and the last-value cache. Reports the @@ -117,3 +117,23 @@ pub enum TimeseriesOp { restart_identity: bool, }, } + +/// What a [`TimeseriesOp::ResolveIngest`] resolves. +#[derive( + Debug, + Clone, + PartialEq, + serde::Serialize, + serde::Deserialize, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub struct TimeseriesResolve { + /// The ingest to resolve. + pub ingest: TimeseriesOp, + /// The encoded schema an earlier ingest of the same transaction into + /// the same collection resolved to. The ingest resolves against it, not + /// against the live schema. `None` resolves against the live schema. + #[serde(default)] + pub base: Option>, +} diff --git a/nodedb-physical/src/physical_plan/vector/op.rs b/nodedb-physical/src/physical_plan/vector/op.rs index b7d10a2e4..2f619365d 100644 --- a/nodedb-physical/src/physical_plan/vector/op.rs +++ b/nodedb-physical/src/physical_plan/vector/op.rs @@ -124,7 +124,9 @@ pub enum VectorOp { /// is not found the op is a no-op (idempotent). DeleteBySurrogate { collection: QualifiedCollection, - surrogate: nodedb_types::Surrogate, + /// The row's bound surrogate. `None` when the key's home binds none: + /// the delete removes nothing and still commits its sync provenance. + surrogate: Option, /// Named vector field; empty = default field. field_name: String, /// Sync provenance: identifies the originating peer and sequence for idempotency. @@ -246,6 +248,12 @@ pub enum VectorOp { field_name: String, /// Surrogate shared by all vectors of the document. document_surrogate: Surrogate, + /// User PK bytes of the document when it has a key; `None` for a + /// headless document, whose surrogate self-keys. Carried on the + /// replication wire so every replica binds `document_surrogate` to + /// this exact key, as `Insert` does. + #[serde(default)] + pk_bytes: Option>, /// Flat vector data: count × dim f32 values. vectors: Vec, /// Number of vectors. diff --git a/nodedb-physical/src/physical_plan/vector/write.rs b/nodedb-physical/src/physical_plan/vector/write.rs index 31927d64b..ab14e1fbc 100644 --- a/nodedb-physical/src/physical_plan/vector/write.rs +++ b/nodedb-physical/src/physical_plan/vector/write.rs @@ -41,8 +41,8 @@ pub enum VectorDirectWriteIntent { zerompk::FromMessagePack, )] pub enum VectorWriteTargets { - /// Surrogates the Control Plane resolved from primary-key equalities. - /// `Surrogate::ZERO` is a key with no binding; it matches no row. + /// Surrogates the Control Plane resolved from primary-key equalities. A + /// key with no binding contributes no entry. Surrogates(Vec), /// Serialized `Vec` the Data Plane evaluates against every /// payload sidecar row. Empty bytes match every row. diff --git a/nodedb-physical/src/surrogate.rs b/nodedb-physical/src/surrogate.rs deleted file mode 100644 index f99dbe1cd..000000000 --- a/nodedb-physical/src/surrogate.rs +++ /dev/null @@ -1,68 +0,0 @@ -// SPDX-License-Identifier: Apache-2.0 - -//! Surrogate-allocation contract used by the shared SqlPlan → PhysicalPlan -//! converter. Origin's WAL-durable, Raft-replicated allocator implements this -//! trait; Lite supplies its own local-monotonic implementation. -//! -//! Synchronous-only: the converter runs on the Control Plane in `Send + Sync` -//! code paths. Origin's async surrogate-fetch work stays internal to its impl -//! and is hidden behind this sync facade. - -use nodedb_types::{CollectionKey, Surrogate, TenantId}; - -/// Errors a [`SurrogateAssigner`] may return. -/// -/// The error surface is deliberately narrow — the converter does not need to -/// distinguish more cases. Origin's rich allocator errors collapse to one of -/// these at the trait boundary; the original error is preserved in -/// [`SurrogateAssignError::Backend`]'s message. -#[derive(Debug, thiserror::Error)] -pub enum SurrogateAssignError { - #[error("surrogate registry lock poisoned")] - LockPoisoned, - #[error("surrogate backend: {0}")] - Backend(String), -} - -/// Allocate stable, cross-engine surrogates for `(collection, pk_bytes)`. -/// -/// Implementations must be: -/// - **idempotent**: repeated calls for the same `(collection, pk_bytes)` -/// return the same `Surrogate`; -/// - **monotonic**: every allocated value is greater than every previously -/// allocated value within the same allocator; -/// - **`Send + Sync`**: the converter holds a reference across `await` -/// points on the Control Plane. -pub trait SurrogateAssigner: Send + Sync { - /// Highest surrogate ever issued by this assigner. `0` on a fresh - /// allocator. Used by CLONE DATABASE to capture an AS-OF cutoff. - fn current_hwm(&self) -> u32; - - /// Resolve `(key, tenant_id, pk_bytes)` to a stable surrogate. Allocate - /// on the first call; return the persisted value on every subsequent - /// call (UPSERT preserves the surrogate). `key` carries the database and - /// the bare catalog name. - fn assign( - &self, - key: CollectionKey<'_>, - tenant_id: TenantId, - pk_bytes: &[u8], - ) -> Result; - - /// Allocate a FRESH, never-before-issued surrogate for a row that has no - /// content primary key — i.e. a collection whose primary key is the - /// auto-generated `_rowid` (no `PRIMARY KEY` was declared at CREATE), or - /// a timeseries row. - /// - /// Every call allocates a new value. There is no `pk_bytes` to - /// content-address on, so repeated calls never collapse onto one - /// surrogate. - /// - /// The returned `String` is the bound identity. The caller uses it - /// verbatim and never re-derives it. - fn assign_fresh( - &self, - key: CollectionKey<'_>, - tenant_id: TenantId, - ) -> Result<(Surrogate, String), SurrogateAssignError>; -} diff --git a/nodedb-query/Cargo.toml b/nodedb-query/Cargo.toml index 8c106abf9..666178a06 100644 --- a/nodedb-query/Cargo.toml +++ b/nodedb-query/Cargo.toml @@ -24,5 +24,6 @@ sonic-rs = { workspace = true } rust_decimal = { workspace = true } rust-stemmers = { workspace = true } unicode-normalization = { workspace = true } +hex = { workspace = true } [dev-dependencies] diff --git a/nodedb-query/src/msgpack_scan/group_key.rs b/nodedb-query/src/msgpack_scan/group_key.rs index 916d22a0d..3ed937692 100644 --- a/nodedb-query/src/msgpack_scan/group_key.rs +++ b/nodedb-query/src/msgpack_scan/group_key.rs @@ -121,11 +121,7 @@ fn append_value_at(buf: &mut String, doc: &[u8], start: usize, end: usize) { let _ = write!(buf, "{n}"); } else { // Complex value (array/map/bin) — hex-encode raw bytes as key. - let bytes = &doc[start..end]; - for b in bytes { - use std::fmt::Write; - let _ = write!(buf, "{b:02x}"); - } + buf.push_str(&hex::encode(&doc[start..end])); } } diff --git a/nodedb-raft/src/error.rs b/nodedb-raft/src/error.rs index 75a331840..4ad3b2707 100644 --- a/nodedb-raft/src/error.rs +++ b/nodedb-raft/src/error.rs @@ -6,8 +6,11 @@ pub type Result = std::result::Result; #[derive(Debug, Error)] pub enum RaftError { - #[error("not leader (leader hint: {leader_hint:?})")] - NotLeader { leader_hint: Option }, + /// `leader_hint` is the leader this node knows at `term`, its current + /// term. A receiver keeps the hint only when `term` is above the term + /// of the hint it holds. + #[error("not leader (leader hint: {leader_hint:?} at term {term})")] + NotLeader { leader_hint: Option, term: u64 }, #[error("log compacted: requested index {requested}, first available {first_available}")] LogCompacted { diff --git a/nodedb-raft/src/log.rs b/nodedb-raft/src/log.rs index 3a146b51f..a9ebb232c 100644 --- a/nodedb-raft/src/log.rs +++ b/nodedb-raft/src/log.rs @@ -94,52 +94,48 @@ impl RaftLog { Ok(&self.entries[start..end]) } - /// Append new entries from a leader's AppendEntries RPC. + /// Append new entries from a leader's AppendEntries RPC. Returns whether + /// storage took a write. /// /// Handles conflict detection per Raft paper §5.3: /// - If an existing entry conflicts with a new one (same index, different /// terms), delete the existing entry and all that follow it. /// - Append any new entries not already in the log. /// - /// Persistence happens BEFORE the in-memory log is mutated. The response to - /// an `AppendEntries` RPC reports `last_index()` from the in-memory log, and - /// the leader treats that number as "durably held by this peer" — it counts - /// toward quorum on success and rewinds `next_index` past it on failure. If - /// the in-memory log were advanced first and the storage write then failed, - /// this node would report entries it does not hold, the leader would never - /// resend them, and they would disappear on restart. Mutating memory only - /// after storage has accepted the write makes that state unreachable: a - /// failed persist leaves `last_index()` covering exactly what is on disk. - pub fn append_entries(&mut self, _prev_index: u64, entries: &[LogEntry]) -> Result<()> { - if entries.is_empty() { - return Ok(()); - } - - // Locate the first conflicting index (same index, different term) without - // mutating anything — detection must not be destructive, because the - // durable writes below may still fail. - let conflict = entries - .iter() - .find(|e| matches!(self.entry_at(e.index), Some(existing) if existing.term != e.term)) + /// An entry this log already holds at the same term is not written again. + /// A leader resends a range until it is acknowledged, and a rewrite of a + /// held range only delays the reply. + /// + /// Persistence happens BEFORE the in-memory log is mutated. A success + /// reply claims the matched entries as durable on this node. A storage + /// write that fails after memory moved leaves a claim on entries storage + /// never took, and a restart loses them. A failed persist therefore + /// leaves memory covering exactly what storage took. + /// Storage that stages writes accepts them at once. When this returns + /// `true`, the caller makes them durable before the reply leaves. + pub fn append_entries(&mut self, _prev_index: u64, entries: &[LogEntry]) -> Result { + // The first entry this log does not hold at the same term. Detection + // mutates nothing: the durable writes below can still fail. + let Some(first_new) = entries.iter().position(|e| { + e.index > self.snapshot_index + && self + .entry_at(e.index) + .is_none_or(|existing| existing.term != e.term) + }) else { + return Ok(false); + }; + let new = &entries[first_new..]; + + let conflict = new + .first() + .filter(|e| self.entry_at(e.index).is_some()) .map(|e| e.index); - if let Some(index) = conflict { self.truncate_from(index)?; } - self.storage.append(entries)?; - - for entry in entries { - if entry.index <= self.snapshot_index { - // Already covered by the snapshot; pushing it would break the - // `entries[0].index == snapshot_index + 1` offset invariant. - continue; - } - if self.entry_at(entry.index).is_none() { - self.entries.push(entry.clone()); - } - // Same index AND same term = already present, nothing to do. - } - Ok(()) + self.storage.append(new)?; + self.entries.extend_from_slice(new); + Ok(true) } /// Append a single entry proposed by the leader. @@ -166,17 +162,43 @@ impl RaftLog { } /// Apply a snapshot: discard all entries up to `last_included_index`. - pub fn apply_snapshot(&mut self, last_included_index: u64, last_included_term: u64) { - // Remove entries already covered by the snapshot. - if last_included_index > self.snapshot_index { - let new_start = last_included_index + 1; - self.entries.retain(|e| e.index >= new_start); - self.snapshot_index = last_included_index; - self.snapshot_term = last_included_term; - let _ = self - .storage - .compact(last_included_index, last_included_term); + /// + /// Storage compacts first. A storage error leaves the in-memory log + /// untouched, so memory never claims a boundary storage does not hold. + pub fn apply_snapshot( + &mut self, + last_included_index: u64, + last_included_term: u64, + ) -> Result<()> { + if last_included_index <= self.snapshot_index { + return Ok(()); + } + self.storage + .compact(last_included_index, last_included_term)?; + let new_start = last_included_index + 1; + self.entries.retain(|e| e.index >= new_start); + self.snapshot_index = last_included_index; + self.snapshot_term = last_included_term; + Ok(()) + } + + /// The last index of this log that storage holds durably. + /// + /// Storage whose writes are durable on return holds the whole log. Storage + /// that stages writes reports its last durable entry `(index, term)`. That + /// entry counts only when this log holds the same entry: Raft logs that + /// agree on one entry agree on every entry before it. A durable entry this + /// log no longer holds, or holds at another term, means a truncation is + /// still on its way to disk. The durable prefix is then not known past the + /// snapshot boundary, and the smaller of the two answers. + pub fn stable_index(&self) -> u64 { + let Some((index, term)) = self.storage.stable_through() else { + return self.last_index(); + }; + if index <= self.last_index() && self.term_at(index) == Some(term) { + return index; } + index.min(self.snapshot_index) } pub fn snapshot_index(&self) -> u64 { @@ -266,7 +288,7 @@ mod tests { log.append(make_entry(1, i)).unwrap(); } - log.apply_snapshot(5, 1); + log.apply_snapshot(5, 1).unwrap(); assert_eq!(log.snapshot_index(), 5); assert_eq!(log.last_index(), 10); // Compacted entries are gone. @@ -361,6 +383,30 @@ mod tests { assert_eq!(log.last_index(), 3); } + /// A resent range this log already holds takes no storage write. Only the + /// entries past it are written. + #[test] + fn a_resent_held_range_writes_nothing() { + let mut log = RaftLog::new(FlakyStorage::default()); + let held = [make_entry(1, 1), make_entry(1, 2)]; + assert!(log.append_entries(0, &held).expect("first delivery")); + + log.storage_mut().fail_append = true; + let wrote = log + .append_entries(0, &held) + .expect("a held range needs no write"); + assert!(!wrote); + + log.storage_mut().fail_append = false; + let wrote = log + .append_entries(0, &[make_entry(1, 1), make_entry(1, 2), make_entry(1, 3)]) + .expect("extend past the held range"); + assert!(wrote); + assert_eq!(log.last_index(), 3); + let persisted = log.storage().load_entries_after(0).expect("load"); + assert_eq!(persisted.len(), 3); + } + /// A failed truncate must leave the conflicting suffix in memory, matching /// what storage still holds, and must not append the overwriting entries. #[test] diff --git a/nodedb-raft/src/message.rs b/nodedb-raft/src/message.rs index 85a1a209c..368a054b3 100644 --- a/nodedb-raft/src/message.rs +++ b/nodedb-raft/src/message.rs @@ -58,6 +58,12 @@ pub struct AppendEntriesRequest { pub leader_commit: u64, /// Raft group ID for Multi-Raft routing. pub group_id: u64, + /// Leader-lease round this request belongs to. The follower echoes it so + /// the leader can anchor its lease at the round's send time. + pub round: u64, + /// The highest log index every voter's log is known to hold (see + /// `RaftNode::replicated_floor`). The follower keeps the highest value. + pub replicated_floor: u64, } #[derive( @@ -76,9 +82,16 @@ pub struct AppendEntriesResponse { pub term: u64, /// True if follower contained entry matching prev_log_index and prev_log_term. pub success: bool, - /// Optimization: on rejection, the follower's last log index. - /// Allows leader to skip back faster than decrementing one-by-one. + /// On success, the last entry the follower shares with the leader and + /// holds durably. The leader takes it as the follower's match index. + /// On rejection, the follower's last log index, so the leader skips back + /// faster than one entry at a time. pub last_log_index: u64, + /// `round` of the request this answers. + pub round: u64, + /// Set on a rejection from a follower that holds no state it can resume + /// the log from. The leader sends it a snapshot instead of entries. + pub needs_snapshot: bool, } /// RequestVote RPC (Raft paper Figure 2). @@ -104,6 +117,10 @@ pub struct RequestVoteRequest { pub last_log_term: u64, /// Raft group ID for Multi-Raft routing. pub group_id: u64, + /// Set by a campaign that a `TimeoutNow` started. Only such a campaign + /// may win a vote from a node that still hears a live leader: the leader + /// asked for its own replacement and stopped serving lease reads first. + pub transfer: bool, } #[derive( @@ -235,6 +252,18 @@ pub struct InstallSnapshotRequest { #[serde(default)] #[msgpack(default)] pub total_size: u64, + /// The group's voters as the leader holds them when it sends the final + /// chunk, the leader included. A snapshot covers the conf changes of its + /// range, and the receiver never applies them, so it takes the + /// membership from here. Empty on every other chunk. + #[serde(default)] + #[msgpack(default)] + pub voters: Vec, + /// The group's learners as the leader holds them when it sends the final + /// chunk. Empty on every other chunk. + #[serde(default)] + #[msgpack(default)] + pub learners: Vec, } #[derive( @@ -279,6 +308,8 @@ mod tests { entries: vec![], leader_commit: 8, group_id: 0, + round: 1, + replicated_floor: 0, }; assert!(req.entries.is_empty()); } @@ -318,10 +349,12 @@ mod tests { last_log_index: 100, last_log_term: 6, group_id: 5, + transfer: true, }; let json = sonic_rs::to_string(&req).unwrap(); let decoded: RequestVoteRequest = sonic_rs::from_str(&json).unwrap(); assert_eq!(req.term, decoded.term); assert_eq!(req.candidate_id, decoded.candidate_id); + assert!(decoded.transfer); } } diff --git a/nodedb-raft/src/node/config.rs b/nodedb-raft/src/node/config.rs index f3194282b..ecf30c563 100644 --- a/nodedb-raft/src/node/config.rs +++ b/nodedb-raft/src/node/config.rs @@ -19,6 +19,14 @@ use std::time::Duration; +/// Fraction of `election_timeout_min` the leader lease gives up to clock-rate +/// drift between the leader and its followers. +/// +/// A follower measures its vote-refusal window on its own clock. A lease that +/// ran the full window on the leader's clock would outlive that window on any +/// follower whose clock runs faster. +pub const MAX_CLOCK_DRIFT_RATIO: f64 = 0.1; + /// Configuration for a Raft node. #[derive(Debug, Clone)] pub struct RaftConfig { @@ -96,6 +104,22 @@ impl RaftConfig { pub fn quorum(&self) -> usize { self.cluster_size() / 2 + 1 } + + /// Slack the leader lease subtracts from `election_timeout_min` for + /// clock-rate drift. + pub fn lease_drift_margin(&self) -> Duration { + self.election_timeout_min.mul_f64(MAX_CLOCK_DRIFT_RATIO) + } + + /// How long a leader lease lasts past its anchor: + /// `election_timeout_min - lease_drift_margin()`. + /// + /// A follower refuses votes for `election_timeout_min` after hearing the + /// leader, so no successor can win inside this window. + pub fn lease_duration(&self) -> Duration { + self.election_timeout_min + .saturating_sub(self.lease_drift_margin()) + } } #[cfg(test)] @@ -135,4 +159,13 @@ mod tests { assert_eq!(c.cluster_size(), 5); assert_eq!(c.quorum(), 3); } + + #[test] + fn lease_duration_leaves_the_drift_margin() { + let c = cfg(vec![2, 3], vec![]); + let margin = c.lease_drift_margin(); + // A tenth of 150ms, within float rounding. + assert!(margin > Duration::from_millis(14) && margin < Duration::from_millis(16)); + assert_eq!(c.lease_duration() + margin, c.election_timeout_min); + } } diff --git a/nodedb-raft/src/node/core.rs b/nodedb-raft/src/node/core.rs index 43a2576fe..c31da6293 100644 --- a/nodedb-raft/src/node/core.rs +++ b/nodedb-raft/src/node/core.rs @@ -21,6 +21,7 @@ use crate::state::{ use crate::storage::LogStorage; use super::config::RaftConfig; +use super::leader_lease::LeaseState; use tracing::info; /// Output actions produced by a tick or RPC handler. @@ -46,6 +47,9 @@ pub struct Ready { /// Peers that need an InstallSnapshot RPC because their next_index /// falls behind the leader's snapshot_index (log compacted). pub snapshots_needed: Vec, + /// Set when the next committed range to deliver is no longer in the log. + /// Delivery for the group halts until a snapshot covers the gap. + pub committed_read_error: Option, } impl Ready { @@ -57,6 +61,7 @@ impl Ready { && self.timeout_now.is_empty() && self.committed_entries.is_empty() && self.snapshots_needed.is_empty() + && self.committed_read_error.is_none() } } @@ -104,14 +109,32 @@ pub struct RaftNode { /// that measure, and a fully caught-up follower in an idle cluster looks /// stale. Heartbeats refresh this even when nothing is being written. pub(super) leader_contact: Option, - /// When a quorum of voters last answered this leader. `None` off the - /// leader path. Drives check-quorum step-down (see - /// [`super::quorum_contact`]). + /// Latest instant by which a quorum of voters answered this term, or the + /// election win before any. `None` off the leader path. Drives + /// check-quorum step-down (see [`super::quorum_contact`]). pub(super) last_quorum_contact: Option, - /// Per-voter `ack_count` readings taken when the current contact window - /// opened. A voter whose count has risen above its reading has answered - /// inside the window. - pub(super) quorum_window: Vec<(u64, u64)>, + /// Per-voter highest lease round acknowledged in the current term, and + /// when the latest answer arrived. Cleared on step-down. + pub(super) quorum_window: Vec, + /// Leader-lease rounds and anchor (see [`super::leader_lease`]). + pub(super) lease: LeaseState, + /// Until this instant the node refuses every vote but a transfer vote. + /// + /// Set to `election_timeout_max` after construction. A restarted node has + /// forgotten its leader contact, and a lease the leader took before the + /// restart can still be live. Kept apart from `leader_contact`, which + /// also feeds the staleness bound: a boot is no contact with any leader. + pub(super) boot_vote_fence: Instant, + /// An outside bound on compaction: the log never discards an entry above + /// it. An archiver that must copy entries before they go raises it. + pub(super) compaction_ceiling: Option>, + /// Whether this follower holds no state it can resume the log from. It + /// then refuses every `AppendEntries` with `needs_snapshot`, and its + /// leader sends a snapshot instead. The driver sets and clears it. + pub(super) snapshot_required: bool, + /// The highest `replicated_floor` a leader sent this node, or this + /// node's own as leader. See [`Self::replicated_floor`]. + pub(super) replicated_floor: u64, } /// A leader's commit index as of its last contact with this node. @@ -155,10 +178,26 @@ impl RaftNode { leader_contact: None, last_quorum_contact: None, quorum_window: Vec::new(), + lease: LeaseState::new(), + boot_vote_fence: now + config.election_timeout_max, + compaction_ceiling: None, + snapshot_required: false, + replicated_floor: 0, config, } } + /// Require a snapshot before this follower takes log entries again, or + /// lift the requirement. See [`Self::snapshot_required`]. + pub fn set_snapshot_required(&mut self, required: bool) { + self.snapshot_required = required; + } + + /// Whether this follower refuses log entries until a snapshot installs. + pub fn snapshot_required(&self) -> bool { + self.snapshot_required + } + /// Restore state from persistent storage. Must be called before ticking. /// /// Seeds `volatile.last_applied` from the durable applied index so @@ -182,6 +221,11 @@ impl RaftNode { self.config.group_id } + /// The log storage this node writes through. + pub fn storage(&self) -> &S { + self.log.storage() + } + pub fn role(&self) -> NodeRole { self.role } @@ -218,6 +262,13 @@ impl RaftNode { } } + /// End the boot-time vote refusal now (for testing). A test that builds + /// nodes and votes at once calls this in place of waiting out + /// `election_timeout_max`. + pub fn expire_boot_vote_fence(&mut self) { + self.boot_vote_fence = Instant::now(); + } + /// Whether a leadership transfer is currently in progress. pub fn leadership_transfer_in_progress(&self) -> bool { self.leadership_transfer.is_some() @@ -231,6 +282,17 @@ impl RaftNode { } } + /// Commit what a quorum holds now that storage made more of this node's + /// log durable. A leader counts its own entries only once they are + /// durable (see [`crate::RaftLog::stable_index`]), so a driver whose + /// storage stages writes calls this after its disk advances. A no-op on a + /// node that does not lead. + pub fn on_storage_progress(&mut self) { + if self.leader_state.is_some() { + self.try_advance_commit_index(); + } + } + /// Take the pending `Ready` output. Caller must execute messages, /// persist hard state, and apply committed entries. pub fn take_ready(&mut self) -> Ready { @@ -264,6 +326,18 @@ impl RaftNode { self.log.snapshot_term() } + /// Term of the entry at `index`, or `None` when the log no longer holds + /// it. + pub fn log_term_at(&self, index: u64) -> Option { + self.log.term_at(index) + } + + /// The entry at `index`, committed or not, or `None` when the log does + /// not hold it. + pub fn log_entry_at(&self, index: u64) -> Option<&crate::message::LogEntry> { + self.log.entry_at(index) + } + /// Return committed log entries in the inclusive range `[lo, hi]`. /// /// Clamps `hi` to `commit_index` so callers that pass `u64::MAX` never @@ -367,6 +441,7 @@ impl RaftNode { } else { None }, + term: self.hard_state.current_term, }); } @@ -387,10 +462,10 @@ impl RaftNode { self.log.append(entry)?; self.replicate_to_all(); - // Single-voter cluster: commit immediately. Learners do not count. + // Single-voter cluster: the entry commits once it is durable here. + // Learners do not count. if self.config.cluster_size() == 1 { - self.volatile.commit_index = index; - self.collect_committed_entries(); + self.try_advance_commit_index(); } Ok(index) @@ -465,7 +540,7 @@ mod tests { } let _ = node.take_ready(); - node.log.apply_snapshot(8, 1); + node.log.apply_snapshot(8, 1).unwrap(); node.replicate_to_all(); let ready = node.take_ready(); diff --git a/nodedb-raft/src/node/durability.rs b/nodedb-raft/src/node/durability.rs index 9593c41f0..9fc90c3fa 100644 --- a/nodedb-raft/src/node/durability.rs +++ b/nodedb-raft/src/node/durability.rs @@ -10,6 +10,7 @@ //! destroys the only source that can rebuild the memory-only engines. use crate::error::{RaftError, Result}; +use crate::message::LogEntry; use crate::node::core::RaftNode; use crate::storage::LogStorage; @@ -19,8 +20,39 @@ impl RaftNode { /// This is the DELIVERY watermark: it advances as entries are handed to /// the state machine, before their effects are necessarily durable. Use /// [`Self::save_durable_applied_index`] for the durability floor. + /// + /// Monotonic: a snapshot install can raise the watermark while the caller + /// applies a batch taken earlier, so a lower `applied_to` is a no-op. pub fn advance_applied(&mut self, applied_to: u64) { - self.volatile.last_applied = applied_to; + self.volatile.last_applied = self.volatile.last_applied.max(applied_to); + } + + /// Return a committed batch taken from `Ready` that the caller did not + /// apply, so the next `take_ready` delivers it again. + /// + /// Entries at or below `last_applied` are dropped: a snapshot installed + /// since the batch was taken covers them. Entries queued after the take + /// that repeat the batch are dropped too, so each index is queued once. + pub fn requeue_committed(&mut self, batch: Vec) { + let applied = self.volatile.last_applied; + let mut merged: Vec = batch + .into_iter() + .filter(|entry| entry.index > applied) + .collect(); + let through = merged.last().map_or(applied, |entry| entry.index); + merged.extend( + self.ready + .committed_entries + .drain(..) + .filter(|entry| entry.index > through), + ); + self.ready.committed_entries = merged; + } + + /// Whether committed entries this node has not applied are gone from its + /// log. Only a snapshot can supply them. + pub fn has_apply_gap(&self) -> bool { + self.volatile.last_applied < self.log.snapshot_index() } /// Highest log index whose apply is durable on this node. @@ -88,18 +120,31 @@ impl RaftNode { /// entries whose redo record is not yet fsynced — losing the only recovery /// source for the memory-only engines. /// + /// It also clamps to `volatile.last_applied`. The data plane can make an + /// entry durable before the tick records its hand-off. Compacting past + /// `last_applied` then would discard entries the next collection still + /// reads. + /// + /// Every request is first clamped to the compaction ceiling, when one is + /// set (see [`Self::set_compaction_ceiling`]). + /// /// Returns `Ok(false)` when there is nothing to compact /// (`up_to_index <= snapshot_index`). Returns /// `Err(RaftError::LogCompacted)` if the term at `up_to_index` is no /// longer available (already compacted away). pub fn compact_log_up_to(&mut self, up_to_index: u64) -> Result { + let up_to_index = match &self.compaction_ceiling { + Some(ceiling) => up_to_index.min(ceiling.load(std::sync::atomic::Ordering::Acquire)), + None => up_to_index, + }; if up_to_index <= self.log.snapshot_index() { return Ok(false); } - if up_to_index > self.durable_applied { + let ceiling = self.durable_applied.min(self.volatile.last_applied); + if up_to_index > ceiling { return Err(RaftError::CompactionAheadOfApplied { requested: up_to_index, - last_applied: self.durable_applied, + last_applied: ceiling, }); } let term = self @@ -109,10 +154,19 @@ impl RaftNode { requested: up_to_index, first_available: self.log.snapshot_index() + 1, })?; - self.log.apply_snapshot(up_to_index, term); + self.log.apply_snapshot(up_to_index, term)?; Ok(true) } + /// Bound every compaction of this log by `ceiling`: no entry above the + /// value it holds is discarded, whoever asks. + pub fn set_compaction_ceiling( + &mut self, + ceiling: std::sync::Arc, + ) { + self.compaction_ceiling = Some(ceiling); + } + /// Check the configured auto-compaction threshold against the /// data-plane applied index `applied_index` and compact the log up to /// `applied_index` if the retained-entry count has reached the @@ -135,8 +189,11 @@ impl RaftNode { if applied_index - snapshot_index < threshold { return Ok(false); } - // Never compact past an entry whose apply is not yet durable. - let up_to = applied_index.min(self.durable_applied); + // Never compact past an entry whose apply is not yet durable, or that + // the delivery watermark has not recorded yet. + let up_to = applied_index + .min(self.durable_applied) + .min(self.volatile.last_applied); self.compact_log_up_to(up_to) } } @@ -211,6 +268,25 @@ mod tests { assert!(node.log.entries_range(1, node.last_log_index()).is_ok()); } + #[test] + fn no_compaction_passes_the_ceiling() { + let mut node = leader_with_applied_noop(test_config(1, vec![])); + let mut last = 0; + for _ in 0..4 { + last = node.propose(b"write".to_vec()).unwrap(); + apply_durably(&mut node, last); + } + let ceiling = std::sync::Arc::new(std::sync::atomic::AtomicU64::new(0)); + node.set_compaction_ceiling(std::sync::Arc::clone(&ceiling)); + assert!( + !node.compact_log_up_to(last).unwrap(), + "a zero ceiling holds every entry" + ); + ceiling.store(2, std::sync::atomic::Ordering::Release); + assert!(node.compact_log_up_to(last).unwrap()); + assert_eq!(node.first_available_index(), 3); + } + #[test] fn compact_log_up_to_rejects_ahead_of_applied() { let mut cfg = test_config(1, vec![]); @@ -248,6 +324,72 @@ mod tests { assert!(node.compact_log_up_to(idx).unwrap()); } + /// A batch taken before a snapshot install finishes after it: its lower + /// watermark must not pull `last_applied` back below the snapshot. + #[test] + fn delivery_watermark_never_regresses() { + let mut node = RaftNode::new(test_config(1, vec![]), MemStorage::new()); + node.advance_applied(10); + assert_eq!(node.last_applied(), 10); + + node.advance_applied(4); + assert_eq!(node.last_applied(), 10); + + node.advance_applied(12); + assert_eq!(node.last_applied(), 12); + } + + /// The data plane can make an entry durable before the tick records its + /// hand-off. Compaction then stops at `last_applied`, so no apply gap + /// opens. + #[test] + fn compaction_never_passes_the_delivery_watermark() { + let mut cfg = test_config(1, vec![]); + cfg.log_compaction_threshold = Some(1); + let mut node = leader_with_applied_noop(cfg); + let idx = node + .propose(b"write".to_vec()) + .expect("single voter commits"); + let _ = node.take_ready(); + node.save_durable_applied_index(idx).expect("save floor"); + + let err = node.compact_log_up_to(idx).expect_err("ahead of delivery"); + assert!(matches!(err, RaftError::CompactionAheadOfApplied { .. })); + node.maybe_compact_log(idx).expect("clamped compaction"); + assert!(node.log_snapshot_index() <= node.last_applied()); + assert!(!node.has_apply_gap()); + } + + /// A requeued batch comes back once, merged with entries queued after it + /// was taken, and without the entries a snapshot has since covered. + #[test] + fn requeued_batch_is_delivered_again_once() { + let mut node = leader_with_applied_noop(test_config(1, vec![])); + let first = node.propose(b"a".to_vec()).expect("single voter commits"); + node.propose(b"b".to_vec()).expect("single voter commits"); + let batch = node.take_ready().committed_entries; + let third = node.propose(b"c".to_vec()).expect("single voter commits"); + + node.requeue_committed(batch.clone()); + let indices: Vec = node + .take_ready() + .committed_entries + .iter() + .map(|entry| entry.index) + .collect(); + assert_eq!(indices, vec![first, first + 1, third]); + + node.advance_applied(first); + node.requeue_committed(batch); + let indices: Vec = node + .take_ready() + .committed_entries + .iter() + .map(|entry| entry.index) + .collect(); + assert_eq!(indices, vec![first + 1]); + } + /// The durable floor never moves backwards, however a caller retries. #[test] fn durable_applied_index_is_monotonic() { diff --git a/nodedb-raft/src/node/internal.rs b/nodedb-raft/src/node/internal.rs index 6af9a3551..978c54400 100644 --- a/nodedb-raft/src/node/internal.rs +++ b/nodedb-raft/src/node/internal.rs @@ -8,7 +8,7 @@ use std::time::{Duration, Instant}; use rand::RngExt; use tracing::{debug, info}; -use crate::error::RaftError; +use crate::error::{RaftError, Result}; use crate::message::{AppendEntriesRequest, LogEntry, PreVoteRequest, RequestVoteRequest}; use crate::state::{LeaderState, NodeRole, PreVoteRound}; use crate::storage::LogStorage; @@ -106,6 +106,7 @@ impl RaftNode { last_log_index: self.log.last_index(), last_log_term: self.log.last_term(), group_id: self.config.group_id, + transfer: false, }, )); } @@ -126,8 +127,18 @@ impl RaftNode { self.role = NodeRole::Follower; } } + // A vote binds for its whole term: a node that steps down within the + // term it voted in never votes in that term again. + if term > self.hard_state.current_term { + self.hard_state.voted_for = 0; + } self.hard_state.current_term = term; - self.hard_state.voted_for = 0; + // The leader of this term is unknown until it reaches this node. A + // caller that knows it (an AppendEntries or InstallSnapshot sender) + // sets it right after. So `leader_id`, when non-zero, always leads + // `current_term`: a stepped-down leader, or a follower that adopted a + // candidate's higher term, never names a leader of an older term. + self.leader_id = 0; self.leader_state = None; self.votes_received.clear(); // Any in-progress leadership transfer is moot once we step down. @@ -188,10 +199,9 @@ impl RaftNode { }; let _ = self.log.append(noop); - // Single-voter cluster: commit the no-op immediately. + // Single-voter cluster: the no-op commits once it is durable here. if self.config.cluster_size() == 1 { - self.volatile.commit_index = self.log.last_index(); - self.collect_committed_entries(); + self.try_advance_commit_index(); } self.replicate_to_all(); @@ -264,6 +274,7 @@ impl RaftNode { vec![] }; + let round = self.stamp_lease_round(); self.ready.messages.push(( peer, AppendEntriesRequest { @@ -274,6 +285,8 @@ impl RaftNode { entries, leader_commit: self.volatile.commit_index, group_id: self.config.group_id, + round, + replicated_floor: self.replicated_floor(), }, )); } @@ -363,6 +376,8 @@ impl RaftNode { entries, leader_commit: self.volatile.commit_index, group_id: self.config.group_id, + round: super::leader_lease::UNTRACKED_ROUND, + replicated_floor: self.replicated_floor(), }, )); @@ -386,6 +401,9 @@ impl RaftNode { }; let last = self.log.last_index(); + // This node acknowledges an entry once its storage holds it durably, + // as every follower does before it answers. + let self_stable = self.log.stable_index(); for n in (self.volatile.commit_index + 1..=last).rev() { let term_at_n = match self.log.term_at(n) { Some(t) => t, @@ -396,7 +414,7 @@ impl RaftNode { continue; } - let mut count = 1u64; // self counts. + let mut count = u64::from(self_stable >= n); for &peer in &self.config.peers { if leader.match_index_for(peer) >= n { count += 1; @@ -411,7 +429,20 @@ impl RaftNode { } } + /// Queue newly committed entries into `Ready`. A range the log no longer + /// holds lands in `Ready::committed_read_error` for the driver to surface. pub(super) fn collect_committed_entries(&mut self) { + if let Err(e) = self.queue_committed_entries() { + self.ready.committed_read_error = Some(e); + } + } + + /// Queue the committed range past everything applied or already queued. + /// + /// Returns [`RaftError::LogCompacted`] when that range starts below + /// `first_available_index`: the state machine lacks entries only a + /// snapshot can supply, so delivery halts rather than skip them. + fn queue_committed_entries(&mut self) -> Result<()> { // Resume from the furthest index already queued into `Ready`, not only // from `last_applied`. This runs on every commit-index advance — a // follower's AppendEntries, a single voter's propose — while the loop @@ -427,11 +458,11 @@ impl RaftNode { let from = self.volatile.last_applied.max(queued_through) + 1; let to = self.volatile.commit_index; if from > to { - return; - } - if let Ok(entries) = self.log.entries_range(from, to) { - self.ready.committed_entries.extend(entries.iter().cloned()); + return Ok(()); } + let entries = self.log.entries_range(from, to)?; + self.ready.committed_entries.extend(entries.iter().cloned()); + Ok(()) } pub(super) fn persist_hard_state(&mut self) { @@ -449,9 +480,12 @@ impl RaftNode { #[cfg(test)] mod tests { + use crate::error::RaftError; use crate::node::core::RaftNode; use crate::storage::MemStorage; - use crate::test_support::{force_election, test_config}; + use crate::test_support::{ + apply_durably, force_election, leader_with_applied_noop, test_config, + }; /// Two commit advances inside one `Ready` window queue each index once: /// the second collect resumes past what the first already queued. @@ -481,4 +515,41 @@ mod tests { "queued indices must be strictly increasing: {indices:?}" ); } + + /// A committed range below the log's first available index is a hard + /// error in `Ready`, never a silently empty delivery. + #[test] + fn compacted_committed_range_surfaces_error() { + let mut node = leader_with_applied_noop(test_config(1, vec![])); + for _ in 0..3 { + let idx = node + .propose(b"write".to_vec()) + .expect("single voter commits"); + let _ = node.take_ready(); + apply_durably(&mut node, idx); + } + let snap = node.last_applied(); + assert!( + node.compact_log_up_to(snap) + .expect("durable prefix compacts") + ); + + // Delivery watermark behind the compacted prefix. + node.volatile.last_applied = 1; + node.propose(b"next".to_vec()) + .expect("single voter commits"); + + let ready = node.take_ready(); + assert!(ready.committed_entries.is_empty()); + match ready.committed_read_error { + Some(RaftError::LogCompacted { + requested, + first_available, + }) => { + assert_eq!(requested, 2); + assert_eq!(first_available, snap + 1); + } + other => panic!("expected LogCompacted, got {other:?}"), + } + } } diff --git a/nodedb-raft/src/node/leader_lease.rs b/nodedb-raft/src/node/leader_lease.rs new file mode 100644 index 000000000..bfc7ca729 --- /dev/null +++ b/nodedb-raft/src/node/leader_lease.rs @@ -0,0 +1,570 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Leader lease: serving a linearizable read without a quorum round. +//! +//! A follower that hears the leader refuses every vote for +//! `election_timeout_min`, except a vote for a transfer campaign (see +//! [`RaftNode::vote_refusal_active`]). A node that boots refuses the same votes +//! for `election_timeout_max`: a restart forgets the leader contact, and a +//! lease taken before the restart can still be live. Once a quorum has answered a request +//! the leader sent at `t`, no successor can win before `t + +//! election_timeout_min` on the followers' clocks. The leader serves reads at +//! its own commit index until `t + RaftConfig::lease_duration()`, which keeps a +//! margin for clock-rate drift. +//! +//! The anchor is the send time of the request, never the arrival time of the +//! response. The follower opened its refusal window when it processed the +//! request, and the response can arrive much later. Check-quorum dates the +//! same answers at their arrival (see [`super::quorum_contact`]). That date +//! only keeps the leader in its role and never extends the lease. +//! +//! Each `AppendEntries` carries a round number, and the follower echoes it. The +//! leader keeps `round -> sent_at` for the rounds no quorum has covered yet, and +//! the highest round each voter has acknowledged. +//! +//! [`RaftConfig::lease_duration()`]: crate::node::config::RaftConfig::lease_duration + +use std::collections::VecDeque; +use std::time::{Duration, Instant}; + +use crate::node::core::RaftNode; +use crate::node::quorum_contact::VoterAck; +use crate::state::NodeRole; +use crate::storage::LogStorage; + +/// Round of a request that no lease tracks, such as one sent to an observer. +/// Also the acknowledged round of a voter that has not answered yet. +pub const UNTRACKED_ROUND: u64 = 0; + +/// Cap on rounds that wait for a quorum. It bounds memory while no quorum +/// answers. Check-quorum steps the leader down long before a normal cluster +/// fills it. Evicting a round only loses lease evidence and never grants any. +const MAX_PENDING_ROUNDS: usize = 4096; + +/// Leader-side lease bookkeeping for one Raft group. +#[derive(Debug)] +pub struct LeaseState { + /// Next round to stamp. Never reset, so a response to a request from an + /// earlier term can never name a round of this term. + pub(super) next_round: u64, + /// First round stamped in the current leadership term. + pub(super) term_first_round: u64, + /// `(round, sent_at)` in ascending round order, for rounds above the last + /// quorum round. + pub(super) sent: VecDeque<(u64, Instant)>, + /// Send time of the highest round a quorum has acknowledged. + pub(super) anchor: Option, + /// Set when a leadership transfer starts. A transfer campaign bypasses + /// vote refusal, and its `RequestVote` can arrive after the transfer + /// aborts, so the lease stays off until the next term. + pub(super) revoked: bool, +} + +impl LeaseState { + pub fn new() -> Self { + Self { + next_round: UNTRACKED_ROUND + 1, + term_first_round: UNTRACKED_ROUND + 1, + sent: VecDeque::new(), + anchor: None, + revoked: false, + } + } + + /// Drop the previous term's evidence when this node becomes leader. + pub(super) fn begin_term(&mut self) { + self.term_first_round = self.next_round; + self.sent.clear(); + self.anchor = None; + self.revoked = false; + } + + /// Allocate the round for a request sent at `now`. + pub(super) fn stamp(&mut self, now: Instant) -> u64 { + let round = self.next_round; + self.next_round = self.next_round.saturating_add(1); + if self.sent.len() >= MAX_PENDING_ROUNDS { + self.sent.pop_front(); + } + self.sent.push_back((round, now)); + round + } + + /// Whether `round` was stamped in the current leadership term. + pub(super) fn is_current(&self, round: u64) -> bool { + round >= self.term_first_round && round < self.next_round + } + + /// Record that a quorum acknowledged `round`, and return the anchor. + pub(super) fn settle(&mut self, round: u64) -> Option { + if let Ok(pos) = self.sent.binary_search_by_key(&round, |&(r, _)| r) { + let sent_at = self.sent[pos].1; + if self.anchor.is_none_or(|anchor| sent_at > anchor) { + self.anchor = Some(sent_at); + } + self.sent.drain(..=pos); + } + self.anchor + } + + pub(super) fn revoke(&mut self) { + self.revoked = true; + } + + /// Forget the anchor. A voter-set change calls this, so the next anchor + /// comes from a quorum of the new set. + pub(super) fn clear_anchor(&mut self) { + self.anchor = None; + } + + /// When the lease ends, or `None` when there is no lease. + pub(super) fn expires_at(&self, duration: Duration) -> Option { + if self.revoked { + return None; + } + self.anchor.map(|anchor| anchor + duration) + } + + /// Send time of pending `round` (for testing). + #[cfg(test)] + pub(super) fn sent_at(&self, round: u64) -> Option { + self.sent + .iter() + .find(|&&(r, _)| r == round) + .map(|&(_, at)| at) + } + + /// Replace the send time of pending `round` (for testing). + #[cfg(test)] + pub(super) fn set_sent_at(&mut self, round: u64, at: Instant) { + if let Some(entry) = self.sent.iter_mut().find(|(r, _)| *r == round) { + entry.1 = at; + } + } +} + +impl Default for LeaseState { + fn default() -> Self { + Self::new() + } +} + +impl RaftNode { + /// Allocate the lease round for an `AppendEntries` sent now. The send is + /// the one place the lease reads the clock itself: the round's anchor is + /// the moment it leaves. Everything downstream takes `now` from its + /// caller. + pub(super) fn stamp_lease_round(&mut self) -> u64 { + self.lease.stamp(Instant::now()) + } + + /// Record that voter `peer` answered `round`, in an answer that arrived at + /// `now`. Rounds from an earlier term are ignored: the follower answered + /// a request of a leadership that no longer exists. + pub(super) fn record_lease_ack(&mut self, peer: u64, round: u64, now: Instant) { + if !self.lease.is_current(round) { + return; + } + match self.quorum_window.iter_mut().find(|ack| ack.peer == peer) { + Some(ack) => { + ack.round = ack.round.max(round); + ack.arrived = ack.arrived.max(now); + } + None => self.quorum_window.push(VoterAck { + peer, + round, + arrived: now, + }), + } + } + + /// Highest round `peer` has acknowledged in this term. + fn acked_round(&self, peer: u64) -> u64 { + self.voter_ack(peer) + .map_or(UNTRACKED_ROUND, |ack| ack.round) + } + + /// Settle the highest round a quorum of voters has acknowledged, and + /// return the lease anchor. `None` for a single voter, which has no + /// round to settle. + pub(super) fn settle_lease(&mut self) -> Option { + // This node acknowledges every round it sends, so a quorum needs + // `quorum - 1` peers. + let peers_needed = self.config.quorum().saturating_sub(1); + if peers_needed == 0 { + return None; + } + let mut acked: Vec = self + .config + .peers + .iter() + .map(|&peer| self.acked_round(peer)) + .collect(); + acked.sort_unstable_by(|a, b| b.cmp(a)); + let quorum_round = *acked.get(peers_needed - 1)?; + if quorum_round == UNTRACKED_ROUND { + return self.lease.anchor; + } + self.lease.settle(quorum_round) + } + + /// Whether this node can serve a linearizable read at its commit index + /// without a quorum round. + /// + /// Needs all of: + /// - leader role; + /// - no leadership transfer in progress or begun this term; + /// - the current-term no-op committed, so the commit index covers every + /// earlier leader's commits; + /// - `now` before the anchor plus `lease_duration()`. + pub fn lease_valid(&self, now: Instant) -> bool { + if self.role != NodeRole::Leader + || self.leadership_transfer.is_some() + || !self.current_term_committed() + { + return false; + } + // A single voter is the whole quorum. No other node can win a vote. + if self.config.peers.is_empty() { + return !self.lease.revoked; + } + self.lease + .expires_at(self.config.lease_duration()) + .is_some_and(|end| now < end) + } + + /// The commit index, when the lease lets a read be served there now. + pub fn lease_read_index(&self, now: Instant) -> Option { + self.lease_valid(now).then_some(self.volatile.commit_index) + } + + /// Whether this node refuses pre-votes and every vote except a transfer + /// vote at `now`. + /// + /// The follower half of the lease. It holds while a leader reached this + /// node within `election_timeout_min`, and until the boot fence passes. + pub(crate) fn vote_refusal_active(&self, now: Instant) -> bool { + if now < self.boot_vote_fence { + return true; + } + self.leader_contact.is_some_and(|contact| { + now.saturating_duration_since(contact.at) < self.config.election_timeout_min + }) + } +} + +#[cfg(test)] +mod tests { + use std::time::{Duration, Instant}; + + use super::{LeaseState, MAX_PENDING_ROUNDS}; + use crate::message::{ + AppendEntriesRequest, AppendEntriesResponse, RequestVoteRequest, RequestVoteResponse, + }; + use crate::node::core::RaftNode; + use crate::state::{HardState, NodeRole}; + use crate::storage::{LogStorage, MemStorage}; + use crate::test_support::{force_election, test_config}; + + const ONE_TICK: Duration = Duration::from_nanos(1); + + /// Node 1 leading peers 2 and 3, with its election output drained. + fn leader() -> RaftNode { + let mut node = RaftNode::new(test_config(1, vec![2, 3]), MemStorage::new()); + force_election(&mut node); + let _ = node.take_ready(); + node.handle_request_vote_response( + 2, + &RequestVoteResponse { + term: 1, + vote_granted: true, + }, + ); + assert_eq!(node.role(), NodeRole::Leader); + let _ = node.take_ready(); + node + } + + /// Send one `AppendEntries` round and return the round sent to peer 2. + fn send_round(node: &mut RaftNode) -> u64 { + node.replicate_to_all(); + let ready = node.take_ready(); + ready + .messages + .iter() + .find(|(peer, _)| *peer == 2) + .map(|(_, req)| req.round) + .expect("the leader sends to peer 2") + } + + fn answer(node: &mut RaftNode, peer: u64, round: u64, last_log_index: u64) { + node.handle_append_entries_response( + peer, + &AppendEntriesResponse { + term: node.current_term(), + success: true, + last_log_index, + round, + needs_snapshot: false, + }, + ); + } + + /// A leader whose no-op is committed, with its lease anchored at the send + /// time of one round. Returns the node and that send time. + fn leased() -> (RaftNode, Instant) { + let mut node = leader(); + let round = send_round(&mut node); + let sent = node.lease.sent_at(round).expect("the round is pending"); + let noop = node.last_log_index(); + answer(&mut node, 2, round, noop); + assert_eq!(node.commit_index(), noop, "the answer commits the no-op"); + (node, sent) + } + + fn lease_duration(node: &RaftNode) -> Duration { + node.config.election_timeout_min - node.config.lease_drift_margin() + } + + #[test] + fn the_lease_is_anchored_at_send_time() { + let mut node = leader(); + let round = send_round(&mut node); + let sent = node.lease.sent_at(round).expect("the round is pending"); + // The request left well before its answer is handled below. + let sent = sent - Duration::from_millis(100); + node.lease.set_sent_at(round, sent); + let last = node.last_log_index(); + answer(&mut node, 2, round, last); + + assert_eq!(node.lease.anchor, Some(sent)); + assert!(node.lease_valid(sent)); + assert!(node.lease_valid(sent + lease_duration(&node) - ONE_TICK)); + } + + #[test] + fn the_lease_expires_at_min_minus_the_drift_margin() { + let (node, sent) = leased(); + let duration = lease_duration(&node); + assert!(node.lease_valid(sent + duration - ONE_TICK)); + assert!(!node.lease_valid(sent + duration)); + assert!(!node.lease_valid(sent + node.config.election_timeout_min)); + } + + #[test] + fn a_late_answer_does_not_move_the_anchor() { + let mut node = leader(); + let first = send_round(&mut node); + let first_sent = node.lease.sent_at(first).expect("the round is pending"); + let second = send_round(&mut node); + let second_sent = first_sent + Duration::from_millis(40); + node.lease.set_sent_at(second, second_sent); + + // The answer to the later round arrives, then a late answer to the + // earlier one. Neither moves the anchor past the later send time. + let last = node.last_log_index(); + answer(&mut node, 2, second, last); + answer(&mut node, 2, first, last); + assert_eq!(node.lease.anchor, Some(second_sent)); + } + + #[test] + fn there_is_no_lease_before_the_noop_commits() { + let mut node = leader(); + let round = send_round(&mut node); + let sent = node.lease.sent_at(round).expect("the round is pending"); + // Peer 2 answers but has not stored the no-op yet. + answer(&mut node, 2, round, 0); + assert_eq!(node.lease.anchor, Some(sent), "a quorum answered the round"); + assert!( + !node.lease_valid(sent), + "the commit index may predate an earlier leader's commits" + ); + assert_eq!(node.lease_read_index(sent), None); + + let round = send_round(&mut node); + let sent = node.lease.sent_at(round).expect("the round is pending"); + let last = node.last_log_index(); + answer(&mut node, 2, round, last); + assert_eq!(node.lease_read_index(sent), Some(node.commit_index())); + } + + #[test] + fn a_transfer_revokes_the_lease_for_the_term() { + let (mut node, _) = leased(); + node.transfer_leadership(2).expect("peer 2 is a voter"); + let round = send_round(&mut node); + let sent = node.lease.sent_at(round).expect("the round is pending"); + let last = node.last_log_index(); + answer(&mut node, 2, round, last); + assert!(!node.lease_valid(sent)); + + // The transfer aborts. Its campaign can still be in flight. + node.transfer_deadline_override(Instant::now() - Duration::from_millis(1)); + node.tick(); + assert!(!node.leadership_transfer_in_progress()); + let round = send_round(&mut node); + let sent = node.lease.sent_at(round).expect("the round is pending"); + let last = node.last_log_index(); + answer(&mut node, 2, round, last); + assert!(!node.lease_valid(sent)); + } + + #[test] + fn an_answer_to_an_earlier_term_does_not_count() { + let mut node = leader(); + let stale = send_round(&mut node); + // Lose and regain leadership: a later term. + node.handle_append_entries_response( + 3, + &AppendEntriesResponse { + term: node.current_term() + 1, + success: false, + last_log_index: 0, + round: stale, + needs_snapshot: false, + }, + ); + assert_eq!(node.role(), NodeRole::Follower); + force_election(&mut node); + let _ = node.take_ready(); + node.handle_request_vote_response( + 2, + &RequestVoteResponse { + term: node.current_term(), + vote_granted: true, + }, + ); + assert_eq!(node.role(), NodeRole::Leader); + let _ = node.take_ready(); + + let last = node.last_log_index(); + answer(&mut node, 2, stale, last); + assert!(node.lease.anchor.is_none()); + } + + #[test] + fn pending_rounds_are_capped() { + let mut lease = LeaseState::new(); + let now = Instant::now(); + for _ in 0..MAX_PENDING_ROUNDS + 10 { + lease.stamp(now); + } + assert_eq!(lease.sent.len(), MAX_PENDING_ROUNDS); + // An evicted round settles nothing. + assert_eq!(lease.settle(1), None); + } + + /// Node 2 following leader 1 at term 1, with its boot fence expired so + /// only leader contact decides. + fn follower() -> RaftNode { + let mut node = RaftNode::new(test_config(2, vec![1, 3]), MemStorage::new()); + node.expire_boot_vote_fence(); + node.handle_append_entries(&AppendEntriesRequest { + term: 1, + leader_id: 1, + prev_log_index: 0, + prev_log_term: 0, + entries: vec![], + leader_commit: 0, + group_id: 1, + round: 1, + replicated_floor: 0, + }); + node + } + + fn vote_request(transfer: bool) -> RequestVoteRequest { + RequestVoteRequest { + term: 2, + candidate_id: 3, + last_log_index: 0, + last_log_term: 0, + group_id: 1, + transfer, + } + } + + /// When leader contact stops blocking votes on `node`. + fn window_end(node: &RaftNode) -> Instant { + let contact = node.leader_contact.expect("the leader reached this node"); + (contact.at + node.config.election_timeout_min).max(node.boot_vote_fence) + } + + #[test] + fn a_vote_is_refused_inside_the_window() { + let mut node = follower(); + let inside = window_end(&node) - ONE_TICK; + let resp = node.handle_request_vote_at(&vote_request(false), inside); + assert!(!resp.vote_granted); + assert_eq!(node.current_term(), 1, "the term is not adopted"); + assert_eq!(resp.term, 1); + assert_eq!(node.leader_id(), 1); + } + + #[test] + fn a_vote_is_granted_once_the_window_closes() { + let mut node = follower(); + let closed = window_end(&node); + let resp = node.handle_request_vote_at(&vote_request(false), closed); + assert!(resp.vote_granted); + assert_eq!(node.current_term(), 2); + } + + #[test] + fn a_transfer_vote_bypasses_the_window() { + let mut node = follower(); + let inside = window_end(&node) - ONE_TICK; + let resp = node.handle_request_vote_at(&vote_request(true), inside); + assert!(resp.vote_granted); + assert_eq!(node.current_term(), 2); + } + + /// A restart forgets the leader contact. The boot fence keeps the node + /// from voting while a lease the old leader took can still be live. + #[test] + fn a_restarted_follower_refuses_a_vote_until_its_boot_fence_passes() { + let mut storage = MemStorage::new(); + storage + .save_hard_state(&HardState { + current_term: 1, + voted_for: 0, + }) + .expect("save hard state"); + let mut node = RaftNode::new(test_config(2, vec![1, 3]), storage); + node.restore().expect("restore"); + assert!( + node.leader_contact.is_none(), + "a restart has no leader contact" + ); + let fence = node.boot_vote_fence; + + let refused = node.handle_request_vote_at(&vote_request(false), fence - ONE_TICK); + assert!(!refused.vote_granted); + assert_eq!(node.current_term(), 1, "the term is not adopted"); + + let granted = node.handle_request_vote_at(&vote_request(false), fence); + assert!(granted.vote_granted); + assert_eq!(node.current_term(), 2); + } + + #[test] + fn a_transfer_vote_bypasses_the_boot_fence() { + let mut node = RaftNode::new(test_config(2, vec![1, 3]), MemStorage::new()); + let inside = node.boot_vote_fence - ONE_TICK; + let resp = node.handle_request_vote_at(&vote_request(true), inside); + assert!(resp.vote_granted); + } + + #[test] + fn a_single_voter_holds_the_lease_alone() { + let mut node = RaftNode::new(test_config(1, vec![]), MemStorage::new()); + node.election_deadline_override(Instant::now() - Duration::from_millis(1)); + node.tick(); + assert_eq!(node.role(), NodeRole::Leader); + assert_eq!( + node.lease_read_index(Instant::now()), + Some(node.commit_index()) + ); + } +} diff --git a/nodedb-raft/src/node/membership.rs b/nodedb-raft/src/node/membership.rs index e64c42564..4626acbca 100644 --- a/nodedb-raft/src/node/membership.rs +++ b/nodedb-raft/src/node/membership.rs @@ -54,6 +54,8 @@ impl RaftNode { } self.config.peers = new_voters; + // The next lease anchor needs a quorum of the new voter set. + self.lease.clear_anchor(); } /// Add a single voter peer to this group. @@ -86,10 +88,15 @@ impl RaftNode { /// forever: it never applies the change, and every view it derives from /// that log stays stale for a group it has already left. /// - /// Call this immediately BEFORE [`Self::remove_peer`], while `peer` is - /// still tracked. No-op unless this node leads and still tracks `peer`. + /// A learner is told the same way: a learner that never applies its own + /// removal keeps listing itself as a replica of the group. + /// + /// Call this immediately BEFORE [`Self::remove_peer`] or + /// [`Self::remove_learner`], while `peer` is still tracked. No-op unless + /// this node leads and still tracks `peer` as a voter or a learner. pub fn notify_removed_peer(&mut self, peer: u64) { - if self.role != NodeRole::Leader || !self.config.peers.contains(&peer) { + let tracked = self.config.peers.contains(&peer) || self.config.learners.contains(&peer); + if self.role != NodeRole::Leader || !tracked { return; } self.send_append_entries(peer); @@ -231,6 +238,7 @@ impl RaftNode { } else { None }, + term: self.hard_state.current_term, }); } @@ -249,6 +257,9 @@ impl RaftNode { emitted: false, deadline, }); + // The target's campaign bypasses vote refusal, so no lease holds from + // here to the end of this term. + self.lease.revoke(); // Emit immediately if the target is already at the frontier; otherwise // the ack hook retries once it catches up. @@ -487,6 +498,30 @@ mod tests { assert!(!node.voters().contains(&2), "peer is removed afterwards"); } + /// A removed learner gets the same final message as a removed voter. + #[test] + fn a_removed_learner_is_told_before_replication_stops() { + let mut node = elect_with_voters(vec![2, 3]); + node.add_learner(4); + node.volatile.commit_index = node.log.last_index(); + + node.notify_removed_peer(4); + node.remove_learner(4); + + let ready = node.take_ready(); + let to_departing: Vec<_> = ready.messages.iter().filter(|(p, _)| *p == 4).collect(); + assert_eq!( + to_departing.len(), + 1, + "the departing learner must get exactly one final AppendEntries" + ); + assert_eq!(to_departing[0].1.leader_commit, node.commit_index()); + assert!( + !node.learners().contains(&4), + "learner is removed afterwards" + ); + } + /// Ordering matters: once the peer is gone from the configuration there is /// no replication state to send from, so a late call must do nothing /// rather than resurrect it. diff --git a/nodedb-raft/src/node/mod.rs b/nodedb-raft/src/node/mod.rs index 50f6e4fa8..75c0ac82b 100644 --- a/nodedb-raft/src/node/mod.rs +++ b/nodedb-raft/src/node/mod.rs @@ -9,6 +9,8 @@ //! - [`durability`]: Applied-index durability floor and log compaction. //! - [`internal`]: Internal state transitions (elections, replication, //! commit advancement) and timeout math. +//! - [`leader_lease`]: Serving a linearizable read from the leader's commit +//! index without a quorum round, and the vote refusal that makes it safe. //! - [`membership`]: Dynamic configuration changes — add/remove voters, //! add/remove/promote learners. //! - [`quorum_contact`]: Check-quorum — tracking when a majority last @@ -25,7 +27,9 @@ pub mod config; pub mod core; pub mod durability; mod internal; +pub mod leader_lease; pub mod membership; +pub mod peer_contact; pub mod quorum_contact; pub mod read_index; pub mod rpc; diff --git a/nodedb-raft/src/node/peer_contact.rs b/nodedb-raft/src/node/peer_contact.rs new file mode 100644 index 000000000..61d986374 --- /dev/null +++ b/nodedb-raft/src/node/peer_contact.rs @@ -0,0 +1,221 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Read-only views of contact with peers: the leader's per-peer response +//! counts, and whether the leader this node names is live. + +use std::time::Instant; + +use crate::node::core::RaftNode; +use crate::state::NodeRole; +use crate::storage::LogStorage; + +impl RaftNode { + /// `AppendEntries` responses received from `peer` in the current term, + /// rejections included. `None` when this node is not the leader. + pub fn peer_ack_count(&self, peer: u64) -> Option { + self.leader_state + .as_ref() + .map(|leader| leader.ack_count_for(peer)) + } + + /// The highest log index every voter's log is known to hold. A log + /// compacted at or below it never sends a voter a snapshot, under this + /// leader or a later one: every voter holds the compacted entries. + /// + /// The leader takes the lowest `match_index` over the other voters, + /// capped at its commit index: a committed entry stays in every log that + /// holds it. It sends the value with each `AppendEntries`. Every replica + /// keeps the highest value it saw, so the floor never moves back. A voter + /// that stops answering holds it where it is. Learners and observers do + /// not count: a new replica takes a snapshot. + pub fn replicated_floor(&self) -> u64 { + let Some(leader) = self.leader_state.as_ref() else { + return self.replicated_floor; + }; + let lowest_voter = self + .config + .peers + .iter() + .map(|&peer| leader.match_index_for(peer)) + .min() + .unwrap_or(u64::MAX); + self.replicated_floor + .max(lowest_voter.min(self.volatile.commit_index)) + } + + /// Whether `peer`'s latest response to this leader asked for a snapshot. + /// Such a peer holds no state to lead from. `false` off the leader. + pub fn peer_awaits_snapshot(&self, peer: u64) -> bool { + self.leader_state + .as_ref() + .is_some_and(|leader| leader.awaiting_snapshot.contains(&peer)) + } + + /// The leader this node names at `now`, when it has proof the leader is + /// live: this node leads, or the leader reached it within + /// `election_timeout_min`. `0` otherwise. + /// + /// A follower keeps naming a crashed leader until its election timeout + /// fires. This view drops that leader as soon as its contact goes stale. + pub fn live_leader(&self, now: Instant) -> u64 { + if self.role == NodeRole::Leader { + return self.config.node_id; + } + if self.leader_id == 0 { + return 0; + } + let fresh = self.leader_contact.is_some_and(|contact| { + now.saturating_duration_since(contact.at) < self.config.election_timeout_min + }); + if fresh { self.leader_id } else { 0 } + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use crate::message::{AppendEntriesRequest, RequestVoteResponse}; + use crate::node::core::RaftNode; + use crate::state::NodeRole; + use crate::storage::MemStorage; + use crate::test_support::{force_election, test_config}; + + fn heartbeat(term: u64, leader_id: u64) -> AppendEntriesRequest { + AppendEntriesRequest { + term, + leader_id, + prev_log_index: 0, + prev_log_term: 0, + entries: vec![], + leader_commit: 0, + group_id: 1, + round: 1, + replicated_floor: 0, + } + } + + /// A leadership transfer moves the term before the new leader reaches + /// this node. In between, this node names no leader: never the old + /// leader at the new term, which a routing hint at that term would keep. + #[test] + fn a_transfer_campaign_leaves_no_leader_named_at_its_term_until_the_winner_reaches_this_node() { + let mut node = RaftNode::new(test_config(2, vec![1, 3]), MemStorage::new()); + let _ = node.handle_append_entries(&heartbeat(1, 1)); + let now = std::time::Instant::now(); + assert_eq!((node.live_leader(now), node.current_term()), (1, 1)); + + // Node 3's transfer campaign at term 2. + let vote = node.handle_request_vote(&crate::message::RequestVoteRequest { + term: 2, + candidate_id: 3, + last_log_index: 0, + last_log_term: 0, + group_id: 1, + transfer: true, + }); + assert!(vote.vote_granted); + assert_eq!(node.current_term(), 2); + assert_eq!(node.leader_id(), 0, "the old leader does not lead term 2"); + assert_eq!(node.live_leader(std::time::Instant::now()), 0); + + let _ = node.handle_append_entries(&heartbeat(2, 3)); + let now = std::time::Instant::now(); + assert_eq!((node.live_leader(now), node.current_term()), (3, 2)); + } + + /// A leader that steps down for lost quorum names no leader. + #[test] + fn a_leader_that_steps_down_names_no_leader() { + let mut node = RaftNode::new(test_config(1, vec![2, 3]), MemStorage::new()); + force_election(&mut node); + let _ = node.take_ready(); + node.handle_request_vote_response( + 2, + &RequestVoteResponse { + term: 1, + vote_granted: true, + }, + ); + assert_eq!(node.leader_id(), 1); + node.become_follower(node.current_term()); + assert_eq!(node.leader_id(), 0); + assert_eq!(node.live_leader(std::time::Instant::now()), 0); + } + + /// The leader's floor is the lowest voter's committed match. A voter + /// that has not answered holds it at 0. + #[test] + fn the_replicated_floor_waits_for_every_voter() { + let mut node = RaftNode::new(test_config(1, vec![2, 3]), MemStorage::new()); + force_election(&mut node); + let _ = node.take_ready(); + node.handle_request_vote_response( + 2, + &RequestVoteResponse { + term: 1, + vote_granted: true, + }, + ); + assert_eq!(node.role(), NodeRole::Leader); + let ack = crate::message::AppendEntriesResponse { + term: 1, + success: true, + last_log_index: 1, + round: 1, + needs_snapshot: false, + }; + node.handle_append_entries_response(2, &ack); + assert_eq!(node.replicated_floor(), 0, "voter 3 holds nothing yet"); + node.handle_append_entries_response(3, &ack); + assert_eq!(node.commit_index(), 1); + assert_eq!(node.replicated_floor(), 1); + } + + /// A follower keeps the highest floor a leader sent it. + #[test] + fn a_follower_keeps_the_highest_replicated_floor() { + let mut node = RaftNode::new(test_config(2, vec![1, 3]), MemStorage::new()); + let heartbeat = |replicated_floor| AppendEntriesRequest { + term: 1, + leader_id: 1, + prev_log_index: 0, + prev_log_term: 0, + entries: vec![], + leader_commit: 0, + group_id: 1, + round: 1, + replicated_floor, + }; + let _ = node.handle_append_entries(&heartbeat(5)); + assert_eq!(node.replicated_floor(), 5); + let _ = node.handle_append_entries(&heartbeat(3)); + assert_eq!(node.replicated_floor(), 5, "the floor never moves back"); + } + + #[test] + fn a_follower_names_the_leader_only_while_its_contact_is_fresh() { + let mut node = RaftNode::new(test_config(2, vec![1, 3]), MemStorage::new()); + assert_eq!(node.live_leader(std::time::Instant::now()), 0); + let _ = node.handle_append_entries(&AppendEntriesRequest { + term: 1, + leader_id: 1, + prev_log_index: 0, + prev_log_term: 0, + entries: vec![], + leader_commit: 0, + group_id: 1, + round: 1, + replicated_floor: 0, + }); + let now = std::time::Instant::now(); + assert_eq!(node.live_leader(now), 1); + let stale = now + node.config.election_timeout_min + Duration::from_millis(1); + assert_eq!(node.live_leader(stale), 0); + assert_eq!( + node.leader_id(), + 1, + "the Raft leader id itself is unchanged" + ); + } +} diff --git a/nodedb-raft/src/node/quorum_contact.rs b/nodedb-raft/src/node/quorum_contact.rs index ef3676678..2b44ac90d 100644 --- a/nodedb-raft/src/node/quorum_contact.rs +++ b/nodedb-raft/src/node/quorum_contact.rs @@ -14,9 +14,9 @@ //! //! Contact is measured from `AppendEntries` **responses**, not from replication //! progress. Any response proves the peer is reachable and still recognises -//! this term, which is the whole question check-quorum asks; whether the peer's -//! log has caught up is a different one. Counting `match_index` instead would -//! break in both directions a healthy cluster routinely hits: +//! this term, which is the whole question check-quorum asks. Whether the +//! peer's log has caught up is a different question. Counting `match_index` +//! instead breaks in two cases a healthy cluster routinely hits: //! //! - A follower rebuilding from a snapshot, or backtracking after a log //! conflict, answers every heartbeat while its `match_index` sits far behind. @@ -24,11 +24,22 @@ //! still in flight, so even fully healthy followers trail it. //! //! In both cases a `match_index` test sees no contact and deposes a leader -//! whose quorum is intact. Responses are therefore counted through -//! [`LeaderState::ack_count_for`], the same monotonic per-peer counter -//! [`super::read_index`] uses, bumped on success and rejection alike. +//! whose quorum is intact. //! -//! [`LeaderState::ack_count_for`]: crate::state::LeaderState::ack_count_for +//! # When contact happened +//! +//! Contact is dated at the **arrival time** of the answers. Check-quorum is a +//! liveness check: it asks whether a quorum still answers this leader. A slow +//! but steady round trip answers it, so queueing and disk waits on the way +//! never depose a leader whose quorum keeps answering. +//! +//! The leader lease is a safety check and keeps the **send time** of the +//! request a quorum answered (see [`super::leader_lease`]). The two clocks are +//! split on purpose. Check-quorum only decides when the leader steps down. It +//! never lets the leader serve a read, so a later date grants no lease. +//! +//! Only answers to rounds stamped in the current leadership term count. An +//! answer to a request of an earlier term is never contact. use std::time::Instant; @@ -36,63 +47,76 @@ use crate::node::core::RaftNode; use crate::state::NodeRole; use crate::storage::LogStorage; +/// What the leader knows of one voter's answers in the current term. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct VoterAck { + /// The voter. + pub(crate) peer: u64, + /// Highest lease round the voter answered. + pub(crate) round: u64, + /// When the latest answer arrived. + pub(crate) arrived: Instant, +} + impl RaftNode { - /// Start a fresh contact window at `now`, treating this instant as proven - /// contact. + /// Open the contact record of a new leadership term at `now`. /// - /// Called on winning an election — a quorum of voters just granted their - /// votes, which is contact by definition — and again each time a new - /// quorum is observed. + /// Called on winning an election. A quorum of voters just granted their + /// votes, which is contact by definition. The lease starts empty: a vote + /// grant does not make a voter refuse other candidates. pub(super) fn arm_quorum_window(&mut self, now: Instant) { self.last_quorum_contact = Some(now); - self.quorum_window = self - .leader_state - .as_ref() - .map(|ls| { - self.config - .peers - .iter() - .map(|&id| (id, ls.ack_count_for(id))) - .collect() - }) - .unwrap_or_default(); + self.quorum_window.clear(); + self.lease.begin_term(); } - /// Re-arm the contact window if a quorum of voters has answered since it - /// opened. No-op for any role but leader. + /// Move contact up to the latest instant by which a quorum of voters + /// answered this term. No-op for any role but leader. /// - /// Called from the `AppendEntries` response path so contact tracks the - /// responses themselves, and from [`RaftNode::tick`] so a single-voter - /// group — which has no peers to answer and so never reaches the response - /// path — still refreshes on its own quorum of one. + /// Called from the `AppendEntries` response path, and from + /// [`RaftNode::tick`] so a single-voter group renews on its own quorum of + /// one. It has no peers to answer and never reaches the response path. pub(super) fn refresh_quorum_contact(&mut self, now: Instant) { if self.role != NodeRole::Leader { return; } - let Some(leader) = self.leader_state.as_ref() else { + // This node answers itself at once, so a quorum needs `quorum - 1` + // peers. + let peers_needed = self.config.quorum().saturating_sub(1); + if peers_needed == 0 { + self.last_quorum_contact = Some(now); return; - }; - // Self is always in contact with itself. - let mut count = 1usize; - for &peer in &self.config.peers { - let baseline = self - .quorum_window - .iter() - .find(|&&(id, _)| id == peer) - .map(|&(_, seen)| seen) - .unwrap_or(0); - if leader.ack_count_for(peer) > baseline { - count += 1; - } } - if count >= self.config.quorum() { - self.arm_quorum_window(now); + let mut arrivals: Vec = self + .config + .peers + .iter() + .filter_map(|&peer| self.voter_ack(peer).map(|ack| ack.arrived)) + .collect(); + arrivals.sort_unstable_by(|a, b| b.cmp(a)); + let Some(&contact) = arrivals.get(peers_needed - 1) else { + return; + }; + if self.last_quorum_contact.is_none_or(|last| contact > last) { + self.last_quorum_contact = Some(contact); } } - /// Push the contact window back to `at` (for testing). + /// What the leader knows of `peer`'s answers in this term. + pub(super) fn voter_ack(&self, peer: u64) -> Option { + self.quorum_window + .iter() + .find(|ack| ack.peer == peer) + .copied() + } + + /// Push the last quorum contact, and every recorded arrival, back to `at` + /// (for testing). The lease anchor is a separate clock and stays. pub fn quorum_contact_at_override(&mut self, at: Instant) { self.last_quorum_contact = Some(at); + for ack in &mut self.quorum_window { + ack.arrived = ack.arrived.min(at); + } } /// Whether the leader has gone an entire election timeout without a @@ -154,6 +178,25 @@ mod tests { node.quorum_contact_at_override(Instant::now() - Duration::from_secs(1)); } + /// The round of the latest `AppendEntries` the leader sent. + fn latest_round(node: &RaftNode) -> u64 { + node.lease.next_round - 1 + } + + /// Peer `peer` answers `round` with success at the leader's last index. + fn answer(node: &mut RaftNode, peer: u64, round: u64) { + node.handle_append_entries_response( + peer, + &AppendEntriesResponse { + term: node.current_term(), + success: true, + last_log_index: node.last_log_index(), + round, + needs_snapshot: false, + }, + ); + } + /// A rejection is contact. A follower backtracking through a log conflict /// answers every round while its `match_index` stays put; deposing that /// leader would be a false positive. @@ -168,6 +211,8 @@ mod tests { term: node.current_term(), success: false, last_log_index: 0, + round: latest_round(&node), + needs_snapshot: false, }, ); @@ -197,6 +242,8 @@ mod tests { term: node.current_term(), success: true, last_log_index: 1, + round: latest_round(&node), + needs_snapshot: false, }, ); @@ -234,6 +281,8 @@ mod tests { term: node.current_term(), success: true, last_log_index: node.last_log_index(), + round: latest_round(&node), + needs_snapshot: false, }, ); @@ -291,6 +340,8 @@ mod tests { }], leader_commit: 1, group_id: 1, + round: 1, + replicated_floor: 0, }); assert!(resp.success); @@ -314,6 +365,8 @@ mod tests { term: node.current_term(), success: true, last_log_index: node.last_log_index(), + round: latest_round(&node), + needs_snapshot: false, }, ); node.tick(); @@ -329,6 +382,104 @@ mod tests { ); } + /// Contact is dated when the answer arrives. The lease keeps the send + /// time of the answered request. + #[test] + fn contact_is_dated_at_arrival_time() { + let mut node = leader(vec![2, 3]); + go_silent(&mut node); + let round = latest_round(&node); + let silent_since = node.last_quorum_contact.expect("leader has contact"); + // The answered request left before the silence began. + let sent = silent_since - Duration::from_millis(100); + node.lease.set_sent_at(round, sent); + + let before = Instant::now(); + answer(&mut node, 2, round); + + let contact = node.last_quorum_contact.expect("leader has contact"); + assert!( + contact >= before, + "contact must be the arrival time, not the send time" + ); + assert_eq!( + node.lease.anchor, + Some(sent), + "the lease keeps the send time" + ); + } + + /// A quorum that answers every round, each answer arriving long after its + /// request left, keeps the leader. The lease stays anchored at the send + /// times and gives no reads. + #[test] + fn a_steady_slow_quorum_keeps_the_leader() { + let mut node = leader(vec![2, 3]); + let round_trip = node.config.election_timeout_max + Duration::from_millis(200); + for _ in 0..4 { + node.replicate_to_all(); + let _ = node.take_ready(); + let round = latest_round(&node); + let sent = Instant::now() - round_trip; + node.lease.set_sent_at(round, sent); + // Everything before this round is older than the step-down bound. + go_silent(&mut node); + + answer(&mut node, 2, round); + node.tick(); + + assert_eq!( + node.role(), + NodeRole::Leader, + "a quorum that keeps answering must keep the leader" + ); + assert_eq!(node.lease.anchor, Some(sent)); + assert!( + !node.lease_valid(Instant::now()), + "a send time older than the lease gives no lease" + ); + } + } + + /// An answer to a round of an earlier leadership term is no contact, even + /// when it carries the current term. + #[test] + fn an_answer_to_an_earlier_term_is_not_contact() { + let mut node = leader(vec![2, 3]); + let stale = latest_round(&node); + node.handle_append_entries_response( + 3, + &AppendEntriesResponse { + term: node.current_term() + 1, + success: false, + last_log_index: 0, + round: stale, + needs_snapshot: false, + }, + ); + assert_eq!(node.role(), NodeRole::Follower); + force_election(&mut node); + let _ = node.take_ready(); + node.handle_request_vote_response( + 2, + &RequestVoteResponse { + term: node.current_term(), + vote_granted: true, + }, + ); + assert_eq!(node.role(), NodeRole::Leader); + let _ = node.take_ready(); + go_silent(&mut node); + + answer(&mut node, 2, stale); + node.tick(); + assert_eq!( + node.role(), + NodeRole::Follower, + "an answer to an earlier term must not renew contact" + ); + } + /// A follower never runs the leader-side check. #[test] fn a_follower_is_never_deposed_by_check_quorum() { @@ -354,6 +505,8 @@ mod tests { term: node.current_term(), success: true, last_log_index: node.last_log_index(), + round: latest_round(&node), + needs_snapshot: false, }, ); @@ -370,10 +523,9 @@ mod tests { #[test] fn contact_is_not_lost_before_the_upper_election_bound() { let node = leader(vec![2, 3]); - let cfg_max = Duration::from_millis(300); - let just_inside = std::time::Instant::now() + cfg_max - Duration::from_millis(10); - assert!(!node.quorum_contact_lost(just_inside)); - let past = std::time::Instant::now() + cfg_max + Duration::from_millis(10); - assert!(node.quorum_contact_lost(past)); + let contact = node.last_quorum_contact.expect("leader has contact"); + let cfg_max = node.config.election_timeout_max; + assert!(!node.quorum_contact_lost(contact + cfg_max - Duration::from_nanos(1))); + assert!(node.quorum_contact_lost(contact + cfg_max)); } } diff --git a/nodedb-raft/src/node/read_index.rs b/nodedb-raft/src/node/read_index.rs index cb8fa6b65..ca3111af0 100644 --- a/nodedb-raft/src/node/read_index.rs +++ b/nodedb-raft/src/node/read_index.rs @@ -6,6 +6,12 @@ //! partition does not notify the old leader. Serving a read on that belief //! returns state the new leader has since moved past. The read index is //! therefore confirmed against a quorum before it is served. +//! +//! A new leader's commit index can also lag entries an earlier leader already +//! committed: Raft only advances it through an entry of the leader's own term. +//! The read index is therefore never below the first entry of the current term +//! (the election no-op), and the probe stays `Pending` until that entry +//! commits. use crate::node::core::RaftNode; use crate::state::NodeRole; @@ -18,9 +24,11 @@ use crate::storage::LogStorage; /// stops immediately and reports that this node cannot serve the read. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum ReadIndexStatus { - /// A quorum has answered — the read may be served. + /// A quorum has answered and the read index is committed. The read may be + /// served. Confirmed, - /// Still leading, still waiting for responses. + /// Still leading, still waiting for responses or for the current term's + /// first entry to commit. Pending, /// No longer leader in the probe's term. The read can never confirm. LeadershipLost, @@ -52,7 +60,7 @@ impl RaftNode { } let leader = self.leader_state.as_ref()?; let probe = ReadIndexProbe { - read_index: self.volatile.commit_index, + read_index: self.term_read_floor(), term: self.hard_state.current_term, acks: self .config @@ -76,6 +84,9 @@ impl RaftNode { let Some(leader) = self.leader_state.as_ref() else { return ReadIndexStatus::LeadershipLost; }; + if self.volatile.commit_index < probe.read_index { + return ReadIndexStatus::Pending; + } let mut count = 1u64; // self counts. for &(peer, seen) in &probe.acks { if leader.ack_count_for(peer) > seen { @@ -94,15 +105,45 @@ impl RaftNode { pub fn read_index_confirmed(&self, probe: &ReadIndexProbe) -> bool { self.read_index_status(probe) == ReadIndexStatus::Confirmed } + + /// Whether the entry at the commit index belongs to the current term. + /// + /// Until it does, entries an earlier leader committed can sit above the + /// commit index, so the commit index is no read index. + pub(super) fn current_term_committed(&self) -> bool { + self.log.term_at(self.volatile.commit_index) == Some(self.hard_state.current_term) + } + + /// Lowest index a read may be served at: the commit index once the + /// current term has committed an entry, else the first entry of the + /// current term. + /// + /// Every entry committed before this term sits below that first entry. A + /// leader whose term has no entry yet answers `last_index + 1`, which + /// holds the read until the term commits something. + fn term_read_floor(&self) -> u64 { + let commit = self.volatile.commit_index; + if self.current_term_committed() { + return commit; + } + let term = self.hard_state.current_term; + let last = self.log.last_index(); + (commit + 1..=last) + .find(|&index| self.log.term_at(index) == Some(term)) + .unwrap_or(last + 1) + } } #[cfg(test)] mod tests { use std::time::{Duration, Instant}; - use crate::message::{AppendEntriesResponse, RequestVoteResponse}; + use crate::message::{ + AppendEntriesRequest, AppendEntriesResponse, LogEntry, RequestVoteResponse, + }; use crate::node::config::RaftConfig; use crate::node::core::RaftNode; + use crate::node::read_index::ReadIndexStatus; use crate::state::NodeRole; use crate::storage::MemStorage; use crate::test_support::force_election; @@ -148,10 +189,20 @@ mod tests { term: node.current_term(), success, last_log_index: node.last_log_index(), + round: node.lease.next_round - 1, + needs_snapshot: false, }, ); } + /// `leader()` with its election no-op committed. + fn settled_leader() -> RaftNode { + let mut node = leader(); + ack(&mut node, 2, true); + assert_eq!(node.commit_index(), node.last_log_index()); + node + } + #[test] fn a_follower_cannot_start_a_read() { let mut node = RaftNode::new(config(1, vec![2, 3]), MemStorage::new()); @@ -181,7 +232,7 @@ mod tests { /// the leadership check needs — so it counts. #[test] fn a_rejected_append_still_confirms_leadership() { - let mut node = leader(); + let mut node = settled_leader(); let probe = node.start_read_index().expect("leader starts a read"); ack(&mut node, 2, false); assert!(node.read_index_confirmed(&probe)); @@ -216,6 +267,8 @@ mod tests { term: node.current_term() + 1, success: false, last_log_index: 0, + round: 0, + needs_snapshot: false, }, ); @@ -243,7 +296,7 @@ mod tests { /// has reached since. #[test] fn the_probe_pins_the_commit_index_it_was_taken_at() { - let mut node = leader(); + let mut node = settled_leader(); let probe = node.start_read_index().expect("leader starts a read"); assert_eq!(probe.read_index, node.commit_index()); } @@ -262,4 +315,58 @@ mod tests { ready.messages ); } + + /// A new leader's commit index can sit below entries an earlier leader + /// committed. Serving at it would miss them, so the probe waits for the + /// current term's no-op. + #[test] + fn a_new_leader_with_a_stale_commit_is_pending() { + let mut node = RaftNode::new(config(1, vec![2, 3]), MemStorage::new()); + let entries = (1..=3) + .map(|index| LogEntry { + term: 1, + index, + data: vec![1], + }) + .collect(); + // Leader 2 replicated three entries but told this node of one commit. + node.handle_append_entries(&AppendEntriesRequest { + term: 1, + leader_id: 2, + prev_log_index: 0, + prev_log_term: 0, + entries, + leader_commit: 1, + group_id: 1, + round: 1, + replicated_floor: 0, + }); + assert_eq!(node.commit_index(), 1); + + force_election(&mut node); + let _ = node.take_ready(); + node.handle_request_vote_response( + 2, + &RequestVoteResponse { + term: node.current_term(), + vote_granted: true, + }, + ); + assert_eq!(node.role(), NodeRole::Leader); + let _ = node.take_ready(); + let noop = node.last_log_index(); + assert_eq!(noop, 4); + + let probe = node.start_read_index().expect("leader starts a read"); + assert_eq!(probe.read_index, noop, "the read index covers the no-op"); + + // A quorum answers, but the no-op is not stored anywhere else yet. + ack(&mut node, 2, false); + assert_eq!(node.commit_index(), 1); + assert_eq!(node.read_index_status(&probe), ReadIndexStatus::Pending); + + ack(&mut node, 3, true); + assert_eq!(node.commit_index(), noop); + assert_eq!(node.read_index_status(&probe), ReadIndexStatus::Confirmed); + } } diff --git a/nodedb-raft/src/node/rpc/append_entries.rs b/nodedb-raft/src/node/rpc/append_entries.rs index a525c230e..95637f3fa 100644 --- a/nodedb-raft/src/node/rpc/append_entries.rs +++ b/nodedb-raft/src/node/rpc/append_entries.rs @@ -17,6 +17,8 @@ impl RaftNode { term: self.hard_state.current_term, success: false, last_log_index: self.log.last_index(), + round: req.round, + needs_snapshot: false, }; } @@ -33,6 +35,9 @@ impl RaftNode { self.leader_id = req.leader_id; self.reset_election_timeout(); + // The leader's knowledge of every voter's log, whether or not this + // log matches. + self.replicated_floor = self.replicated_floor.max(req.replicated_floor); // Recorded before the log checks: contact happened and the leader's // commit index is authoritative whether or not our log matches. A // mismatched follower is behind, which is exactly what the staleness @@ -42,6 +47,31 @@ impl RaftNode { at: std::time::Instant::now(), }); + // A follower with no state to resume from takes no entries. The + // leader answers the refusal with a snapshot. + if self.snapshot_required { + return AppendEntriesResponse { + term: self.hard_state.current_term, + success: false, + last_log_index: self.log.last_index(), + round: req.round, + needs_snapshot: true, + }; + } + + // Committed entries this node never applied are gone from its log. + // Rejecting at `last_applied` walks the leader's `next_index` below its + // compacted prefix, where the leader sends InstallSnapshot instead. + if self.has_apply_gap() { + return AppendEntriesResponse { + term: self.hard_state.current_term, + success: false, + last_log_index: self.volatile.last_applied, + round: req.round, + needs_snapshot: false, + }; + } + // Check prev_log consistency. if req.prev_log_index > 0 { match self.log.term_at(req.prev_log_index) { @@ -51,29 +81,59 @@ impl RaftNode { term: self.hard_state.current_term, success: false, last_log_index: self.log.last_index(), + round: req.round, + needs_snapshot: false, }; } } } - if let Err(e) = self.log.append_entries(req.prev_log_index, &req.entries) { - warn!(group = self.config.group_id, error = %e, "append_entries failed"); - return AppendEntriesResponse { - term: self.hard_state.current_term, - success: false, - last_log_index: self.log.last_index(), - }; - } + let wrote = match self.log.append_entries(req.prev_log_index, &req.entries) { + Ok(wrote) => wrote, + Err(e) => { + warn!(group = self.config.group_id, error = %e, "append_entries failed"); + return AppendEntriesResponse { + term: self.hard_state.current_term, + success: false, + last_log_index: self.log.last_index(), + round: req.round, + needs_snapshot: false, + }; + } + }; - if req.leader_commit > self.volatile.commit_index { - self.volatile.commit_index = req.leader_commit.min(self.log.last_index()); + // The last entry this log now shares with the leader. Entries past it + // can be a stale suffix of an earlier term. + let matched = req.entries.last().map_or(req.prev_log_index, |e| e.index); + let commit = req.leader_commit.min(matched); + if commit > self.volatile.commit_index { + self.volatile.commit_index = commit; self.collect_committed_entries(); } AppendEntriesResponse { term: self.hard_state.current_term, success: true, - last_log_index: self.log.last_index(), + last_log_index: self.claimable_match(matched, wrote), + round: req.round, + needs_snapshot: false, + } + } + + /// The match index a success reply claims, from the last entry shared + /// with the leader. + /// + /// The leader counts the claim as entries durable on this node. When the + /// request took a storage write, the caller makes every write staged so + /// far durable before the reply leaves. Every held entry is then durable, + /// so the claim is `matched`. Otherwise the reply waits on no disk write, + /// and the claim stops at the durable prefix. A later round reports the + /// rest once the disk holds it. + fn claimable_match(&self, matched: u64, wrote: bool) -> u64 { + if wrote { + matched + } else { + matched.min(self.log.stable_index()) } } @@ -117,11 +177,13 @@ impl RaftNode { } else { if let Some(state) = leader.observer_state_mut(peer) { let new_next = resp.last_log_index + 1; - if new_next < state.next_index { - state.next_index = new_next.max(1); + let backed_off = if new_next < state.next_index { + new_next } else { - state.next_index = state.next_index.saturating_sub(1).max(1); - } + state.next_index.saturating_sub(1) + }; + // A stale rejection never moves below the match. + state.next_index = backed_off.max(state.match_index.saturating_add(1)).max(1); state.pending_count = state.pending_count.saturating_sub(1); } self.send_append_entries_to_observer(peer); @@ -139,16 +201,26 @@ impl RaftNode { // recognises this term, which is all a leadership check needs. leader.record_ack(peer); - // Same signal drives check-quorum: this peer has answered, so a - // majority may now have answered inside the current contact window. - if peer_is_voter { - self.refresh_quorum_contact(std::time::Instant::now()); + // Same signal drives check-quorum and the lease. A rejection counts: + // the follower recorded this leader's contact before its log check. + // The lease settles at the round's send time. Check-quorum dates the + // answer at its arrival. + if peer_is_voter && resp.term == self.hard_state.current_term { + let now = std::time::Instant::now(); + self.record_lease_ack(peer, resp.round, now); + self.settle_lease(); + self.refresh_quorum_contact(now); } let leader = match self.leader_state.as_mut() { Some(ls) => ls, None => return, }; + if resp.needs_snapshot { + leader.awaiting_snapshot.insert(peer); + } else { + leader.awaiting_snapshot.remove(&peer); + } if resp.success { let new_match = resp.last_log_index; if new_match > leader.match_index_for(peer) { @@ -157,15 +229,28 @@ impl RaftNode { } if peer_is_voter { self.try_advance_commit_index(); + self.replicated_floor = self.replicated_floor(); + } + } else if resp.needs_snapshot { + // The peer holds no state to resume from: entries from any index + // leave it as it is, so it gets a snapshot. + if !self.ready.snapshots_needed.contains(&peer) { + self.ready.snapshots_needed.push(peer); } } else { let new_next = resp.last_log_index + 1; let current_next = leader.next_index_for(peer); - if new_next < current_next { - leader.set_next_index(peer, new_next.max(1)); + let backed_off = if new_next < current_next { + new_next } else { - leader.set_next_index(peer, current_next.saturating_sub(1).max(1)); - } + current_next.saturating_sub(1) + }; + // The peer holds every entry through its `match_index`. A stale + // rejection, from a request sent before a later one matched, never + // moves the next index below it: that would send entries the peer + // holds, or a snapshot once the log compacted them. + let floor = leader.match_index_for(peer).saturating_add(1); + leader.set_next_index(peer, backed_off.max(floor).max(1)); self.send_append_entries(peer); } @@ -185,6 +270,7 @@ mod tests { }; use crate::node::config::RaftConfig; use crate::node::core::RaftNode; + use crate::node::leader_lease::UNTRACKED_ROUND; use crate::node::rpc::test_helpers::{setup_leader_with_observer, test_config}; use crate::state::NodeRole; use crate::storage::MemStorage; @@ -204,6 +290,8 @@ mod tests { entries: vec![], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }; let resp = node.handle_append_entries(&req); @@ -235,6 +323,8 @@ mod tests { ], leader_commit: 1, group_id: 1, + round: 1, + replicated_floor: 0, }; let resp = node.handle_append_entries(&req); @@ -262,6 +352,8 @@ mod tests { }], leader_commit: 1, group_id: 1, + round: 1, + replicated_floor: 0, }; let resp = node.handle_append_entries(&req); @@ -272,6 +364,105 @@ mod tests { assert_eq!(node.leader_id(), 1); } + /// A follower that requires a snapshot refuses entries with + /// `needs_snapshot`, and its leader flags it for a snapshot even though + /// the leader's log holds every entry. + #[test] + fn a_follower_that_requires_a_snapshot_gets_one() { + let mut follower = RaftNode::new(test_config(2, vec![1]), MemStorage::new()); + follower.set_snapshot_required(true); + let req = AppendEntriesRequest { + term: 1, + leader_id: 1, + prev_log_index: 0, + prev_log_term: 0, + entries: vec![LogEntry { + term: 1, + index: 1, + data: b"x".to_vec(), + }], + leader_commit: 1, + group_id: 1, + round: 1, + replicated_floor: 0, + }; + let refusal = follower.handle_append_entries(&req); + assert!(!refusal.success); + assert!(refusal.needs_snapshot); + assert_eq!(follower.last_log_index(), 0, "no entry was appended"); + assert_eq!(follower.leader_id(), 1, "the leader's contact still counts"); + + let mut leader = RaftNode::new(test_config(1, vec![2]), MemStorage::new()); + force_election(&mut leader); + leader.handle_request_vote_response( + 2, + &RequestVoteResponse { + term: 1, + vote_granted: true, + }, + ); + assert_eq!(leader.role(), NodeRole::Leader); + let _ = leader.take_ready(); + leader.handle_append_entries_response(2, &refusal); + assert!(leader.take_ready().snapshots_needed.contains(&2)); + + follower.set_snapshot_required(false); + assert!(follower.handle_append_entries(&req).success); + } + + /// A stale rejection, answering a request sent before a later one + /// matched, never moves the peer's next index below its match. + #[test] + fn a_stale_rejection_keeps_the_next_index_above_the_match() { + let mut node = RaftNode::new(test_config(1, vec![2, 3]), MemStorage::new()); + force_election(&mut node); + let _ = node.take_ready(); + node.handle_request_vote_response( + 2, + &RequestVoteResponse { + term: 1, + vote_granted: true, + }, + ); + assert_eq!(node.role(), NodeRole::Leader); + let answer = |success, last_log_index| AppendEntriesResponse { + term: 1, + success, + last_log_index, + round: 1, + needs_snapshot: false, + }; + node.handle_append_entries_response(2, &answer(true, 1)); + node.handle_append_entries_response(2, &answer(false, 0)); + let leader = node.leader_state.as_ref().expect("leader state"); + assert_eq!(leader.match_index_for(2), 1); + assert_eq!(leader.next_index_for(2), 2); + } + + /// A node that steps down within the term it voted in keeps that vote. + /// It never votes twice in one term. + #[test] + fn a_step_down_within_a_term_keeps_its_vote() { + let mut node = RaftNode::new(test_config(1, vec![2, 3]), MemStorage::new()); + force_election(&mut node); + let _ = node.take_ready(); + node.handle_request_vote_response( + 2, + &RequestVoteResponse { + term: 1, + vote_granted: true, + }, + ); + assert_eq!(node.role(), NodeRole::Leader); + + node.become_follower(1); + assert_eq!(node.current_term(), 1); + assert_eq!(node.hard_state.voted_for, 1, "the vote of term 1 stands"); + + node.become_follower(2); + assert_eq!(node.hard_state.voted_for, 0, "a new term frees the vote"); + } + #[test] fn leader_steps_down_on_higher_term() { let config = test_config(1, vec![2, 3]); @@ -294,6 +485,8 @@ mod tests { entries: vec![], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }; node.handle_append_entries(&req); assert_eq!(node.role(), NodeRole::Follower); @@ -324,6 +517,8 @@ mod tests { entries: vec![], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }; node.handle_append_entries(&req); @@ -368,6 +563,8 @@ mod tests { term: 1, success: true, last_log_index: 2, + round: UNTRACKED_ROUND, + needs_snapshot: false, }; node.handle_append_entries_response(4, &ae_ok); assert_eq!( @@ -388,6 +585,7 @@ mod tests { let mut node1 = RaftNode::new(config1, MemStorage::new()); let mut node2 = RaftNode::new(config2, MemStorage::new()); + node2.expire_boot_vote_fence(); force_election(&mut node1); let ready = node1.take_ready(); @@ -470,6 +668,8 @@ mod tests { term: 1, success: true, last_log_index: idx, + round: UNTRACKED_ROUND, + needs_snapshot: false, }; leader.handle_append_entries_response(2, &ae_ok); assert_eq!(leader.commit_index(), idx); @@ -522,6 +722,8 @@ mod tests { term: 1, success: true, last_log_index: idx, + round: UNTRACKED_ROUND, + needs_snapshot: false, }; node1.handle_append_entries_response(5, &obs_ack); assert_eq!( @@ -546,6 +748,8 @@ mod tests { term: 1, success: true, last_log_index: idx, + round: UNTRACKED_ROUND, + needs_snapshot: false, }; leader.handle_append_entries_response(2, &voter_ack); assert_eq!( @@ -555,4 +759,136 @@ mod tests { ); let _ = ready; } + + /// Storage that takes every write at once and reports a durable prefix + /// the test sets. + #[derive(Default)] + struct StagingStorage { + inner: MemStorage, + stable: (u64, u64), + } + + impl crate::storage::LogStorage for StagingStorage { + fn append(&mut self, entries: &[LogEntry]) -> crate::error::Result<()> { + self.inner.append(entries) + } + fn truncate(&mut self, index: u64) -> crate::error::Result<()> { + self.inner.truncate(index) + } + fn load_entries_after(&self, snapshot_index: u64) -> crate::error::Result> { + self.inner.load_entries_after(snapshot_index) + } + fn compact(&mut self, index: u64, term: u64) -> crate::error::Result<()> { + self.inner.compact(index, term) + } + fn snapshot_metadata(&self) -> (u64, u64) { + self.inner.snapshot_metadata() + } + fn save_hard_state(&mut self, state: &crate::state::HardState) -> crate::error::Result<()> { + self.inner.save_hard_state(state) + } + fn load_hard_state(&self) -> crate::error::Result { + self.inner.load_hard_state() + } + fn save_applied_index(&mut self, index: u64) -> crate::error::Result<()> { + self.inner.save_applied_index(index) + } + fn load_applied_index(&self) -> crate::error::Result { + self.inner.load_applied_index() + } + fn stable_through(&self) -> Option<(u64, u64)> { + Some(self.stable) + } + } + + fn entry(term: u64, index: u64) -> LogEntry { + LogEntry { + term, + index, + data: vec![index as u8], + } + } + + fn append_request( + term: u64, + prev: (u64, u64), + entries: Vec, + leader_commit: u64, + ) -> AppendEntriesRequest { + AppendEntriesRequest { + term, + leader_id: 2, + prev_log_index: prev.0, + prev_log_term: prev.1, + entries, + leader_commit, + group_id: 1, + round: 1, + replicated_floor: 0, + } + } + + /// A request that writes entries claims them: its caller makes them + /// durable before the reply leaves. A resend of the same range writes + /// nothing and waits on no disk, so it claims only the durable prefix. + #[test] + fn a_reply_without_a_write_claims_only_the_durable_prefix() { + let mut node = RaftNode::new(test_config(1, vec![2, 3]), StagingStorage::default()); + let req = append_request(1, (0, 0), vec![entry(1, 1), entry(1, 2)], 0); + + let resp = node.handle_append_entries(&req); + assert!(resp.success); + assert_eq!(resp.last_log_index, 2, "the written entries are claimed"); + + let resp = node.handle_append_entries(&req); + assert!(resp.success); + assert_eq!(resp.last_log_index, 0, "nothing is durable yet"); + + node.log.storage_mut().stable = (1, 1); + let heartbeat = append_request(1, (2, 1), vec![], 0); + let resp = node.handle_append_entries(&heartbeat); + assert!(resp.success); + assert_eq!( + resp.last_log_index, 1, + "the claim stops at the durable entry" + ); + + node.log.storage_mut().stable = (2, 1); + let resp = node.handle_append_entries(&heartbeat); + assert_eq!(resp.last_log_index, 2); + } + + /// Entries past the request's last entry can be a stale suffix of an + /// earlier term. A success reply never claims them, and the follower + /// never commits them. + #[test] + fn a_reply_never_claims_a_stale_suffix() { + let mut node = RaftNode::new(test_config(1, vec![2, 3]), MemStorage::new()); + let old = append_request(1, (0, 0), vec![entry(1, 1), entry(1, 2), entry(1, 3)], 0); + assert!(node.handle_append_entries(&old).success); + + // A leader of term 2 shares only entry 1 with this log. + let heartbeat = append_request(2, (1, 1), vec![], 3); + let resp = node.handle_append_entries(&heartbeat); + assert!(resp.success); + assert_eq!( + resp.last_log_index, 1, + "entries 2 and 3 are not the leader's" + ); + assert_eq!(node.commit_index(), 1); + } + + /// A reordered request that matches less of the log never pulls the + /// commit index back. + #[test] + fn a_reordered_request_never_lowers_the_commit_index() { + let mut node = RaftNode::new(test_config(1, vec![2, 3]), MemStorage::new()); + let full = append_request(1, (0, 0), vec![entry(1, 1), entry(1, 2), entry(1, 3)], 3); + assert!(node.handle_append_entries(&full).success); + assert_eq!(node.commit_index(), 3); + + let late = append_request(1, (0, 0), vec![entry(1, 1)], 4); + assert!(node.handle_append_entries(&late).success); + assert_eq!(node.commit_index(), 3); + } } diff --git a/nodedb-raft/src/node/rpc/install_snapshot.rs b/nodedb-raft/src/node/rpc/install_snapshot.rs index 0e1c49d3b..278c757b6 100644 --- a/nodedb-raft/src/node/rpc/install_snapshot.rs +++ b/nodedb-raft/src/node/rpc/install_snapshot.rs @@ -36,33 +36,287 @@ impl RaftNode { self.leader_id = req.leader_id; self.reset_election_timeout(); - if req.done && req.last_included_index > self.log.snapshot_index() { - info!( - node = self.config.node_id, - group = self.config.group_id, - snapshot_index = req.last_included_index, - snapshot_term = req.last_included_term, - "applying installed snapshot" - ); - - self.log - .apply_snapshot(req.last_included_index, req.last_included_term); - - if self.volatile.commit_index < req.last_included_index { - self.volatile.commit_index = req.last_included_index; - } - if self.volatile.last_applied < req.last_included_index { - self.volatile.last_applied = req.last_included_index; - } - // Move the durable floor with the snapshot boundary. The entries - // the snapshot subsumes are gone from the log, so a restart that - // resumed from the pre-snapshot floor would replay from an index - // the log can no longer serve. - self.save_durable_applied_index(req.last_included_index)?; + if req.done { + self.adopt_snapshot_boundary(req.last_included_index, req.last_included_term)?; } Ok(InstallSnapshotResponse { term: self.hard_state.current_term, }) } + + /// Move the log boundary, commit index, applied index, and durable floor + /// to a snapshot the state machine already holds. + /// + /// No term check: a snapshot carries only committed state, so adopting it + /// is safe whichever term sent it. Boot recovery calls this to complete an + /// install whose state-machine apply finished before a crash. A no-op when + /// both the log boundary and `last_applied` are at or past + /// `last_included_index`. + /// + /// A snapshot at or below the log boundary but above `last_applied` still + /// moves the applied index: it fills part of an apply gap. + /// + /// The CALLER MUST have restored the snapshot into the state machine + /// first, for the same reason as [`Self::handle_install_snapshot`]. + pub fn adopt_snapshot_boundary( + &mut self, + last_included_index: u64, + last_included_term: u64, + ) -> Result<()> { + let moves_boundary = last_included_index > self.log.snapshot_index(); + if !moves_boundary && last_included_index <= self.volatile.last_applied { + return Ok(()); + } + info!( + node = self.config.node_id, + group = self.config.group_id, + snapshot_index = last_included_index, + snapshot_term = last_included_term, + "applying installed snapshot" + ); + + if moves_boundary { + self.log + .apply_snapshot(last_included_index, last_included_term)?; + } + + if self.volatile.commit_index < last_included_index { + self.volatile.commit_index = last_included_index; + } + if self.volatile.last_applied < last_included_index { + self.volatile.last_applied = last_included_index; + } + // The snapshot already holds every effect at or below its index, so + // queued entries in that range must not reach the state machine again. + self.ready + .committed_entries + .retain(|entry| entry.index > last_included_index); + if !self.has_apply_gap() { + self.ready.committed_read_error = None; + } + // Move the durable floor with the snapshot boundary. The entries + // the snapshot subsumes are gone from the log, so a restart that + // resumed from the pre-snapshot floor would replay from an index + // the log can no longer serve. + self.save_durable_applied_index(last_included_index) + } + + /// Record on the leader that `peer` installed a snapshot through + /// `last_included_index`, so replication to it resumes after that index. + /// + /// Without this the leader keeps `peer`'s next index at or below its own + /// log boundary and flags it for another snapshot on every heartbeat. + /// No-op when this node does not lead, or `peer` already matched past + /// the index. + pub fn record_snapshot_installed(&mut self, peer: u64, last_included_index: u64) { + let Some(leader) = self.leader_state.as_mut() else { + return; + }; + if last_included_index > leader.match_index_for(peer) { + leader.set_match_index(peer, last_included_index); + leader.set_next_index(peer, last_included_index + 1); + } + } +} + +#[cfg(test)] +mod tests { + use std::ops::RangeInclusive; + + use crate::message::{ + AppendEntriesRequest, AppendEntriesResponse, InstallSnapshotRequest, LogEntry, + RequestVoteResponse, + }; + use crate::node::core::RaftNode; + use crate::node::leader_lease::UNTRACKED_ROUND; + use crate::state::NodeRole; + use crate::storage::MemStorage; + use crate::test_support::{apply_durably, force_election}; + + use super::super::test_helpers::test_config; + + const TERM: u64 = 1; + + fn follower() -> RaftNode { + RaftNode::new(test_config(2, vec![1, 3]), MemStorage::new()) + } + + /// Node 1 elected leader of voters {1, 2, 3} at `TERM`. + fn leader() -> RaftNode { + let mut node = RaftNode::new(test_config(1, vec![2, 3]), MemStorage::new()); + force_election(&mut node); + let _ = node.take_ready(); + node.handle_request_vote_response( + 2, + &RequestVoteResponse { + term: TERM, + vote_granted: true, + }, + ); + assert_eq!(node.role(), NodeRole::Leader); + let _ = node.take_ready(); + node + } + + /// Deliver leader 1's entries `indices` after `prev_log_index`, committed + /// through `leader_commit`. + fn append( + node: &mut RaftNode, + prev_log_index: u64, + indices: RangeInclusive, + leader_commit: u64, + ) { + let entries = indices + .map(|index| LogEntry { + term: TERM, + index, + data: b"write".to_vec(), + }) + .collect(); + let resp = node.handle_append_entries(&AppendEntriesRequest { + term: TERM, + leader_id: 1, + prev_log_index, + prev_log_term: if prev_log_index == 0 { 0 } else { TERM }, + entries, + leader_commit, + group_id: 1, + round: 0, + replicated_floor: 0, + }); + assert!(resp.success, "append after {prev_log_index} must succeed"); + } + + fn install(node: &mut RaftNode, last_included_index: u64) { + node.handle_install_snapshot(&InstallSnapshotRequest { + term: TERM, + leader_id: 1, + last_included_index, + last_included_term: TERM, + offset: 0, + data: Vec::new(), + done: true, + group_id: 1, + total_size: 0, + voters: Vec::new(), + learners: Vec::new(), + }) + .expect("snapshot install succeeds"); + } + + fn committed_indices(node: &mut RaftNode) -> Vec { + let ready = node.take_ready(); + assert!( + ready.committed_read_error.is_none(), + "delivery must not stall: {:?}", + ready.committed_read_error + ); + ready.committed_entries.iter().map(|e| e.index).collect() + } + + /// Entries queued before an install that the snapshot covers are dropped. + #[test] + fn install_prunes_queued_entries_the_snapshot_covers() { + let mut node = follower(); + append(&mut node, 0, 1..=5, 5); + + install(&mut node, 3); + + assert_eq!(node.last_applied(), 3); + assert_eq!(committed_indices(&mut node), vec![4, 5]); + } + + /// The driver takes a batch, a snapshot installs past it, and the batch + /// finishes afterwards. Delivery resumes past the snapshot. + #[test] + fn interleaved_install_does_not_stall_delivery() { + let mut node = follower(); + append(&mut node, 0, 1..=5, 5); + let in_flight = committed_indices(&mut node); + assert_eq!(in_flight, vec![1, 2, 3, 4, 5]); + + install(&mut node, 8); + node.advance_applied(5); + assert_eq!(node.last_applied(), 8); + + append(&mut node, 8, 9..=10, 10); + assert_eq!(committed_indices(&mut node), vec![9, 10]); + } + + /// A follower whose log was compacted past its applied index rejects at + /// that index. The leader then needs a snapshot for it, and after the + /// install the follower applies new entries again. + #[test] + fn follower_with_apply_gap_gets_a_snapshot_and_resumes() { + let mut leader = leader(); + for _ in 0..11 { + leader.propose(b"write".to_vec()).expect("leader accepts"); + } + let tip = leader.last_log_index(); + leader.handle_append_entries_response( + 3, + &AppendEntriesResponse { + term: TERM, + success: true, + last_log_index: tip, + round: UNTRACKED_ROUND, + needs_snapshot: false, + }, + ); + assert_eq!(leader.commit_index(), tip); + apply_durably(&mut leader, tip); + let leader_snapshot = tip - 2; + assert!( + leader + .compact_log_up_to(leader_snapshot) + .expect("durable prefix compacts") + ); + let _ = leader.take_ready(); + + // Follower log compacted through 8, applied only through 4. Compaction + // never passes `last_applied`, so the gap is forced directly. + let mut node = follower(); + append(&mut node, 0, 1..=10, 10); + let _ = node.take_ready(); + apply_durably(&mut node, 8); + assert!(node.compact_log_up_to(8).expect("compacts")); + node.volatile.last_applied = 4; + assert!(node.has_apply_gap()); + + let resp = node.handle_append_entries(&AppendEntriesRequest { + term: TERM, + leader_id: 1, + prev_log_index: tip, + prev_log_term: TERM, + entries: Vec::new(), + leader_commit: tip, + group_id: 1, + round: 0, + replicated_floor: 0, + }); + assert!(!resp.success); + assert_eq!(resp.last_log_index, 4); + + leader.handle_append_entries_response(2, &resp); + assert!(leader.take_ready().snapshots_needed.contains(&2)); + + // Once the leader records the install, replication to the peer + // resumes after the snapshot instead of flagging another one. + leader.record_snapshot_installed(2, leader_snapshot); + assert_eq!(leader.match_index_for(2), Some(leader_snapshot)); + leader.propose(b"after".to_vec()).expect("leader accepts"); + assert!(!leader.take_ready().snapshots_needed.contains(&2)); + + install(&mut node, leader_snapshot); + assert!(!node.has_apply_gap()); + assert_eq!(node.last_applied(), leader_snapshot); + let _ = node.take_ready(); + + append(&mut node, leader_snapshot, leader_snapshot + 1..=tip, tip); + assert_eq!( + committed_indices(&mut node), + (leader_snapshot + 1..=tip).collect::>() + ); + } } diff --git a/nodedb-raft/src/node/rpc/pre_vote.rs b/nodedb-raft/src/node/rpc/pre_vote.rs index 6712d01b0..751c0f013 100644 --- a/nodedb-raft/src/node/rpc/pre_vote.rs +++ b/nodedb-raft/src/node/rpc/pre_vote.rs @@ -7,6 +7,8 @@ //! recorded, no election deadline is reset. Only a granting quorum promotes //! the prober to a real election. +use std::time::Instant; + use tracing::debug; use crate::message::{PreVoteRequest, PreVoteResponse}; @@ -21,6 +23,12 @@ impl RaftNode { /// `leader_id`, or the role. Resetting the deadline for a probe would let /// any peer suppress a legitimate election just by probing. pub fn handle_pre_vote(&mut self, req: &PreVoteRequest) -> PreVoteResponse { + self.handle_pre_vote_at(req, Instant::now()) + } + + /// [`Self::handle_pre_vote`] with the vote-refusal window measured at + /// `now`. + pub fn handle_pre_vote_at(&mut self, req: &PreVoteRequest, now: Instant) -> PreVoteResponse { let term = self.hard_state.current_term; let refuse = PreVoteResponse { term, @@ -45,12 +53,10 @@ impl RaftNode { // Leader stickiness, bounded by TIME. A leader that is still reaching // us needs no replacement. The bound is what makes the check safe: once // a real leader crashes, contact ages past `election_timeout_min` and - // every follower starts granting again. A node that has never heard a - // leader is never blocked here, so a fresh cluster still elects. - let leader_is_live = self - .leader_contact - .is_some_and(|c| c.at.elapsed() < self.config.election_timeout_min); - if leader_is_live { + // every follower starts granting again. A freshly booted node refuses + // until `election_timeout_max` after boot, the same window as the real + // vote, so a probe it grants is never followed by a vote it refuses. + if self.vote_refusal_active(now) { return refuse; } @@ -153,8 +159,12 @@ mod tests { entries: vec![], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }); assert_eq!(node.role(), NodeRole::Follower); + // Leader contact, not the boot fence, is what these tests exercise. + node.expire_boot_vote_fence(); node } @@ -183,6 +193,7 @@ mod tests { assert_eq!(ready.pre_vote_requests[0].1.term, 1); let mut peer = RaftNode::new(test_config(2, vec![1, 3]), MemStorage::new()); + peer.expire_boot_vote_fence(); let resp = peer.handle_pre_vote(&ready.pre_vote_requests[0].1); assert!(resp.vote_granted); assert_eq!(peer.current_term(), 0, "answering must not adopt a term"); @@ -213,6 +224,8 @@ mod tests { entries: vec![], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }); prober.election_deadline_override(Instant::now() - Duration::from_millis(1)); prober.tick(); @@ -249,9 +262,25 @@ mod tests { assert_eq!(follower.current_term(), 1); } + /// A booted node has no record of the leader it may have followed before + /// a restart. It refuses until the boot fence passes, then grants. + #[test] + fn a_booted_node_refuses_until_its_fence_passes() { + let mut node = RaftNode::new(test_config(2, vec![1, 3]), MemStorage::new()); + let fence = node.boot_vote_fence; + let just_before = fence - Duration::from_nanos(1); + assert!( + !node + .handle_pre_vote_at(&probe(1, 0, 0), just_before) + .vote_granted + ); + assert!(node.handle_pre_vote_at(&probe(1, 0, 0), fence).vote_granted); + } + #[test] fn a_node_that_never_heard_a_leader_grants() { let mut node = RaftNode::new(test_config(2, vec![1, 3]), MemStorage::new()); + node.expire_boot_vote_fence(); let deadline_before = node.election_deadline; assert!(node.handle_pre_vote(&probe(1, 0, 0)).vote_granted); @@ -308,8 +337,11 @@ mod tests { }], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }); node.leader_contact_at_override(Instant::now() - Duration::from_secs(1)); + node.expire_boot_vote_fence(); // Older last term. assert!(!node.handle_pre_vote(&probe(3, 5, 1)).vote_granted); @@ -322,12 +354,14 @@ mod tests { #[test] fn a_hypothetical_term_at_or_below_the_current_one_is_refused() { let mut node = RaftNode::new(test_config(2, vec![1, 3]), MemStorage::new()); + node.expire_boot_vote_fence(); node.handle_request_vote(&RequestVoteRequest { term: 5, candidate_id: 1, last_log_index: 0, last_log_term: 0, group_id: 1, + transfer: false, }); assert_eq!(node.current_term(), 5); diff --git a/nodedb-raft/src/node/rpc/request_vote.rs b/nodedb-raft/src/node/rpc/request_vote.rs index 260c5334d..81a72aaae 100644 --- a/nodedb-raft/src/node/rpc/request_vote.rs +++ b/nodedb-raft/src/node/rpc/request_vote.rs @@ -2,6 +2,8 @@ //! `RequestVote` request and response handlers. +use std::time::Instant; + use tracing::debug; use crate::message::{RequestVoteRequest, RequestVoteResponse}; @@ -15,7 +17,40 @@ impl RaftNode { /// Learners and observers never grant votes: by definition they are not /// members of the voting set for this term, and granting a vote could /// let an incorrect quorum form. + /// + /// While a leader reached this node within `election_timeout_min`, or + /// within `election_timeout_max` of boot, a voter refuses every vote but a + /// transfer vote, and does not adopt the candidate's term. The leader + /// lease depends on this refusal (see [`crate::node::leader_lease`]). pub fn handle_request_vote(&mut self, req: &RequestVoteRequest) -> RequestVoteResponse { + self.handle_request_vote_at(req, Instant::now()) + } + + /// [`Self::handle_request_vote`] with the vote-refusal window measured + /// at `now`. + pub fn handle_request_vote_at( + &mut self, + req: &RequestVoteRequest, + now: Instant, + ) -> RequestVoteResponse { + let is_voter = matches!( + self.role, + NodeRole::Follower | NodeRole::Candidate | NodeRole::Leader + ); + if is_voter && !req.transfer && self.vote_refusal_active(now) { + debug!( + node = self.config.node_id, + group = self.config.group_id, + candidate = req.candidate_id, + term = req.term, + "refused vote: leader still live" + ); + return RequestVoteResponse { + term: self.hard_state.current_term, + vote_granted: false, + }; + } + if req.term > self.hard_state.current_term { self.become_follower(req.term); } @@ -115,6 +150,7 @@ mod tests { fn vote_grant_and_reject() { let config = test_config(1, vec![2, 3]); let mut node = RaftNode::new(config, MemStorage::new()); + node.expire_boot_vote_fence(); let req = RequestVoteRequest { term: 1, @@ -122,6 +158,7 @@ mod tests { last_log_index: 0, last_log_term: 0, group_id: 1, + transfer: false, }; let resp = node.handle_request_vote(&req); assert!(resp.vote_granted); @@ -132,6 +169,7 @@ mod tests { last_log_index: 0, last_log_term: 0, group_id: 1, + transfer: false, }; let resp2 = node.handle_request_vote(&req2); assert!(!resp2.vote_granted); @@ -150,6 +188,7 @@ mod tests { last_log_index: 10, last_log_term: 4, group_id: 1, + transfer: false, }; let resp = node.handle_request_vote(&req); assert!( @@ -193,6 +232,8 @@ mod tests { let mut node1 = RaftNode::new(config1, MemStorage::new()); let mut node2 = RaftNode::new(config2, MemStorage::new()); let mut node3 = RaftNode::new(config3, MemStorage::new()); + node2.expire_boot_vote_fence(); + node3.expire_boot_vote_fence(); force_election(&mut node1); assert_eq!(node1.role(), NodeRole::Candidate); @@ -221,6 +262,7 @@ mod tests { last_log_index: 100, last_log_term: 9, group_id: 1, + transfer: false, }; let resp = obs.handle_request_vote(&req); assert!(!resp.vote_granted, "observer must never grant a vote"); diff --git a/nodedb-raft/src/node/rpc/timeout_now.rs b/nodedb-raft/src/node/rpc/timeout_now.rs index 29e7b3978..60fcdd492 100644 --- a/nodedb-raft/src/node/rpc/timeout_now.rs +++ b/nodedb-raft/src/node/rpc/timeout_now.rs @@ -21,12 +21,22 @@ impl RaftNode { /// Deliberately calls `start_election` rather than `start_pre_election`: /// a transfer needs an immediate, guaranteed term bump, and the outgoing /// leader has already confirmed the target is caught up. + /// + /// The campaign's vote requests carry `transfer`, so voters that still + /// hear the outgoing leader grant them. That leader stopped serving lease + /// reads when the transfer began. pub fn handle_timeout_now(&mut self, req: &TimeoutNowRequest) { if self.role == NodeRole::Follower && req.term == self.hard_state.current_term && req.leader_id == self.leader_id { self.start_election(); + let term = self.hard_state.current_term; + for (_, vote) in self.ready.vote_requests.iter_mut() { + if vote.term == term { + vote.transfer = true; + } + } } } } @@ -41,6 +51,7 @@ mod tests { }; use crate::node::config::RaftConfig; use crate::node::core::RaftNode; + use crate::node::leader_lease::UNTRACKED_ROUND; use crate::state::NodeRole; use crate::storage::MemStorage; use crate::test_support::force_election; @@ -89,6 +100,8 @@ mod tests { entries: vec![], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }); assert_eq!(node.role(), NodeRole::Follower); assert_eq!(node.current_term(), 1); @@ -101,6 +114,8 @@ mod tests { term: 1, success: true, last_log_index: idx, + round: UNTRACKED_ROUND, + needs_snapshot: false, } } @@ -127,6 +142,29 @@ mod tests { target.handle_timeout_now(&req); assert_eq!(target.role(), NodeRole::Candidate); assert_eq!(target.current_term(), term_before + 1); + + // Voters still hear leader 1, and grant only because this is a + // transfer campaign. + let votes = target.take_ready().vote_requests; + assert_eq!(votes.len(), 2); + assert!(votes.iter().all(|(_, vote)| vote.transfer)); + let mut voter = RaftNode::new(cfg(3, vec![1, 2], vec![]), MemStorage::new()); + voter.handle_append_entries(&AppendEntriesRequest { + term: 1, + leader_id: 1, + prev_log_index: 0, + prev_log_term: 0, + entries: vec![], + leader_commit: 0, + group_id: 1, + round: 1, + replicated_floor: 0, + }); + let (_, vote) = votes + .iter() + .find(|(peer, _)| *peer == 3) + .expect("the target asks voter 3"); + assert!(voter.handle_request_vote(vote).vote_granted); } // t2: transfer to a lagging target defers; the trigger fires once the @@ -222,6 +260,8 @@ mod tests { term: 5, success: false, last_log_index: 0, + round: UNTRACKED_ROUND, + needs_snapshot: false, }, ); assert_eq!(leader.role(), NodeRole::Follower); diff --git a/nodedb-raft/src/node/staleness.rs b/nodedb-raft/src/node/staleness.rs index 528921ed2..98ba6be76 100644 --- a/nodedb-raft/src/node/staleness.rs +++ b/nodedb-raft/src/node/staleness.rs @@ -106,6 +106,8 @@ mod tests { entries: vec![], leader_commit, group_id: 1, + round: 1, + replicated_floor: 0, }); } diff --git a/nodedb-raft/src/state.rs b/nodedb-raft/src/state.rs index 89b78d3f9..18faf94b4 100644 --- a/nodedb-raft/src/state.rs +++ b/nodedb-raft/src/state.rs @@ -183,6 +183,9 @@ pub struct LeaderState { /// timed: leadership is confirmed by a quorum of these rising after a /// heartbeat round. pub ack_count: Vec<(u64, u64)>, + /// Peers whose latest `AppendEntries` response asked for a snapshot. + /// Such a peer holds no state to lead from. + pub awaiting_snapshot: std::collections::HashSet, } impl LeaderState { @@ -192,6 +195,7 @@ impl LeaderState { next_index: peers.iter().map(|&id| (id, last_log_index + 1)).collect(), match_index: peers.iter().map(|&id| (id, 0)).collect(), ack_count: peers.iter().map(|&id| (id, 0)).collect(), + awaiting_snapshot: std::collections::HashSet::new(), observer_states: observers .iter() .map(|&id| { @@ -250,6 +254,7 @@ impl LeaderState { self.next_index.retain(|&(id, _)| id != peer); self.match_index.retain(|&(id, _)| id != peer); self.ack_count.retain(|&(id, _)| id != peer); + self.awaiting_snapshot.remove(&peer); } /// Responses received from `peer` in this term. diff --git a/nodedb-raft/src/storage.rs b/nodedb-raft/src/storage.rs index efb0b93a4..4a8fabc30 100644 --- a/nodedb-raft/src/storage.rs +++ b/nodedb-raft/src/storage.rs @@ -9,7 +9,8 @@ use crate::state::HardState; /// Implementors handle durability. The `nodedb-cluster` crate provides /// a production implementation backed by `nodedb-wal`. pub trait LogStorage: Send { - /// Persist log entries (must be durable before returning). + /// Persist log entries: durable before returning, or staged and reported + /// through [`Self::stable_through`] once durable. fn append(&mut self, entries: &[LogEntry]) -> Result<()>; /// Truncate log entries from `index` onward (inclusive). @@ -44,6 +45,19 @@ pub trait LogStorage: Send { /// retained log — the safe direction for storage written before this index /// existed. fn load_applied_index(&self) -> Result; + + /// The last log entry this storage holds durably, as `(index, term)`. + /// + /// `None`, the default, for storage whose every write is durable before + /// it returns. Storage that stages writes and makes them durable later + /// reports how far its disk has come. The node then counts only durable + /// entries toward its own acknowledgement of a commit (see + /// [`crate::RaftLog::stable_index`]). A caller of such storage makes a + /// write durable before it sends a reply or a vote request that depends + /// on it. + fn stable_through(&self) -> Option<(u64, u64)> { + None + } } /// In-memory storage for testing. diff --git a/nodedb-raft/tests/election.rs b/nodedb-raft/tests/election.rs index cf76bc3d7..fabd645a4 100644 --- a/nodedb-raft/tests/election.rs +++ b/nodedb-raft/tests/election.rs @@ -60,6 +60,7 @@ fn force_election(node: &mut RaftNode) { #[test] fn vote_grant_is_idempotent_for_same_candidate() { let mut node = RaftNode::new(config(1, vec![2, 3]), MemStorage::new()); + node.expire_boot_vote_fence(); let req = RequestVoteRequest { term: 1, @@ -67,6 +68,7 @@ fn vote_grant_is_idempotent_for_same_candidate() { last_log_index: 0, last_log_term: 0, group_id: 1, + transfer: false, }; let r1 = node.handle_request_vote(&req); assert!(r1.vote_granted); @@ -86,6 +88,7 @@ fn vote_grant_is_idempotent_for_same_candidate() { #[test] fn stale_term_request_vote_rejected_with_current_term() { let mut node = RaftNode::new(config(1, vec![2, 3]), MemStorage::new()); + node.expire_boot_vote_fence(); // Drive the node up to term 5 via a higher-term AE. let bump = AppendEntriesRequest { @@ -96,9 +99,13 @@ fn stale_term_request_vote_rejected_with_current_term() { entries: vec![], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }; let _ = node.handle_append_entries(&bump); assert_eq!(node.current_term(), 5); + // Age the leader contact so the term rule, not vote refusal, decides. + node.leader_contact_at_override(Instant::now() - Duration::from_secs(1)); let stale = RequestVoteRequest { term: 3, @@ -106,6 +113,7 @@ fn stale_term_request_vote_rejected_with_current_term() { last_log_index: 0, last_log_term: 0, group_id: 1, + transfer: false, }; let resp = node.handle_request_vote(&stale); assert!(!resp.vote_granted); @@ -121,6 +129,7 @@ fn stale_term_request_vote_rejected_with_current_term() { #[test] fn vote_denied_when_candidate_log_not_up_to_date() { let mut node = RaftNode::new(config(1, vec![2, 3]), MemStorage::new()); + node.expire_boot_vote_fence(); // Seed the voter's log with two entries at term 2 via AE. let seed = AppendEntriesRequest { @@ -142,9 +151,13 @@ fn vote_denied_when_candidate_log_not_up_to_date() { ], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }; assert!(node.handle_append_entries(&seed).success); assert_eq!(node.current_term(), 2); + // Age the leader contact so the up-to-date rule, not vote refusal, decides. + node.leader_contact_at_override(Instant::now() - Duration::from_secs(1)); // Candidate at term 3 but with a shorter log (last_log at term 1). let stale_log = RequestVoteRequest { @@ -153,6 +166,7 @@ fn vote_denied_when_candidate_log_not_up_to_date() { last_log_index: 5, // long, but... last_log_term: 1, // ...older term — loses to our term-2 tail group_id: 1, + transfer: false, }; let resp = node.handle_request_vote(&stale_log); assert!( @@ -167,6 +181,7 @@ fn vote_denied_when_candidate_log_not_up_to_date() { last_log_index: 1, last_log_term: 2, group_id: 1, + transfer: false, }; let resp = node.handle_request_vote(&shorter); assert!(!resp.vote_granted, "same term, shorter log must lose"); @@ -178,6 +193,7 @@ fn vote_denied_when_candidate_log_not_up_to_date() { last_log_index: 2, last_log_term: 2, group_id: 1, + transfer: false, }; let resp = node.handle_request_vote(&equal); assert!( @@ -205,6 +221,8 @@ fn candidate_steps_down_on_higher_term_append_entries() { entries: vec![], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }; let resp = node.handle_append_entries(&intruder); assert!(resp.success); @@ -244,6 +262,7 @@ fn consecutive_failed_elections_strictly_increment_term() { #[test] fn second_candidate_rejected_in_same_term_after_vote_already_cast() { let mut node = RaftNode::new(config(1, vec![2, 3]), MemStorage::new()); + node.expire_boot_vote_fence(); let a = RequestVoteRequest { term: 1, @@ -251,6 +270,7 @@ fn second_candidate_rejected_in_same_term_after_vote_already_cast() { last_log_index: 0, last_log_term: 0, group_id: 1, + transfer: false, }; assert!(node.handle_request_vote(&a).vote_granted); @@ -260,6 +280,7 @@ fn second_candidate_rejected_in_same_term_after_vote_already_cast() { last_log_index: 0, last_log_term: 0, group_id: 1, + transfer: false, }; let resp = node.handle_request_vote(&b); assert!( @@ -295,6 +316,7 @@ fn candidate_steps_down_on_higher_term_vote_response() { #[test] fn restart_does_not_double_vote_in_same_term() { let mut node = RaftNode::new(config(1, vec![2, 3]), MemStorage::new()); + node.expire_boot_vote_fence(); // Node 1 votes for candidate 2 in term 1. let vote_for_a = RequestVoteRequest { @@ -303,6 +325,7 @@ fn restart_does_not_double_vote_in_same_term() { last_log_index: 0, last_log_term: 0, group_id: 1, + transfer: false, }; assert!(node.handle_request_vote(&vote_for_a).vote_granted); assert_eq!(node.current_term(), 1); @@ -327,6 +350,7 @@ fn restart_does_not_double_vote_in_same_term() { last_log_index: 0, last_log_term: 0, group_id: 1, + transfer: false, }; let resp = node.handle_request_vote(&vote_for_b); assert!( diff --git a/nodedb-raft/tests/log_consistency.rs b/nodedb-raft/tests/log_consistency.rs index c07f4bc5f..c3799c001 100644 --- a/nodedb-raft/tests/log_consistency.rs +++ b/nodedb-raft/tests/log_consistency.rs @@ -113,6 +113,8 @@ fn ae_prev_log_mismatch_returns_backtrack_hint() { entries: vec![entry_with(1, 1, b"x")], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }; let resp = node.handle_append_entries(&bootstrap); assert!(resp.success); @@ -126,6 +128,8 @@ fn ae_prev_log_mismatch_returns_backtrack_hint() { entries: vec![entry_with(1, 6, b"y")], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }; let resp = node.handle_append_entries(&stale); assert!(!resp.success, "AE with bad prev must be rejected"); @@ -149,6 +153,8 @@ fn ae_prev_log_term_mismatch_returns_backtrack_hint() { entries: vec![entry_with(2, 1, b"a"), entry_with(2, 2, b"b")], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }; assert!(node.handle_append_entries(&bootstrap).success); @@ -161,6 +167,8 @@ fn ae_prev_log_term_mismatch_returns_backtrack_hint() { entries: vec![entry_with(5, 3, b"c")], leader_commit: 0, group_id: 1, + round: 1, + replicated_floor: 0, }; let resp = node.handle_append_entries(&bad_term); assert!(!resp.success); @@ -177,7 +185,8 @@ fn append_then_snapshot_then_appendentries_post_boundary() { for i in 1..=5 { log.append(entry(1, i)).unwrap(); } - log.apply_snapshot(3, 1); + log.apply_snapshot(3, 1) + .expect("MemStorage compacts to the snapshot boundary"); // Boundary semantics. assert_eq!(log.snapshot_index(), 3); @@ -239,6 +248,8 @@ fn leader_commit_clamped_to_last_index() { entries: vec![entry(1, 1), entry(1, 2)], leader_commit: 99, // way past tail group_id: 1, + round: 1, + replicated_floor: 0, }; let resp = node.handle_append_entries(&req); assert!(resp.success); diff --git a/nodedb-raft/tests/membership.rs b/nodedb-raft/tests/membership.rs index ae1849ac6..57ecd2089 100644 --- a/nodedb-raft/tests/membership.rs +++ b/nodedb-raft/tests/membership.rs @@ -191,6 +191,8 @@ fn learner_catchup_then_promotion_lifecycle() { term: 1, success: true, last_log_index: idx, + round: 0, + needs_snapshot: false, }; node.handle_append_entries_response(2, &voter_ack); assert_eq!(node.commit_index(), idx); @@ -200,6 +202,8 @@ fn learner_catchup_then_promotion_lifecycle() { term: 1, success: true, last_log_index: idx, + round: 0, + needs_snapshot: false, }; node.handle_append_entries_response(3, &learner_ack); assert_eq!(node.match_index_for(3), Some(idx)); diff --git a/nodedb-sql/Cargo.toml b/nodedb-sql/Cargo.toml index 33cc6d669..f06dc6ab9 100644 --- a/nodedb-sql/Cargo.toml +++ b/nodedb-sql/Cargo.toml @@ -21,3 +21,4 @@ thiserror = { workspace = true } chrono = { workspace = true } sonic-rs = { workspace = true } rust_decimal = { workspace = true } +hex = { workspace = true } diff --git a/nodedb-sql/src/ddl_ast/parse/collection/body.rs b/nodedb-sql/src/ddl_ast/parse/collection/body.rs index 0c6762130..238c21484 100644 --- a/nodedb-sql/src/ddl_ast/parse/collection/body.rs +++ b/nodedb-sql/src/ddl_ast/parse/collection/body.rs @@ -4,6 +4,7 @@ use super::column_list::{extract_column_pairs, find_column_list_paren_end}; use super::engine_suffix::extract_engine_suffix; +use super::flags::extract_flags; use super::with_clause::{extract_balanced_raw, extract_with_options}; use crate::error::SqlError; use nodedb_types::find_ascii_case_insensitive; @@ -77,8 +78,6 @@ pub(super) fn parse_collection_body(trimmed: &str, name: &str) -> Result Result None, }; - let mut flags: Vec = Vec::new(); - if upper_body.contains("APPEND_ONLY") { - flags.push("APPEND_ONLY".to_string()); - } - if upper_body.contains("HASH_CHAIN") { - flags.push("HASH_CHAIN".to_string()); - } - if upper_body.contains("BITEMPORAL") { - flags.push("BITEMPORAL".to_string()); - } - if upper_body.contains("SIGNED_DELTAS") { - flags.push("SIGNED_DELTAS".to_string()); - } + let column_list = if columns.is_empty() { + None + } else { + body.find('(').zip(find_column_list_paren_end(body)) + }; + let flags = extract_flags(body, column_list); let balanced_raw = extract_balanced_raw(body); @@ -131,6 +123,31 @@ pub(super) fn parse_collection_body(trimmed: &str, name: &str) -> Result` sets the flag only for a true value (`true`, `1`, `on`, +//! `yes`, quoted or bare). + +/// Every recognised flag, in the order the parsed list reports them. +const FLAGS: [&str; 4] = ["APPEND_ONLY", "HASH_CHAIN", "BITEMPORAL", "SIGNED_DELTAS"]; + +/// The flags `body` sets. `column_list` is the byte span of the column-list +/// parentheses, both ends inclusive, when the body has one. +pub(super) fn extract_flags(body: &str, column_list: Option<(usize, usize)>) -> Vec { + let bytes = body.as_bytes(); + let mut set = [false; FLAGS.len()]; + let mut i = 0usize; + while i < bytes.len() { + if let Some((start, end)) = column_list + && i == start + { + i = end + 1; + continue; + } + match bytes[i] { + quote @ (b'\'' | b'"' | b'`') => i = skip_quoted(bytes, i, quote), + b if is_ident_byte(b) => { + let start = i; + while i < bytes.len() && is_ident_byte(bytes[i]) { + i += 1; + } + let word = &body[start..i]; + if let Some(pos) = FLAGS.iter().position(|f| word.eq_ignore_ascii_case(f)) { + let (enabled, next) = flag_value(body, i); + set[pos] |= enabled; + i = next; + } + } + _ => i += 1, + } + } + FLAGS + .iter() + .zip(set) + .filter(|(_, enabled)| *enabled) + .map(|(flag, _)| (*flag).to_string()) + .collect() +} + +/// Identifier bytes. Non-ASCII bytes count, so a Unicode identifier that +/// contains a flag's letters stays one token. +fn is_ident_byte(b: u8) -> bool { + b.is_ascii_alphanumeric() || b == b'_' || b >= 0x80 +} + +/// The index after the quoted run opening at `open`. A doubled quote is an +/// escaped quote. An unterminated run ends the body. +fn skip_quoted(bytes: &[u8], open: usize, quote: u8) -> usize { + let mut i = open + 1; + while i < bytes.len() { + if bytes[i] == quote { + if bytes.get(i + 1) == Some("e) { + i += 2; + continue; + } + return i + 1; + } + i += 1; + } + bytes.len() +} + +/// Whether the flag keyword ending at `after` is set, and where scanning +/// resumes. A bare keyword is set. `= value` is set only for a true value. +fn flag_value(body: &str, after: usize) -> (bool, usize) { + let rest = &body[after..]; + let Some(assigned) = rest.trim_start().strip_prefix('=') else { + return (true, after); + }; + let value = assigned.trim_start(); + let value_start = after + (rest.len() - value.len()); + let token = match value.as_bytes().first() { + Some("e @ (b'\'' | b'"')) => { + let end = skip_quoted(value.as_bytes(), 0, quote); + let inner_end = end.saturating_sub(1).max(1); + (value.get(1..inner_end).unwrap_or(""), end) + } + _ => { + let end = value + .bytes() + .position(|b| !is_ident_byte(b)) + .unwrap_or(value.len()); + (&value[..end], end) + } + }; + let (word, consumed) = token; + let enabled = ["true", "1", "on", "yes"] + .iter() + .any(|truthy| word.eq_ignore_ascii_case(truthy)); + (enabled, value_start + consumed) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn flags(body: &str) -> Vec { + extract_flags(body, None) + } + + #[test] + fn trailing_flags_after_a_with_clause_are_set() { + assert_eq!( + flags("(id STRING) WITH (engine='document_schemaless') APPEND_ONLY HASH_CHAIN"), + vec!["APPEND_ONLY", "HASH_CHAIN"] + ); + } + + #[test] + fn a_flag_inside_an_identifier_sets_nothing() { + assert!(flags("WITH (engine='kv') not_append_only_view").is_empty()); + assert!(flags("WITH (engine='kv') hash_chains").is_empty()); + } + + #[test] + fn a_flag_inside_a_literal_or_quoted_identifier_sets_nothing() { + assert!(flags("WITH (note='HASH_CHAIN') \"BITEMPORAL\"").is_empty()); + assert!(flags("WITH (note='it''s APPEND_ONLY')").is_empty()); + } + + #[test] + fn a_column_named_like_a_flag_sets_nothing() { + let body = "(append_only BOOL, hash_chain TEXT) WITH (engine='document_strict')"; + let close = body.find(')').expect("column list"); + assert!(extract_flags(body, Some((0, close))).is_empty()); + } + + #[test] + fn an_assigned_flag_follows_its_value() { + assert_eq!(flags("WITH (bitemporal=true)"), vec!["BITEMPORAL"]); + assert_eq!(flags("WITH (bitemporal = 'TRUE')"), vec!["BITEMPORAL"]); + assert!(flags("WITH (bitemporal=false)").is_empty()); + assert!(flags("WITH (append_only = 'no')").is_empty()); + } +} diff --git a/nodedb-sql/src/ddl_ast/parse/collection/mod.rs b/nodedb-sql/src/ddl_ast/parse/collection/mod.rs index 3b1d8c458..59d804136 100644 --- a/nodedb-sql/src/ddl_ast/parse/collection/mod.rs +++ b/nodedb-sql/src/ddl_ast/parse/collection/mod.rs @@ -15,6 +15,7 @@ mod body; mod column_list; mod dispatcher; mod engine_suffix; +mod flags; mod with_clause; pub(super) use dispatcher::try_parse; diff --git a/nodedb-sql/src/ddl_ast/parse/database/backup_restore.rs b/nodedb-sql/src/ddl_ast/parse/database/backup_restore.rs index 2aa64a322..d9806e3c7 100644 --- a/nodedb-sql/src/ddl_ast/parse/database/backup_restore.rs +++ b/nodedb-sql/src/ddl_ast/parse/database/backup_restore.rs @@ -1,27 +1,52 @@ // SPDX-License-Identifier: Apache-2.0 -//! `BACKUP DATABASE TO ` and `RESTORE DATABASE FROM `. +//! `BACKUP DATABASE TO ''` and +//! `RESTORE DATABASE FROM '' [FORCE] [DRY RUN]`. use crate::ddl_ast::statement::{DatabaseStmt, NodedbStatement}; use crate::error::SqlError; -pub(super) fn parse_backup_database(parts: &[&str]) -> Result { - let name = parts +fn parse_error(detail: impl Into) -> SqlError { + SqlError::Parse { + detail: detail.into(), + } +} + +/// The database name at `parts[2]`. +fn database_name(parts: &[&str], statement: &str) -> Result { + parts .get(2) - .copied() - .ok_or_else(|| SqlError::Parse { - detail: "BACKUP DATABASE requires a name".into(), - })? - .trim_matches('"') - .to_string(); - let to_idx = parts + .map(|name| name.trim_matches('"').to_string()) + .ok_or_else(|| parse_error(format!("{statement} requires a database name"))) +} + +/// The URI that follows the keyword `keyword`, with its quotes removed. +fn uri_after<'a>( + parts: &'a [&'a str], + keyword: &str, + statement: &str, +) -> Result<(String, &'a [&'a str]), SqlError> { + let at = parts .iter() - .position(|w| w.to_uppercase() == "TO") - .ok_or_else(|| SqlError::Parse { - detail: "BACKUP DATABASE requires TO ".into(), - })?; - let uri = parts[to_idx + 1..].join(" ").trim_matches('\'').to_string(); + .position(|w| w.eq_ignore_ascii_case(keyword)) + .ok_or_else(|| parse_error(format!("{statement} requires {keyword} ''")))?; + let uri = parts + .get(at + 1) + .map(|uri| uri.trim_matches('\'').to_string()) + .filter(|uri| !uri.is_empty()) + .ok_or_else(|| parse_error(format!("{statement} requires {keyword} ''")))?; + Ok((uri, parts.get(at + 2..).unwrap_or(&[]))) +} +pub(super) fn parse_backup_database(parts: &[&str]) -> Result { + let statement = "BACKUP DATABASE"; + let name = database_name(parts, statement)?; + let (uri, rest) = uri_after(parts, "TO", statement)?; + if let Some(extra) = rest.first() { + return Err(parse_error(format!( + "{statement}: unexpected '{extra}' after the URI" + ))); + } Ok(NodedbStatement::Database(DatabaseStmt::BackupDatabase { name, uri, @@ -29,27 +54,30 @@ pub(super) fn parse_backup_database(parts: &[&str]) -> Result Result { - let name = parts - .get(2) - .copied() - .ok_or_else(|| SqlError::Parse { - detail: "RESTORE DATABASE requires a name".into(), - })? - .trim_matches('"') - .to_string(); - let from_idx = parts - .iter() - .position(|w| w.to_uppercase() == "FROM") - .ok_or_else(|| SqlError::Parse { - detail: "RESTORE DATABASE requires FROM ".into(), - })?; - let uri = parts[from_idx + 1..] - .join(" ") - .trim_matches('\'') - .to_string(); - + let statement = "RESTORE DATABASE"; + let name = database_name(parts, statement)?; + let (uri, mut rest) = uri_after(parts, "FROM", statement)?; + let (mut force, mut dry_run) = (false, false); + while let Some(word) = rest.first() { + if word.eq_ignore_ascii_case("FORCE") && !force { + force = true; + rest = &rest[1..]; + } else if word.eq_ignore_ascii_case("DRY") + && rest.get(1).is_some_and(|w| w.eq_ignore_ascii_case("RUN")) + && !dry_run + { + dry_run = true; + rest = &rest[2..]; + } else { + return Err(parse_error(format!( + "{statement}: unexpected '{word}' after the URI; expected FORCE or DRY RUN" + ))); + } + } Ok(NodedbStatement::Database(DatabaseStmt::RestoreDatabase { name, uri, + force, + dry_run, })) } diff --git a/nodedb-sql/src/ddl_ast/statement/collection.rs b/nodedb-sql/src/ddl_ast/statement/collection.rs index 9a51f9d6a..420dbb1aa 100644 --- a/nodedb-sql/src/ddl_ast/statement/collection.rs +++ b/nodedb-sql/src/ddl_ast/statement/collection.rs @@ -12,7 +12,7 @@ pub enum CloneAsOf { /// explicit `… AS OF SYSTEM TIME LATEST` form. Latest, /// Use the LSN corresponding to the given milliseconds-since-epoch - /// timestamp, resolved via the `LsnMsAnchor` mechanism. + /// timestamp, resolved from the WAL time anchors. /// /// Corresponds to `… AS OF SYSTEM TIME `. SystemTimeMs(i64), diff --git a/nodedb-sql/src/ddl_ast/statement/types/database.rs b/nodedb-sql/src/ddl_ast/statement/types/database.rs index f3f16c8b7..eafe826fd 100644 --- a/nodedb-sql/src/ddl_ast/statement/types/database.rs +++ b/nodedb-sql/src/ddl_ast/statement/types/database.rs @@ -129,19 +129,20 @@ pub enum DatabaseStmt { /// Filter to a specific mirror by name, or `None` to show all mirrors. name: Option, }, - /// `BACKUP DATABASE TO ` - /// - /// Returns `FEATURE_NOT_YET_IMPLEMENTED` until the backup subsystem lands. + /// `BACKUP DATABASE TO ''`: every tenant's rows in the + /// database, written to an object-store URI. BackupDatabase { name: String, uri: String, }, - /// `RESTORE DATABASE FROM ` - /// - /// Returns `FEATURE_NOT_YET_IMPLEMENTED` until the restore subsystem lands. + /// `RESTORE DATABASE FROM '' [FORCE] [DRY RUN]`. RestoreDatabase { name: String, uri: String, + /// Overwrite writes newer than the backup. + force: bool, + /// Check the backup and write nothing. + dry_run: bool, }, // ── Backup / restore ───────────────────────────────────────── diff --git a/nodedb-sql/src/parser/database_stmt/parse.rs b/nodedb-sql/src/parser/database_stmt/parse.rs index f7616b177..3fb84230e 100644 --- a/nodedb-sql/src/parser/database_stmt/parse.rs +++ b/nodedb-sql/src/parser/database_stmt/parse.rs @@ -226,7 +226,7 @@ mod tests { match ok("BACKUP DATABASE mydb TO 's3://bucket/path'") { NodedbStatement::Database(DatabaseStmt::BackupDatabase { name, uri }) => { assert_eq!(name, "mydb"); - assert!(!uri.is_empty()); + assert_eq!(uri, "s3://bucket/path"); } other => panic!("unexpected: {other:?}"), } @@ -235,12 +235,48 @@ mod tests { #[test] fn parse_restore_database() { match ok("RESTORE DATABASE mydb FROM 's3://bucket/path'") { - NodedbStatement::Database(DatabaseStmt::RestoreDatabase { name, uri }) => { + NodedbStatement::Database(DatabaseStmt::RestoreDatabase { + name, + uri, + force, + dry_run, + }) => { assert_eq!(name, "mydb"); - assert!(!uri.is_empty()); + assert_eq!(uri, "s3://bucket/path"); + assert!(!force && !dry_run); + } + other => panic!("unexpected: {other:?}"), + } + } + + #[test] + fn parse_restore_database_flags() { + match ok("RESTORE DATABASE mydb FROM 'file:///b/x' FORCE DRY RUN") { + NodedbStatement::Database(DatabaseStmt::RestoreDatabase { + uri, + force, + dry_run, + .. + }) => { + assert_eq!(uri, "file:///b/x"); + assert!(force && dry_run); } other => panic!("unexpected: {other:?}"), } + match ok("RESTORE DATABASE mydb FROM 'file:///b/x' DRY RUN") { + NodedbStatement::Database(DatabaseStmt::RestoreDatabase { force, dry_run, .. }) => { + assert!(!force && dry_run); + } + other => panic!("unexpected: {other:?}"), + } + } + + #[test] + fn restore_database_refuses_an_unknown_trailing_word() { + assert!( + try_parse_database_statement("RESTORE DATABASE mydb FROM 'file:///b/x' LATER").is_err() + ); + assert!(try_parse_database_statement("BACKUP DATABASE mydb TO").is_err()); } #[test] diff --git a/nodedb-sql/src/parser/preprocess/literal.rs b/nodedb-sql/src/parser/preprocess/literal.rs index ddcf7ea84..c4f182cd2 100644 --- a/nodedb-sql/src/parser/preprocess/literal.rs +++ b/nodedb-sql/src/parser/preprocess/literal.rs @@ -29,10 +29,7 @@ pub fn value_to_sql_literal(value: &nodedb_types::Value) -> String { let inner: Vec = items.iter().map(value_to_sql_literal).collect(); format!("ARRAY[{}]", inner.join(", ")) } - nodedb_types::Value::Bytes(b) => { - let hex: String = b.iter().map(|byte| format!("{byte:02x}")).collect(); - format!("'\\x{hex}'") - } + nodedb_types::Value::Bytes(b) => format!("'\\x{}'", hex::encode(b)), nodedb_types::Value::Object(map) => { let json = super::function_args::value_map_to_json(map); format!("'{}'", json.replace('\'', "''")) diff --git a/nodedb-sql/src/planner/dml_helpers/ast_extract.rs b/nodedb-sql/src/planner/dml_helpers/ast_extract.rs index a9ffcb61a..4bdcc5d3f 100644 --- a/nodedb-sql/src/planner/dml_helpers/ast_extract.rs +++ b/nodedb-sql/src/planner/dml_helpers/ast_extract.rs @@ -35,47 +35,58 @@ pub fn extract_point_keys(selection: Option<&ast::Expr>, info: &CollectionInfo) }; let mut keys = Vec::new(); - collect_pk_equalities(expr, &pk, &mut keys); - keys + // The keys stand for the WHERE only when they cover every row it matches. + // A disjunct that is no key equality (`id = 'd' OR v < 2`) matches rows no + // key names, so the statement takes the predicate path. + if collect_pk_equalities(expr, &pk, &mut keys) { + keys + } else { + Vec::new() + } } -fn collect_pk_equalities(expr: &ast::Expr, pk: &str, keys: &mut Vec) { +/// Push the keys `expr` names into `keys`. Returns `false` when `expr` matches +/// a row no pushed key names: a disjunct that is no key equality, or a key +/// value that is no literal. +fn collect_pk_equalities(expr: &ast::Expr, pk: &str, keys: &mut Vec) -> bool { match expr { ast::Expr::BinaryOp { left, op: ast::BinaryOperator::Eq, right, } => { - if is_column(left, pk) - && let Ok(v) = expr_to_sql_value(right) - { - keys.push(v); - } else if is_column(right, pk) - && let Ok(v) = expr_to_sql_value(left) - { - keys.push(v); + let value = if is_column(left, pk) { + right + } else if is_column(right, pk) { + left + } else { + return false; + }; + match expr_to_sql_value(value) { + Ok(v) => { + keys.push(v); + true + } + Err(_) => false, } } ast::Expr::BinaryOp { left, op: ast::BinaryOperator::Or, right, - } => { - collect_pk_equalities(left, pk, keys); - collect_pk_equalities(right, pk, keys); - } + } => collect_pk_equalities(left, pk, keys) && collect_pk_equalities(right, pk, keys), ast::Expr::InList { expr: inner, list, negated: false, - } if is_column(inner, pk) => { - for item in list { - if let Ok(v) = expr_to_sql_value(item) { - keys.push(v); - } + } if is_column(inner, pk) => list.iter().all(|item| match expr_to_sql_value(item) { + Ok(v) => { + keys.push(v); + true } - } - _ => {} + Err(_) => false, + }), + _ => false, } } diff --git a/nodedb-sql/src/planner/dml_helpers/vector_primary_insert.rs b/nodedb-sql/src/planner/dml_helpers/vector_primary_insert.rs index 08d88e256..f5df1f448 100644 --- a/nodedb-sql/src/planner/dml_helpers/vector_primary_insert.rs +++ b/nodedb-sql/src/planner/dml_helpers/vector_primary_insert.rs @@ -95,7 +95,6 @@ pub(crate) fn build_vector_primary_insert_plan( })?; result_rows.push(VectorPrimaryRow { - surrogate: nodedb_types::Surrogate::ZERO, vector, payload_fields, }); diff --git a/nodedb-sql/src/types/plan/row_types.rs b/nodedb-sql/src/types/plan/row_types.rs index 6d0cb2dc8..57a7c35aa 100644 --- a/nodedb-sql/src/types/plan/row_types.rs +++ b/nodedb-sql/src/types/plan/row_types.rs @@ -6,13 +6,10 @@ use crate::types_expr::SqlValue; /// A single row for a vector-primary INSERT. /// -/// The surrogate is allocated by the Control Plane before the op reaches -/// the Data Plane; the Data Plane only stores the binding. +/// The row carries no surrogate: the Control Plane binds one from the row's +/// primary key when it converts the plan. #[derive(Debug, Clone)] pub struct VectorPrimaryRow { - /// Global surrogate allocated by the Control Plane (`Surrogate::ZERO` - /// is a sentinel meaning "not yet assigned"). - pub surrogate: nodedb_types::Surrogate, /// FP32 vector extracted from the vector-field column. pub vector: Vec, /// Payload fields (non-vector columns that may feed bitmap indexes). diff --git a/nodedb-test-support/src/array_sync.rs b/nodedb-test-support/src/array_sync.rs index 356ed3249..83da01d6d 100644 --- a/nodedb-test-support/src/array_sync.rs +++ b/nodedb-test-support/src/array_sync.rs @@ -18,18 +18,15 @@ use nodedb_array::types::ArrayId; use nodedb_array::types::domain::{Domain, DomainBound}; use nodedb_types::TenantId; -#[allow(dead_code)] pub fn rep(id: u64) -> ReplicaId { ReplicaId::new(id) } -#[allow(dead_code)] pub fn hlc(ms: u64, rid: u64) -> Hlc { Hlc::new(ms, 0, rep(rid)).expect("valid HLC") } /// One-dimensional Int64 schema over [0, 99] with attribute "v" (Float64). -#[allow(dead_code)] pub fn simple_schema(name: &str) -> ArraySchema { ArraySchema { name: name.into(), @@ -47,7 +44,6 @@ pub fn simple_schema(name: &str) -> ArraySchema { /// Build a real Loro snapshot for `array_name` and return /// `(snapshot_bytes, schema_hlc)` suitable for `import_snapshot` calls. -#[allow(dead_code)] pub fn build_schema_snapshot(array_name: &str) -> (Vec, Hlc) { let hlc_gen = HlcGenerator::new(rep(1)); let schema = simple_schema(array_name); @@ -59,7 +55,6 @@ pub fn build_schema_snapshot(array_name: &str) -> (Vec, Hlc) { /// Import a pre-built snapshot on `shared`. Use this with a single /// `(bytes, schema_hlc)` tuple shared across all nodes so every replica /// observes the same remote HLC. -#[allow(dead_code)] pub fn import_schema_snapshot( shared: &Arc, array_name: &str, @@ -74,7 +69,6 @@ pub fn import_schema_snapshot( /// Register an array catalog entry on `shared` so the Data Plane can open /// the array when applying ops. Independent of schema-CRDT registration. -#[allow(dead_code)] pub fn register_catalog_entry(shared: &Arc, array_name: &str) { let schema = simple_schema(array_name); let schema_msgpack = zerompk::to_msgpack_vec(&schema).expect("encode schema"); @@ -87,6 +81,8 @@ pub fn register_catalog_entry(shared: &Arc, array_name: &str) { prefix_bits: 8, audit_retain_ms: None, minimum_audit_retain_ms: None, + modification_hlc: nodedb_types::Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, }; let mut cat = shared.array_catalog.write().expect("array catalog lock"); if cat.lookup_by_name(array_name).is_none() { @@ -97,7 +93,6 @@ pub fn register_catalog_entry(shared: &Arc, array_name: &str) { /// Build one snapshot and import it on every node, returning the shared /// schema HLC. Use this HLC when stamping ops so the schema-gating check /// in `OriginArrayInbound` accepts them on every replica. -#[allow(dead_code)] pub fn register_schema_on_all(shareds: &[&Arc], array_name: &str) -> Hlc { let (bytes, schema_hlc) = build_schema_snapshot(array_name); for shared in shareds { @@ -106,15 +101,3 @@ pub fn register_schema_on_all(shareds: &[&Arc], array_name: &str) - } schema_hlc } - -/// Register a single node with a freshly built snapshot. Used by tests that -/// add a new node mid-flight (e.g. snapshot-install learner). Returns the -/// schema HLC, but tests should typically use the HLC produced by the -/// initial `register_schema_on_all` call instead. -#[allow(dead_code)] -pub fn register_schema(shared: &Arc, array_name: &str) -> Hlc { - let (bytes, schema_hlc) = build_schema_snapshot(array_name); - import_schema_snapshot(shared, array_name, &bytes, schema_hlc); - register_catalog_entry(shared, array_name); - schema_hlc -} diff --git a/nodedb-test-support/src/booted_state.rs b/nodedb-test-support/src/booted_state.rs new file mode 100644 index 000000000..68984ba5d --- /dev/null +++ b/nodedb-test-support/src/booted_state.rs @@ -0,0 +1,284 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A `SharedState` served by a one-node cluster that runs on its own thread. +//! +//! Some tests drive the Control Plane directly: they call DDL dispatch, an +//! HTTP router, or a listener's accept loop against a `SharedState`. A DDL +//! proposes through the metadata group, so the state needs a booted cluster. +//! [`BootedState::boot`] boots one through [`crate::single_node`], with a Data +//! Plane core, a response poller and, unless the options turn it off, an +//! Event Plane, on a dedicated thread +//! and runtime. The call is synchronous, so a sync fixture can return it. +//! Dropping the handle stops the node and removes its directory. + +use std::sync::Arc; +use std::time::Duration; + +use nodedb::bridge::dispatch::Dispatcher; +use nodedb::control::state::SharedState; +use nodedb::event::{EventPlane, EventPlaneConfig, create_event_bus}; +use nodedb::wal::WalManager; + +use crate::single_node::{BootError, OneNodeRaft}; + +/// Stack size of the node runtime's worker threads. The metadata applier and +/// the post-apply steps run there, with deep call stacks in debug builds. +const NODE_THREAD_STACK: usize = 16 * 1024 * 1024; + +/// How the booted node's state is set up before anything shares it. +pub struct BootOptions { + /// Create the built-in `default` database in the catalog. + pub default_database: bool, + /// Spawn the node's Event Plane. A node runs exactly one Event Plane: a + /// test that spawns its own plane on the state boots with `false`, so its + /// plane is the node's only one. + pub event_plane: bool, + /// Applied to the state before the gateway install. The installed + /// gateway holds a back-reference, so no field is settable afterward. + pub configure: Box, +} + +impl Default for BootOptions { + fn default() -> Self { + Self { + default_database: true, + event_plane: true, + configure: Box::new(|_| {}), + } + } +} + +/// A running one-node cluster and the state it serves. +pub struct BootedState { + state: Arc, + stop_tx: Option>, + thread: Option>, +} + +impl std::ops::Deref for BootedState { + type Target = Arc; + + fn deref(&self) -> &Self::Target { + &self.state + } +} + +impl BootedState { + /// Boot a one-node cluster in a fresh directory and return its state + /// once it serves requests. + /// + /// Panics when the boot fails. + pub fn boot(options: BootOptions) -> Self { + let (state_tx, state_rx) = std::sync::mpsc::channel::, String>>(); + let (stop_tx, stop_rx) = tokio::sync::oneshot::channel::<()>(); + let thread = std::thread::Builder::new() + .name("booted-state".into()) + .spawn(move || run_node(options, state_tx, stop_rx)) + .expect("spawn the booted-state thread"); + let state = match state_rx.recv() { + Ok(Ok(state)) => state, + Ok(Err(error)) => panic!("one-node cluster boot failed: {error}"), + Err(_) => panic!("one-node cluster boot thread exited before it booted"), + }; + Self { + state, + stop_tx: Some(stop_tx), + thread: Some(thread), + } + } +} + +impl Drop for BootedState { + fn drop(&mut self) { + if let Some(stop_tx) = self.stop_tx.take() { + let _ = stop_tx.send(()); + } + if let Some(thread) = self.thread.take() { + let _ = thread.join(); + } + } +} + +/// Everything the node thread stops on shutdown. +struct Node { + state: Arc, + shutdown_bus: nodedb::control::shutdown::ShutdownBus, + poller_shutdown_tx: tokio::sync::watch::Sender, + poller: tokio::task::JoinHandle<()>, + core_stop_tx: std::sync::mpsc::Sender<()>, + core: tokio::task::JoinHandle<()>, + /// `None` when the test spawns the node's Event Plane itself. + event_plane: Option, + raft: OneNodeRaft, + _dir: tempfile::TempDir, +} + +/// The node thread: boot, hand the state over, wait for the stop signal, +/// then shut down. +fn run_node( + options: BootOptions, + state_tx: std::sync::mpsc::Sender, String>>, + stop_rx: tokio::sync::oneshot::Receiver<()>, +) { + let runtime = match tokio::runtime::Builder::new_multi_thread() + .worker_threads(2) + .thread_stack_size(NODE_THREAD_STACK) + .enable_all() + .build() + { + Ok(runtime) => runtime, + Err(error) => { + let _ = state_tx.send(Err(format!("build the node runtime: {error}"))); + return; + } + }; + runtime.block_on(async move { + let node = match boot_node(options).await { + Ok(node) => node, + Err(error) => { + let _ = state_tx.send(Err(error.to_string())); + return; + } + }; + if state_tx.send(Ok(Arc::clone(&node.state))).is_err() { + node.shutdown().await; + return; + } + // A dropped sender stops the node too. + let _ = stop_rx.await; + node.shutdown().await; + }); +} + +async fn boot_node(options: BootOptions) -> Result { + let BootOptions { + default_database, + event_plane: spawn_event_plane, + configure, + } = options; + let dir = tempfile::tempdir()?; + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("test.wal"))?); + let credentials = Arc::new( + nodedb::control::security::credential::store::CredentialStore::open( + &dir.path().join("system.redb"), + )?, + ); + if default_database { + credentials.catalog().bootstrap_default_database()?; + } + let (dispatcher, data_sides) = Dispatcher::new(1, 64); + let (event_producers, event_consumers) = create_event_bus(1); + let mut state = + SharedState::new_with_credentials(dispatcher, Arc::clone(&wal), credentials, false)?; + let cluster = crate::single_node::init(dir.path()).await?; + { + let wired = Arc::get_mut(&mut state).ok_or("shared state is cloned before wiring")?; + crate::single_node::wire(wired, &cluster, dir.path())?; + configure(wired); + } + nodedb::bootstrap::state_wiring::install_gateway(&state)?; + + let data_side = data_sides.into_iter().next().ok_or("no data side")?; + let event_producer = event_producers + .into_iter() + .next() + .ok_or("no event producer")?; + let (core_stop_tx, core_stop_rx) = std::sync::mpsc::channel::<()>(); + let core = crate::core_loop_runner::spawn_core_loop(crate::core_loop_runner::CoreLoopSpawn { + idx: 0, + num_cores: 1, + data_side, + core_dir: dir.path().to_path_buf(), + core_array_catalog: state.array_catalog.clone(), + event_producer, + core_metrics: state.system_metrics.clone(), + governor: state.governor.clone(), + replay: None, + graph_tuning: nodedb_types::config::tuning::GraphTuning::default(), + query_tuning: nodedb_types::config::tuning::QueryTuning::default(), + timeseries_tuning: nodedb_types::config::tuning::TimeseriesToning::default(), + doc_config_seed: nodedb::bootstrap::data_plane::load_doc_config_registry_from( + state.credentials.catalog(), + ), + event_interest: crate::core_loop_runner::event_interest_for(&state), + stop_rx: core_stop_rx, + }); + + let poll_state = Arc::clone(&state); + let (poller_shutdown_tx, mut poller_shutdown_rx) = tokio::sync::watch::channel(false); + let poller = tokio::spawn(async move { + loop { + poll_state.poll_and_route_responses(); + tokio::select! { + _ = tokio::time::sleep(Duration::from_millis(1)) => {} + _ = poller_shutdown_rx.changed() => break, + } + } + }); + + let (shutdown_bus, _) = + nodedb::control::shutdown::ShutdownBus::new(Arc::clone(&state.shutdown)); + let event_plane = if spawn_event_plane { + let watermark_store = Arc::new(nodedb::event::watermark::WatermarkStore::open(dir.path())?); + let trigger_dlq = Arc::new(std::sync::Mutex::new( + nodedb::event::trigger::TriggerDlq::open(dir.path())?, + )); + Some(EventPlane::spawn(EventPlaneConfig { + consumers_rx: event_consumers, + wal: Arc::clone(&wal), + watermark_store, + shared_state: Arc::clone(&state), + trigger_dlq, + cdc_router: Arc::clone(&state.cdc_router), + shutdown: Arc::clone(&state.shutdown), + shutdown_bus: shutdown_bus.clone(), + })) + } else { + // The core's events have no consumer here. The test's own plane + // consumes the events it emits on its own bus. + drop(event_consumers); + None + }; + + let raft = crate::single_node::start(&cluster, &state, dir.path()).await?; + Ok(Node { + state, + shutdown_bus, + poller_shutdown_tx, + poller, + core_stop_tx, + core, + event_plane, + raft, + _dir: dir, + }) +} + +impl Node { + async fn shutdown(self) { + let Node { + state, + shutdown_bus, + poller_shutdown_tx, + poller, + core_stop_tx, + core, + event_plane, + mut raft, + _dir, + } = self; + shutdown_bus.initiate(); + let _ = poller_shutdown_tx.send(true); + let _ = poller.await; + let _ = core_stop_tx.send(()); + let _ = core.await; + if let Some(event_plane) = event_plane { + event_plane.shutdown_and_join().await; + } + state + .loop_registry + .shutdown_all(Duration::from_secs(5)) + .await; + raft.shutdown(&state).await; + } +} diff --git a/nodedb-test-support/src/catalog_fixtures.rs b/nodedb-test-support/src/catalog_fixtures.rs new file mode 100644 index 000000000..9a87aa37d --- /dev/null +++ b/nodedb-test-support/src/catalog_fixtures.rs @@ -0,0 +1,25 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Catalog rows an integration test writes straight to the catalog. +//! +//! The catalog refuses a collection row with no incarnation. A fixture that +//! bypasses the proposer builds its row here, stamped as the proposer stamps +//! a create. + +use std::sync::OnceLock; + +use nodedb::control::security::catalog::StoredCollection; +use nodedb_types::HlcClock; + +/// A collection stamped as the proposer stamps a create: a fresh incarnation +/// and descriptor version 1. Each call names a new incarnation. +pub fn stamped_collection(tenant_id: u64, name: &str, owner: &str) -> StoredCollection { + static CLOCK: OnceLock = OnceLock::new(); + let hlc = CLOCK.get_or_init(HlcClock::new).now(); + StoredCollection { + descriptor_version: 1, + modification_hlc: hlc, + incarnation: hlc, + ..StoredCollection::new(tenant_id, name, owner) + } +} diff --git a/nodedb-test-support/src/cluster_harness/cluster/bringup.rs b/nodedb-test-support/src/cluster_harness/cluster/bringup.rs index cda7141b3..d1bea0ea7 100644 --- a/nodedb-test-support/src/cluster_harness/cluster/bringup.rs +++ b/nodedb-test-support/src/cluster_harness/cluster/bringup.rs @@ -8,7 +8,7 @@ use std::time::Duration; use nodedb_types::config::tuning::ClusterTransportTuning; use super::TestCluster; -use super::types::ClusterSpawnConfig; +use super::types::{ClusterSpawnConfig, DEFAULT_NUM_GROUPS}; use crate::cluster_harness::node::TestClusterNode; impl TestCluster { @@ -31,9 +31,136 @@ impl TestCluster { num_cores, log_compaction_threshold, replication_factor, + num_groups: DEFAULT_NUM_GROUPS, single_node_calvin: false, + backup_storage: None, + pitr: None, + node_timeseries_tuning: std::collections::HashMap::new(), }; + Self::spawn_three_with_config(config).await + } + + /// Spawn a 3-node cluster with `num_groups` data groups, a low Raft + /// `log_compaction_threshold` and `replication_factor`. More groups give + /// the rendezvous placement more distinct replica sets to choose from. + pub async fn spawn_three_with_groups_compaction_threshold_and_rf( + num_groups: u64, + threshold: u64, + replication_factor: usize, + ) -> Result> { + Self::spawn_three_with_config(ClusterSpawnConfig { + tuning: super::types::fast_cluster_tuning(), + graph_tuning: nodedb_types::config::tuning::GraphTuning::default(), + query_tuning: nodedb_types::config::tuning::QueryTuning::default(), + num_cores: 1, + log_compaction_threshold: Some(threshold), + replication_factor, + num_groups, + single_node_calvin: false, + backup_storage: None, + pitr: None, + node_timeseries_tuning: std::collections::HashMap::new(), + }) + .await + } + + /// Spawn a 3-node cluster with `num_groups` data groups, + /// `replication_factor`, and `num_cores` Data-Plane cores per node. + pub async fn spawn_three_with_groups_rf_and_cores( + num_groups: u64, + replication_factor: usize, + num_cores: usize, + ) -> Result> { + Self::spawn_three_with_config(ClusterSpawnConfig { + tuning: super::types::fast_cluster_tuning(), + graph_tuning: nodedb_types::config::tuning::GraphTuning::default(), + query_tuning: nodedb_types::config::tuning::QueryTuning::default(), + num_cores, + log_compaction_threshold: None, + replication_factor, + num_groups, + single_node_calvin: false, + backup_storage: None, + pitr: None, + node_timeseries_tuning: std::collections::HashMap::new(), + }) + .await + } + + /// Spawn a 3-node cluster where node `node_id` runs `timeseries_tuning` + /// and the other nodes run the default. Uses the standard fast-election + /// tuning and 1 core per node. + pub async fn spawn_three_with_node_timeseries_tuning( + node_id: u64, + timeseries_tuning: nodedb_types::config::tuning::TimeseriesToning, + ) -> Result> { + Self::spawn_three_with_config(ClusterSpawnConfig { + tuning: super::types::fast_cluster_tuning(), + graph_tuning: nodedb_types::config::tuning::GraphTuning::default(), + query_tuning: nodedb_types::config::tuning::QueryTuning::default(), + num_cores: 1, + log_compaction_threshold: None, + replication_factor: 3, + num_groups: DEFAULT_NUM_GROUPS, + single_node_calvin: false, + backup_storage: None, + pitr: None, + node_timeseries_tuning: std::collections::HashMap::from([(node_id, timeseries_tuning)]), + }) + .await + } + + /// Spawn a 3-node cluster whose nodes share one `[backup_storage]` + /// `local_root`, so a `file://` backup URI names the same file on every + /// node. Uses the standard fast-election tuning and 1 core per node. + pub async fn spawn_three_with_backup_root( + local_root: std::path::PathBuf, + ) -> Result> { + Self::spawn_three_with_config(ClusterSpawnConfig { + tuning: super::types::fast_cluster_tuning(), + graph_tuning: nodedb_types::config::tuning::GraphTuning::default(), + query_tuning: nodedb_types::config::tuning::QueryTuning::default(), + num_cores: 1, + log_compaction_threshold: None, + replication_factor: 3, + num_groups: DEFAULT_NUM_GROUPS, + single_node_calvin: false, + backup_storage: Some(nodedb::config::server::BackupStorageSettings { + local_root: Some(local_root), + ..Default::default() + }), + pitr: None, + node_timeseries_tuning: std::collections::HashMap::new(), + }) + .await + } + /// Spawn a 3-node cluster whose nodes share `pitr`'s cold store, + /// snapshot store and WAL key. Uses the standard fast-election tuning and + /// 1 core per node. + pub async fn spawn_three_with_pitr( + pitr: crate::cluster_harness::pitr::PitrStorage, + ) -> Result> { + Self::spawn_three_with_config(ClusterSpawnConfig { + tuning: super::types::fast_cluster_tuning(), + graph_tuning: nodedb_types::config::tuning::GraphTuning::default(), + query_tuning: nodedb_types::config::tuning::QueryTuning::default(), + num_cores: 1, + log_compaction_threshold: None, + replication_factor: 3, + num_groups: DEFAULT_NUM_GROUPS, + single_node_calvin: false, + backup_storage: None, + pitr: Some(pitr), + node_timeseries_tuning: std::collections::HashMap::new(), + }) + .await + } + + /// Spawn node 1, then nodes 2 and 3 joining it, all with `config`. + async fn spawn_three_with_config( + config: ClusterSpawnConfig, + ) -> Result> { let node1 = TestClusterNode::spawn_with_full_config(1, vec![], &config).await?; // Wait until node 1 has bootstrapped (topology shows itself) @@ -71,7 +198,9 @@ impl TestCluster { spawn_config: config, }; - cluster.await_ready().await; + cluster + .await_ready(super::ready::LeaderBar::Preferred) + .await; Ok(cluster) } diff --git a/nodedb-test-support/src/cluster_harness/cluster/mod.rs b/nodedb-test-support/src/cluster_harness/cluster/mod.rs index dcafb7432..63fd6d1e4 100644 --- a/nodedb-test-support/src/cluster_harness/cluster/mod.rs +++ b/nodedb-test-support/src/cluster_harness/cluster/mod.rs @@ -15,5 +15,6 @@ mod restart; mod spawn_variants; mod types; -pub(crate) use types::ClusterSpawnConfig; +pub use restart::{StoppedCluster, StoppedMember, StoppedNodeInfo}; pub use types::TestCluster; +pub(crate) use types::{ClusterSpawnConfig, DEFAULT_NUM_GROUPS}; diff --git a/nodedb-test-support/src/cluster_harness/cluster/ready.rs b/nodedb-test-support/src/cluster_harness/cluster/ready.rs index 0dca12245..b4b0c1cd1 100644 --- a/nodedb-test-support/src/cluster_harness/cluster/ready.rs +++ b/nodedb-test-support/src/cluster_harness/cluster/ready.rs @@ -1,19 +1,33 @@ // SPDX-License-Identifier: BUSL-1.1 //! The convergence barriers a cluster passes before a test issues anything: -//! topology size, rolling-upgrade compat-mode exit, metadata-group leader -//! stability, and per-group Raft leader stability. A fresh bringup and an -//! in-place restart both wait here. +//! topology size, metadata-group leader stability, and data groups settled +//! on their placement and leader. A fresh bringup and an in-place restart +//! both wait here. +use std::collections::HashMap; use std::time::Duration; use super::TestCluster; -use crate::cluster_harness::wait::wait_for; +use crate::cluster_harness::TestClusterNode; +use crate::cluster_harness::wait::{wait_for, wait_for_report}; + +/// Which leader a settled data group must have. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum LeaderBar { + /// The group's preferred leader. The leader balance moves every group's + /// leader there as the cluster forms. + Preferred, + /// Any leader. A restart elects its leaders against other voters, and + /// the leader balance holds such a win for ten election timeouts. + Elected, +} impl TestCluster { - /// Wait until every node agrees on the topology, every node left compat - /// mode, and every Raft group has one leader every node sees. - pub(super) async fn await_ready(&self) { + /// Wait until every node agrees on the topology, the metadata group has + /// one leader every node sees, and every data group has settled with a + /// leader that meets `bar`. + pub(super) async fn await_ready(&self, bar: LeaderBar) { let node_count = self.nodes.len(); wait_for( "every node reports the full topology", @@ -23,49 +37,11 @@ impl TestCluster { ) .await; - // CRITICAL: wait for every node to exit rolling-upgrade - // compat mode before letting the test issue any DDL. - // - // `metadata_proposer::propose_catalog_entry` consults - // `cluster_version_view().can_activate_feature(DISTRIBUTED_CATALOG_VERSION)` - // and, while even one node still reports a lower wire - // version, returns `Ok(0)` without going through the raft - // group. The pgwire DDL handlers (CREATE USER, etc.) then - // fall through to a LEGACY path that writes the record - // directly on the proposing node — **with zero - // replication** to followers. Any subsequent - // `has_active_user` check on a follower returns false and - // the test flakes. - // - // Topology has three members the moment the join request - // completes, but the `wire_version` field on each node's - // topology entry is updated asynchronously by the gossip - // path. That's why `topology_size == 3` converges fast yet - // `can_activate_feature(...)` can still be false for - // several hundred milliseconds afterwards. Waiting here - // closes the window deterministically — no retries, no - // flakes, no compat-mode fallback silently breaking - // replication. - wait_for( - "every node exits rolling-upgrade compat mode", - Duration::from_secs(30), - Duration::from_millis(20), - || { - self.nodes.iter().all(|n| { - n.shared.cluster_version_view().can_activate_feature( - nodedb::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION, - ) - }) - }, - ) - .await; - // CRITICAL: wait for the metadata Raft group to elect a leader // and for every node's local view to agree on the same leader id. // - // Topology convergence + rolling-upgrade exit only guarantees - // membership and wire version are agreed; they say nothing about - // election state. Under heavy host load (e.g. running this test + // Topology convergence only guarantees membership is agreed; it + // says nothing about election state. Under heavy host load (e.g. running this test // immediately after another full-suite cluster test exits and // the unit-test pool ramps back up), the initial Raft heartbeat // window can be missed and the first `acquire`/`propose` issued @@ -74,8 +50,7 @@ impl TestCluster { // descriptor-lease or DDL call. // // Waiting until every node reports the same non-zero leader id - // closes the window deterministically. Symmetric to the - // rolling-upgrade wait above: no retries, no flakes, no + // closes the window deterministically: no retries, no flakes, no // wasted CI minutes on cleanup of a doomed cluster bringup. wait_for( "metadata group has stable leader visible on every node", @@ -93,70 +68,205 @@ impl TestCluster { ) .await; - // CRITICAL: wait for EVERY data Raft group to elect a stable - // leader visible on every node. Without this barrier, the - // first data-group write after `spawn_three()` returns can - // race a still-electing group: + // CRITICAL: wait for EVERY data Raft group to settle before the + // test issues anything. Without this barrier, the first data-group + // write can race a still-electing group: a proposer that thinks it + // leads gets a `log_index` that never commits, and an unrelated + // entry at that index wakes its waiter. // - // 1. Proposer's local `propose()` runs on a node that thinks - // it's leader (stale routing-table hint), gets an Ok back - // with a `log_index` that was never actually committed. - // 2. `ProposeTracker::register((group_id, log_index))`. - // 3. Some unrelated entry that *does* commit at that index - // (e.g., a leadership-change no-op) fires `tracker.complete`, - // waking the waiter with `Ok([])` even though the user's - // `INSERT` row was never replicated. - // 4. `simple_query` returns success; the row is permanently - // lost. + // A data group has settled once: + // - its replicas are exactly its placement, as every replica's + // routing view records it, and no other node hosts a replica. + // With a replication factor below the node count, the nodes + // outside a group's placement host none of it; + // - every replica's Raft names one leader that meets `bar`; + // - every replica's routing hint names that leader, and every + // other node's hint names a replica; + // - every node's routing view lists the replicas as the group's + // voters, with no learners. A node outside the group learns its + // membership from the group leader's answer to its leader probe. // - // The metadata-group-only wait above is insufficient because - // data groups elect independently and lag the metadata group - // by hundreds of milliseconds under load. Waiting until every - // group on every node reports a non-zero leader closes the - // window deterministically. - wait_for( - "every Raft group has a stable leader visible on every node", + // The Calvin sequencer group is not part of the routing topology. + // Calvin tests gate on it separately (`wait_for_sequencer_leader`). + wait_for_report( + "every data group has settled on its placement and leader", Duration::from_secs(30), Duration::from_millis(20), - || { - // Snapshot every node's per-group leader view. A group - // is "ready" iff every node reports the same non-zero - // leader for it. - let per_node: Vec> = - self.nodes.iter().map(|n| n.all_group_leaders()).collect(); - if per_node.iter().any(|v| v.is_empty()) { - return false; - } - // The Calvin sequencer group is an internal Raft group that is - // not part of the data/metadata routing topology. Cluster - // readiness for data operations does not depend on it, and its - // leader is surfaced to the observer on a slower/independent path - // than the routing groups — so gating general cluster startup on - // it makes every test (Calvin or not) flake when the sequencer - // group's observed leader lags. Calvin tests gate on the - // sequencer separately (`wait_for_sequencer_leader`). Exclude it - // from the general readiness gate. - let group_ids: std::collections::BTreeSet = per_node - .iter() - .flat_map(|v| v.iter().map(|(gid, _)| *gid)) - .filter(|gid| *gid != nodedb_cluster::calvin::SEQUENCER_GROUP_ID) - .collect(); - if group_ids.is_empty() { - return false; - } - group_ids.iter().all(|gid| { - let leaders: Vec = per_node - .iter() - .filter_map(|v| v.iter().find(|(g, _)| g == gid).map(|(_, l)| *l)) - .collect(); - if leaders.len() != per_node.len() { - return false; - } - let first = leaders[0]; - first != 0 && leaders.iter().all(|&l| l == first) - }) - }, + || self.data_groups_settled(bar), ) .await; } + + /// Every data group's leader, as a replica's Raft reports it: + /// `group_id → leader`. A group no replica knows a leader of is absent. + /// + /// No single node answers this. With a replication factor below the + /// node count, a node hosts only the groups placed on it. + pub fn data_group_leaders(&self) -> HashMap { + let mut leaders = HashMap::new(); + for node in &self.nodes { + for (group_id, leader) in node.all_group_leaders() { + if group_id == nodedb_cluster::METADATA_GROUP_ID + || group_id == nodedb_cluster::calvin::SEQUENCER_GROUP_ID + || leader == 0 + || !node.replicates_data_group(group_id) + { + continue; + } + leaders.entry(group_id).or_insert(leader); + } + } + leaders + } + + /// `Ok` once every data group has settled, as [`Self::await_ready`] + /// describes. `Err` names each group and node that has not, and why. + fn data_groups_settled(&self, bar: LeaderBar) -> Result<(), String> { + let routing = self + .nodes + .first() + .and_then(|n| n.shared.cluster_routing.as_ref()) + .ok_or("node 1 has no routing table")?; + let group_ids: Vec = routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_ids() + .into_iter() + .filter(|gid| { + *gid != nodedb_cluster::METADATA_GROUP_ID + && *gid != nodedb_cluster::calvin::SEQUENCER_GROUP_ID + }) + .collect(); + if group_ids.is_empty() { + return Err("node 1's routing table holds no data group".into()); + } + let reasons: Vec = group_ids + .into_iter() + .filter_map(|gid| { + self.data_group_settled(gid, bar) + .err() + .map(|why| format!("group {gid}: {why}")) + }) + .collect(); + if reasons.is_empty() { + Ok(()) + } else { + Err(reasons.join("; ")) + } + } + + fn data_group_settled(&self, group_id: u64, bar: LeaderBar) -> Result<(), String> { + let replicas: Vec<&TestClusterNode> = self + .nodes + .iter() + .filter(|n| n.replicates_data_group(group_id)) + .collect(); + if replicas.is_empty() { + return Err("no node replicates it".into()); + } + let stale: Vec = self + .nodes + .iter() + .filter(|n| n.hosts_data_group(group_id) && !n.replicates_data_group(group_id)) + .map(|n| n.node_id) + .collect(); + if !stale.is_empty() { + return Err(format!("nodes {stale:?} host a replica they left")); + } + let mut replica_ids: Vec = replicas.iter().map(|n| n.node_id).collect(); + replica_ids.sort_unstable(); + + let leader = raft_leader(replicas[0], group_id); + if leader == 0 { + return Err(format!( + "replica {} knows no leader; replicas {replica_ids:?}", + replicas[0].node_id + )); + } + for replica in &replicas { + let node = replica.node_id; + // Raft status locks `MultiRaft`, which reads routing under that lock. + // It must be read before this routing guard is taken. + let raft = raft_leader(replica, group_id); + let routing = replica + .shared + .cluster_routing + .as_ref() + .ok_or(format!("node {node} has no routing table"))?; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let info = routing + .group_info(group_id) + .ok_or(format!("node {node} has no routing entry"))?; + let mut placement = routing.effective_placement(group_id); + placement.sort_unstable(); + let mut members = info.members.clone(); + members.sort_unstable(); + if placement != replica_ids || members != replica_ids || !info.learners.is_empty() { + return Err(format!( + "node {node}: replicas {replica_ids:?}, placement {placement:?}, members \ + {members:?}, learners {:?}", + info.learners + )); + } + if raft != leader || info.leader != leader { + return Err(format!( + "node {node}: raft leader {raft}, hint ({}, term {}), replica {} names {leader}", + info.leader, info.leader_term, replicas[0].node_id + )); + } + if bar == LeaderBar::Preferred { + let preferred = nodedb_cluster::rebalancer::preferred_leaders(&routing) + .get(&group_id) + .copied(); + if preferred != Some(leader) { + return Err(format!( + "node {node}: leader {leader}, preferred leader {preferred:?}" + )); + } + } + } + for node in &self.nodes { + let view = node.shared.cluster_routing.as_ref().and_then(|routing| { + routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_info(group_id) + .map(|info| { + let mut members = info.members.clone(); + members.sort_unstable(); + ( + info.leader, + info.leader_term, + members, + info.learners.clone(), + ) + }) + }); + let Some((leader, term, members, learners)) = view else { + return Err(format!("node {}: no routing entry", node.node_id)); + }; + if !replica_ids.contains(&leader) { + return Err(format!( + "node {}: hint ({leader}, term {term}) names no replica of {replica_ids:?}", + node.node_id + )); + } + if members != replica_ids || !learners.is_empty() { + return Err(format!( + "node {}: members {members:?}, learners {learners:?}, replicas \ + {replica_ids:?}", + node.node_id + )); + } + } + Ok(()) + } +} + +/// The leader of `group_id` as `node`'s Raft reports it, `0` when none. +fn raft_leader(node: &TestClusterNode, group_id: u64) -> u64 { + node.all_group_leaders() + .into_iter() + .find(|&(group, _)| group == group_id) + .map_or(0, |(_, leader)| leader) } diff --git a/nodedb-test-support/src/cluster_harness/cluster/restart.rs b/nodedb-test-support/src/cluster_harness/cluster/restart.rs index a46d6ea2e..b80566f7d 100644 --- a/nodedb-test-support/src/cluster_harness/cluster/restart.rs +++ b/nodedb-test-support/src/cluster_harness/cluster/restart.rs @@ -1,30 +1,60 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Restart every node of a [`TestCluster`] in place. +//! Stop every node of a [`TestCluster`] and start each again in place, or +//! stop one member and bring it back later. use super::TestCluster; +use super::types::ClusterSpawnConfig; use crate::cluster_harness::node::TestClusterNode; +use crate::cluster_harness::node::lifecycle::StoppedNode; -impl TestCluster { - /// Stop every node, then bring every one back on its node id, listen - /// address and data directory, and wait until the cluster is ready. - /// - /// Every node stops before any restarts, so nothing survives in memory: - /// what each node serves afterwards comes from its own disk. The nodes - /// restart together, since each Raft group needs a quorum to elect. - pub async fn restart_all(self) -> Result> { - let TestCluster { +/// A member stopped by [`TestCluster::stop_member`]. It keeps the member's +/// data directory until [`TestCluster::restart_member`] brings it back. +pub struct StoppedMember { + index: usize, + node: StoppedNode, +} + +/// Every node of a [`TestCluster`], stopped by [`TestCluster::stop_all`]. +/// Each keeps its data directory until [`Self::start_all`]. +pub struct StoppedCluster { + nodes: Vec, + spawn_config: ClusterSpawnConfig, +} + +/// One stopped node, as a caller that rewrites its data directory sees it. +#[derive(Debug, Clone)] +pub struct StoppedNodeInfo { + pub node_id: u64, + pub listen_addr: std::net::SocketAddr, + pub data_dir: std::path::PathBuf, +} + +impl StoppedCluster { + /// Every stopped node, in cluster order. + pub fn nodes(&self) -> Vec { + self.nodes + .iter() + .map(|node| StoppedNodeInfo { + node_id: node.node_id(), + listen_addr: node.listen_addr(), + data_dir: node.data_dir().to_path_buf(), + }) + .collect() + } + + /// Bring every node back on its node id, listen address and data + /// directory, together, since each Raft group needs a quorum to elect, + /// and wait until the cluster is ready. + pub async fn start_all(self) -> Result> { + let StoppedCluster { nodes, spawn_config, } = self; - let mut stopped = Vec::with_capacity(nodes.len()); - for node in nodes { - stopped.push(node.stop_for_restart().await?); - } let seeds: Vec = - stopped.iter().map(|node| node.listen_addr()).collect(); + nodes.iter().map(|node| node.listen_addr()).collect(); let nodes = futures::future::try_join_all( - stopped + nodes .into_iter() .map(|node| TestClusterNode::restart(node, seeds.clone(), &spawn_config)), ) @@ -33,7 +63,65 @@ impl TestCluster { nodes, spawn_config, }; - cluster.await_ready().await; + cluster.await_ready(super::ready::LeaderBar::Elected).await; Ok(cluster) } } + +impl TestCluster { + /// Stop every node, keeping each data directory. Every node stops before + /// [`StoppedCluster::start_all`] restarts any, so nothing survives in + /// memory: what each node serves afterwards comes from its own disk. + pub async fn stop_all( + self, + ) -> Result> { + let TestCluster { + nodes, + spawn_config, + } = self; + let mut stopped = Vec::with_capacity(nodes.len()); + for node in nodes { + stopped.push(node.stop_for_restart().await?); + } + Ok(StoppedCluster { + nodes: stopped, + spawn_config, + }) + } + + /// Stop every node, then bring every one back on its node id, listen + /// address and data directory, and wait until the cluster is ready. + pub async fn restart_all(self) -> Result> { + self.stop_all().await?.start_all().await + } + + /// Stop the member at `index` and take it out of [`Self::nodes`]. The + /// rest of the cluster keeps running and commits without it. + pub async fn stop_member( + &mut self, + index: usize, + ) -> Result> { + let node = self.nodes.remove(index); + Ok(StoppedMember { + index, + node: node.stop_for_restart().await?, + }) + } + + /// Bring `stopped` back on its node id, listen address and data + /// directory, at its former index, and wait until the cluster is ready. + /// It catches up with what the cluster committed while it was down. + pub async fn restart_member( + &mut self, + stopped: StoppedMember, + ) -> Result<(), Box> { + let StoppedMember { index, node } = stopped; + let mut seeds: Vec = + self.nodes.iter().map(|member| member.listen_addr).collect(); + seeds.push(node.listen_addr()); + let restarted = TestClusterNode::restart(node, seeds, &self.spawn_config).await?; + self.nodes.insert(index, restarted); + self.await_ready(super::ready::LeaderBar::Elected).await; + Ok(()) + } +} diff --git a/nodedb-test-support/src/cluster_harness/cluster/spawn_variants.rs b/nodedb-test-support/src/cluster_harness/cluster/spawn_variants.rs index f67106008..a1c57240d 100644 --- a/nodedb-test-support/src/cluster_harness/cluster/spawn_variants.rs +++ b/nodedb-test-support/src/cluster_harness/cluster/spawn_variants.rs @@ -165,6 +165,26 @@ impl TestCluster { .await } + /// [`Self::spawn_three_with_compaction_threshold_and_rf`] with + /// `num_cores` Data-Plane cores per node, including any learner added + /// later. A snapshot install then has to place each row on its owning + /// core. + pub async fn spawn_three_with_compaction_threshold_rf_and_cores( + threshold: u64, + replication_factor: usize, + num_cores: usize, + ) -> Result> { + Self::spawn_three_inner( + fast_cluster_tuning(), + nodedb_types::config::tuning::GraphTuning::default(), + nodedb_types::config::tuning::QueryTuning::default(), + num_cores, + Some(threshold), + replication_factor, + ) + .await + } + /// Spawn a 3-node cluster whose data groups each place /// `replication_factor` of the three nodes. With a factor below 3 some /// node replicates no copy of a group, so a test can act on a node that @@ -185,6 +205,41 @@ impl TestCluster { .await } + /// [`Self::spawn_three_with_replication_factor`] with `num_cores` + /// Data-Plane cores per node, so the vShards one node leads spread over + /// several cores. + pub async fn spawn_three_with_replication_factor_and_cores( + replication_factor: usize, + num_cores: usize, + ) -> Result> { + Self::spawn_three_inner( + fast_cluster_tuning(), + nodedb_types::config::tuning::GraphTuning::default(), + nodedb_types::config::tuning::QueryTuning::default(), + num_cores, + None, + replication_factor, + ) + .await + } + + /// [`Self::spawn_three_with_replication_factor`] with `graph_tuning` on + /// every node's cores and Control Plane. + pub async fn spawn_three_with_replication_factor_and_graph_tuning( + replication_factor: usize, + graph_tuning: nodedb_types::config::tuning::GraphTuning, + ) -> Result> { + Self::spawn_three_inner( + fast_cluster_tuning(), + graph_tuning, + nodedb_types::config::tuning::QueryTuning::default(), + 1, + None, + replication_factor, + ) + .await + } + /// Spawn a 3-node cluster with custom cluster-transport, graph engine tuning, /// query execution tuning, and a specific core count per node. /// diff --git a/nodedb-test-support/src/cluster_harness/cluster/types.rs b/nodedb-test-support/src/cluster_harness/cluster/types.rs index 3d149acf1..23983f3cd 100644 --- a/nodedb-test-support/src/cluster_harness/cluster/types.rs +++ b/nodedb-test-support/src/cluster_harness/cluster/types.rs @@ -7,6 +7,9 @@ use nodedb_types::config::tuning::ClusterTransportTuning; use super::super::node::TestClusterNode; +/// Data Raft groups a test cluster bootstraps with by default. +pub(crate) const DEFAULT_NUM_GROUPS: u64 = 2; + /// The spawn configuration used to bring up every node in a cluster. /// /// Captured at spawn so that a later [`TestCluster::add_learner_node`] @@ -25,13 +28,40 @@ pub(crate) struct ClusterSpawnConfig { /// `min(replication_factor, node_count)`). Defaults to 3 for every /// spawn entry point except [`TestCluster::spawn_three_with_compaction_threshold_and_rf`]. pub(crate) replication_factor: usize, + /// Data Raft groups the cluster bootstraps with. Every spawn entry + /// point uses [`DEFAULT_NUM_GROUPS`] except + /// [`TestCluster::spawn_three_with_groups_compaction_threshold_and_rf`]. + pub(crate) num_groups: u64, /// When `true`, the node acquires its cluster handle from - /// `init_single_node_calvin` (the flag-gated standalone Calvin - /// synthesis) instead of building explicit `ClusterSettings` and - /// calling `init_cluster_with_transport`. Used only by - /// [`TestClusterNode::spawn_single_node_calvin`]. Defaults to `false` - /// for every multi-node spawn path. + /// `init_single_node_calvin` (the one-node cluster synthesis production + /// boot runs when `[cluster]` is absent) instead of building explicit + /// `ClusterSettings` and calling `init_cluster_with_transport`. Only + /// [`TestClusterNode::spawn_single_node_calvin`] uses it. Every multi-node + /// spawn path sets `false`. pub(crate) single_node_calvin: bool, + /// `[backup_storage]` installed on every node. `None` leaves every + /// `file://` backup URI refused. + pub(crate) backup_storage: Option, + /// Shared PITR storage. `Some` opens every node's WAL encrypted at + /// `/wal` and wires PITR before its Raft groups start. + pub(crate) pitr: Option, + /// Timeseries tuning of the node with each listed id. Every other node + /// runs `TimeseriesToning::default()`. + pub(crate) node_timeseries_tuning: + std::collections::HashMap, +} + +impl ClusterSpawnConfig { + /// The timeseries tuning node `node_id` runs. + pub(crate) fn timeseries_tuning_for( + &self, + node_id: u64, + ) -> nodedb_types::config::tuning::TimeseriesToning { + self.node_timeseries_tuning + .get(&node_id) + .cloned() + .unwrap_or_default() + } } /// An in-process cluster of `TestClusterNode`s. diff --git a/nodedb-test-support/src/cluster_harness/mod.rs b/nodedb-test-support/src/cluster_harness/mod.rs index 72e600585..ebabcb540 100644 --- a/nodedb-test-support/src/cluster_harness/mod.rs +++ b/nodedb-test-support/src/cluster_harness/mod.rs @@ -1,7 +1,5 @@ // SPDX-License-Identifier: BUSL-1.1 -#![allow(dead_code, unused_imports)] // Not every test file uses every helper. - //! Multi-node in-process cluster harness for nodedb-crate integration tests. //! //! One `TestClusterNode` owns a full NodeDB server stack: temp data dir, @@ -19,10 +17,13 @@ pub mod cluster; pub mod node; +pub mod pitr; pub mod retriable; +pub mod shared_steps; pub mod wait; -pub use cluster::TestCluster; +pub use cluster::{StoppedCluster, StoppedNodeInfo, TestCluster}; pub use node::TestClusterNode; +pub use pitr::PitrStorage; pub use retriable::{is_no_serving_leader, read_once_a_leader_exists}; -pub use wait::{wait_for, wait_for_async}; +pub use wait::{wait_for, wait_for_async, wait_for_report}; diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/catalog.rs b/nodedb-test-support/src/cluster_harness/node/inspect/catalog.rs index 2b1bd3276..34d459f35 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/catalog.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/catalog.rs @@ -40,17 +40,6 @@ impl TestClusterNode { ) } - /// Read the current counter of a default-database sequence from this - /// node's in-memory registry, if present. - pub fn sequence_current_value(&self, tenant_id: u64, name: &str) -> Option { - self.shared - .sequence_registry - .list(nodedb_types::DatabaseId::DEFAULT.as_u64(), tenant_id) - .into_iter() - .find(|(n, _, _)| n == name) - .map(|(_, current, _)| current) - } - /// The value the next `nextval` call on this sequence returns: /// `current_value + increment`. Holds whether or not the sequence has /// been called yet — a restart stores `value - increment` with @@ -240,34 +229,4 @@ impl TestClusterNode { .flatten() .map(|coll| (coll.descriptor_version, coll.modification_hlc)) } - - /// Same as [`collection_descriptor`] for stored functions. - pub fn function_descriptor( - &self, - tenant_id: u64, - name: &str, - ) -> Option<(u64, nodedb_types::Hlc)> { - self.shared - .credentials - .catalog() - .get_function(tenant_id, name) - .ok() - .flatten() - .map(|f| (f.descriptor_version, f.modification_hlc)) - } - - /// Same as [`collection_descriptor`] for stored procedures. - pub fn procedure_descriptor( - &self, - tenant_id: u64, - name: &str, - ) -> Option<(u64, nodedb_types::Hlc)> { - self.shared - .credentials - .catalog() - .get_procedure(tenant_id, name) - .ok() - .flatten() - .map(|p| (p.descriptor_version, p.modification_hlc)) - } } diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/core_placement.rs b/nodedb-test-support/src/cluster_harness/node/inspect/core_placement.rs new file mode 100644 index 000000000..cdaa555ad --- /dev/null +++ b/nodedb-test-support/src/cluster_harness/node/inspect/core_placement.rs @@ -0,0 +1,94 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-core row placement on [`TestClusterNode`]: which Data-Plane core holds +//! which documents, and which core a collection routes to. + +use std::sync::atomic::{AtomicU64, Ordering}; + +use nodedb::bridge::envelope::{Priority, Request, Status}; +use nodedb::event::EventSource; +use nodedb::types::{DatabaseId, ReadConsistency, RequestId, TenantDataSnapshot, TenantId}; +use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; + +use crate::cluster_harness::node::lifecycle::TestClusterNode; + +/// Request ids for this file, on a base no other harness file uses. +static PLACEMENT_REQUEST_ID: AtomicU64 = AtomicU64::new(1 << 50); + +impl TestClusterNode { + /// Number of Data-Plane cores on this node. + pub fn num_cores(&self) -> usize { + self.shared + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()) + .num_cores() + } + + /// The core that reads and writes of `collection` in the default + /// database route to. + pub fn home_core_of(&self, collection: &str) -> usize { + let vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard(); + self.shared + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()) + .router() + .resolve(vshard) + .expect("every vshard routes to a core") + } + + /// The document keys of `tenant` in the default database that core + /// `core_id` stores, from that core's own tenant snapshot. + pub async fn document_keys_on_core(&self, core_id: usize, tenant: TenantId) -> Vec { + let request_id = RequestId::new(PLACEMENT_REQUEST_ID.fetch_add(1, Ordering::Relaxed)); + let request = Request { + request_id, + tenant_id: tenant, + database_id: DatabaseId::DEFAULT, + vshard_id: nodedb::types::VShardId::new(core_id as u32), + plan: PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id: tenant.as_u64(), + cut_watermark: None, + cut_capture: None, + arrays: false, + }), + deadline: std::time::Instant::now() + std::time::Duration::from_secs(10), + priority: Priority::Normal, + trace_id: nodedb_types::TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + commit_hlc: None, + admission: nodedb::bridge::envelope::Admission::Exempt( + nodedb::bridge::envelope::ExemptReason::Read, + ), + }; + let mut rx = self.shared.tracker.register(request_id); + self.shared + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()) + .dispatch_to_core(core_id, request) + .expect("dispatch the per-core snapshot"); + let response = tokio::time::timeout(std::time::Duration::from_secs(10), rx.recv()) + .await + .expect("core answers within 10s") + .expect("response channel open"); + assert_eq!( + response.status, + Status::Ok, + "core {core_id} snapshot failed" + ); + let snap: TenantDataSnapshot = + zerompk::from_msgpack(response.payload.as_bytes()).expect("core snapshot decodes"); + snap.documents.into_iter().map(|(k, _)| k).collect() + } +} diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/crdt.rs b/nodedb-test-support/src/cluster_harness/node/inspect/crdt.rs index 0c2519152..24df83b47 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/crdt.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/crdt.rs @@ -56,6 +56,7 @@ impl TestClusterNode { txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: nodedb::bridge::envelope::Admission::Exempt( nodedb::bridge::envelope::ExemptReason::Read, ), @@ -129,6 +130,7 @@ impl TestClusterNode { txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: nodedb::bridge::envelope::Admission::Admitted, }; diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/lease.rs b/nodedb-test-support/src/cluster_harness/node/inspect/lease.rs index 557bc65c8..e2a7dc705 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/lease.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/lease.rs @@ -19,18 +19,6 @@ impl TestClusterNode { .is_draining(descriptor_id, min_version) } - /// Total number of leases (across all descriptors and node_ids) - /// in this node's `MetadataCache.leases` map. Includes expired - /// records — for filtered counts use [`active_lease_count`]. - pub fn lease_count(&self) -> usize { - let cache = self - .shared - .metadata_cache - .read() - .unwrap_or_else(|p| p.into_inner()); - cache.leases.len() - } - /// Number of leases whose `expires_at` is strictly greater /// than this node's current HLC peek. pub fn active_lease_count(&self) -> usize { @@ -100,11 +88,6 @@ impl TestClusterNode { } /// Acquire a lease on this node via the SharedState facade. - /// Called directly from the test's tokio runtime worker so the - /// `block_in_place` inside `acquire_descriptor_lease` lands on - /// a real runtime thread (which is what `block_in_place` - /// requires — it cannot be called from a `spawn_blocking` - /// worker). pub async fn acquire_lease( &self, kind: nodedb_cluster::DescriptorKind, @@ -121,9 +104,36 @@ impl TestClusterNode { ); self.shared .acquire_descriptor_lease(id, version, duration) + .await .map_err(|e| format!("acquire failed: {e}")) } + /// Admit a statement on this node that holds a lease on `name` at + /// `version` until the returned scope drops. + /// + /// A bare [`Self::acquire_lease`] leaves an idle lease that a drain start + /// releases at once. A held scope is what a drain waits for. + pub async fn hold_lease( + &self, + kind: nodedb_cluster::DescriptorKind, + tenant_id: u64, + name: &str, + version: u64, + ) -> Result { + let id = nodedb_cluster::DescriptorId::new( + nodedb_types::DatabaseId::DEFAULT.as_u64(), + tenant_id, + kind, + name.to_string(), + ); + let mut versions = nodedb::control::planner::descriptor_set::DescriptorVersionSet::new(); + versions.record(id, version); + self.shared + .acquire_plan_lease_scope(&versions) + .await + .map_err(|e| format!("hold failed: {e}")) + } + /// Release a batch of leases on this node via the SharedState facade. pub async fn release_leases( &self, @@ -131,6 +141,7 @@ impl TestClusterNode { ) -> Result<(), String> { self.shared .release_descriptor_leases(descriptor_ids) + .await .map_err(|e| format!("release failed: {e}")) } } diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/mod.rs b/nodedb-test-support/src/cluster_harness/node/inspect/mod.rs index c7cf16a3c..6f340e141 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/mod.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/mod.rs @@ -9,7 +9,9 @@ //! spawn/shutdown. mod catalog; +mod core_placement; mod crdt; mod lease; mod snapshot; +mod timeseries; mod topology; diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs b/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs index 412d5d6a7..1a25df725 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/snapshot.rs @@ -35,6 +35,23 @@ impl TestClusterNode { /// (this op is not collection-scoped). Returns an empty `Vec` on dispatch /// error or non-Ok response. pub async fn create_tenant_snapshot(&self, tenant: TenantId) -> Vec { + self.snapshot_tenant(tenant, false).await + } + + /// Every array cell version of `tenant` in the default database that + /// this node's core holds, exported the way a backup exports them. + /// Empty on dispatch error or non-Ok response. + pub async fn array_cells(&self, tenant: TenantId) -> Vec { + let bytes = self.snapshot_tenant(tenant, true).await; + if bytes.is_empty() { + return Vec::new(); + } + zerompk::from_msgpack::(&bytes) + .map(|snap| snap.arrays) + .unwrap_or_default() + } + + async fn snapshot_tenant(&self, tenant: TenantId, arrays: bool) -> Vec { let request_id = RequestId::new(SNAPSHOT_REQUEST_ID.fetch_add(1, Ordering::Relaxed)); let vshard_id = VShardId::new(vshard_for_collection( nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "__system"), @@ -47,6 +64,8 @@ impl TestClusterNode { plan: PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: tenant.as_u64(), cut_watermark: None, + cut_capture: None, + arrays, }), deadline: std::time::Instant::now() + std::time::Duration::from_secs(5), priority: Priority::Normal, @@ -60,6 +79,7 @@ impl TestClusterNode { txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: nodedb::bridge::envelope::Admission::Exempt( nodedb::bridge::envelope::ExemptReason::Read, ), @@ -88,16 +108,15 @@ impl TestClusterNode { /// Restore a captured tenant snapshot into this node's engines via /// `MetaOp::RestoreTenantSnapshot`. /// - /// Test hook for proving U6's restore path (`restore.rs`). Mirrors the - /// production `DataPlaneSnapshotApplier` dispatch shape: `tenant_id: 0` - /// is only the routing/dispatch key (the merged Raft snapshot applies - /// with dispatch tenant 0), and `replace_mode: true` because this - /// simulates a Raft `InstallSnapshot` apply, which must overwrite local - /// state rather than fail against it. The restore handler installs each - /// snapshot entry (including `crdt_constraints`) by its own - /// tenant-explicit fields, independent of the dispatch tenant. Routed to - /// the `"__system"` vshard. Returns `true` iff the response status is - /// `Ok`. + /// Test hook for proving U6's restore path (`restore.rs`) on ONE core: + /// the core the `"__system"` vshard routes to. `tenant_id: 0` is only the + /// dispatch key, and `replace_mode: true` because this simulates a Raft + /// `InstallSnapshot` apply, which must overwrite local state rather than + /// fail against it. The restore handler installs each snapshot entry + /// (including `crdt_constraints`) by its own tenant-explicit fields, + /// independent of the dispatch tenant. The production applier splits a + /// snapshot per owning core instead. Returns `true` iff the response + /// status is `Ok`. pub async fn restore_tenant_snapshot(&self, snapshot_bytes: Vec) -> bool { let request_id = RequestId::new(SNAPSHOT_REQUEST_ID.fetch_add(1, Ordering::Relaxed)); let vshard_id = VShardId::new(vshard_for_collection( @@ -112,8 +131,8 @@ impl TestClusterNode { tenant_id: 0, snapshot: snapshot_bytes, replace_mode: true, - clear_vshards: Vec::new(), collections_to_clear: Vec::new(), + group_vshards: Vec::new(), }), deadline: std::time::Instant::now() + std::time::Duration::from_secs(5), priority: Priority::Normal, @@ -127,6 +146,7 @@ impl TestClusterNode { txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: nodedb::bridge::envelope::Admission::Exempt( nodedb::bridge::envelope::ExemptReason::Read, ), diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/timeseries.rs b/nodedb-test-support/src/cluster_harness/node/inspect/timeseries.rs new file mode 100644 index 000000000..e16519549 --- /dev/null +++ b/nodedb-test-support/src/cluster_harness/node/inspect/timeseries.rs @@ -0,0 +1,107 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Local timeseries reads on [`TestClusterNode`]: the rows this node's own +//! replica stores, read from its Data Plane without routing to a leader. + +use std::sync::atomic::{AtomicU64, Ordering}; + +use nodedb::bridge::envelope::{Priority, Request, Status}; +use nodedb::event::EventSource; +use nodedb::types::{DatabaseId, ReadConsistency, RequestId, TenantId, VShardId}; +use nodedb_physical::physical_plan::{PhysicalPlan, TimeseriesOp, UNBOUNDED_TIME_RANGE}; + +use crate::cluster_harness::node::lifecycle::TestClusterNode; + +/// Request ids for this file, on a base no other harness file uses. +static TIMESERIES_REQUEST_ID: AtomicU64 = AtomicU64::new(1 << 51); + +/// The most rows one local scan returns. A test collection holds far fewer. +const LOCAL_SCAN_LIMIT: usize = 100_000; + +impl TestClusterNode { + /// Every row of the timeseries `collection` in the default database of + /// `tenant` that this node's own replica stores, memtable and flushed + /// partitions both, as JSON objects. The scan runs on the collection's + /// home core of this node and never leaves it. It returns at most + /// `LOCAL_SCAN_LIMIT` rows. + /// + /// Panics with the scan's status when the core answers with an error. + pub async fn timeseries_rows_local( + &self, + tenant: TenantId, + collection: &str, + ) -> Vec { + let request_id = RequestId::new(TIMESERIES_REQUEST_ID.fetch_add(1, Ordering::Relaxed)); + let vshard = + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection).vshard(); + let request = Request { + request_id, + tenant_id: tenant, + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(vshard.as_u32()), + plan: PhysicalPlan::Timeseries(TimeseriesOp::Scan { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, collection), + time_range: UNBOUNDED_TIME_RANGE, + projection: Vec::new(), + limit: LOCAL_SCAN_LIMIT, + filters: Vec::new(), + sort_keys: Vec::new(), + bucket_interval_ms: 0, + group_by: Vec::new(), + aggregates: Vec::new(), + gap_fill: String::new(), + computed_columns: Vec::new(), + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + }), + deadline: std::time::Instant::now() + std::time::Duration::from_secs(10), + priority: Priority::Normal, + trace_id: nodedb_types::TraceId::ZERO, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: EventSource::User, + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + commit_hlc: None, + admission: nodedb::bridge::envelope::Admission::Exempt( + nodedb::bridge::envelope::ExemptReason::Read, + ), + }; + let core_id = self.home_core_of(collection); + let mut rx = self.shared.tracker.register(request_id); + self.shared + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()) + .dispatch_to_core(core_id, request) + .expect("dispatch the local timeseries scan"); + let response = tokio::time::timeout(std::time::Duration::from_secs(10), rx.recv()) + .await + .expect("the local timeseries scan answers within 10s") + .expect("the local timeseries scan answers"); + assert_eq!( + response.status, + Status::Ok, + "node {} local scan of '{collection}' failed: {:?}", + self.node_id, + response.error_code + ); + let json = nodedb::data::executor::response_codec::decode_payload_to_json( + response.payload.as_bytes(), + ); + match sonic_rs::from_str::(&json) { + Ok(serde_json::Value::Array(rows)) => rows, + Ok(serde_json::Value::Null) => Vec::new(), + Ok(other) => vec![other], + Err(e) => panic!( + "node {} local scan payload is not JSON: {e}: {json}", + self.node_id + ), + } + } +} diff --git a/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs b/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs index f01c71164..8cee7b434 100644 --- a/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs +++ b/nodedb-test-support/src/cluster_harness/node/inspect/topology.rs @@ -279,23 +279,6 @@ impl TestClusterNode { cache.applied_index } - /// Force the routing table on this node to point `group_id` at `fake_leader`, - /// creating a stale route. - /// - /// When the gateway on this node next dispatches to `group_id`, it will send - /// the request to `fake_leader` instead of the real leader. The remote node - /// (which is NOT the leader for that group) will return `TypedClusterError::NotLeader`, - /// causing `retry_not_leader` to update the routing table and retry against - /// the real leader. This is the canonical way to exercise the NotLeader retry - /// path in tests without needing a real leadership change (which is slow and - /// flaky). - pub fn force_stale_route_for_test(&self, group_id: u64, fake_leader: u64) { - if let Some(ref routing) = self.shared.cluster_routing { - let mut table = routing.write().unwrap_or_else(|p| p.into_inner()); - table.set_leader(group_id, fake_leader); - } - } - /// Read the current `not_leader_retry_count` from this node's shared gateway. /// /// Returns 0 if the gateway has not been constructed yet (shouldn't happen diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/mod.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/mod.rs index dc0df838d..0bc645c83 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/mod.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/mod.rs @@ -36,5 +36,7 @@ mod spawn_full; mod spawn_variants; mod teardown; mod types; +mod wire_state; +pub(crate) use restart::StoppedNode; pub use types::{HARNESS_SUPERUSER, TestClusterNode}; diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/restart.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/restart.rs index 8893a4fba..f416e2696 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/restart.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/restart.rs @@ -23,6 +23,15 @@ impl StoppedNode { pub(crate) fn listen_addr(&self) -> SocketAddr { self.listen_addr } + + pub(crate) fn node_id(&self) -> u64 { + self.node_id + } + + /// The data directory the node restarts on. + pub(crate) fn data_dir(&self) -> &std::path::Path { + self.data_dir.path() + } } impl TestClusterNode { diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs index 5c0498945..566288684 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_full.rs @@ -17,9 +17,9 @@ use nodedb::config::server::ClusterSettings; use nodedb::control::server::pgwire::listener::PgListener; use nodedb::control::state::SharedState; use nodedb::event::{EventPlane, EventPlaneConfig, create_event_bus}; -use nodedb::wal::WalManager; use crate::cluster_harness::cluster::ClusterSpawnConfig; +use crate::cluster_harness::pitr::{open_node_wal, wire_boot}; use super::client_slot::ClusterTestClient; use super::types::{DataDir, HARNESS_SUPERUSER, TestClusterNode}; @@ -101,9 +101,7 @@ impl TestClusterNode { // Open WAL + dispatcher + event bus. Replay whatever is already on // disk (empty for a fresh directory) so every core can rebuild its // in-memory-only structures before it starts ticking. - let wal = Arc::new(WalManager::open_for_testing( - &data_dir_path.join("test.wal"), - )?); + let wal = open_node_wal(config.pitr.as_ref(), &data_dir_path)?; let wal_records: Arc<[nodedb_wal::WalRecord]> = Arc::from(wal.replay()?.into_boxed_slice()); let replay_tombstones = nodedb_wal::extract_tombstones(&wal_records).unwrap(); let (dispatcher, data_sides) = Dispatcher::new(num_cores, DATA_PLANE_QUEUE_CAPACITY); @@ -124,7 +122,7 @@ impl TestClusterNode { // Static, deployment-time surrogate-registry mode — same predicate as // production's `config.cluster.is_some()`. `single_node_calvin` here // takes the `init_single_node_calvin` synthesis below (never joinable, - // mirrors production's default standalone path); every other branch + // mirrors production boot without `[cluster]`); every other branch // builds a real `ClusterSettings` and joins/bootstraps a genuine Raft // group, so it must start in `Cluster` mode from construction — this // node was never going to promote into it later. @@ -170,7 +168,7 @@ impl TestClusterNode { node_id, listen: listen_addr, seed_nodes: seeds, - num_groups: 2, + num_groups: config.num_groups, replication_factor, force_bootstrap: false, tls: None, @@ -181,6 +179,9 @@ impl TestClusterNode { log_compaction_threshold, join_retry_max_attempts: 8, join_retry_max_backoff_secs: 32, + // `listen` port + 1 can belong to another test's socket, so + // take an OS-assigned port on the same IP. It is advertised. + swim_listen: Some(std::net::SocketAddr::new(listen_addr.ip(), 0)), }; // Initialise the cluster using the pre-bound transport. @@ -194,54 +195,36 @@ impl TestClusterNode { (handle, listen_addr) }; - // Wire cluster handles into SharedState (mirrors main.rs). - // `Arc::get_mut` is valid here: `shared` has not been cloned. + // Wire cluster handles into SharedState with the function production + // boot runs. `Arc::get_mut` is valid here: `shared` has not been + // cloned. Before `EventPlane::spawn` below: the Event Plane starts the + // cross-shard drain only when its sender is wired. if let Some(state) = Arc::get_mut(&mut shared) { - state.node_id = handle.node_id; - state.cluster_topology = Some(Arc::clone(&handle.topology)); - state.cluster_routing = Some(Arc::clone(&handle.routing)); - state.cluster_transport = Some(Arc::clone(&handle.transport)); - state.metadata_cache = Arc::clone(&handle.metadata_cache); - state.group_watchers = Arc::clone(&handle.group_watchers); - // Wire the cross-shard event SENDER (mirrors production - // `bootstrap::state_wiring::wire_state`) so an AFTER-trigger body - // writing to a remote-homed collection is dispatched to the owning - // node instead of silently mis-written locally. Must be set BEFORE - // `EventPlane::spawn` below, whose gate spawns the dispatcher drain - // task only when these fields are `Some`. The receiver builds its - // own HWM store at Raft group setup, so `hwm_store` stays `None`. - let cross_shard_metrics = - Arc::new(nodedb::event::cross_shard::CrossShardMetrics::new()); - state.cross_shard_dispatcher = Some(Arc::new( - nodedb::event::cross_shard::CrossShardDispatcher::new( - handle.node_id, - Arc::clone(&cross_shard_metrics), - ), - )); - state.cross_shard_dlq = Some(Arc::new(std::sync::Mutex::new( - nodedb::event::cross_shard::CrossShardDlq::open(&data_dir_path)?, - ))); - state.cross_shard_metrics = Some(cross_shard_metrics); + // Consumer offsets, job history, MV state and the array-sync + // stores live in the data directory in the production layout. A + // restart reopens them there, since no metadata entry below the + // applied floor re-applies to rebuild them. + state.open_disk_stores_at(&data_dir_path)?; + nodedb::bootstrap::state_wiring::wire_cluster_handle(state, &handle, &data_dir_path)?; + // The Control Plane runs with the graph and query tuning the + // cores run with, as a production node reads both from its one + // `[tuning]` section. A walk coordinator's visit cap reads it. + state.tuning.graph = graph_tuning.clone(); + state.tuning.query = query_tuning.clone(); // Fixed test KEK so backup tests produce encrypted envelopes. state.backup_kek = Some(Arc::new([0x42u8; 32])); - // Durable producer registry, sharing the credential store's - // already-open catalog (mirrors production `SharedState::open`). - // Required for sync handshake fencing to replicate via the - // metadata Raft group on cluster nodes. - let catalog = state.credentials.catalog().clone(); - match nodedb::control::sync_producer::registry::SyncProducerRegistry::open(Arc::new( - catalog, - )) { - Ok(reg) => state.producer_registry = Some(Arc::new(reg)), - Err(e) => { - return Err( - format!("SyncProducerRegistry::open failed in test harness: {e}").into(), - ); - } + state.backup_storage = config.backup_storage.clone().map(Arc::new); + super::wire_state::open_producer_registry(state)?; + // Drains, pending DDL records, and the DDL preparation owner load + // from their rows (mirrors production `SharedState::open`). + nodedb::control::cluster::metadata_applier::seed_host_tables(state)?; + if let Some(pitr) = &config.pitr { + pitr.install_stores(state)?; } } else { return Err("SharedState already cloned before cluster wire-up".into()); } + wire_boot(config.pitr.as_ref(), &shared, &handle.catalog).await?; // Start one Data-Plane core loop per core. Each core gets its own SPSC // data side and event producer; per-core stores live under the shared @@ -272,6 +255,7 @@ impl TestClusterNode { replay, graph_tuning: graph_tuning.clone(), query_tuning: query_tuning.clone(), + timeseries_tuning: config.timeseries_tuning_for(node_id), // Seeded from the SAME durable catalog production reads, so a // harness restart reconstructs cores the way a real one does. // An empty catalog yields an empty seed, which is exactly what @@ -279,6 +263,7 @@ impl TestClusterNode { doc_config_seed: nodedb::bootstrap::data_plane::load_doc_config_registry_from( shared.credentials.catalog(), ), + event_interest: crate::core_loop_runner::event_interest_for(&shared), stop_rx: core_stop_rx, }); core_stop_txs.push(core_stop_tx); @@ -307,6 +292,10 @@ impl TestClusterNode { )); let (pg_shutdown_bus, _) = nodedb::control::shutdown::ShutdownBus::new(Arc::clone(&shared.shutdown)); + // Sink delivery managers, wired as the server's background loops + // wire them, so every node runs a task for every sink stream. + shared.webhook_manager.set_state(&shared); + shared.kafka_manager.set_state(&shared); let event_plane = EventPlane::spawn(EventPlaneConfig { consumers_rx: event_consumers, wal: Arc::clone(&wal), @@ -320,7 +309,8 @@ impl TestClusterNode { // Start Raft + install MetadataCommitApplier. let (cluster_shutdown_tx, cluster_shutdown_rx) = tokio::sync::watch::channel(false); - nodedb::control::cluster::start_raft(&handle, Arc::clone(&shared), &data_dir_path, tuning)?; + nodedb::control::cluster::start_raft(&handle, Arc::clone(&shared), &data_dir_path, tuning) + .await?; // `start_raft` spawns the cluster subsystems (SWIM, reachability, // decommission, rebalancer) and stashes the resulting @@ -355,34 +345,19 @@ impl TestClusterNode { .constraint_reconcile_interval_ms, ); - // Spawn the descriptor lease renewal loop on the same - // shutdown channel as raft so cluster shutdown stops it - // cleanly. Returns None on single-node clusters that - // never wired metadata_raft (the harness always wires it, - // so this returns Some in practice for cluster tests). - let lease_renewal_handle = nodedb::control::lease::LeaseRenewalLoop::spawn( - Arc::clone(&shared), - tuning, - cluster_shutdown_rx, - ) - .map(|(join, metrics)| { - shared.loop_metrics_registry.register(metrics); - join - }); + // Spawn the descriptor lease renewal loop on the cluster shutdown + // channel so cluster shutdown stops it cleanly. `start_raft` above + // installed the metadata Raft handle it needs. + let (lease_renewal_handle, lease_metrics) = + nodedb::control::lease::LeaseRenewalLoop::spawn( + Arc::clone(&shared), + tuning, + cluster_shutdown_rx, + )?; + shared.loop_metrics_registry.register(lease_metrics); - // Construct the gateway and install it (plus its DDL invalidator) on - // SharedState, mirroring what main.rs does before listeners bind. The - // fields are `OnceLock`s, set through `&self`, so no `Arc::get_mut` - // (which `shared` being already cloned would defeat) or raw-pointer - // write is needed. - { - let gateway = Arc::new(nodedb::control::gateway::Gateway::new(Arc::clone(&shared))); - let invalidator = Arc::new(nodedb::control::gateway::PlanCacheInvalidator::new( - &gateway.plan_cache, - )); - let _ = shared.gateway.set(gateway); - let _ = shared.gateway_invalidator.set(invalidator); - } + // The gateway install production boot runs before listeners bind. + nodedb::bootstrap::state_wiring::install_gateway(&shared)?; // pgwire listener. // In the test harness, use the startup gate already on SharedState @@ -453,11 +428,10 @@ impl TestClusterNode { // The node plans permission-checked statements only under an // authorization lease, as a production node opens its gateway only // once it holds one. - if let Some(timing) = shared.authorization_fence.timing() { + if shared.authorization_fence.timing().is_some() { nodedb::control::security::auth_lease::await_planning_admitted( &shared, Duration::from_secs(15), - timing.renew_every, ) .await .map_err(|e| format!("node {node_id}: {e}"))?; @@ -481,7 +455,7 @@ impl TestClusterNode { _poller_handle: Some(poller_handle), _core_handles: core_handles, _event_plane: Some(event_plane), - _lease_renewal_handle: lease_renewal_handle, + _lease_renewal_handle: Some(lease_renewal_handle), _running_cluster: running_cluster, }) } diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_variants.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_variants.rs index dffccc7ca..1f46d3508 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_variants.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/spawn_variants.rs @@ -8,7 +8,7 @@ use std::path::PathBuf; use nodedb_types::config::tuning::ClusterTransportTuning; -use crate::cluster_harness::cluster::ClusterSpawnConfig; +use crate::cluster_harness::cluster::{ClusterSpawnConfig, DEFAULT_NUM_GROUPS}; use super::types::TestClusterNode; @@ -103,18 +103,22 @@ impl TestClusterNode { num_cores, log_compaction_threshold: None, replication_factor: 3, + num_groups: DEFAULT_NUM_GROUPS, single_node_calvin: false, + backup_storage: None, + pitr: None, + node_timeseries_tuning: std::collections::HashMap::new(), }; Self::spawn_with_full_config(node_id, seed_nodes, &config).await } - /// Spawn a standalone node with the flag-gated single-node Calvin stack - /// (`server.single_node_calvin = true`), exercising the same - /// `init_single_node_calvin` synthesis the production standalone boot uses. + /// Spawn a node with the single-node Calvin stack, exercising the same + /// `init_single_node_calvin` synthesis production boot runs when + /// `[cluster]` is absent. /// /// The node stands up its own sequencer Raft group and per-vShard - /// schedulers, so `calvin_available` becomes true and a cross-vShard - /// transaction traverses the deterministic Calvin path — all on one node. + /// schedulers, so a cross-vShard transaction traverses the deterministic + /// Calvin path — all on one node. /// `num_cores` should be `>= 2` so distinct vShards map to distinct cores. pub async fn spawn_single_node_calvin( num_cores: usize, @@ -157,7 +161,11 @@ impl TestClusterNode { num_cores, log_compaction_threshold: None, replication_factor: 1, + num_groups: DEFAULT_NUM_GROUPS, single_node_calvin: true, + backup_storage: None, + pitr: None, + node_timeseries_tuning: std::collections::HashMap::new(), }; Self::spawn_with_full_config_at(1, vec![], &config, data_dir_path, None).await } diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/teardown.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/teardown.rs index 0c77dc508..a8df19f71 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/teardown.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/teardown.rs @@ -61,10 +61,9 @@ impl TestClusterNode { // directory. let _ = self.shared.wal.sync(); - // Abort+await the lease-renewal loop so its `Arc` clone - // releases before we return. Previously this `JoinHandle` was bound - // to a local and dropped (detached, not cancelled) when - // `spawn_with_full_config_at` returned. + // Abort and await the lease-renewal loop, so its `Arc` + // clone drops before this function returns. A dropped `JoinHandle` + // detaches its task and never cancels it, so the node keeps this one. if let Some(h) = self._lease_renewal_handle.take() { h.abort(); let _ = h.await; @@ -150,29 +149,28 @@ impl TestClusterNode { ); } - // `start_raft` fans out to background tasks (raft apply loop, tick - // loop, sequencer service, RPC server, health monitor, per-vShard - // Calvin schedulers, reconcile loop) that each hold an - // `Arc` clone. They are fire-and-forget inside production - // code — the harness has no `JoinHandle` to await — but they were all - // signaled to stop via `cluster_shutdown_tx.send(true)` at the top of - // this function and exit asynchronously after their next `.await`. - // Until every one of them drops its clone, the catalog redb `Database` - // (owned transitively by `SharedState`) stays open and the next - // `spawn_single_node_calvin_on_path` on this directory fails with - // "Database already open. Cannot acquire lock." Condition-wait (NOT a - // fixed sleep) for the strong count to fall to 1 — meaning `self.shared` - // is the last surviving clone — so `self` dropping below actually - // releases every redb file lock. + // `start_raft` spawns tasks the harness holds no `JoinHandle` for: + // the raft apply and tick loops, the sequencer service, the RPC + // server, the health monitor, and the per-vShard Calvin schedulers. + // Each holds an `Arc` clone and exits after the shutdown + // signal sent above. `SharedState` owns the catalog redb and the QUIC + // endpoint, so both stay open until its last clone drops. Wait for + // `self.shared` to be the only clone, then drop it with `self`. + // + // A clone that survives the wait is a leak: a task that ignores + // shutdown, or a reference cycle through `SharedState`. The node + // fails here, at the leak, never later as a locked catalog or a + // bound port on the next open. let poll_deadline = tokio::time::Instant::now() + Duration::from_secs(5); while std::sync::Arc::strong_count(&self.shared) > 1 { if tokio::time::Instant::now() >= poll_deadline { - eprintln!( - "graceful_shutdown_wal_only: SharedState still has {} strong refs after 5s \ - — a background task did not release its clone", + panic!( + "graceful_shutdown_wal_only: node {} SharedState still has {} strong refs \ + 5s after shutdown, where 1 is expected: a task or a reference cycle still \ + holds SharedState, which keeps its redb files and QUIC endpoint open", + self.node_id, std::sync::Arc::strong_count(&self.shared) ); - break; } tokio::time::sleep(Duration::from_millis(25)).await; } diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/types.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/types.rs index f4098adf2..51565fbf2 100644 --- a/nodedb-test-support/src/cluster_harness/node/lifecycle/types.rs +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/types.rs @@ -56,12 +56,10 @@ pub struct TestClusterNode { pub(in crate::cluster_harness::node) _poller_handle: Option>, pub(in crate::cluster_harness::node) _core_handles: Vec>, pub(in crate::cluster_harness::node) _event_plane: Option, - /// `LeaseRenewalLoop::spawn`'s `JoinHandle` — previously bound to a local - /// (`_lease_renewal`) and dropped/detached when `spawn_with_full_config_at` - /// returned. Retained here so `graceful_shutdown_wal_only` can abort+await - /// it, releasing its `Arc` clone before returning. `None` on - /// single-node clusters that never wire `metadata_raft` - /// (`LeaseRenewalLoop::spawn` returns `None` in that case). + /// `LeaseRenewalLoop::spawn`'s `JoinHandle`, retained so + /// `graceful_shutdown_wal_only` can abort+await it, releasing its + /// `Arc` clone before returning. `Option` so shutdown can + /// `.take()` it without violating the `Drop` impl. pub(in crate::cluster_harness::node) _lease_renewal_handle: Option>, /// Cluster subsystem tasks (SWIM, reachability, decommission, /// rebalancer) started by `start_raft` and stashed on the diff --git a/nodedb-test-support/src/cluster_harness/node/lifecycle/wire_state.rs b/nodedb-test-support/src/cluster_harness/node/lifecycle/wire_state.rs new file mode 100644 index 000000000..0183c61b6 --- /dev/null +++ b/nodedb-test-support/src/cluster_harness/node/lifecycle/wire_state.rs @@ -0,0 +1,24 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Host state a cluster node's `SharedState` needs before anything clones it: +//! the durable producer registry. + +use std::sync::Arc; + +use nodedb::control::state::SharedState; + +/// Open the durable producer registry on the credential store's catalog +/// (mirrors production `SharedState::open`). Sync handshake fencing needs it +/// to replicate through the metadata Raft group on cluster nodes. +pub(super) fn open_producer_registry( + state: &mut SharedState, +) -> Result<(), Box> { + let catalog = state.credentials.catalog().clone(); + match nodedb::control::sync_producer::registry::SyncProducerRegistry::open(Arc::new(catalog)) { + Ok(reg) => { + state.producer_registry = Some(Arc::new(reg)); + Ok(()) + } + Err(e) => Err(format!("SyncProducerRegistry::open failed in test harness: {e}").into()), + } +} diff --git a/nodedb-test-support/src/cluster_harness/pitr.rs b/nodedb-test-support/src/cluster_harness/pitr.rs new file mode 100644 index 000000000..1dbb51ce9 --- /dev/null +++ b/nodedb-test-support/src/cluster_harness/pitr.rs @@ -0,0 +1,266 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Point-in-time recovery storage every node of a test cluster shares: one +//! cold store, one snapshot store and one WAL key, as the nodes of a real +//! cluster share them. A node spawned with it opens its WAL encrypted at +//! `/wal`, the production layout a restore writes. + +use std::io::Write as _; +use std::net::SocketAddr; +use std::path::{Path, PathBuf}; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use nodedb::ServerConfig; +use nodedb::config::server::{ + ClusterSettings, ColdStorageSettings, EncryptionSettings, SnapshotStorageSettings, +}; +use nodedb::control::state::SharedState; +use nodedb::ctl::restore::RestoreScope; +use nodedb::wal::WalManager; +use nodedb_cluster::METADATA_GROUP_ID; +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_wal::record::{RecordType, RestorePointPayload}; + +use crate::cluster_harness::cluster::StoppedNodeInfo; +use crate::cluster_harness::node::TestClusterNode; + +type HarnessResult = Result>; + +/// How long a node can take to record and archive a restore point. +const POINT_ARCHIVE_TIMEOUT: Duration = Duration::from_secs(60); + +/// The shared PITR storage under one root directory. +#[derive(Clone, Debug)] +pub struct PitrStorage { + root: PathBuf, +} + +impl PitrStorage { + /// Lay out the stores under `root` and write the WAL key. + pub fn create(root: &Path) -> HarnessResult { + std::fs::create_dir_all(root.join("cold"))?; + std::fs::create_dir_all(root.join("snapshots"))?; + let storage = Self { + root: root.to_path_buf(), + }; + let mut key = std::fs::OpenOptions::new(); + key.write(true).create_new(true); + #[cfg(unix)] + { + use std::os::unix::fs::OpenOptionsExt as _; + key.mode(0o600); + } + key.open(storage.key_path())?.write_all(&[0x5C; 32])?; + Ok(storage) + } + + pub fn key_path(&self) -> PathBuf { + self.root.join("wal.key") + } + + pub fn key(&self) -> HarnessResult { + Ok(nodedb_wal::crypto::WalEncryptionKey::from_file( + &self.key_path(), + )?) + } + + pub fn cold_settings(&self) -> HarnessResult { + Ok(serde_json::from_value(serde_json::json!({ + "local_dir": self.root.join("cold"), + }))?) + } + + pub fn snapshot_settings(&self) -> HarnessResult { + Ok(serde_json::from_value(serde_json::json!({ + "local_dir": self.root.join("snapshots"), + }))?) + } + + /// The server config of node `node_id` at `data_dir`, as `nodedb restore` + /// reads it. + pub fn server_config( + &self, + node_id: u64, + data_dir: &Path, + listen: SocketAddr, + ) -> HarnessResult { + let mut config = ServerConfig::default(); + config.server.data_dir = data_dir.to_path_buf(); + config.encryption = Some(EncryptionSettings { + key_path: self.key_path(), + }); + config.cold_storage = Some(self.cold_settings()?); + config.snapshot_storage = Some(self.snapshot_settings()?); + config.pitr.enabled = true; + config.cluster = Some(ClusterSettings { + node_id, + listen, + seed_nodes: vec![listen], + num_groups: 2, + replication_factor: 3, + force_bootstrap: false, + tls: None, + max_active_sessions: 0, + login_attempts_per_ip_per_min: 30, + login_attempts_per_user_per_min: 10, + insecure_transport: true, + log_compaction_threshold: None, + join_retry_max_attempts: 8, + join_retry_max_backoff_secs: 32, + swim_listen: None, + }); + Ok(config) + } + + /// Wait until `node` recorded restore point `id` for every group it + /// hosts, and cold storage holds those records and the metadata log + /// through `id`. + pub async fn await_point_archived(&self, node: &TestClusterNode, id: u64) -> HarnessResult<()> { + let deadline = Instant::now() + POINT_ARCHIVE_TIMEOUT; + loop { + let last = point_archived(&node.shared, id).await; + if matches!(last, Ok(true)) { + return Ok(()); + } + if Instant::now() >= deadline { + return Err(format!( + "node {} did not archive restore point {id} within {POINT_ARCHIVE_TIMEOUT:?}: \ + {last:?}", + node.node_id + ) + .into()); + } + tokio::time::sleep(Duration::from_millis(200)).await; + } + } + + /// Rewrite a stopped node's data directory as its part of a cluster + /// restore to `restore_point`: empty it except `tls/`, then restore as + /// `nodedb restore --cluster --restore-point` does. Returns the report. + pub async fn restore_node( + &self, + node: &StoppedNodeInfo, + restore_point: u64, + ) -> HarnessResult { + for entry in std::fs::read_dir(&node.data_dir)? { + let entry = entry?; + if entry.file_name() == "tls" { + continue; + } + if entry.file_type()?.is_dir() { + std::fs::remove_dir_all(entry.path())?; + } else { + std::fs::remove_file(entry.path())?; + } + } + let config = self.server_config(node.node_id, &node.data_dir, node.listen_addr)?; + let restored = nodedb::ctl::restore::restore_with_config( + &config, + &RestoreScope::Cluster { restore_point }, + None, + false, + ) + .await?; + Ok(restored.to_string()) + } + + /// Install the shared cold and snapshot stores on `state`, which the + /// node spawn already rooted at its data directory with its node-level + /// stores in the production layout. + pub(crate) fn install_stores(&self, state: &mut SharedState) -> HarnessResult<()> { + let cold = nodedb::storage::cold::ColdStorage::new( + self.cold_settings()?.to_cold_storage_config(), + )?; + state.cold_storage = Some(Arc::new(cold)); + state.snapshot_storage = nodedb::storage::snapshot_writer::build_snapshot_store( + &self.snapshot_settings()?.to_snapshot_storage_config(), + &state.data_dir, + )?; + Ok(()) + } +} + +/// Open a node's WAL: with `pitr`, encrypted at `/wal`, the +/// production layout a restore writes; without, at `/test.wal`. +pub(crate) fn open_node_wal( + pitr: Option<&PitrStorage>, + data_dir: &Path, +) -> HarnessResult> { + let Some(pitr) = pitr else { + return Ok(Arc::new(WalManager::open_for_testing( + &data_dir.join("test.wal"), + )?)); + }; + let mut wal = WalManager::open_for_testing(&data_dir.join("wal"))?; + wal.set_encryption_ring(nodedb_wal::crypto::KeyRing::new(pitr.key()?))?; + Ok(Arc::new(wal)) +} + +/// The boot steps after a node's state is wired and before its Raft groups +/// start: read back the cut barriers its catalog holds, and with `pitr`, +/// seal a restored generation and resolve the node life. +pub(crate) async fn wire_boot( + pitr: Option<&PitrStorage>, + shared: &SharedState, + cluster_catalog: &Arc, +) -> HarnessResult<()> { + shared + .pitr + .install_recorded_cuts(nodedb::control::pitr::restore_point::load_recorded_cuts( + shared.credentials.catalog(), + )?); + if pitr.is_some() { + nodedb::control::pitr::seal_restored_generation(shared).await?; + nodedb::control::pitr::wire_pitr(shared, Arc::clone(cluster_catalog)).await?; + } + Ok(()) +} + +/// Whether `shared` recorded restore point `id` for every group it hosts and +/// archived the records and the metadata log through `id`. Seals and uploads +/// the WAL on the way. +async fn point_archived(shared: &Arc, id: u64) -> HarnessResult { + let recorded: std::collections::BTreeSet = shared + .wal + .replay()? + .iter() + .filter(|record| { + RecordType::from_raw(record.logical_record_type()) == Some(RecordType::RestorePoint) + }) + .filter_map(|record| RestorePointPayload::from_bytes(&record.payload).ok()) + .filter(|point| point.id == id) + .map(|point| point.group_id) + .collect(); + let mut expected: std::collections::BTreeSet = [METADATA_GROUP_ID, SEQUENCER_GROUP_ID] + .into_iter() + .collect(); + if let Some(routing) = &shared.cluster_routing { + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + for (group_id, info) in routing.group_members() { + if info.members.contains(&shared.node_id) { + expected.insert(*group_id); + } + } + } + if !expected.is_subset(&recorded) { + return Ok(false); + } + if !nodedb::control::pitr::archive_wal_now(shared).await? { + return Ok(false); + } + let life = shared.pitr.life().ok_or("PITR wired no node life")?; + let cold = shared.cold_storage.as_ref().ok_or("no cold storage")?; + let timeline = shared.credentials.catalog().load_metadata_timeline()?; + let chunks = nodedb::storage::raft_log_archive::list_chunks( + &cold.object_store(), + cold.prefix(), + nodedb::storage::raft_log_archive::ArchiveLife { + timeline, + node_id: life.node_id, + incarnation: life.incarnation.as_str(), + }, + ) + .await?; + Ok(chunks.last().is_some_and(|chunk| chunk.last >= id)) +} diff --git a/nodedb-test-support/src/cluster_harness/shared_steps.rs b/nodedb-test-support/src/cluster_harness/shared_steps.rs new file mode 100644 index 000000000..d29581f50 --- /dev/null +++ b/nodedb-test-support/src/cluster_harness/shared_steps.rs @@ -0,0 +1,301 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Steps that several cluster test cases share: backup and restore over +//! pgwire `COPY`, database switching, group and leader lookups, and name +//! search. + +use std::sync::atomic::Ordering; +use std::time::Duration; + +use bytes::Bytes; +use futures::{SinkExt, StreamExt}; +use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry, encode_entry}; +use nodedb_types::id::VShardId; +use nodedb_types::{DatabaseId, TenantId}; + +use super::{TestCluster, TestClusterNode, wait_for}; + +/// Wait for the single-node sequencer and metadata Raft groups to elect +/// `node` as leader. Every spawn, fresh boot and restart alike, needs both +/// before DDL against group 0 can proceed. +pub async fn wait_for_single_node_ready(node: &TestClusterNode) { + wait_for( + "single-node sequencer leader elected", + Duration::from_secs(10), + Duration::from_millis(50), + || node.sequencer_leader() == node.node_id, + ) + .await; + wait_for( + "single-node metadata leader elected", + Duration::from_secs(10), + Duration::from_millis(50), + || node.shared.is_metadata_leader(), + ) + .await; +} + +/// Render a pgwire error as `SQLSTATE: message`, or its display text when +/// the server sent no database error. +pub fn db_detail(e: &tokio_postgres::Error) -> String { + match e.as_db_error() { + Some(db) => format!("{}: {}", db.code().code(), db.message()), + None => format!("{e}"), + } +} + +/// Take a backup of `tenant` over `client` and return the envelope bytes. +/// Panics with the server's error when the backup fails. +pub async fn drain_backup(client: &tokio_postgres::Client, tenant: u64) -> Vec { + let stream = client + .copy_out(&format!("COPY (BACKUP TENANT {tenant}) TO STDOUT")) + .await + .unwrap_or_else(|e| panic!("BACKUP TENANT {tenant}: {}", db_detail(&e))); + let mut bytes = Vec::new(); + let mut stream = Box::pin(stream); + while let Some(chunk) = stream.next().await { + bytes.extend_from_slice( + &chunk.unwrap_or_else(|e| panic!("backup chunk: {}", db_detail(&e))), + ); + } + bytes +} + +/// Restore `envelope` into `tenant` over `client`. The error carries the +/// server's error text. +pub async fn try_push_restore( + client: &tokio_postgres::Client, + tenant: u64, + envelope: Vec, +) -> Result<(), String> { + let sink = client + .copy_in::<_, Bytes>(&format!("COPY tenant_restore({tenant}) FROM STDIN")) + .await + .map_err(|e| db_detail(&e))?; + let mut sink = Box::pin(sink); + sink.as_mut() + .send(Bytes::from(envelope)) + .await + .map_err(|e| db_detail(&e))?; + sink.as_mut() + .finish() + .await + .map(|_| ()) + .map_err(|e| db_detail(&e)) +} + +/// Restore `envelope` into `tenant` over `client`. Panics with the server's +/// error when the restore fails. +pub async fn push_restore(client: &tokio_postgres::Client, tenant: u64, envelope: Vec) { + try_push_restore(client, tenant, envelope) + .await + .unwrap_or_else(|e| panic!("RESTORE tenant {tenant}: {e}")); +} + +/// Switch every node's harness session to `database`. +pub async fn use_database(cluster: &TestCluster, database: &str) { + for node in &cluster.nodes { + node.exec(&format!("USE DATABASE {database}")) + .await + .unwrap_or_else(|e| panic!("USE DATABASE {database} on node {}: {e}", node.node_id)); + } +} + +/// The id `node`'s catalog holds for the database `name`. +pub fn database_id(node: &TestClusterNode, name: &str) -> DatabaseId { + node.shared + .credentials + .catalog() + .get_database_id_by_name(name) + .expect("look up database id") + .unwrap_or_else(|| panic!("node {} lacks database '{name}'", node.node_id)) +} + +/// Whether `node` leads vShard 0's group under a valid leader lease. +pub fn holds_vshard0_lease(node: &TestClusterNode) -> bool { + let group = node + .shared + .cluster_routing + .as_ref() + .expect("cluster routing") + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(0) + .expect("vShard 0 maps to a group"); + node.shared + .raft_read_gate + .get() + .is_some_and(|gate| gate.holds_leader_lease(group)) +} + +/// Whether any Data Plane core of `node` has fail-stopped. +pub fn fail_stopped(node: &TestClusterNode) -> bool { + node.shared + .system_metrics + .as_ref() + .is_some_and(|metrics| metrics.core_fail_stops.is_stopped()) +} + +/// The number of transactions the sequencer on `node` has admitted, or 0 +/// when `node` runs no sequencer. +pub fn sequencer_admitted(node: &TestClusterNode) -> u64 { + node.shared + .sequencer_metrics + .get() + .map(|m| m.admitted_total.load(Ordering::Relaxed)) + .unwrap_or(0) +} + +/// Propose `entry` to group 0 and wait for it to apply on `node`. +pub async fn propose_and_apply(node: &TestClusterNode, entry: &MetadataEntry) { + let handle = node + .shared + .metadata_raft + .get() + .expect("metadata raft handle installed"); + let index = handle + .propose_async(encode_entry(entry).expect("encode metadata entry")) + .await + .expect("propose metadata entry"); + let watcher = node.shared.applied_index_watcher(METADATA_GROUP_ID); + wait_for( + "metadata entry applied", + Duration::from_secs(10), + Duration::from_millis(20), + || watcher.current() >= index, + ) + .await; +} + +/// The rows of the timeseries `collection` in the default database of +/// `tenant` that `node`'s own replica stores, each rendered as JSON text and +/// sorted. +pub async fn local_timeseries_rows( + node: &TestClusterNode, + tenant: u64, + collection: &str, +) -> Vec { + let mut rows: Vec = node + .timeseries_rows_local(TenantId::new(tenant), collection) + .await + .iter() + .map(|row| sonic_rs::to_string(row).expect("render a row")) + .collect(); + rows.sort(); + rows +} + +/// The data group `collection` maps to on `node`. +pub fn group_of(node: &TestClusterNode, collection: &str) -> u64 { + node.group_id_for_collection(collection) + .unwrap_or_else(|| panic!("no group mapping for {collection}")) +} + +/// The data group `key` hashes to in `node`'s routing table. +pub fn group_of_key(node: &TestClusterNode, key: &str) -> u64 { + let routing = node + .shared + .cluster_routing + .as_ref() + .expect("cluster routing"); + let table = routing.read().unwrap_or_else(|p| p.into_inner()); + table + .group_for_vshard(VShardId::from_key(key.as_bytes()).as_u32()) + .expect("vshard maps to a group") +} + +/// The member node ids of `group_id` in `node`'s routing table. +pub fn group_members(node: &TestClusterNode, group_id: u64) -> Vec { + let routing = node + .shared + .cluster_routing + .as_ref() + .expect("cluster routing"); + let table = routing.read().unwrap_or_else(|p| p.into_inner()); + table + .group_info(group_id) + .map(|info| info.members.clone()) + .unwrap_or_default() +} + +/// `node`'s Raft status of `group_id`, when it hosts the group. +pub fn group_status(node: &TestClusterNode, group_id: u64) -> Option { + node.shared + .cluster_observer + .get()? + .group_status + .upgrade()? + .group_statuses() + .into_iter() + .find(|g| g.group_id == group_id) +} + +/// The collection name inside a document storage key, `None` when the key +/// has no collection part. +pub fn key_collection(key: &str) -> Option<&str> { + let rest = key.splitn(3, ':').nth(2)?; + rest.split([':', '\u{0}']).next() +} + +/// The index in `cluster.nodes` of the node that leads `collection`'s data +/// group. Panics when no live node leads it. +pub fn leader_index_of(cluster: &TestCluster, collection: &str) -> usize { + let probe = &cluster.nodes[0]; + let group = group_of(probe, collection); + let leader = leader_of(probe, group); + cluster + .nodes + .iter() + .position(|node| node.node_id == leader) + .unwrap_or_else(|| panic!("no live node leads {collection}'s group {group}")) +} + +/// Shut down `cluster.nodes[idx]`, remove it from the cluster, and wait for +/// the survivors to elect a live leader in every group. +pub async fn kill_node(cluster: &mut TestCluster, idx: usize) { + let dead = cluster.nodes.remove(idx); + let dead_id = dead.node_id; + dead.shutdown().await; + wait_for( + "the survivors elect new leaders", + Duration::from_secs(30), + Duration::from_millis(100), + || { + cluster.nodes.iter().all(|node| { + node.all_group_leaders() + .into_iter() + .all(|(_, leader)| leader != 0 && leader != dead_id) + }) + }, + ) + .await; +} + +/// The leader `node` sees for `group_id`, or 0 when it sees none. +pub fn leader_of(node: &TestClusterNode, group_id: u64) -> u64 { + node.all_group_leaders() + .into_iter() + .find(|(group, _)| *group == group_id) + .map(|(_, leader)| leader) + .unwrap_or(0) +} + +/// The number of rows `SELECT id` returns for `collection` on `node`. +pub async fn row_count(node: &TestClusterNode, collection: &str) -> usize { + let rows = node + .client + .simple_query(&format!("SELECT id FROM {collection}")) + .await + .unwrap_or_else(|e| panic!("SELECT from {collection}: {e}")); + rows.iter() + .filter(|msg| matches!(msg, tokio_postgres::SimpleQueryMessage::Row(_))) + .count() +} + +/// The first collection name `_` for which `pick` holds. +pub fn name_where(prefix: &str, pick: impl Fn(&str) -> bool) -> String { + (0..4096u32) + .map(|i| format!("{prefix}_{i}")) + .find(|name| pick(name)) + .unwrap_or_else(|| panic!("no collection name for prefix {prefix}")) +} diff --git a/nodedb-test-support/src/cluster_harness/wait.rs b/nodedb-test-support/src/cluster_harness/wait.rs index 4a43e832f..5643a5771 100644 --- a/nodedb-test-support/src/cluster_harness/wait.rs +++ b/nodedb-test-support/src/cluster_harness/wait.rs @@ -23,6 +23,26 @@ pub async fn wait_for bool>( panic!("timed out after {:?} waiting for: {}", deadline, desc); } +/// [`wait_for`] with a check that says why it does not hold yet. The +/// timeout message carries the last reason. +pub async fn wait_for_report Result<(), String>>( + desc: &str, + deadline: Duration, + step: Duration, + mut check: F, +) { + let start = Instant::now(); + let mut last = String::new(); + while start.elapsed() < deadline { + match check() { + Ok(()) => return, + Err(reason) => last = reason, + } + tokio::time::sleep(step).await; + } + panic!("timed out after {deadline:?} waiting for: {desc}; last check: {last}"); +} + /// Async predicate variant of [`wait_for`]. Awaits the future returned /// by `pred` directly so callers don't need `block_in_place` / /// `Handle::block_on` gymnastics inside an async context. diff --git a/nodedb-test-support/src/core_loop_runner.rs b/nodedb-test-support/src/core_loop_runner.rs index a17642c8a..ef19a6922 100644 --- a/nodedb-test-support/src/core_loop_runner.rs +++ b/nodedb-test-support/src/core_loop_runner.rs @@ -29,8 +29,10 @@ use std::time::Duration; use nodedb::bridge::dispatch::CoreChannelDataSide; use nodedb::control::array_catalog::ArrayCatalog; use nodedb::control::metrics::SystemMetrics; +use nodedb::control::state::SharedState; use nodedb::data::executor::core_loop::CoreLoop; use nodedb::event::EventProducer; +use nodedb::event::interest::EventInterest; use nodedb_mem::MemoryGovernor; use nodedb_types::OrdinalClock; use nodedb_wal::{TombstoneSet, WalRecord}; @@ -91,6 +93,11 @@ pub struct CoreLoopSpawn { /// unless a test overrides e.g. `columnar_flush_threshold` to drive flush /// on a small dataset. pub query_tuning: nodedb_types::config::tuning::QueryTuning, + /// Timeseries engine tuning wired via `core.set_timeseries_tuning`. + /// Production (`data::runtime`) wires this from the node's config. The + /// harness passes `TimeseriesToning::default()` unless a test lowers one + /// node's memtable budget or tag cardinality. + pub timeseries_tuning: nodedb_types::config::tuning::TimeseriesToning, /// Catalog-sourced `doc_configs` seed, applied BEFORE WAL replay exactly as /// production's `seed_catalog_state` does. /// @@ -101,10 +108,21 @@ pub struct CoreLoopSpawn { /// `timestamp`. That made every restart test here quietly weaker than it /// looked — none of them exercised the seeded path production always takes. pub doc_config_seed: Vec, + /// The collections some Event Plane consumer reads, as production boot + /// wires them. Build it with [`event_interest_for`]. + pub event_interest: Arc, /// Stop signal for the tick loop. Sender lives in the harness shutdown path. pub stop_rx: std::sync::mpsc::Receiver<()>, } +/// The consumed-collection set of `shared`'s registries, as production boot +/// installs it once `SharedState` exists. +pub fn event_interest_for(shared: &SharedState) -> Arc { + let interest = EventInterest::new(); + interest.install(shared.event_interest_sources()); + interest +} + /// Spawn a `CoreLoop` for one Data Plane core inside `tokio::spawn_blocking`. /// /// Returns the `JoinHandle` for the blocking task. The inner OS thread is @@ -123,7 +141,9 @@ pub fn spawn_core_loop(spawn: CoreLoopSpawn) -> tokio::task::JoinHandle<()> { replay, graph_tuning, query_tuning, + timeseries_tuning, doc_config_seed, + event_interest, stop_rx, } = spawn; @@ -143,9 +163,11 @@ pub fn spawn_core_loop(spawn: CoreLoopSpawn) -> tokio::task::JoinHandle<()> { ) .expect("CoreLoop::open_with_array_catalog"); core.set_event_producer(event_producer); + core.set_event_interest(event_interest); core.set_num_cores(num_cores); core.set_query_tuning(query_tuning); core.set_graph_tuning(graph_tuning); + core.set_timeseries_tuning(timeseries_tuning); if let Some(m) = core_metrics { core.set_metrics(m); } @@ -159,7 +181,8 @@ pub fn spawn_core_loop(spawn: CoreLoopSpawn) -> tokio::task::JoinHandle<()> { num_cores: replay_num_cores, }) = replay { - core.replay_all_wal(&records, replay_num_cores, &tombstones); + core.replay_all_wal(&records, replay_num_cores, &tombstones) + .expect("WAL replay of the test core"); } while matches!( stop_rx.try_recv(), diff --git a/nodedb-test-support/src/ilp_client.rs b/nodedb-test-support/src/ilp_client.rs index 71cdb6288..dabeb3815 100644 --- a/nodedb-test-support/src/ilp_client.rs +++ b/nodedb-test-support/src/ilp_client.rs @@ -12,8 +12,6 @@ //! ILP lines are newline-delimited raw text written directly to the stream: //! no further framing, and no per-line acknowledgement. -#![allow(dead_code)] // Not every test binary exercises every helper. - use tokio::io::AsyncWriteExt; use tokio::net::TcpStream; @@ -44,6 +42,37 @@ pub async fn connect_and_auth_from( addr: std::net::SocketAddr, username: &str, password: &str, +) -> TcpStream { + connect_with( + source, + addr, + AuthMethod::Password { + username: username.into(), + password: password.into(), + }, + ) + .await +} + +/// Like [`connect_and_auth`], against a server in trust mode: the Auth frame +/// names `username` and carries no password. +pub async fn connect_trust(addr: std::net::SocketAddr, username: &str) -> TcpStream { + connect_with( + None, + addr, + AuthMethod::Trust { + username: username.into(), + }, + ) + .await +} + +/// Complete the Hello + Auth prelude with `auth` and return the stream ready +/// for raw ILP lines. +async fn connect_with( + source: Option, + addr: std::net::SocketAddr, + auth: AuthMethod, ) -> TcpStream { let (mut stream, _ack) = do_handshake_from(source, addr, &HelloFrame::current()) .await @@ -54,10 +83,7 @@ pub async fn connect_and_auth_from( 1, OpCode::Auth, TextFields { - auth: Some(AuthMethod::Password { - username: username.into(), - password: password.into(), - }), + auth: Some(auth), ..Default::default() }, ) diff --git a/nodedb-test-support/src/kv_rows.rs b/nodedb-test-support/src/kv_rows.rs new file mode 100644 index 000000000..8679cb7c8 --- /dev/null +++ b/nodedb-test-support/src/kv_rows.rs @@ -0,0 +1,19 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Bound identities for KV rows a test writes straight to the Data Plane. + +use nodedb_types::Surrogate; + +/// A bound surrogate for a KV test row, derived from its key (FNV-1a). +/// +/// The same key always maps to the same value, and distinct keys map to +/// distinct values with overwhelming likelihood. The value is never +/// `Surrogate::ZERO`, which every KV write path refuses. +pub fn kv_row_surrogate(key: &[u8]) -> Surrogate { + const FNV_OFFSET: u32 = 0x811c_9dc5; + const FNV_PRIME: u32 = 0x0100_0193; + let hash = key.iter().fold(FNV_OFFSET, |hash, byte| { + (hash ^ u32::from(*byte)).wrapping_mul(FNV_PRIME) + }); + Surrogate::new(hash.max(1)) +} diff --git a/nodedb-test-support/src/lib.rs b/nodedb-test-support/src/lib.rs index 6a79b6b09..f2c785607 100644 --- a/nodedb-test-support/src/lib.rs +++ b/nodedb-test-support/src/lib.rs @@ -3,15 +3,19 @@ //! Shared helpers for integration tests. pub mod array_sync; +pub mod booted_state; +pub mod catalog_fixtures; pub mod cluster_harness; pub mod core_loop_runner; pub mod ilp_client; pub mod insert_returning_engines; pub mod jwks_fixture; +pub mod kv_rows; pub mod native_harness; pub mod occ_shuffle; pub mod pgwire_auth_helpers; pub mod pgwire_harness; +pub mod single_node; pub mod sync_client; pub mod test_tracing; pub mod tx_batch_helpers; @@ -21,7 +25,6 @@ use nodedb::event::cdc::event::CdcEvent; use nodedb_types::DatabaseId; /// Current time in milliseconds since UNIX epoch. -#[allow(dead_code)] pub fn now_ms() -> u64 { std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) @@ -30,7 +33,6 @@ pub fn now_ms() -> u64 { } /// Create a [`CdcEvent`] with sensible test defaults. -#[allow(dead_code)] pub fn make_cdc_event( database_id: DatabaseId, seq: u64, @@ -46,6 +48,8 @@ pub fn make_cdc_event( row_id: format!("r-{seq}"), event_time: now_ms(), lsn: seq * 10, + index: seq * 10, + epoch: 0, database_id, tenant_id: 1, new_value: Some(serde_json::json!({"id": seq})), diff --git a/nodedb-test-support/src/native_harness/frames.rs b/nodedb-test-support/src/native_harness/frames.rs index 0ccbd8588..f99c9f2b7 100644 --- a/nodedb-test-support/src/native_harness/frames.rs +++ b/nodedb-test-support/src/native_harness/frames.rs @@ -12,7 +12,7 @@ use nodedb_types::protocol::request_fields::RequestFields; use nodedb_types::protocol::text_fields::TextFields; use nodedb_types::protocol::{ AuthMethod, FRAME_HEADER_LEN, HELLO_ACK_MAGIC, HELLO_ERROR_MAGIC_U32, HelloAckFrame, - HelloErrorFrame, HelloFrame, NativeRequest, NativeResponse, OpCode, + HelloErrorFrame, HelloFrame, NativeRequest, NativeResponse, OpCode, ResponseStatus, }; /// Perform the handshake with a custom `HelloFrame`. @@ -223,6 +223,31 @@ pub async fn send_api_key_auth(stream: &mut TcpStream, seq: u64, token: String) .await } +/// Open a native connection to `127.0.0.1:port` and authenticate it as the +/// trust-mode user `username`. The JSON Auth request also selects JSON +/// framing for the rest of the session. Panics when the handshake or the +/// authentication fails. +pub async fn open_trust_session(port: u16, username: &str) -> TcpStream { + let addr = format!("127.0.0.1:{port}").parse().expect("native addr"); + let (mut stream, _ack) = do_handshake(addr, &HelloFrame::current()) + .await + .unwrap_or_else(|e| panic!("native handshake: {e:?}")); + let auth = send_request( + &mut stream, + 1, + OpCode::Auth, + TextFields { + auth: Some(AuthMethod::Trust { + username: username.to_string(), + }), + ..Default::default() + }, + ) + .await; + assert_eq!(auth.status, ResponseStatus::Ok, "native auth: {auth:?}"); + stream +} + /// Send a `SHOW`/SQL statement over an established JSON-encoding session and /// decode the `NativeResponse`. Assumes the session's first frame already /// selected JSON (see `json_request_gets_json_response`) — callers that open diff --git a/nodedb-test-support/src/native_harness/mod.rs b/nodedb-test-support/src/native_harness/mod.rs index f0ac31cbb..cb0df1c10 100644 --- a/nodedb-test-support/src/native_harness/mod.rs +++ b/nodedb-test-support/src/native_harness/mod.rs @@ -10,7 +10,7 @@ mod frames; mod server; pub use frames::{ - do_handshake, do_handshake_from, read_frame, send_api_key_auth, send_request, send_sql, - write_frame, + do_handshake, do_handshake_from, open_trust_session, read_frame, send_api_key_auth, + send_request, send_sql, write_frame, }; pub use server::NativeTestServer; diff --git a/nodedb-test-support/src/native_harness/server.rs b/nodedb-test-support/src/native_harness/server.rs index 3bf526426..44908e9a3 100644 --- a/nodedb-test-support/src/native_harness/server.rs +++ b/nodedb-test-support/src/native_harness/server.rs @@ -14,6 +14,9 @@ use nodedb::data::executor::core_loop::CoreLoop; use nodedb::event::{EventPlane, EventPlaneConfig, create_event_bus}; use nodedb::wal::WalManager; +/// The backup KEK every native test server wraps backups with. +pub const NATIVE_TEST_BACKUP_KEK: [u8; 32] = [0x24u8; 32]; + /// A running native-protocol test server. pub struct NativeTestServer { pub addr: std::net::SocketAddr, @@ -27,6 +30,8 @@ pub struct NativeTestServer { pub(super) _poller_handle: tokio::task::JoinHandle<()>, pub(super) _core_handle: tokio::task::JoinHandle<()>, pub(super) _event_plane: EventPlane, + /// The one-node cluster's Raft side: its lease loop and subsystems. + pub(super) raft: crate::single_node::OneNodeRaft, pub(super) _dir: tempfile::TempDir, } @@ -51,7 +56,7 @@ impl NativeTestServer { let (event_producers, event_consumers) = create_event_bus(1); // Use catalog-backed credential store (mirrors pgwire_harness::start) - // so DDL apply (`apply_locally_if_needed`) and planner reads + // so the proposer's apply on this node and planner reads // (`OriginCatalog::get_collection`) resolve against a real catalog and // collections created over the native protocol are visible. let catalog_path = dir.path().join("system.redb"); @@ -74,11 +79,29 @@ impl NativeTestServer { // Ensure the built-in `default` database (id 0) is present in the // catalog so the default connection database works in tests. let _ = credentials.catalog().bootstrap_default_database(); - let shared = + let mut shared = SharedState::new_with_credentials(dispatcher, Arc::clone(&wal), credentials, false) .expect("build shared state"); + // What production boot wires from `[backup_encryption]` and + // `[backup_storage]`: `file://` backup URIs resolve inside the + // server's `backups` directory. + let backup_root = dir.path().join("backups"); + std::fs::create_dir_all(&backup_root).expect("create backup root"); + let cluster = crate::single_node::init(dir.path()) + .await + .expect("init the one-node cluster"); + { + let state = Arc::get_mut(&mut shared).expect("state is not shared yet"); + crate::single_node::wire(state, &cluster, dir.path()) + .expect("wire the one-node cluster"); + state.backup_kek = Some(Arc::new(NATIVE_TEST_BACKUP_KEK)); + state.backup_storage = Some(Arc::new(nodedb::config::server::BackupStorageSettings { + local_root: Some(backup_root), + ..Default::default() + })); + } // The same gateway install production boot runs. - nodedb::bootstrap::state_wiring::install_gateway(&shared); + nodedb::bootstrap::state_wiring::install_gateway(&shared).expect("install gateway"); let data_side = data_sides.into_iter().next().expect("data side"); let core_dir = dir.path().to_path_buf(); @@ -144,6 +167,10 @@ impl NativeTestServer { shutdown_bus: shutdown_bus.clone(), }); + let raft = crate::single_node::start(&cluster, &shared, dir.path()) + .await + .expect("start the one-node cluster"); + let listener = Listener::bind("127.0.0.1:0".parse().expect("addr")) .await .expect("bind"); @@ -183,15 +210,25 @@ impl NativeTestServer { _poller_handle, _core_handle, _event_plane, + raft, _dir: dir, } } + /// A `file://` URI of `name` inside this server's backup root. + pub fn backup_uri(&self, name: &str) -> String { + format!( + "file://{}/{name}", + self._dir.path().join("backups").display() + ) + } + /// Shut down the server and give background tasks time to unwind. - pub async fn shutdown(self) { + pub async fn shutdown(mut self) { self.shutdown_bus.initiate(); let _ = self.poller_shutdown_tx.send(true); let _ = self.core_stop_tx.send(()); + self.raft.shutdown(&self.shared).await; tokio::time::sleep(Duration::from_millis(50)).await; } } diff --git a/nodedb-test-support/src/pgwire_auth_helpers.rs b/nodedb-test-support/src/pgwire_auth_helpers.rs index 1546a8dae..4d74bd506 100644 --- a/nodedb-test-support/src/pgwire_auth_helpers.rs +++ b/nodedb-test-support/src/pgwire_auth_helpers.rs @@ -2,65 +2,51 @@ //! Shared fixtures for `pgwire_auth_*` integration tests. //! -//! Each split test file needs: a minimal `SharedState`, two canonical +//! Each split test file needs: a `SharedState` that serves DDL, two canonical //! identities (superuser + readonly), and two DDL runners (expect ok / //! expect err). Keeping them here avoids copy-paste drift across files. +//! +//! A DDL proposes through the metadata group, so every state here is served +//! by a booted one-node cluster ([`BootedState`]). #![allow(dead_code)] -use std::sync::Arc; - -use nodedb::bridge::dispatch::Dispatcher; use nodedb::control::security::identity::{AuthMethod, AuthenticatedIdentity, DatabaseSet, Role}; use nodedb::control::server::pgwire::ddl_encode; use nodedb::control::server::shared::ddl; use nodedb::control::server::shared::session::DetachedTxnScope; use nodedb::control::state::SharedState; use nodedb::types::TenantId; -use nodedb::wal::WalManager; - -/// Shorten the Data Plane dispatch deadline on a fixture whose data side was -/// dropped at construction. -/// -/// These fixtures own no live core, so any dispatch can only ever time out. -/// The production deadline turns each such test into a 30s wall wait. -fn shorten_dead_core_deadline(state: &mut Arc) { - if let Some(state) = Arc::get_mut(state) { - state.tuning.network.default_deadline_secs = 1; - } + +use crate::booted_state::BootOptions; +pub use crate::booted_state::BootedState; + +/// A booted state whose catalog holds no database yet. +pub fn make_state() -> BootedState { + BootedState::boot(BootOptions { + default_database: false, + ..BootOptions::default() + }) } -/// Create a minimal `SharedState` (no Data Plane needed for DDL tests). -pub fn make_state() -> Arc { - let dir = tempfile::tempdir().unwrap(); - let wal_path = dir.path().join("test.wal"); - let wal = Arc::new(WalManager::open_for_testing(&wal_path).unwrap()); - let (dispatcher, _data_sides) = Dispatcher::new(1, 64); - let mut state = SharedState::new(dispatcher, wal).expect("build shared state"); - shorten_dead_core_deadline(&mut state); - state +/// A booted state whose catalog holds the built-in `default` database. Use +/// this for DDL tests that resolve database names (e.g. +/// `FOR DATABASE default`). +pub fn make_state_with_catalog() -> BootedState { + BootedState::boot(BootOptions::default()) } -/// Create a `SharedState` whose `CredentialStore` is backed by a real redb -/// catalog and the built-in `default` database is bootstrapped. Use this for -/// DDL tests that resolve database names (e.g. `FOR DATABASE default`). -pub fn make_state_with_catalog() -> Arc { - let dir = tempfile::tempdir().unwrap(); - let wal_path = dir.path().join("test.wal"); - let wal = Arc::new(WalManager::open_for_testing(&wal_path).unwrap()); - let catalog_path = dir.path().join("system.redb"); - let credentials = Arc::new( - nodedb::control::security::credential::store::CredentialStore::open(&catalog_path).unwrap(), - ); - let _ = credentials.catalog().bootstrap_default_database(); - let (dispatcher, _data_sides) = Dispatcher::new(1, 64); - let mut state = SharedState::new_with_credentials(dispatcher, wal, credentials, false) - .expect("build shared state"); - shorten_dead_core_deadline(&mut state); - state - // `dir` drops here. On Linux, file handles held by `wal` and the redb - // catalog keep both files readable for the test's lifetime even after - // the directory entry is removed (open-then-unlink semantics). +/// [`make_state_with_catalog`], with `configure` applied to the state before +/// the gateway install. The installed gateway holds a back-reference, so no +/// `Arc::get_mut` succeeds after this returns: set every field here. +pub fn make_state_with_catalog_configured( + configure: impl FnOnce(&mut SharedState) + Send + 'static, +) -> BootedState { + BootedState::boot(BootOptions { + default_database: true, + configure: Box::new(configure), + ..BootOptions::default() + }) } /// Superuser identity for DDL tests. @@ -90,16 +76,20 @@ pub fn readonly_user() -> AuthenticatedIdentity { /// Run DDL, expect success. pub async fn ddl_ok(state: &SharedState, identity: &AuthenticatedIdentity, sql: &str) { + ddl_ok_in(state, identity, sql, nodedb_types::id::DatabaseId::DEFAULT).await; +} + +/// Run DDL with `database_id` as the session database, expect success. +pub async fn ddl_ok_in( + state: &SharedState, + identity: &AuthenticatedIdentity, + sql: &str, + database_id: nodedb_types::id::DatabaseId, +) { let scope = DetachedTxnScope::new(); - let result = ddl::dispatch( - state, - identity, - sql, - nodedb_types::id::DatabaseId::DEFAULT, - &scope.ctx(), - ) - .await - .map(ddl_encode::ddl_results_to_pgwire); + let result = ddl::dispatch(state, identity, sql, database_id, &scope.ctx()) + .await + .map(ddl_encode::ddl_results_to_pgwire); assert!(result.is_some(), "DDL not recognized: {sql}"); result .unwrap() diff --git a/nodedb-test-support/src/pgwire_harness/backup.rs b/nodedb-test-support/src/pgwire_harness/backup.rs new file mode 100644 index 000000000..3e8f1249e --- /dev/null +++ b/nodedb-test-support/src/pgwire_harness/backup.rs @@ -0,0 +1,50 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `TestServer` with a backup root: `[backup_storage] local_root` set, so +//! `BACKUP DATABASE ... TO 'file://...'`, `RESTORE DATABASE` and scheduled +//! backups write and read inside the server's data directory. + +use std::path::PathBuf; + +use super::start::StartConfig; +use super::types::TestServer; + +/// The backup root, relative to the server's data directory. +pub(super) const BACKUP_DIR: &str = "backups"; + +impl TestServer { + /// Spawn a single-core server whose `file://` backup URIs resolve inside + /// [`Self::backup_root`]. All other settings stay at their defaults. + pub async fn start_with_backup_root() -> Self { + Self::start_with_config(StartConfig { + backup_root: true, + ..Default::default() + }) + .await + } + + /// [`Self::start_with_backup_root`] with usage metering replaced by + /// `metering`, so a backup or restore runs quota admission and charges. + pub async fn start_with_metering_and_backup_root( + metering: nodedb::control::security::metering::config::MeteringConfig, + ) -> Self { + Self::start_with_config(StartConfig { + backup_root: true, + metering: Some(metering), + ..Default::default() + }) + .await + } + + /// The server's `[backup_storage] local_root`, canonical so it matches + /// the path in every `file://` URI the server accepts. + pub fn backup_root(&self) -> PathBuf { + let root = self._dir.path().join(BACKUP_DIR); + root.canonicalize().unwrap_or(root) + } + + /// A `file://` URI of `name` inside [`Self::backup_root`]. + pub fn backup_uri(&self, name: &str) -> String { + format!("file://{}/{name}", self.backup_root().display()) + } +} diff --git a/nodedb-test-support/src/pgwire_harness/mod.rs b/nodedb-test-support/src/pgwire_harness/mod.rs index 47aa90af5..55ddde016 100644 --- a/nodedb-test-support/src/pgwire_harness/mod.rs +++ b/nodedb-test-support/src/pgwire_harness/mod.rs @@ -3,12 +3,13 @@ //! Shared pgwire end-to-end test harness. //! //! Spawns a full NodeDB server (Data Plane core + pgwire listener + response -//! poller) and provides a connected `tokio_postgres::Client` for SQL execution. +//! poller) on a one-node Raft cluster booted by [`crate::single_node`], and +//! provides a connected `tokio_postgres::Client` for SQL execution. +mod backup; mod multicore; mod query; pub mod raw_pgwire; -mod read_gate; mod restart; mod start; mod support; diff --git a/nodedb-test-support/src/pgwire_harness/multicore.rs b/nodedb-test-support/src/pgwire_harness/multicore.rs index 0f71f6873..24b04b9bb 100644 --- a/nodedb-test-support/src/pgwire_harness/multicore.rs +++ b/nodedb-test-support/src/pgwire_harness/multicore.rs @@ -18,7 +18,6 @@ use nodedb::wal::WalManager; use super::support::{bind_http_listener, bind_native_listener, init_test_memory_governor}; use super::types::{TestClient, TestServer}; -#[allow(dead_code)] impl TestServer { /// Spawn an N-core NodeDB server and connect via pgwire. /// @@ -46,14 +45,19 @@ impl TestServer { let mut shared = SharedState::new_with_credentials(dispatcher, Arc::clone(&wal), credentials, false) .expect("build shared state"); - if let Some(s) = Arc::get_mut(&mut shared) { + let cluster = crate::single_node::init(dir.path()) + .await + .expect("init the one-node cluster"); + { + let s = Arc::get_mut(&mut shared).expect("shared state is not cloned yet"); + crate::single_node::wire(s, &cluster, dir.path()).expect("wire the one-node cluster"); s.backup_kek = Some(Arc::new([0x42u8; 32])); s.governor = init_test_memory_governor(); } let shared = shared; // The same gateway install production boot runs, after every // `Arc::get_mut` above. - nodedb::bootstrap::state_wiring::install_gateway(&shared); + nodedb::bootstrap::state_wiring::install_gateway(&shared).expect("install gateway"); let mut core_stop_txs = Vec::new(); let mut core_handles = Vec::new(); @@ -74,6 +78,7 @@ impl TestServer { replay: None, graph_tuning: nodedb_types::config::tuning::GraphTuning::default(), query_tuning: nodedb_types::config::tuning::QueryTuning::default(), + timeseries_tuning: nodedb_types::config::tuning::TimeseriesToning::default(), // Seeded from the SAME durable catalog production reads, so a // harness restart reconstructs cores the way a real one does. // An empty catalog yields an empty seed, which is exactly what @@ -81,6 +86,7 @@ impl TestServer { doc_config_seed: nodedb::bootstrap::data_plane::load_doc_config_registry_from( shared.credentials.catalog(), ), + event_interest: crate::core_loop_runner::event_interest_for(&shared), stop_rx: core_stop_rx, }); core_stop_txs.push(core_stop_tx); @@ -122,6 +128,10 @@ impl TestServer { shutdown_bus: shutdown_bus.clone(), }); + let raft = crate::single_node::start(&cluster, &shared, dir.path()) + .await + .expect("start the one-node cluster"); + let pg_listener = PgListener::bind("127.0.0.1:0".parse().unwrap()) .await .unwrap(); @@ -180,6 +190,7 @@ impl TestServer { poller_handle: Some(poller_handle), core_handles: Some(core_handles), event_plane: Some(event_plane), + raft: Some(raft), _dir: dir, } } diff --git a/nodedb-test-support/src/pgwire_harness/query.rs b/nodedb-test-support/src/pgwire_harness/query.rs index ae94f22cf..6bdf6c8c4 100644 --- a/nodedb-test-support/src/pgwire_harness/query.rs +++ b/nodedb-test-support/src/pgwire_harness/query.rs @@ -5,7 +5,6 @@ use super::types::TestServer; -#[allow(dead_code)] impl TestServer { /// Execute a SQL statement, returning the text of each row's first column. pub async fn query_text(&self, sql: &str) -> Result, String> { diff --git a/nodedb-test-support/src/pgwire_harness/raw_pgwire.rs b/nodedb-test-support/src/pgwire_harness/raw_pgwire.rs index 0d7d70a4b..b2c186068 100644 --- a/nodedb-test-support/src/pgwire_harness/raw_pgwire.rs +++ b/nodedb-test-support/src/pgwire_harness/raw_pgwire.rs @@ -103,6 +103,31 @@ impl RawPgConn { } } +/// The message (`M`) field of every `NoticeResponse` (`N`) in `messages`, in +/// wire order. +pub fn notice_messages(messages: &[(u8, Vec)]) -> Vec { + messages + .iter() + .filter(|(tag, _)| *tag == b'N') + .filter_map(|(_, body)| { + // Fields are `code byte, NUL-terminated string`, closed by a NUL + // code byte. + let mut rest = body.as_slice(); + while let Some((&code, tail)) = rest.split_first() { + if code == 0 { + break; + } + let end = tail.iter().position(|b| *b == 0).unwrap_or(tail.len()); + if code == b'M' { + return Some(String::from_utf8_lossy(&tail[..end]).into_owned()); + } + rest = tail.get(end + 1..).unwrap_or_default(); + } + None + }) + .collect() +} + /// `CommandComplete` (`C`) tag strings in `messages`, in wire order. pub fn command_tags(messages: &[(u8, Vec)]) -> Vec { messages diff --git a/nodedb-test-support/src/pgwire_harness/read_gate.rs b/nodedb-test-support/src/pgwire_harness/read_gate.rs deleted file mode 100644 index e501b7646..000000000 --- a/nodedb-test-support/src/pgwire_harness/read_gate.rs +++ /dev/null @@ -1,73 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! The Raft read gate for a harness that runs as a single-node cluster. -//! -//! Production `start_raft` publishes a gate backed by the Raft loop. A -//! harness started with a routing table runs no Raft loop. Without a gate, -//! every linearizable read refuses with "no leader is currently serving -//! this range". This gate answers both questions the way a group with one -//! voter answers them. - -use std::sync::{Arc, RwLock}; -use std::time::Duration; - -use nodedb::control::cluster::{RaftReadGate, ReadIndexRefusal}; -use nodedb::control::state::SharedState; -use nodedb_cluster::RoutingTable; - -/// Read gate for groups whose only voter is this node. -struct SingleVoterReadGate { - node_id: u64, - routing: Arc>, -} - -impl SingleVoterReadGate { - /// Whether this node is the sole voter and the leader of `group_id`. - fn is_sole_leader(&self, group_id: u64) -> bool { - let routing = self.routing.read().unwrap_or_else(|p| p.into_inner()); - routing - .group_info(group_id) - .is_some_and(|info| info.leader == self.node_id && info.members == [self.node_id]) - } -} - -#[async_trait::async_trait] -impl RaftReadGate for SingleVoterReadGate { - /// A sole voter is its own quorum, so it confirms leadership at once. - /// - /// The harness keeps no Raft log, so the read index is `0`. The caller - /// serves the read from local state. - async fn confirm_leader( - &self, - group_id: u64, - _timeout: Duration, - ) -> Result { - if self.is_sole_leader(group_id) { - Ok(0) - } else { - Err(ReadIndexRefusal::NotLeader) - } - } - - /// A sole voter holds the only copy, so it is never behind. - fn within_staleness_bound(&self, group_id: u64, _max_staleness: Duration) -> bool { - self.is_sole_leader(group_id) - } -} - -/// Publish the single-voter read gate when `shared` carries a routing table. -/// -/// `raft_read_gate` is a `OnceLock`, so this runs once, after every -/// `Arc::get_mut` install. -pub(super) fn install_single_voter_read_gate(shared: &SharedState) { - let Some(routing) = shared.cluster_routing.as_ref() else { - return; - }; - let gate: Arc = Arc::new(SingleVoterReadGate { - node_id: shared.node_id, - routing: Arc::clone(routing), - }); - if shared.raft_read_gate.set(gate).is_err() { - panic!("harness raft_read_gate installed twice"); - } -} diff --git a/nodedb-test-support/src/pgwire_harness/restart.rs b/nodedb-test-support/src/pgwire_harness/restart.rs index 4a0e1f7bf..860b419a6 100644 --- a/nodedb-test-support/src/pgwire_harness/restart.rs +++ b/nodedb-test-support/src/pgwire_harness/restart.rs @@ -18,7 +18,6 @@ use nodedb::wal::WalManager; use super::support::{bind_http_listener, bind_native_listener, init_test_memory_governor}; use super::types::{TestClient, TestDataDir, TestServer}; -#[allow(dead_code)] impl TestServer { /// Consume the server, send shutdown signals, and await all core threads. /// @@ -103,6 +102,25 @@ impl TestServer { .loop_registry .shutdown_all(std::time::Duration::from_secs(5)) .await; + // The lease loop, the cluster subsystems and the transport. The Raft + // loops saw the shutdown signal the bus sent above. + if let Some(mut raft) = self.raft.take() { + raft.shutdown(&self.shared).await; + } + // Every Raft task holds an `Arc` until it observes the + // signal. The catalog redb stays locked until the last clone drops, + // so wait for this handle to be the only one left. + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + while Arc::strong_count(&self.shared) > 1 { + if tokio::time::Instant::now() >= deadline { + eprintln!( + "pgwire_harness: SharedState still has {} strong refs after 5s", + Arc::strong_count(&self.shared) + ); + break; + } + tokio::time::sleep(Duration::from_millis(25)).await; + } // conn_handle, shared, _dir all drop here, releasing the remaining // Arc and CredentialStore redb handle. } @@ -195,33 +213,25 @@ impl TestServer { let mut shared = SharedState::new_with_credentials(dispatcher, Arc::clone(&wal), credentials, false) .expect("build shared state"); - if let Some(s) = Arc::get_mut(&mut shared) { + // The directory's cluster catalog restarts the same one-node cluster. + let cluster = crate::single_node::init(dir_path) + .await + .expect("restart the one-node cluster"); + { + let s = Arc::get_mut(&mut shared).expect("shared state is not cloned yet"); + crate::single_node::wire(s, &cluster, dir_path).expect("wire the one-node cluster"); s.backup_kek = Some(Arc::new([0x42u8; 32])); s.governor = init_test_memory_governor(); } let shared = shared; // The same gateway install production boot runs, after every // `Arc::get_mut` above. - nodedb::bootstrap::state_wiring::install_gateway(&shared); + nodedb::bootstrap::state_wiring::install_gateway(&shared).expect("install gateway"); nodedb::bootstrap::credentials::replay_surrogate_wal( &shared, &wal_records, &replay_tombstones, ); - // Restore in-memory synonym registry from the persisted catalog. - let catalog = shared.credentials.catalog(); - if let Err(e) = shared.synonym_registry.reload_from_catalog(catalog) { - eprintln!("pgwire_harness: failed to reload synonym groups: {e}"); - } - let catalog = shared.credentials.catalog(); - if let Ok(entries) = catalog.load_all_arrays() - && let Ok(mut guard) = shared.array_catalog.write() - { - for entry in entries { - let _ = guard.register(entry); - } - } - let mut core_stop_txs = Vec::new(); let mut core_handles = Vec::new(); for (idx, (data_side, event_producer)) in @@ -254,6 +264,7 @@ impl TestServer { } qt }, + timeseries_tuning: nodedb_types::config::tuning::TimeseriesToning::default(), // Seeded from the SAME durable catalog production reads, so a // harness restart reconstructs cores the way a real one does. // An empty catalog yields an empty seed, which is exactly what @@ -261,6 +272,7 @@ impl TestServer { doc_config_seed: nodedb::bootstrap::data_plane::load_doc_config_registry_from( shared.credentials.catalog(), ), + event_interest: crate::core_loop_runner::event_interest_for(&shared), stop_rx: core_stop_rx, }); core_stop_txs.push(core_stop_tx); @@ -284,26 +296,6 @@ impl TestServer { } }); - nodedb::bootstrap::schema_rehydrate::rehydrate_schema_registry(&shared) - .await - .expect("schema rehydration on restart"); - - // Re-register every persisted continuous aggregate on the local - // Data Plane manager: the registry is per-core in-memory state - // and is otherwise lost across restart. - nodedb::control::server::shared::ddl::neutral::continuous_agg::register_persisted_continuous_aggregates( - &shared, - ) - .await; - - // Rehydrate the AFTER-trigger registry from the catalog before the - // Event Plane starts — it is per-process in-memory state and is - // otherwise empty after a restart, so replayed/live events would match - // no trigger. Mirrors the production boot sequence. - shared - .trigger_registry - .load_all(shared.credentials.catalog()); - let watermark_store = Arc::new(nodedb::event::watermark::WatermarkStore::open(dir_path).unwrap()); let trigger_dlq = Arc::new(std::sync::Mutex::new( @@ -322,11 +314,13 @@ impl TestServer { shutdown_bus: shutdown_bus.clone(), }); - // Load grants and hierarchy edges before the listener opens, as the - // production boot does once the data groups replayed. - nodedb::bootstrap::permission_tree_load::load_permission_trees(&shared) + // Raft replays the metadata and data groups. The readiness wait then + // rehydrates the schema registry and the continuous aggregates and + // loads the permission trees, as production boot does before it + // opens a listener. + let raft = crate::single_node::start(&cluster, &shared, dir_path) .await - .expect("permission tree load on restart"); + .expect("restart the one-node cluster's Raft"); let pg_listener = PgListener::bind("127.0.0.1:0".parse().unwrap()) .await @@ -400,6 +394,7 @@ impl TestServer { poller_handle: Some(poller_handle), core_handles: Some(core_handles), event_plane: Some(event_plane), + raft: Some(raft), _dir: placeholder_dir, } } diff --git a/nodedb-test-support/src/pgwire_harness/start.rs b/nodedb-test-support/src/pgwire_harness/start.rs index 782b8a4a0..86c27cdcd 100644 --- a/nodedb-test-support/src/pgwire_harness/start.rs +++ b/nodedb-test-support/src/pgwire_harness/start.rs @@ -1,7 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Single-core `TestServer::start`, plus `take_dir` for handing the data -//! directory to a subsequent restart. +//! Single-core `TestServer::start` on a one-node Raft cluster, plus +//! `take_dir` for handing the data directory to a subsequent restart. use std::sync::Arc; use std::time::Duration; @@ -13,14 +13,11 @@ use nodedb::control::state::SharedState; use nodedb::event::{EventPlane, EventPlaneConfig, create_event_bus}; use nodedb::wal::WalManager; -use super::read_gate::install_single_voter_read_gate; -use super::support::{ - bind_http_listener, bind_native_listener, init_test_memory_governor, single_routing_leader, -}; +use super::support::{bind_http_listener, bind_native_listener, init_test_memory_governor}; use super::types::{TestClient, TestDataDir, TestServer}; -/// Knobs for spawning a `TestServer`. `Default` reproduces the historical -/// `TestServer::start` behaviour: trust-mode auth, lockout disabled. +/// Knobs for spawning a `TestServer`. `Default` is the +/// `TestServer::start` config: trust-mode auth, lockout disabled. pub(super) struct StartConfig { /// pgwire authentication mode. pub auth_mode: AuthMode, @@ -36,12 +33,6 @@ pub(super) struct StartConfig { /// system default (65 536 rows). Set to a small value (e.g. `4`) in tests /// that need to observe segment-flush behaviour without inserting 65k rows. pub columnar_flush_threshold: Option, - /// When `Some`, installs a cluster routing table on the node's - /// `SharedState` (`cluster_routing`) before the state is shared. A - /// single-node `TestServer` is normally `cluster_routing == None`; the - /// Raft snapshot builder requires a routing table to resolve a group's - /// vShards, so round-trip tests inject one here. - pub routing: Option, /// Idle session timeout in seconds applied to the node's `SharedState` /// before it is shared. `0` (the default) leaves the idle watchdog /// disabled; a small value (e.g. `1`) lets tests exercise the pgwire @@ -63,6 +54,13 @@ pub(super) struct StartConfig { /// catalog-backed `quota_manager` built by `new_with_credentials` is left /// in place, so quota definitions stay durable. pub metering: Option, + /// When `true`, installs `[backup_storage] local_root` at the data + /// directory's `backups` subdirectory, so `file://` backup URIs resolve + /// there. See [`TestServer::backup_root`]. + pub backup_root: bool, + /// When `Some`, the graph tuning of the core and of `SharedState`. `None` + /// keeps the defaults. + pub graph_tuning: Option, } impl Default for StartConfig { @@ -72,16 +70,16 @@ impl Default for StartConfig { provision_superuser: true, lockout: None, columnar_flush_threshold: None, - routing: None, idle_timeout_secs: 0, session_absolute_timeout_secs: 0, jwks_registry: None, metering: None, + backup_root: false, + graph_tuning: None, } } } -#[allow(dead_code)] impl TestServer { /// Spawn a single-core NodeDB server and connect via pgwire (trust mode). pub async fn start() -> Self { @@ -169,18 +167,14 @@ impl TestServer { .await } - /// Spawn a single-core NodeDB server with a cluster routing table - /// installed on `SharedState::cluster_routing`. - /// - /// Single-node `TestServer`s are normally `cluster_routing == None`, but - /// the production Raft snapshot builder/applier resolve a group's vShards - /// through the routing table. Snapshot round-trip tests inject one with - /// `RoutingTable::uniform(...)` so the builder can filter and the applier - /// can rebind. All other settings stay at their defaults (trust-mode auth, - /// lockout disabled). - pub async fn start_with_routing(routing: nodedb_cluster::RoutingTable) -> Self { + /// Spawn a single-core NodeDB server whose core and Control Plane run with + /// `graph_tuning`. All other settings stay at their defaults (trust-mode + /// auth, lockout disabled). + pub async fn start_with_graph_tuning( + graph_tuning: nodedb_types::config::tuning::GraphTuning, + ) -> Self { Self::start_with_config(StartConfig { - routing: Some(routing), + graph_tuning: Some(graph_tuning), ..Default::default() }) .await @@ -188,6 +182,7 @@ impl TestServer { /// Spawn a single-core NodeDB server and connect via pgwire. pub(super) async fn start_with_config(cfg: StartConfig) -> Self { + let graph_tuning = cfg.graph_tuning.clone().unwrap_or_default(); let dir = tempfile::tempdir().unwrap(); let wal_path = dir.path().join("test.wal"); let wal = Arc::new(WalManager::open_for_testing(&wal_path).unwrap()); @@ -225,22 +220,31 @@ impl TestServer { let mut shared = SharedState::new_with_credentials(dispatcher, Arc::clone(&wal), credentials, false) .expect("build shared state"); - // Inject a fixed test KEK so backup tests produce encrypted envelopes. - // Deterministic 32-byte key — same value every test run. - if let Some(s) = Arc::get_mut(&mut shared) { + // The one-node cluster a server with no `[cluster]` section runs. + let cluster = crate::single_node::init(dir.path()) + .await + .expect("init the one-node cluster"); + { + let s = Arc::get_mut(&mut shared).expect("shared state is not cloned yet"); + crate::single_node::wire(s, &cluster, dir.path()).expect("wire the one-node cluster"); + // Inject a fixed test KEK so backup tests produce encrypted envelopes. + // Deterministic 32-byte key — same value every test run. s.backup_kek = Some(Arc::new([0x42u8; 32])); s.governor = init_test_memory_governor(); - if let Some(routing) = cfg.routing { - // Production takes `node_id` from the cluster handle that owns - // the routing table. A single-node server runs as the one node - // that leads every group in it, so the gateway routes locally. - s.node_id = single_routing_leader(&routing); - s.cluster_routing = Some(std::sync::Arc::new(std::sync::RwLock::new(routing))); - } + s.tuning.graph = graph_tuning.clone(); s.jwks_registry = cfg.jwks_registry; if let Some(metering) = cfg.metering { s.metering_config = metering; } + if cfg.backup_root { + let root = dir.path().join(super::backup::BACKUP_DIR); + std::fs::create_dir_all(&root).expect("create backup root"); + let root = root.canonicalize().unwrap_or(root); + s.backup_storage = Some(Arc::new(nodedb::config::server::BackupStorageSettings { + local_root: Some(root), + ..Default::default() + })); + } s.set_session_timeouts_for_test( cfg.idle_timeout_secs, cfg.session_absolute_timeout_secs, @@ -249,9 +253,7 @@ impl TestServer { let shared = shared; // The same gateway install production boot runs, after every // `Arc::get_mut` above. - nodedb::bootstrap::state_wiring::install_gateway(&shared); - // Production `start_raft` publishes the read gate for a routed node. - install_single_voter_read_gate(&shared); + nodedb::bootstrap::state_wiring::install_gateway(&shared).expect("install gateway"); // Data Plane core. Share the SharedState's array_catalog so DDL // mutations made by the SQL converter are visible to the handler @@ -277,7 +279,7 @@ impl TestServer { core_metrics: shared.system_metrics.clone(), governor: shared.governor.clone(), replay: None, - graph_tuning: nodedb_types::config::tuning::GraphTuning::default(), + graph_tuning: graph_tuning.clone(), query_tuning: { let mut qt = nodedb_types::config::tuning::QueryTuning::default(); if let Some(threshold) = cfg.columnar_flush_threshold { @@ -285,6 +287,7 @@ impl TestServer { } qt }, + timeseries_tuning: nodedb_types::config::tuning::TimeseriesToning::default(), // Seeded from the SAME durable catalog production reads, so a // harness restart reconstructs cores the way a real one does. // An empty catalog yields an empty seed, which is exactly what @@ -292,6 +295,7 @@ impl TestServer { doc_config_seed: nodedb::bootstrap::data_plane::load_doc_config_registry_from( shared.credentials.catalog(), ), + event_interest: crate::core_loop_runner::event_interest_for(&shared), stop_rx: core_stop_rx, }); core_stop_txs.push(core_stop_tx); @@ -336,6 +340,12 @@ impl TestServer { shutdown_bus: shutdown_bus.clone(), }); + // Raft, the lease loop, and the readiness production waits for + // before it opens a listener. + let raft = crate::single_node::start(&cluster, &shared, dir.path()) + .await + .expect("start the one-node cluster"); + // PgWire listener. let pg_listener = PgListener::bind("127.0.0.1:0".parse().unwrap()) .await @@ -411,6 +421,7 @@ impl TestServer { poller_handle: Some(poller_handle), core_handles: Some(core_handles), event_plane: Some(event_plane), + raft: Some(raft), _dir: dir, } } diff --git a/nodedb-test-support/src/pgwire_harness/support.rs b/nodedb-test-support/src/pgwire_harness/support.rs index 13f7de794..f150057f5 100644 --- a/nodedb-test-support/src/pgwire_harness/support.rs +++ b/nodedb-test-support/src/pgwire_harness/support.rs @@ -27,24 +27,6 @@ pub(super) fn init_test_memory_governor() -> Arc { nodedb::memory::init_governor(ceiling, &budgets).expect("harness governor config is valid") } -/// The one node that leads every group in a single-node routing table. -/// -/// Panics when the table names no leader, or more than one: a single-node -/// harness cannot serve a group another node leads. -pub(super) fn single_routing_leader(routing: &nodedb_cluster::RoutingTable) -> u64 { - let mut leaders: Vec = routing - .group_members() - .values() - .map(|group| group.leader) - .collect(); - leaders.sort_unstable(); - leaders.dedup(); - match leaders.as_slice() { - [leader] if *leader != 0 => *leader, - other => panic!("single-node harness routing must name one leader, got {other:?}"), - } -} - /// Bind a native (MessagePack) protocol listener on `127.0.0.1:0` and /// spawn its accept loop. Returns the listener's local port plus the /// handle to await on shutdown. diff --git a/nodedb-test-support/src/pgwire_harness/types.rs b/nodedb-test-support/src/pgwire_harness/types.rs index 42033c0a6..bd70d5ae7 100644 --- a/nodedb-test-support/src/pgwire_harness/types.rs +++ b/nodedb-test-support/src/pgwire_harness/types.rs @@ -52,7 +52,6 @@ pub struct TestServer { /// Underlying shared state — exposed so integration tests can drive /// store-level side effects (e.g. seeding a session handle with a /// specific `ClientFingerprint`) before hitting the wire. - #[allow(dead_code)] pub shared: Arc, pub(super) conn_handle: Option>, // Fields wrapped in Option so that `graceful_shutdown(self)` can `.take()` @@ -67,6 +66,8 @@ pub struct TestServer { pub(super) poller_handle: Option>, pub(super) core_handles: Option>>, pub(super) event_plane: Option, + /// The one-node cluster's Raft side: its lease loop and subsystems. + pub(super) raft: Option, pub(super) _dir: tempfile::TempDir, } @@ -119,5 +120,8 @@ impl Drop for TestServer { h.abort(); } } + if let Some(raft) = self.raft.take() { + raft.abort(); + } } } diff --git a/nodedb-test-support/src/single_node.rs b/nodedb-test-support/src/single_node.rs new file mode 100644 index 000000000..0934489a7 --- /dev/null +++ b/nodedb-test-support/src/single_node.rs @@ -0,0 +1,156 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The one-node cluster boot every in-process harness runs. +//! +//! A server with no `[cluster]` section boots a one-node Calvin cluster: +//! `init_single_node_calvin`, `wire_cluster_handle`, `start_raft`, the +//! descriptor lease loop, then `await_cluster_ready`. A harness runs the same +//! production functions in the same order: +//! +//! 1. [`init`] synthesizes the cluster before `SharedState` is shared. +//! 2. [`wire`] installs its handles on the state before anything clones it. +//! 3. The harness starts its cores, response poller and Event Plane. +//! 4. [`start`] starts Raft and the lease loop, then waits for the boot +//! readiness production waits for before it opens a listener. +//! 5. The harness binds its listeners. +//! +//! [`OneNodeRaft::shutdown`] stops what [`start`] started. + +use std::path::Path; +use std::sync::Arc; +use std::time::Duration; + +use nodedb::control::cluster::ClusterHandle; +use nodedb::control::startup::{StartupPhase, StartupSequencer}; +use nodedb::control::state::SharedState; +use nodedb_types::config::tuning::ClusterTransportTuning; + +/// Error type of every boot step. +pub type BootError = Box; + +/// How long each shutdown step can take. +const SHUTDOWN_STEP: Duration = Duration::from_secs(5); + +/// Transport tuning for a cluster whose only voter is this node. +/// +/// A sole voter never waits on a peer's vote, so a short election timeout +/// elects it at once. The boot vote fence refuses votes to other nodes only, +/// so it never delays a sole voter. +pub fn tuning() -> ClusterTransportTuning { + ClusterTransportTuning { + election_timeout_min_ms: 150, + election_timeout_max_ms: 300, + ..ClusterTransportTuning::default() + } +} + +/// Synthesize the one-node cluster rooted at `data_dir`, as boot does for a +/// server with no `[cluster]` section. A directory that already holds a +/// cluster catalog restarts that cluster. +pub async fn init(data_dir: &Path) -> Result { + Ok(nodedb::control::cluster::init_single_node_calvin(data_dir, &tuning()).await?) +} + +/// Install `handle` on `state` and load the host tables from the catalog. +/// +/// Runs before `state` is shared and before the Event Plane spawns: the +/// Event Plane starts the cross-shard drain only when its sender is wired. +pub fn wire( + state: &mut SharedState, + handle: &ClusterHandle, + data_dir: &Path, +) -> Result<(), BootError> { + nodedb::bootstrap::state_wiring::wire_cluster_handle(state, handle, data_dir)?; + nodedb::control::cluster::metadata_applier::seed_host_tables(state)?; + Ok(()) +} + +/// The Raft side of a running one-node cluster. +pub struct OneNodeRaft { + running: Option, + lease_renewal: Option>, + lease_shutdown_tx: tokio::sync::watch::Sender, +} + +/// Start Raft and the descriptor lease loop on `shared`, then wait for the +/// readiness production boot waits for before it opens a listener. +/// +/// The cores and the response poller run already: the readiness steps +/// dispatch to them. The wait covers the metadata group's first apply, the +/// data groups' replay, the schema rehydration and the first authorization +/// lease. +pub async fn start( + handle: &ClusterHandle, + shared: &Arc, + data_dir: &Path, +) -> Result { + let tuning = tuning(); + let ready_rx = + nodedb::control::cluster::start_raft(handle, Arc::clone(shared), data_dir, &tuning).await?; + let running = handle + .running_cluster + .lock() + .unwrap_or_else(|p| p.into_inner()) + .take(); + let (lease_shutdown_tx, lease_shutdown_rx) = tokio::sync::watch::channel(false); + let (lease_renewal, lease_metrics) = nodedb::control::lease::LeaseRenewalLoop::spawn( + Arc::clone(shared), + &tuning, + lease_shutdown_rx, + )?; + shared.loop_metrics_registry.register(lease_metrics); + let raft = OneNodeRaft { + running, + lease_renewal: Some(lease_renewal), + lease_shutdown_tx, + }; + + let (sequencer, _gate) = StartupSequencer::new(); + let gates = nodedb::bootstrap::cluster_ready::ClusterReadyGates { + raft_gate: sequencer.register_gate(StartupPhase::RaftMetadataReplay, "raft"), + schema_gate: sequencer.register_gate(StartupPhase::SchemaCacheWarmup, "schema"), + sanity_gate: sequencer.register_gate(StartupPhase::CatalogSanityCheck, "sanity"), + data_groups_gate: sequencer.register_gate(StartupPhase::DataGroupsReplay, "data-groups"), + transport_gate: sequencer.register_gate(StartupPhase::TransportBind, "transport"), + warm_peers_gate: sequencer.register_gate(StartupPhase::WarmPeers, "warm-peers"), + health_loop_gate: sequencer.register_gate(StartupPhase::HealthLoopStart, "health-loop"), + gateway_enable_gate: sequencer.register_gate(StartupPhase::GatewayEnable, "gateway"), + }; + // The harness cores replay their WAL before they report ready, so no + // replay receiver is owed here. + nodedb::bootstrap::cluster_ready::await_cluster_ready(shared, ready_rx, Vec::new(), gates) + .await?; + Ok(raft) +} + +impl OneNodeRaft { + /// Stop the lease loop and the cluster subsystems, then close the + /// transport. `shared.shutdown` stops the Raft loops: the caller signals + /// it first. + pub async fn shutdown(&mut self, shared: &SharedState) { + let _ = self.lease_shutdown_tx.send(true); + if let Some(handle) = self.lease_renewal.take() { + handle.abort(); + let _ = handle.await; + } + if let Some(running) = self.running.take() { + let errors = running.shutdown_all(SHUTDOWN_STEP).await; + if !errors.is_empty() { + eprintln!("one-node cluster: subsystem shutdown errors: {errors:?}"); + } + } + if let Some(transport) = shared.cluster_transport.clone() { + // A one-node cluster has no peer to acknowledge the close. + let _ = transport.close(SHUTDOWN_STEP).await; + } + } + + /// Stop the lease loop without waiting, for a harness dropped without + /// its async shutdown. + pub fn abort(&self) { + let _ = self.lease_shutdown_tx.send(true); + if let Some(handle) = self.lease_renewal.as_ref() { + handle.abort(); + } + } +} diff --git a/nodedb-test-support/src/tx_batch_helpers.rs b/nodedb-test-support/src/tx_batch_helpers.rs index 4ab576dca..510aa1304 100644 --- a/nodedb-test-support/src/tx_batch_helpers.rs +++ b/nodedb-test-support/src/tx_batch_helpers.rs @@ -2,7 +2,6 @@ //! Plan builders and a single-core commit driver shared by the transaction //! cross-engine tests. -#![allow(dead_code)] use std::sync::Arc; use std::time::{Duration, Instant}; @@ -62,6 +61,7 @@ pub fn make_request(plan: PhysicalPlan) -> Request { txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: nodedb::bridge::envelope::Admission::Admitted, } } @@ -114,7 +114,7 @@ pub fn doc_put(collection: &str, doc_id: &str, val: &[u8]) -> PhysicalPlan { collection: qualify(collection), document_id: doc_id.into(), value: val.to_vec(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: crate::kv_rows::kv_row_surrogate(doc_id.as_bytes()), pk_bytes: Vec::new(), returning: None, rls_filters: Vec::new(), @@ -129,7 +129,7 @@ pub fn doc_get(collection: &str, doc_id: &str) -> PhysicalPlan { rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: Some(crate::kv_rows::kv_row_surrogate(doc_id.as_bytes())), pk_bytes: Vec::new(), }) } @@ -140,7 +140,7 @@ pub fn doc_conflict(collection: &str, doc_id: &str) -> PhysicalPlan { collection: qualify(collection), document_id: doc_id.into(), value: b"conflict".to_vec(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: crate::kv_rows::kv_row_surrogate(doc_id.as_bytes()), if_absent: false, returning: None, rls_filters: Vec::new(), @@ -156,8 +156,8 @@ pub fn edge_put(collection: &str, src: &str, dst: &str) -> PhysicalPlan { label: "REL".into(), dst_id: dst.into(), properties: Vec::new(), - src_surrogate: nodedb_types::Surrogate::ZERO, - dst_surrogate: nodedb_types::Surrogate::ZERO, + src_surrogate: crate::kv_rows::kv_row_surrogate(src.as_bytes()), + dst_surrogate: crate::kv_rows::kv_row_surrogate(dst.as_bytes()), }) } @@ -177,7 +177,7 @@ pub fn kv_put(key: &[u8], value: &[u8]) -> PhysicalPlan { key: key.to_vec(), value: value.to_vec(), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: crate::kv_rows::kv_row_surrogate(key), returning: None, rls_filters: Vec::new(), provenance: None, @@ -281,20 +281,6 @@ pub fn timeseries_scan(collection: &str) -> PhysicalPlan { }) } -pub fn crdt_apply(collection: &str, doc_id: &str) -> PhysicalPlan { - PhysicalPlan::Crdt(CrdtOp::Apply { - collection: qualify(collection), - document_id: doc_id.into(), - delta: vec![0u8; 8], - peer_id: 1, - mutation_id: 42, - surrogate: nodedb_types::Surrogate::ZERO, - provenance: None, - constraint_version_required: 0, - expected_frontier_digest: None, - }) -} - /// A vector-primary direct insert of a 3-dimensional vector. pub fn vector_direct_insert(collection: &str, surrogate: u32) -> PhysicalPlan { let mut payload = std::collections::HashMap::new(); @@ -370,24 +356,6 @@ pub fn assert_kv_absent( ); } -/// Assert that a document is absent (NotFound or empty payload). -pub fn assert_doc_absent( - core: &mut CoreLoop, - tx: &mut Producer, - rx: &mut Consumer, - collection: &str, - doc_id: &str, -) { - let r = send_raw(core, tx, rx, doc_get(collection, doc_id)); - let is_absent = r.status == Status::Error || r.payload.is_empty() || r.payload.len() <= 3; - assert!( - is_absent, - "doc {collection}/{doc_id} must be absent after rollback; status={:?} payload_len={}", - r.status, - r.payload.len() - ); -} - /// Assert that graph node `src` has no REL neighbors. pub fn assert_edge_absent( core: &mut CoreLoop, diff --git a/nodedb-types/Cargo.toml b/nodedb-types/Cargo.toml index 575d693db..646fcb604 100644 --- a/nodedb-types/Cargo.toml +++ b/nodedb-types/Cargo.toml @@ -38,6 +38,7 @@ bytemuck = { workspace = true } aes-gcm = { workspace = true } getrandom = { workspace = true } sha2 = { workspace = true } +hex = { workspace = true } [dev-dependencies] toml = { workspace = true } diff --git a/nodedb-types/build.rs b/nodedb-types/build.rs new file mode 100644 index 000000000..499189687 --- /dev/null +++ b/nodedb-types/build.rs @@ -0,0 +1,76 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Emits `NODEDB_WIRE_BUILD_ID`: the short git commit hash, or +//! `CARGO_PKG_VERSION` when git is unavailable (a crates.io build). +//! +//! Commit only — never the dirty-tree state, which would force a rebuild of +//! every dependent crate on every uncommitted edit. + +use std::process::Command; + +fn main() { + let build_id = + git_commit().unwrap_or_else(|| std::env::var("CARGO_PKG_VERSION").unwrap_or_default()); + println!("cargo:rustc-env=NODEDB_WIRE_BUILD_ID={build_id}"); + + for path in git_head_ref_paths() { + println!("cargo:rerun-if-changed={path}"); + } +} + +/// Short hash of the current commit, or `None` when git is unavailable. +fn git_commit() -> Option { + let output = Command::new("git") + .args(["rev-parse", "--short", "HEAD"]) + .output() + .ok()?; + if !output.status.success() { + return None; + } + let commit = String::from_utf8(output.stdout).ok()?.trim().to_owned(); + if commit.is_empty() { + None + } else { + Some(commit) + } +} + +/// Paths, resolved via `git`, whose mtime tracks the current commit: the +/// git-dir `HEAD` file, and — when `HEAD` is a symbolic ref — the ref file +/// it points at. Empty when git is unavailable. +fn git_head_ref_paths() -> Vec { + let Some(git_dir) = git_dir() else { + return Vec::new(); + }; + let mut paths = vec![format!("{git_dir}/HEAD")]; + if let Some(head_ref) = symbolic_ref() { + paths.push(format!("{git_dir}/{head_ref}")); + } + paths +} + +fn git_dir() -> Option { + let output = Command::new("git") + .args(["rev-parse", "--git-dir"]) + .output() + .ok()?; + if !output.status.success() { + return None; + } + let dir = String::from_utf8(output.stdout).ok()?.trim().to_owned(); + if dir.is_empty() { None } else { Some(dir) } +} + +/// The ref `HEAD` points at (e.g. `refs/heads/main`), or `None` on a +/// detached `HEAD`. +fn symbolic_ref() -> Option { + let output = Command::new("git") + .args(["symbolic-ref", "-q", "HEAD"]) + .output() + .ok()?; + if !output.status.success() { + return None; + } + let r = String::from_utf8(output.stdout).ok()?.trim().to_owned(); + if r.is_empty() { None } else { Some(r) } +} diff --git a/nodedb-types/src/backup_envelope/database.rs b/nodedb-types/src/backup_envelope/database.rs new file mode 100644 index 000000000..6d74855c5 --- /dev/null +++ b/nodedb-types/src/backup_envelope/database.rs @@ -0,0 +1,29 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! A database backup: one encrypted envelope whose sections are whole tenant +//! envelopes. +//! +//! The outer envelope's `meta.tenant_id` is [`DATABASE_BACKUP_TENANT`]. Its +//! first section, [`SECTION_ORIGIN_DATABASE_MANIFEST`], names the database and +//! lists its tenants. Every other section's `origin_node_id` is a tenant id, +//! and its body is that tenant's envelope, covering this database only. The +//! outer authentication tag covers the whole set, so a dropped tenant fails +//! the parse. + +/// `meta.tenant_id` of a database backup's outer envelope. +pub const DATABASE_BACKUP_TENANT: u64 = u64::MAX; + +/// Section carrying a database backup's [`DatabaseBackupManifest`]. +pub const SECTION_ORIGIN_DATABASE_MANIFEST: u64 = 0xFFFF_FFFF_FFFF_FFE0; + +/// What a database backup holds. +#[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct DatabaseBackupManifest { + /// Name of the backed-up database. A restore targets the same name. + pub database: String, + /// HLC wall time, in nanoseconds, of the one consistent cut every tenant + /// was captured at. Every row in the backup committed at or below it. + pub cut_hlc: u64, + /// Every tenant with a section, in section order. + pub tenants: Vec, +} diff --git a/nodedb-types/src/backup_envelope/mod.rs b/nodedb-types/src/backup_envelope/mod.rs index 0f2d8816e..3ef8cf4fa 100644 --- a/nodedb-types/src/backup_envelope/mod.rs +++ b/nodedb-types/src/backup_envelope/mod.rs @@ -1,19 +1,29 @@ // SPDX-License-Identifier: Apache-2.0 pub mod crypto; +pub mod database; pub mod read; pub mod types; +pub mod verification; pub mod write; pub use crypto::parse_encrypted; +pub use database::{ + DATABASE_BACKUP_TENANT, DatabaseBackupManifest, SECTION_ORIGIN_DATABASE_MANIFEST, +}; pub use read::parse; pub use types::{ - DEFAULT_MAX_SECTION_BYTES, DEFAULT_MAX_TOTAL_BYTES, HEADER_LEN, MAGIC, - SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_DATABASES, SECTION_ORIGIN_SOURCE_TOMBSTONES, - SECTION_ORIGIN_SURROGATE_PK, SECTION_OVERHEAD, TRAILER_LEN, VERSION, + ArrayCatalogBlob, DatabaseBlob, DatabaseDataSection, Envelope, EnvelopeError, EnvelopeMeta, + Section, SourceTombstoneEntry, StoredCollectionBlob, SurrogateBindBlob, }; pub use types::{ - DatabaseBlob, DatabaseDataSection, Envelope, EnvelopeError, EnvelopeMeta, Section, - SourceTombstoneEntry, StoredCollectionBlob, SurrogateBindBlob, + DEFAULT_MAX_SECTION_BYTES, DEFAULT_MAX_TOTAL_BYTES, HEADER_LEN, MAGIC, + SECTION_ORIGIN_ARRAY_CATALOG, SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_DATABASES, + SECTION_ORIGIN_SOURCE_TOMBSTONES, SECTION_ORIGIN_SURROGATE_PK, SECTION_ORIGIN_VERIFICATION, + SECTION_OVERHEAD, TRAILER_LEN, VERSION, +}; +pub use verification::{ + CollectionVerification, VerificationMismatch, VerificationPhase, VerifiedPart, VerifiedTally, + verification_failure_message, }; pub use write::EnvelopeWriter; diff --git a/nodedb-types/src/backup_envelope/read.rs b/nodedb-types/src/backup_envelope/read.rs index fd038a4e3..357105554 100644 --- a/nodedb-types/src/backup_envelope/read.rs +++ b/nodedb-types/src/backup_envelope/read.rs @@ -308,7 +308,7 @@ mod tests { assert!(parse(truncated, DEFAULT_MAX_TOTAL_BYTES).is_err()); } - /// Asserts `NDBB` magic at [0..4], VERSION == 2 at [4], and that the + /// Asserts `NDBB` magic at [0..4], VERSION == 3 at [4], and that the /// header CRC at [48..52] covers header bytes [0..48]. #[test] fn golden_backup_envelope_format() { @@ -319,9 +319,9 @@ mod tests { // Magic at [0..4]. assert_eq!(&bytes[0..4], MAGIC.as_slice(), "magic mismatch"); - // VERSION == 2 at [4]. + // VERSION == 3 at [4]. assert_eq!(bytes[4], VERSION, "version mismatch"); - assert_eq!(bytes[4], 2u8, "expected VERSION == 2"); + assert_eq!(bytes[4], 3u8, "expected VERSION == 3"); // Header CRC at [48..52] covers [0..48]. assert!(bytes.len() >= HEADER_LEN, "envelope too short for header"); diff --git a/nodedb-types/src/backup_envelope/types.rs b/nodedb-types/src/backup_envelope/types.rs index 1f514a943..5476ec297 100644 --- a/nodedb-types/src/backup_envelope/types.rs +++ b/nodedb-types/src/backup_envelope/types.rs @@ -10,10 +10,11 @@ pub const MAGIC: &[u8; 4] = b"NDBB"; /// Plaintext and encrypted envelopes carry the same version. The crypto /// block (68 bytes after the header) distinguishes them. /// -/// Version 2 scopes every section body to a database: data sections carry a +/// Every section body is scoped to a database: data sections carry a /// [`DatabaseDataSection`], and the metadata sections name the database of -/// each entry. A version-1 envelope is refused. -pub const VERSION: u8 = 2; +/// each entry. Every envelope carries one `SECTION_ORIGIN_VERIFICATION` +/// section. An envelope of any other version is refused. +pub const VERSION: u8 = 3; /// Header is fixed-size — 52 bytes (48 framed + 4 crc). /// @@ -48,6 +49,14 @@ pub const SECTION_ORIGIN_SURROGATE_PK: u64 = 0xFFFF_FFFF_FFFF_FFF2; /// is a msgpack-encoded `Vec`. Restore reads it first: every /// other section names its database by the id recorded here. pub const SECTION_ORIGIN_DATABASES: u64 = 0xFFFF_FFFF_FFFF_FFF3; +/// Section carrying the per-collection row counts and digests restore checks. +/// The body is a msgpack-encoded `Vec`. Every +/// envelope carries exactly one. +pub const SECTION_ORIGIN_VERIFICATION: u64 = 0xFFFF_FFFF_FFFF_FFF4; +/// Section carrying the catalog row of each of the tenant's arrays. The body +/// is a msgpack-encoded `Vec`. The cells travel in the data +/// sections. +pub const SECTION_ORIGIN_ARRAY_CATALOG: u64 = 0xFFFF_FFFF_FFFF_FFF5; /// One database of the backed-up tenant, carried in a /// `SECTION_ORIGIN_DATABASES` section. @@ -92,6 +101,16 @@ pub struct StoredCollectionBlob { pub bytes: Vec, } +/// One array's catalog row in a `SECTION_ORIGIN_ARRAY_CATALOG` section. +#[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct ArrayCatalogBlob { + /// Source id of the database the array lives in. + pub database_id: u64, + pub name: String, + /// zerompk-encoded `ArrayCatalogEntry` from the `nodedb` crate. + pub bytes: Vec, +} + /// Single source-side tombstone entry. `purge_lsn` is the Origin WAL /// LSN at which the hard-delete committed — restore uses it as a /// per-collection replay barrier so rows older than the purge don't diff --git a/nodedb-types/src/backup_envelope/verification.rs b/nodedb-types/src/backup_envelope/verification.rs new file mode 100644 index 000000000..fb332967b --- /dev/null +++ b/nodedb-types/src/backup_envelope/verification.rs @@ -0,0 +1,289 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Per-collection verification records of a backup envelope. +//! +//! A backup records, per collection and engine part, the number of canonical +//! rows and an order-independent digest of them. A restore recomputes both +//! from the envelope before it writes, and from the destination after it +//! writes. +//! +//! The digest is the sum, modulo 2^256, of the SHA-256 of every row: +//! +//! * A sum is order-independent, so rows add in any order and the partial +//! digests of several nodes merge by addition. +//! * A duplicated row changes the sum. Under XOR two copies cancel. +//! * A row's contribution subtracts out, so a restore can drop a TTL row that +//! expired after the backup. A sorted Merkle hash cannot, and it needs every +//! row hash in memory to sort. + +use std::fmt; + +/// The engine part of a collection a verification record covers. A collection +/// with graph edges has an `Edges` record next to its `Documents` record. +#[derive( + Debug, + Clone, + Copy, + PartialEq, + Eq, + PartialOrd, + Ord, + Hash, + zerompk::ToMessagePack, + zerompk::FromMessagePack, +)] +pub enum VerifiedPart { + Documents, + Edges, + KeyValue, + Vectors, + Timeseries, + Columnar, + Crdt, + /// Array cell versions. + Array, +} + +impl VerifiedPart { + /// The byte that separates the parts' row hashes. + pub fn tag(self) -> u8 { + match self { + Self::Documents => 1, + Self::Edges => 2, + Self::KeyValue => 3, + Self::Vectors => 4, + Self::Timeseries => 5, + Self::Columnar => 6, + Self::Crdt => 7, + Self::Array => 8, + } + } + + pub fn as_str(self) -> &'static str { + match self { + Self::Documents => "documents", + Self::Edges => "edges", + Self::KeyValue => "kv", + Self::Vectors => "vectors", + Self::Timeseries => "timeseries", + Self::Columnar => "columnar", + Self::Crdt => "crdt", + Self::Array => "array", + } + } +} + +impl fmt::Display for VerifiedPart { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(self.as_str()) + } +} + +/// A row count and the sum of the rows' SHA-256, a little-endian 256-bit +/// integer. +#[derive( + Debug, Clone, Copy, PartialEq, Eq, Default, zerompk::ToMessagePack, zerompk::FromMessagePack, +)] +pub struct VerifiedTally { + pub count: u64, + pub digest: [u8; 32], +} + +impl VerifiedTally { + /// Add one row by its SHA-256. + pub fn add(&mut self, row: &[u8; 32]) { + self.count = self.count.wrapping_add(1); + let mut carry = 0u16; + for (sum, byte) in self.digest.iter_mut().zip(row) { + let total = u16::from(*sum) + u16::from(*byte) + carry; + *sum = total as u8; + carry = total >> 8; + } + } + + /// Remove one row [`Self::add`] added. + pub fn remove(&mut self, row: &[u8; 32]) { + self.count = self.count.wrapping_sub(1); + let mut borrow = 0i16; + for (sum, byte) in self.digest.iter_mut().zip(row) { + let total = i16::from(*sum) - i16::from(*byte) - borrow; + borrow = i16::from(total < 0); + *sum = total.rem_euclid(256) as u8; + } + } +} + +impl fmt::Display for VerifiedTally { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + let tail: Vec = self.digest.iter().rev().take(8).copied().collect(); + write!(f, "{} rows, digest {}", self.count, hex::encode(tail)) + } +} + +/// The body of a `SECTION_ORIGIN_VERIFICATION` section is a +/// msgpack-encoded `Vec`: one record per collection +/// part that holds at least one row. +#[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct CollectionVerification { + /// Source id of the database the collection lives in. + pub database_id: u64, + /// Bare catalog name of the collection. + pub collection: String, + pub part: VerifiedPart, + pub tally: VerifiedTally, +} + +/// Which check a verification ran as. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum VerificationPhase { + /// The envelope's rows against the tallies it records. Runs before the + /// first write. + Envelope, + /// The destination's rows against the envelope's. Runs after the last + /// re-issue. + Destination, + /// A MOVE TENANT target's rows against the source capture. Runs after the + /// re-issue and before the catalog moves. + Move, +} + +/// One collection part whose recomputed tally differs from the expected one. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct VerificationMismatch { + /// The database the collection lives in: the source id in the envelope + /// phase, the destination id in the destination phase. + pub database_id: u64, + pub collection: String, + pub part: VerifiedPart, + pub expected: VerifiedTally, + pub found: VerifiedTally, +} + +impl fmt::Display for VerificationMismatch { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!( + f, + "collection '{}' in database {} ({}): expected {}, found {}", + self.collection, self.database_id, self.part, self.expected, self.found + ) + } +} + +/// The message of a failed verification. It names every mismatched +/// collection and states what the restore or move left behind. +pub fn verification_failure_message( + phase: &VerificationPhase, + mismatches: &[VerificationMismatch], +) -> String { + let list = mismatches + .iter() + .map(ToString::to_string) + .collect::>() + .join("; "); + match phase { + VerificationPhase::Envelope => format!( + "restore verification failed: the backup's rows do not match the counts and \ + digests it records for {list}; nothing was restored" + ), + VerificationPhase::Destination => format!( + "restore verification failed: the destination does not hold the backed-up rows \ + of {list}; the restore is failed and not rolled back, and the restored data \ + stays in place for inspection" + ), + VerificationPhase::Move => format!( + "move verification failed: the target does not hold the moved rows of {list}; \ + the move was not applied and the source is untouched" + ), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn row(seed: u8) -> [u8; 32] { + let mut bytes = [0u8; 32]; + for (i, b) in bytes.iter_mut().enumerate() { + *b = seed + .wrapping_mul(31) + .wrapping_add(i as u8) + .wrapping_mul(0x9D); + } + bytes + } + + #[test] + fn the_sum_ignores_row_order() { + let mut forward = VerifiedTally::default(); + let mut backward = VerifiedTally::default(); + for seed in 0..50 { + forward.add(&row(seed)); + } + for seed in (0..50).rev() { + backward.add(&row(seed)); + } + assert_eq!(forward, backward); + assert_eq!(forward.count, 50); + } + + #[test] + fn a_duplicated_row_changes_the_sum() { + let mut once = VerifiedTally::default(); + once.add(&row(1)); + once.add(&row(2)); + let mut twice = once; + twice.add(&row(2)); + twice.remove(&row(1)); + assert_eq!(once.count, twice.count); + assert_ne!(once.digest, twice.digest); + } + + #[test] + fn remove_undoes_add_across_carries() { + let full = [0xFFu8; 32]; + let mut tally = VerifiedTally::default(); + tally.add(&row(7)); + let before = tally; + tally.add(&full); + tally.add(&full); + tally.remove(&full); + tally.remove(&full); + assert_eq!(tally, before); + } + + #[test] + fn a_record_round_trips() { + let mut tally = VerifiedTally::default(); + tally.add(&row(3)); + let records = vec![CollectionVerification { + database_id: 1025, + collection: "orders".into(), + part: VerifiedPart::Edges, + tally, + }]; + let bytes = zerompk::to_msgpack_vec(&records).expect("encode"); + let decoded: Vec = zerompk::from_msgpack(&bytes).expect("decode"); + assert_eq!(decoded, records); + } + + #[test] + fn the_message_names_every_collection_and_the_outcome() { + let mismatch = |collection: &str| VerificationMismatch { + database_id: 0, + collection: collection.into(), + part: VerifiedPart::Documents, + expected: VerifiedTally::default(), + found: VerifiedTally::default(), + }; + let both = [mismatch("orders"), mismatch("users")]; + let message = verification_failure_message(&VerificationPhase::Destination, &both); + assert!(message.contains("'orders'") && message.contains("'users'")); + assert!(message.contains("not rolled back")); + let message = verification_failure_message(&VerificationPhase::Envelope, &both); + assert!(message.contains("nothing was restored")); + let message = verification_failure_message(&VerificationPhase::Move, &both); + assert!(message.starts_with("move verification failed")); + assert!(message.contains("'orders'") && message.contains("'users'")); + assert!(message.contains("the move was not applied and the source is untouched")); + } +} diff --git a/nodedb-types/src/calvin.rs b/nodedb-types/src/calvin.rs index 7569edb5a..003662c56 100644 --- a/nodedb-types/src/calvin.rs +++ b/nodedb-types/src/calvin.rs @@ -111,6 +111,16 @@ pub enum EngineKeySet { edges: SortedVec<(u32, u32)>, home_vshards: SortedVec, }, + /// Array engine: the array's cells on each vShard in `vshards`. + /// + /// An array is tile-partitioned: its cells live on the vShards their + /// tiles hash to, not on the array's collection home. Each vShard in + /// `vshards` participates, and the transaction locks the whole array on + /// it. Appended last: the encoding is positional. + Array { + collection: String, + vshards: SortedVec, + }, } impl EngineKeySet { @@ -130,6 +140,8 @@ impl EngineKeySet { Self::Kv { keys, .. } => keys.iter().map(|k| k.len()).sum(), // Edge: two u32 per edge = 8 bytes each. Self::Edge { edges, .. } => edges.len() * 8, + // Array: one u32 vShard each. + Self::Array { vshards, .. } => vshards.len() * 4, } } @@ -139,7 +151,8 @@ impl EngineKeySet { Self::Document { collection, .. } | Self::Vector { collection, .. } | Self::Kv { collection, .. } - | Self::Edge { collection, .. } => collection, + | Self::Edge { collection, .. } + | Self::Array { collection, .. } => collection, } } @@ -150,6 +163,7 @@ impl EngineKeySet { Self::Vector { surrogates, .. } => surrogates.is_empty(), Self::Kv { keys, .. } => keys.is_empty(), Self::Edge { edges, .. } => edges.is_empty(), + Self::Array { vshards, .. } => vshards.is_empty(), } } } @@ -277,6 +291,16 @@ pub struct VersionedReadEntry { pub key: ReadKeyIdent, /// The responding shard's write-LSN watermark at read time. pub read_lsn: Lsn, + /// The vShard whose write versions validate this read. `None` homes the + /// read to its collection's vShard. A graph read names the key vShard it + /// read edges on, because edges live on their endpoints' vShards. A homed + /// read with an empty `collection` observed every collection there, so it + /// validates against the shard's core watermark. + pub home_vshard: Option, + /// The node that served the read. `read_lsn` is a position in that node's + /// WAL, so a participant on any other node treats the read as changed. + /// `0` when no one node is known to have served it. + pub served_by: u64, } /// The LSN-versioned read-set of a Calvin transaction. diff --git a/nodedb-types/src/config/tuning/config.rs b/nodedb-types/src/config/tuning/config.rs index 65bc5795f..aac570e7f 100644 --- a/nodedb-types/src/config/tuning/config.rs +++ b/nodedb-types/src/config/tuning/config.rs @@ -100,6 +100,7 @@ mod tests { assert_eq!(parsed.cluster_transport.ghost_sweep_interval_secs, 1800); assert_eq!(parsed.cluster_transport.health_ping_interval_secs, 5); assert_eq!(parsed.cluster_transport.health_failure_threshold, 3); + assert_eq!(parsed.cluster_transport.change_feed_heartbeat_ms, 1_000); // New QueryTuning fields. assert_eq!(parsed.query.doc_cache_entries, 4096); assert_eq!(parsed.query.columnar_flush_threshold, 65_536); diff --git a/nodedb-types/src/config/tuning/engines.rs b/nodedb-types/src/config/tuning/engines.rs index 3164b411f..c895a2087 100644 --- a/nodedb-types/src/config/tuning/engines.rs +++ b/nodedb-types/src/config/tuning/engines.rs @@ -105,6 +105,11 @@ pub const DEFAULT_VARLEN_MAX_RESULTS: usize = 100_000; /// expansion before it must page via cross-shard resume. pub const DEFAULT_VARLEN_MAX_FRONTIER: usize = 100_000; +/// Default cap on the edges one gathered graph algorithm run collects. Each +/// gathered edge holds its three names and a weight, around 100 bytes with +/// typical keys, so the default bounds one run near 1 GB before its CSR. +pub const DEFAULT_MAX_GATHERED_ALGO_EDGES: usize = 10_000_000; + /// Graph engine tuning (traversal limits, LCC algorithm). #[derive(Debug, Clone, Serialize, Deserialize)] pub struct GraphTuning { @@ -129,6 +134,12 @@ pub struct GraphTuning { /// dense / bidirectional traversals. #[serde(default = "default_varlen_max_frontier")] pub varlen_max_frontier: usize, + /// Hard cap on the distinct edges a gathered graph algorithm collects from + /// every owner before it runs on one core. The algorithm needs every edge + /// in one CSR, so a run over more edges is refused with the edge count + /// rather than answered from part of the graph. + #[serde(default = "default_max_gathered_algo_edges")] + pub max_gathered_algo_edges: usize, } impl Default for GraphTuning { @@ -140,6 +151,7 @@ impl Default for GraphTuning { lcc_sample_pairs: default_lcc_sample_pairs(), varlen_max_results: default_varlen_max_results(), varlen_max_frontier: default_varlen_max_frontier(), + max_gathered_algo_edges: default_max_gathered_algo_edges(), } } } @@ -162,6 +174,9 @@ fn default_varlen_max_results() -> usize { fn default_varlen_max_frontier() -> usize { DEFAULT_VARLEN_MAX_FRONTIER } +fn default_max_gathered_algo_edges() -> usize { + DEFAULT_MAX_GATHERED_ALGO_EDGES +} /// Timeseries engine tuning (memtable budgets, block sizes). #[derive(Debug, Clone, Serialize, Deserialize)] diff --git a/nodedb-types/src/config/tuning/network.rs b/nodedb-types/src/config/tuning/network.rs index 06b0d9cd7..6f2ef37c6 100644 --- a/nodedb-types/src/config/tuning/network.rs +++ b/nodedb-types/src/config/tuning/network.rs @@ -222,6 +222,12 @@ pub struct ClusterTransportTuning { /// earlier in the lease lifecycle. #[serde(default = "default_descriptor_lease_renewal_threshold_pct")] pub descriptor_lease_renewal_threshold_pct: u8, + /// Longest interval, in milliseconds, a data-group leader lets a node + /// that does not replicate the group go without the group's change-feed + /// position. A settled run with events carries the position at once; an + /// idle feed carries it on a heartbeat at this interval. + #[serde(default = "default_change_feed_heartbeat_ms")] + pub change_feed_heartbeat_ms: u64, } impl ClusterTransportTuning { @@ -272,10 +278,15 @@ impl Default for ClusterTransportTuning { default_descriptor_lease_renewal_check_interval_secs(), descriptor_lease_renewal_threshold_pct: default_descriptor_lease_renewal_threshold_pct( ), + change_feed_heartbeat_ms: default_change_feed_heartbeat_ms(), } } } +fn default_change_feed_heartbeat_ms() -> u64 { + 1_000 +} + fn default_descriptor_lease_duration_secs() -> u64 { 300 } diff --git a/nodedb-types/src/error/code.rs b/nodedb-types/src/error/code.rs index 0d0a3343b..b78b74ea7 100644 --- a/nodedb-types/src/error/code.rs +++ b/nodedb-types/src/error/code.rs @@ -79,7 +79,6 @@ error_codes! { // Query (1200–1299) PLAN_ERROR = 1200; - FAN_OUT_EXCEEDED = 1201; SQL_NOT_ENABLED = 1202; /// A function call names no registered scalar/aggregate/window function. UNDEFINED_FUNCTION = 1203; diff --git a/nodedb-types/src/error/code_table.rs b/nodedb-types/src/error/code_table.rs index 7bd5cfe6c..bd78afa23 100644 --- a/nodedb-types/src/error/code_table.rs +++ b/nodedb-types/src/error/code_table.rs @@ -92,7 +92,6 @@ error_code_table! { // Query. PLAN_ERROR => PlanError { phase: "remote".into(), detail: message.to_owned() }, - FAN_OUT_EXCEEDED => FanOutExceeded { shards_touched: 0, limit: 0 }, SQL_NOT_ENABLED => SqlNotEnabled, UNDEFINED_FUNCTION => UndefinedFunction { name: String::new() }, UNDEFINED_COLUMN => UndefinedColumn { column: String::new() }, diff --git a/nodedb-types/src/error/ctors/read_query_auth.rs b/nodedb-types/src/error/ctors/read_query_auth.rs index 19c769a07..332b3f499 100644 --- a/nodedb-types/src/error/ctors/read_query_auth.rs +++ b/nodedb-types/src/error/ctors/read_query_auth.rs @@ -99,18 +99,6 @@ impl NodeDbError { } } - pub fn fan_out_exceeded(shards_touched: u16, limit: u16) -> Self { - Self { - code: ErrorCode::FAN_OUT_EXCEEDED, - message: format!("query fan-out exceeded: {shards_touched} shards > limit {limit}"), - details: ErrorDetails::FanOutExceeded { - shards_touched, - limit, - }, - cause: None, - } - } - pub fn sql_not_enabled() -> Self { Self { code: ErrorCode::SQL_NOT_ENABLED, diff --git a/nodedb-types/src/error/details.rs b/nodedb-types/src/error/details.rs index 04af07aea..3d86ffc00 100644 --- a/nodedb-types/src/error/details.rs +++ b/nodedb-types/src/error/details.rs @@ -119,8 +119,6 @@ pub enum ErrorDetails { // Query #[serde(rename = "plan_error")] PlanError { phase: String, detail: String }, - #[serde(rename = "fan_out_exceeded")] - FanOutExceeded { shards_touched: u16, limit: u16 }, #[serde(rename = "sql_not_enabled")] SqlNotEnabled, /// A function call names no registered scalar/aggregate/window function. diff --git a/nodedb-types/src/error/msgpack/constants.rs b/nodedb-types/src/error/msgpack/constants.rs index 893da1017..7490aa8bd 100644 --- a/nodedb-types/src/error/msgpack/constants.rs +++ b/nodedb-types/src/error/msgpack/constants.rs @@ -27,7 +27,6 @@ // | 19 | CollectionDraining | // | 20 | CollectionDeactivated | // | 21 | PlanError | -// | 22 | FanOutExceeded | // | 23 | SqlNotEnabled | // | 24 | AuthorizationDenied | // | 25 | AuthExpired | @@ -113,7 +112,6 @@ pub(super) const TAG_DOCUMENT_NOT_FOUND: u16 = 18; pub(super) const TAG_COLLECTION_DRAINING: u16 = 19; pub(super) const TAG_COLLECTION_DEACTIVATED: u16 = 20; pub(super) const TAG_PLAN_ERROR: u16 = 21; -pub(super) const TAG_FAN_OUT_EXCEEDED: u16 = 22; pub(super) const TAG_SQL_NOT_ENABLED: u16 = 23; pub(super) const TAG_AUTHORIZATION_DENIED: u16 = 24; pub(super) const TAG_AUTH_EXPIRED: u16 = 25; diff --git a/nodedb-types/src/error/msgpack/decode/from_messagepack.rs b/nodedb-types/src/error/msgpack/decode/from_messagepack.rs index 94ce8ad8f..229074d6f 100644 --- a/nodedb-types/src/error/msgpack/decode/from_messagepack.rs +++ b/nodedb-types/src/error/msgpack/decode/from_messagepack.rs @@ -6,9 +6,9 @@ use zerompk::{FromMessagePack, Read}; use super::readers::{ - read_collection_deactivated, read_fan_out, read_header, read_segment_corrupted_tolerant, - read_string_vec, read_sync_delta_rejected, read_u8_field, read1_str, read2_str, - read2_str_tolerant, read2_u32, read2_u64, read3_str_tolerant, skip_fields, + read_collection_deactivated, read_header, read_segment_corrupted_tolerant, read_string_vec, + read_sync_delta_rejected, read_u8_field, read1_str, read2_str, read2_str_tolerant, read2_u32, + read2_u64, read3_str_tolerant, skip_fields, }; use crate::error::details::ErrorDetails; use crate::error::msgpack::constants::*; @@ -136,13 +136,6 @@ impl<'a> FromMessagePack<'a> for ErrorDetails { let (phase, detail) = read2_str_tolerant(reader, field_count)?; Ok(ErrorDetails::PlanError { phase, detail }) } - TAG_FAN_OUT_EXCEEDED => { - let (shards_touched, limit) = read_fan_out(reader, field_count)?; - Ok(ErrorDetails::FanOutExceeded { - shards_touched, - limit, - }) - } TAG_SQL_NOT_ENABLED => { skip_fields(reader, field_count)?; Ok(ErrorDetails::SqlNotEnabled) @@ -813,15 +806,6 @@ mod tests { assert_eq!(roundtrip(&v), v); } - #[test] - fn fan_out_exceeded_roundtrip() { - let v = ErrorDetails::FanOutExceeded { - shards_touched: 100, - limit: 50, - }; - assert_eq!(roundtrip(&v), v); - } - #[test] fn sync_delta_rejected_with_hint_roundtrip() { let v = ErrorDetails::SyncDeltaRejected { diff --git a/nodedb-types/src/error/msgpack/decode/readers.rs b/nodedb-types/src/error/msgpack/decode/readers.rs index 1cfb2a487..01e0afda5 100644 --- a/nodedb-types/src/error/msgpack/decode/readers.rs +++ b/nodedb-types/src/error/msgpack/decode/readers.rs @@ -183,24 +183,6 @@ pub(super) fn read_string_vec<'a, R: Read<'a>>( Ok(out) } -pub(super) fn read_fan_out<'a, R: Read<'a>>( - reader: &mut R, - field_count: usize, -) -> zerompk::Result<(u16, u16)> { - if field_count < 2 { - return Err(zerompk::Error::InvalidMarker(0)); - } - let _k1 = reader.read_u8()?; - let shards_touched = reader.read_u16()?; - let _k2 = reader.read_u8()?; - let limit = reader.read_u16()?; - for _ in 2..field_count { - reader.read_u8()?; - skip_one(reader)?; - } - Ok((shards_touched, limit)) -} - /// Read 2 string fields, tolerating `field_count < 2` by filling missing /// fields with `"unspecified"`. pub(super) fn read2_str_tolerant<'a, R: Read<'a>>( diff --git a/nodedb-types/src/error/msgpack/encode.rs b/nodedb-types/src/error/msgpack/encode.rs index 77f26b7e1..c14df8ee8 100644 --- a/nodedb-types/src/error/msgpack/encode.rs +++ b/nodedb-types/src/error/msgpack/encode.rs @@ -169,10 +169,6 @@ impl ToMessagePack for ErrorDetails { ErrorDetails::PlanError { phase, detail } => { write2(writer, TAG_PLAN_ERROR, phase, detail) } - ErrorDetails::FanOutExceeded { - shards_touched, - limit, - } => write2(writer, TAG_FAN_OUT_EXCEEDED, shards_touched, limit), ErrorDetails::SqlNotEnabled => write_unit(writer, TAG_SQL_NOT_ENABLED), ErrorDetails::UndefinedFunction { name } => { write1(writer, TAG_UNDEFINED_FUNCTION, name) diff --git a/nodedb-types/src/error/sqlstate.rs b/nodedb-types/src/error/sqlstate.rs index 79aa86d17..3d2b55525 100644 --- a/nodedb-types/src/error/sqlstate.rs +++ b/nodedb-types/src/error/sqlstate.rs @@ -220,7 +220,7 @@ pub const CONFIGURATION_LIMIT_EXCEEDED: &str = "53400"; /// `54000` — `program_limit_exceeded` (generic over-cap) pub const PROGRAM_LIMIT_EXCEEDED: &str = "54000"; -/// `54001` — `statement_too_complex` (fan-out / rate limit exceeded) +/// `54001` — `statement_too_complex` pub const STATEMENT_TOO_COMPLEX: &str = "54001"; // ── Class 55 — Object Not In Prerequisite State ────────────────────────────── diff --git a/nodedb-types/src/lib.rs b/nodedb-types/src/lib.rs index 452c538d3..d43404d4e 100644 --- a/nodedb-types/src/lib.rs +++ b/nodedb-types/src/lib.rs @@ -121,8 +121,8 @@ pub use quota::{ pub use result::{QueryResult, SearchResult, SubGraph}; pub use rls_write_check::{RlsWriteCheck, WriteGateDecision}; pub use row_identity::{ - DEFAULT_IDENTITY_COLUMN, HEADLESS_SENTINEL_PREFIX, RowIdentity, StorageKey, extract_pk_value, - value_to_pk_string, + DEFAULT_IDENTITY_COLUMN, HEADLESS_SENTINEL_PREFIX, ROWID_COLUMN, RowIdentity, StorageKey, + extract_pk_value, value_to_pk_string, }; pub use sparse_vector::{SparseVector, SparseVectorError}; pub use sql_quote::{quote_ident, quote_literal}; @@ -133,9 +133,9 @@ pub use sync::shape::{ShapeDefinition, ShapeType}; pub use sync::violation::ViolationType; pub use sync::wire::{SyncFrame, SyncMessageType}; pub use temporal::{ - BitemporalFilter, BitemporalInterval, LsnMapError, LsnMsAnchor, LsnMsMap, NANOS_PER_MS, - OPEN_UPPER, OrdinalClock, SystemTimeScope, ValidTimePredicate, lsn_to_ms, ms_to_ordinal_upper, - ordinal_to_ms, + BitemporalFilter, BitemporalInterval, LsnTimeAnchor, LsnTimeError, LsnTimeMap, + MAX_POSITIONS_PER_EPOCH, NANOS_PER_MS, OPEN_UPPER, OrdinalClock, SystemTimeScope, + ValidTimePredicate, calvin_txn_ordinal, ms_to_ordinal_upper, ordinal_to_ms, }; pub use text_search::{Bm25Params, QueryMode, TextSearchParams}; pub use trace::{SpanId, TraceId}; diff --git a/nodedb-types/src/row_identity.rs b/nodedb-types/src/row_identity.rs index e7d1b1311..fd97daa92 100644 --- a/nodedb-types/src/row_identity.rs +++ b/nodedb-types/src/row_identity.rs @@ -34,6 +34,12 @@ use crate::{Surrogate, Value}; /// `PRIMARY KEY`. INSERT and every stored-row identity derivation use it. pub const DEFAULT_IDENTITY_COLUMN: &str = "id"; +/// The system column a strict collection with no declared `PRIMARY KEY` +/// stores its row id in. The first write fills it from the row's surrogate, +/// and a copy keeps it, so the row keeps its identity under a new surrogate +/// after a clone, restore, or tenant move. +pub const ROWID_COLUMN: &str = "_rowid"; + /// Prefix of a rendered key for a row with no surrogate binding, used by both /// planes: the Data Plane's `HybridFusionKey::Headless` and the Control /// Plane's hybrid response decoders that must recognize the same sentinel. @@ -110,14 +116,36 @@ impl RowIdentity { /// /// The identity column is `declared_primary_key`, else /// [`DEFAULT_IDENTITY_COLUMN`]. A body carrying that column yields its - /// value. A body without it yields the decimal surrogate of `key`. + /// value. A body without it yields its [`ROWID_COLUMN`] value, and a body + /// with neither yields the decimal surrogate of `key`. A first write sets + /// `_rowid` to that same surrogate, so the two agree until a copy moves + /// the row to a new surrogate. pub fn of_stored_row( body: &[u8], declared_primary_key: Option<&str>, key: StorageKey, + ) -> RowIdentity { + match crate::value_from_msgpack(body) { + Ok(row) => Self::of_row_value(&row, declared_primary_key, key), + Err(_) => key.to_identity(), + } + } + + /// [`Self::of_stored_row`] applied to a decoded row. A row that is not + /// an object yields the decimal surrogate of `key`. A scan that projects + /// the identity columns returns a row this reads the same identity from. + pub fn of_row_value( + row: &Value, + declared_primary_key: Option<&str>, + key: StorageKey, ) -> RowIdentity { let column = declared_primary_key.unwrap_or(DEFAULT_IDENTITY_COLUMN); - extract_pk_value(body, column) + let Value::Object(obj) = row else { + return key.to_identity(); + }; + obj.get(column) + .and_then(value_to_pk_string) + .or_else(|| obj.get(ROWID_COLUMN).and_then(value_to_pk_string)) .map(RowIdentity::from_user_key) .unwrap_or_else(|| key.to_identity()) } @@ -287,4 +315,24 @@ mod tests { "user-declared-id" ); } + /// A copied strict row keeps the identity its `_rowid` carries, not its + /// new storage surrogate. + #[test] + fn of_stored_row_keeps_the_rowid_identity_under_a_new_surrogate() { + let copied = body(&[ + ("_rowid", Value::Integer(123)), + ("name", Value::String("a".into())), + ]); + let new_key = StorageKey::for_surrogate(Surrogate::new(456)); + assert_eq!( + RowIdentity::of_stored_row(&copied, None, new_key).as_str(), + "123" + ); + let declared = body(&[("sku", Value::Integer(9)), ("_rowid", Value::Integer(123))]); + assert_eq!( + RowIdentity::of_stored_row(&declared, Some("sku"), new_key).as_str(), + "9", + "a declared key wins over _rowid" + ); + } } diff --git a/nodedb-types/src/sync/violation.rs b/nodedb-types/src/sync/violation.rs index ecd53f94f..cb96f7d85 100644 --- a/nodedb-types/src/sync/violation.rs +++ b/nodedb-types/src/sync/violation.rs @@ -38,6 +38,9 @@ pub enum ViolationType { /// Foreign key reference missing. #[serde(rename = "foreign_key_missing")] ForeignKeyMissing { referenced_id: String }, + /// NOT NULL violation: a required field is absent or null. + #[serde(rename = "not_null_violation")] + NotNullViolation { field: String }, /// Permission denied (no write access to target resource). #[serde(rename = "permission_denied")] PermissionDenied, @@ -71,6 +74,7 @@ impl std::fmt::Display for ViolationType { Self::ForeignKeyMissing { referenced_id } => { write!(f, "fk_missing:{referenced_id}") } + Self::NotNullViolation { field } => write!(f, "not_null:{field}"), Self::PermissionDenied => write!(f, "permission_denied"), Self::RateLimited => write!(f, "rate_limited"), Self::TokenExpired => write!(f, "token_expired"), @@ -109,6 +113,10 @@ impl ViolationType { field: field.clone(), reason: reason.clone(), }, + Self::NotNullViolation { field } => CompensationHint::SchemaViolation { + field: field.clone(), + reason: "required field missing".into(), + }, Self::ConstraintViolation { detail } => CompensationHint::Custom { constraint: "constraint".into(), detail: detail.clone(), @@ -140,6 +148,13 @@ mod tests { .to_string(), "unique:email=x@y.com" ); + assert_eq!( + ViolationType::NotNullViolation { + field: "name".into() + } + .to_string(), + "not_null:name" + ); } #[test] @@ -222,6 +237,7 @@ mod tests { ViolationType::ForeignKeyMissing { referenced_id: "r".into(), }, + ViolationType::NotNullViolation { field: "f".into() }, ViolationType::RlsPolicyViolation { policy_name: "p".into(), }, diff --git a/nodedb-types/src/temporal/lsn_map.rs b/nodedb-types/src/temporal/lsn_map.rs index e46c9de78..0ad0e7ae2 100644 --- a/nodedb-types/src/temporal/lsn_map.rs +++ b/nodedb-types/src/temporal/lsn_map.rs @@ -1,22 +1,31 @@ // SPDX-License-Identifier: Apache-2.0 -//! LSN ↔ wall-clock milliseconds interpolator. +//! LSN ↔ commit-time map built from WAL time anchors. //! -//! Bitemporal reads need a `system_from_ms` value for every LSN, but storing -//! wall time next to every write amplifies WAL volume. Instead, the WAL writer -//! emits periodic anchor records (`RecordType::LsnMsAnchor`) and this table -//! interpolates between them linearly. +//! The WAL writer appends one time anchor to every group-commit batch, inside +//! the batch's own write. An anchor names the batch's last LSN and the HLC wall +//! time (ns since the Unix epoch) the batch committed at. Every record at or +//! below an anchor's LSN committed at or before the anchor's time. //! -//! Anchors are strictly monotonic in both LSN and wall-clock time (enforced -//! on insert). Lookup is O(log n) binary search. - -use std::cmp::Ordering; +//! Anchors are strictly increasing in both LSN and time. The map holds at most +//! `cap` anchors. When full, it first drops anchors that share a millisecond +//! with their successor, which no millisecond-granular lookup can return. If +//! that frees too little, it drops every other anchor in the older half, so +//! recent history stays dense and old history gets coarser. use serde::{Deserialize, Serialize}; -use crate::lsn::Lsn; +const NANOS_PER_MS: u64 = 1_000_000; + +/// Default anchor capacity: 65,536 anchors, 1 MiB. +pub const DEFAULT_ANCHOR_CAP: usize = 1 << 16; -/// A single anchor point mapping an LSN to a wall-clock millisecond. +/// Smallest capacity a map accepts. Downsampling needs room to keep the first +/// and last anchors plus a thinned middle. +const MIN_ANCHOR_CAP: usize = 4; + +/// One WAL time anchor: every record at or below `lsn` committed at or before +/// `hlc_wall_ns`. #[derive( Debug, Clone, @@ -28,51 +37,79 @@ use crate::lsn::Lsn; zerompk::ToMessagePack, zerompk::FromMessagePack, )] -pub struct LsnMsAnchor { +pub struct LsnTimeAnchor { pub lsn: u64, - pub wall_ms: i64, + /// HLC wall component, in nanoseconds since the Unix epoch. + pub hlc_wall_ns: u64, } -impl LsnMsAnchor { - pub const fn new(lsn: u64, wall_ms: i64) -> Self { - Self { lsn, wall_ms } +impl LsnTimeAnchor { + pub const fn new(lsn: u64, hlc_wall_ns: u64) -> Self { + Self { lsn, hlc_wall_ns } } } -/// Error produced by the LSN↔ms map. +/// Error from the LSN ↔ time map. #[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] #[non_exhaustive] -pub enum LsnMapError { - /// Attempted to insert an anchor that is not strictly monotonic in LSN or wall time. +pub enum LsnTimeError { + /// An anchor with a higher LSN carried a time at or below the last anchor's. #[error( - "non-monotonic anchor: prev=(lsn={prev_lsn}, ms={prev_ms}), \ - new=(lsn={new_lsn}, ms={new_ms})" + "time anchor is not monotonic: last=(lsn={last_lsn}, ns={last_ns}), \ + new=(lsn={new_lsn}, ns={new_ns})" )] NonMonotonic { - prev_lsn: u64, - prev_ms: i64, + last_lsn: u64, + last_ns: u64, new_lsn: u64, - new_ms: i64, + new_ns: u64, }, - /// The map is empty — cannot interpolate. - #[error("LSN→ms map is empty")] - Empty, + /// The map holds no anchor, so no time maps to an LSN. + #[error("no WAL time anchor is known; no commit time maps to an LSN")] + NoAnchors, + + /// The target time is before the oldest retained anchor. + #[error( + "time {target_ns}ns predates the oldest retained WAL time anchor \ + ({first_anchor_ns}ns); no committed state is known before it" + )] + BeforeFirstAnchor { + target_ns: u64, + first_anchor_ns: u64, + }, } -/// In-memory LSN ↔ wall-ms interpolation table. +/// Bounded, ordered LSN ↔ commit-time map. /// -/// Not `Send + Sync` by itself — callers wrap in whatever concurrency primitive -/// fits their plane (e.g. `Mutex` in Control Plane, per-core cell in Data Plane). -#[derive(Debug, Clone, Default)] -pub struct LsnMsMap { - anchors: Vec, +/// Not synchronized. The owner wraps it in the primitive its plane needs. +#[derive(Debug, Clone)] +pub struct LsnTimeMap { + anchors: Vec, + cap: usize, } -impl LsnMsMap { +impl Default for LsnTimeMap { + fn default() -> Self { + Self::new() + } +} + +impl LsnTimeMap { pub const fn new() -> Self { + Self::with_cap(DEFAULT_ANCHOR_CAP) + } + + /// A map holding at most `cap` anchors (raised to 4 when smaller). + pub const fn with_cap(cap: usize) -> Self { + let cap = if cap < MIN_ANCHOR_CAP { + MIN_ANCHOR_CAP + } else { + cap + }; Self { anchors: Vec::new(), + cap, } } @@ -84,173 +121,253 @@ impl LsnMsMap { self.anchors.is_empty() } - /// Return the anchors (oldest first) — useful for persistence/replay. - pub fn anchors(&self) -> &[LsnMsAnchor] { + pub fn cap(&self) -> usize { + self.cap + } + + /// The retained anchors, oldest first. + pub fn anchors(&self) -> &[LsnTimeAnchor] { &self.anchors } - /// Insert an anchor. Must be strictly monotonic relative to the last - /// anchor on both axes. - pub fn push(&mut self, anchor: LsnMsAnchor) -> Result<(), LsnMapError> { - if let Some(&last) = self.anchors.last() - && (anchor.lsn <= last.lsn || anchor.wall_ms <= last.wall_ms) - { - return Err(LsnMapError::NonMonotonic { - prev_lsn: last.lsn, - prev_ms: last.wall_ms, - new_lsn: anchor.lsn, - new_ms: anchor.wall_ms, - }); + pub fn last(&self) -> Option { + self.anchors.last().copied() + } + + /// Add an anchor. Returns `Ok(false)` when its LSN is at or below the last + /// anchor's: replay re-reads anchors the map already holds. + pub fn push(&mut self, anchor: LsnTimeAnchor) -> Result { + if let Some(last) = self.anchors.last().copied() { + if anchor.lsn <= last.lsn { + return Ok(false); + } + if anchor.hlc_wall_ns <= last.hlc_wall_ns { + return Err(LsnTimeError::NonMonotonic { + last_lsn: last.lsn, + last_ns: last.hlc_wall_ns, + new_lsn: anchor.lsn, + new_ns: anchor.hlc_wall_ns, + }); + } } self.anchors.push(anchor); - Ok(()) + if self.anchors.len() > self.cap { + self.downsample(); + } + Ok(true) } - /// Interpolate the wall-clock millisecond for a given LSN. + /// The highest anchored LSN whose commit time is at or before `target_ns`. + /// + /// A target past the last anchor returns the last anchor's LSN. Records + /// above it have no completed commit yet, so they are not part of any + /// past state. /// - /// - LSN < first anchor → clamps to first anchor's wall_ms. - /// - LSN > last anchor → linearly extrapolates using the last two anchors, - /// or returns the last anchor's wall_ms if only one anchor exists. - /// - LSN between two anchors → linear interpolation. - /// - LSN exactly equals an anchor → that anchor's wall_ms. - pub fn wall_ms_for_lsn(&self, lsn: Lsn) -> Result { - let target = lsn.as_u64(); - match self.anchors.as_slice() { - [] => Err(LsnMapError::Empty), - [only] => Ok(only.wall_ms), - _ => { - let search = self.anchors.binary_search_by(|a| a.lsn.cmp(&target)); - match search { - Ok(idx) => Ok(self.anchors[idx].wall_ms), - Err(idx) => { - if idx == 0 { - Ok(self.anchors[0].wall_ms) - } else if idx >= self.anchors.len() { - let n = self.anchors.len(); - Ok(Self::interpolate( - self.anchors[n - 2], - self.anchors[n - 1], - target, - )) - } else { - Ok(Self::interpolate( - self.anchors[idx - 1], - self.anchors[idx], - target, - )) - } - } - } + /// A target before the oldest anchor returns LSN 0 when that anchor names + /// LSN 0: nothing had committed by its time, so nothing had committed + /// before it either. Before an oldest anchor above LSN 0, the records at + /// or below it have no known commit time, and the lookup is an error. + pub fn lsn_at_or_before(&self, target_ns: u64) -> Result { + let first = self.anchors.first().ok_or(LsnTimeError::NoAnchors)?; + let idx = self.anchors.partition_point(|a| a.hlc_wall_ns <= target_ns); + if idx == 0 { + if first.lsn == 0 { + return Ok(0); } + return Err(LsnTimeError::BeforeFirstAnchor { + target_ns, + first_anchor_ns: first.hlc_wall_ns, + }); } + Ok(self.anchors[idx - 1].lsn) } - fn interpolate(lo: LsnMsAnchor, hi: LsnMsAnchor, lsn: u64) -> i64 { - let lsn_span = hi.lsn.saturating_sub(lo.lsn) as i128; - if lsn_span == 0 { - return lo.wall_ms; + /// [`Self::lsn_at_or_before`] for a millisecond target. The whole + /// millisecond `target_ms` counts, so a commit at `target_ms` + 0.5 ms is + /// included. A negative target is before every anchor. + pub fn lsn_at_or_before_ms(&self, target_ms: i64) -> Result { + let first = self.anchors.first().ok_or(LsnTimeError::NoAnchors)?; + let Ok(ms) = u64::try_from(target_ms) else { + if first.lsn == 0 { + return Ok(0); + } + return Err(LsnTimeError::BeforeFirstAnchor { + target_ns: 0, + first_anchor_ns: first.hlc_wall_ns, + }); + }; + let target_ns = ms + .saturating_mul(NANOS_PER_MS) + .saturating_add(NANOS_PER_MS - 1); + self.lsn_at_or_before(target_ns) + } + + /// Commit time of the batch holding `lsn`: the time of the first anchor at + /// or above it. `None` when no anchor covers `lsn` yet. + pub fn commit_ns_of(&self, lsn: u64) -> Option { + let idx = self.anchors.partition_point(|a| a.lsn < lsn); + self.anchors.get(idx).map(|a| a.hlc_wall_ns) + } + + fn downsample(&mut self) { + // Keep an anchor only when the next one falls in a later millisecond. + // A millisecond target resolves to the last anchor of its millisecond, + // so the dropped anchors were unreachable at that granularity. + let n = self.anchors.len(); + let mut write = 0; + for read in 0..n { + let keep = read + 1 == n + || self.anchors[read].hlc_wall_ns / NANOS_PER_MS + != self.anchors[read + 1].hlc_wall_ns / NANOS_PER_MS; + if keep { + self.anchors[write] = self.anchors[read]; + write += 1; + } + } + self.anchors.truncate(write); + + // Leave a quarter of the cap free so thinning is not re-run per push. + let target = self.cap - self.cap / 4; + if self.anchors.len() <= target { + return; } - let ms_span = hi.wall_ms as i128 - lo.wall_ms as i128; - let delta_lsn = (lsn as i128) - (lo.lsn as i128); - let delta_ms = ms_span * delta_lsn / lsn_span; - let result = lo.wall_ms as i128 + delta_ms; - match result.cmp(&(i64::MAX as i128)) { - Ordering::Greater => i64::MAX, - _ if result < i64::MIN as i128 => i64::MIN, - _ => result as i64, + let half = self.anchors.len() / 2; + let mut write = 0; + for read in 0..self.anchors.len() { + if read >= half || read % 2 == 0 { + self.anchors[write] = self.anchors[read]; + write += 1; + } } + self.anchors.truncate(write); } } -/// Convenience wrapper: look up wall-ms for a given LSN using an owning map. -pub fn lsn_to_ms(map: &LsnMsMap, lsn: Lsn) -> Result { - map.wall_ms_for_lsn(lsn) -} - #[cfg(test)] mod tests { use super::*; - #[test] - fn empty_map_errors() { - let m = LsnMsMap::new(); - assert!(matches!( - m.wall_ms_for_lsn(Lsn::new(5)), - Err(LsnMapError::Empty) - )); + const MS: u64 = NANOS_PER_MS; + + fn map_of(anchors: &[(u64, u64)]) -> LsnTimeMap { + let mut m = LsnTimeMap::new(); + for &(lsn, ns) in anchors { + assert!(m.push(LsnTimeAnchor::new(lsn, ns)).unwrap()); + } + m } #[test] - fn single_anchor_returns_its_wall_ms() { - let mut m = LsnMsMap::new(); - m.push(LsnMsAnchor::new(10, 1_000)).unwrap(); - assert_eq!(m.wall_ms_for_lsn(Lsn::new(5)).unwrap(), 1_000); - assert_eq!(m.wall_ms_for_lsn(Lsn::new(10)).unwrap(), 1_000); - assert_eq!(m.wall_ms_for_lsn(Lsn::new(999)).unwrap(), 1_000); + fn empty_map_is_a_typed_error() { + let m = LsnTimeMap::new(); + assert_eq!(m.lsn_at_or_before(5), Err(LsnTimeError::NoAnchors)); + assert_eq!(m.lsn_at_or_before_ms(5), Err(LsnTimeError::NoAnchors)); + assert_eq!(m.commit_ns_of(1), None); } #[test] - fn interpolates_between_anchors() { - let mut m = LsnMsMap::new(); - m.push(LsnMsAnchor::new(0, 1_000)).unwrap(); - m.push(LsnMsAnchor::new(100, 2_000)).unwrap(); - assert_eq!(m.wall_ms_for_lsn(Lsn::new(0)).unwrap(), 1_000); - assert_eq!(m.wall_ms_for_lsn(Lsn::new(50)).unwrap(), 1_500); - assert_eq!(m.wall_ms_for_lsn(Lsn::new(100)).unwrap(), 2_000); + fn floor_is_exact_at_batch_edges() { + let m = map_of(&[(5, 100), (9, 200), (14, 300)]); + assert_eq!(m.lsn_at_or_before(100).unwrap(), 5); + assert_eq!(m.lsn_at_or_before(199).unwrap(), 5); + assert_eq!(m.lsn_at_or_before(200).unwrap(), 9); + assert_eq!(m.lsn_at_or_before(299).unwrap(), 9); + assert_eq!(m.lsn_at_or_before(300).unwrap(), 14); + assert_eq!(m.lsn_at_or_before(u64::MAX).unwrap(), 14); } #[test] - fn clamps_below_first_anchor() { - let mut m = LsnMsMap::new(); - m.push(LsnMsAnchor::new(100, 5_000)).unwrap(); - m.push(LsnMsAnchor::new(200, 6_000)).unwrap(); - assert_eq!(m.wall_ms_for_lsn(Lsn::new(50)).unwrap(), 5_000); + fn target_before_first_anchor_is_an_error() { + let m = map_of(&[(5, 100), (9, 200)]); + assert_eq!( + m.lsn_at_or_before(99), + Err(LsnTimeError::BeforeFirstAnchor { + target_ns: 99, + first_anchor_ns: 100, + }) + ); + assert!(matches!( + m.lsn_at_or_before_ms(-1), + Err(LsnTimeError::BeforeFirstAnchor { .. }) + )); } #[test] - fn extrapolates_beyond_last_anchor() { - let mut m = LsnMsMap::new(); - m.push(LsnMsAnchor::new(0, 0)).unwrap(); - m.push(LsnMsAnchor::new(100, 1_000)).unwrap(); - assert_eq!(m.wall_ms_for_lsn(Lsn::new(150)).unwrap(), 1_500); + fn target_before_an_lsn_zero_anchor_is_the_empty_state() { + let m = map_of(&[(0, 100), (4, 200)]); + assert_eq!(m.lsn_at_or_before(99).unwrap(), 0); + assert_eq!(m.lsn_at_or_before(0).unwrap(), 0); + assert_eq!(m.lsn_at_or_before_ms(-1).unwrap(), 0); + assert_eq!(m.lsn_at_or_before(200).unwrap(), 4); } #[test] - fn non_monotonic_rejected() { - let mut m = LsnMsMap::new(); - m.push(LsnMsAnchor::new(10, 1_000)).unwrap(); + fn millisecond_target_covers_the_whole_millisecond() { + let m = map_of(&[(5, 7 * MS + 1), (9, 7 * MS + 999_999), (12, 8 * MS)]); assert!(matches!( - m.push(LsnMsAnchor::new(10, 2_000)), - Err(LsnMapError::NonMonotonic { .. }) - )); - assert!(matches!( - m.push(LsnMsAnchor::new(20, 1_000)), - Err(LsnMapError::NonMonotonic { .. }) + m.lsn_at_or_before_ms(6), + Err(LsnTimeError::BeforeFirstAnchor { .. }) )); + assert_eq!(m.lsn_at_or_before_ms(7).unwrap(), 9); + assert_eq!(m.lsn_at_or_before_ms(8).unwrap(), 12); + } + + #[test] + fn commit_time_is_the_covering_anchor() { + let m = map_of(&[(5, 100), (9, 200)]); + assert_eq!(m.commit_ns_of(1), Some(100)); + assert_eq!(m.commit_ns_of(5), Some(100)); + assert_eq!(m.commit_ns_of(6), Some(200)); + assert_eq!(m.commit_ns_of(9), Some(200)); + assert_eq!(m.commit_ns_of(10), None); + } + + #[test] + fn replayed_anchor_is_ignored_and_regression_is_rejected() { + let mut m = map_of(&[(10, 1_000)]); + assert!(!m.push(LsnTimeAnchor::new(10, 1_000)).unwrap()); + assert!(!m.push(LsnTimeAnchor::new(5, 500)).unwrap()); assert!(matches!( - m.push(LsnMsAnchor::new(5, 500)), - Err(LsnMapError::NonMonotonic { .. }) + m.push(LsnTimeAnchor::new(20, 1_000)), + Err(LsnTimeError::NonMonotonic { .. }) )); + assert_eq!(m.len(), 1); } #[test] - fn free_function_matches_method() { - let mut m = LsnMsMap::new(); - m.push(LsnMsAnchor::new(0, 0)).unwrap(); - m.push(LsnMsAnchor::new(100, 1_000)).unwrap(); - assert_eq!( - lsn_to_ms(&m, Lsn::new(50)).unwrap(), - m.wall_ms_for_lsn(Lsn::new(50)).unwrap() - ); + fn same_millisecond_anchors_are_dropped_first() { + let mut m = LsnTimeMap::with_cap(8); + // Two anchors per millisecond: the first of each pair is unreachable + // by a millisecond lookup. + for i in 0..9u64 { + m.push(LsnTimeAnchor::new(i + 1, (i / 2) * MS + (i % 2) * 10)) + .unwrap(); + } + assert!(m.len() <= m.cap()); + for ms in 0..4i64 { + assert_eq!(m.lsn_at_or_before_ms(ms).unwrap(), 2 * ms as u64 + 2); + } + assert_eq!(m.lsn_at_or_before_ms(4).unwrap(), 9); } #[test] - fn exact_match_binary_search() { - let mut m = LsnMsMap::new(); - m.push(LsnMsAnchor::new(0, 0)).unwrap(); - m.push(LsnMsAnchor::new(100, 1_000)).unwrap(); - m.push(LsnMsAnchor::new(200, 3_000)).unwrap(); - assert_eq!(m.wall_ms_for_lsn(Lsn::new(100)).unwrap(), 1_000); - assert_eq!(m.wall_ms_for_lsn(Lsn::new(200)).unwrap(), 3_000); + fn map_stays_bounded_and_keeps_both_ends() { + let mut m = LsnTimeMap::with_cap(64); + for i in 1..=10_000u64 { + m.push(LsnTimeAnchor::new(i, i * MS)).unwrap(); + assert!(m.len() <= m.cap()); + } + let anchors = m.anchors(); + assert_eq!(anchors[0], LsnTimeAnchor::new(1, MS)); + assert_eq!( + anchors[anchors.len() - 1], + LsnTimeAnchor::new(10_000, 10_000 * MS) + ); + assert!(anchors.windows(2).all(|w| w[0].lsn < w[1].lsn)); + // Recent history keeps full resolution. + assert_eq!(m.lsn_at_or_before_ms(9_999).unwrap(), 9_999); + // Old history is coarser, but the floor never overshoots. + let lsn = m.lsn_at_or_before_ms(500).unwrap(); + assert!(lsn <= 500); } } diff --git a/nodedb-types/src/temporal/mod.rs b/nodedb-types/src/temporal/mod.rs index d62523d2e..09e45a602 100644 --- a/nodedb-types/src/temporal/mod.rs +++ b/nodedb-types/src/temporal/mod.rs @@ -8,6 +8,9 @@ pub mod system_time; pub use filter::{BitemporalFilter, ValidTimePredicate}; pub use interval::{BitemporalInterval, OPEN_UPPER}; -pub use lsn_map::{LsnMapError, LsnMsAnchor, LsnMsMap, lsn_to_ms}; -pub use ordinal::{NANOS_PER_MS, OrdinalClock, ms_to_ordinal_upper, ordinal_to_ms}; +pub use lsn_map::{DEFAULT_ANCHOR_CAP, LsnTimeAnchor, LsnTimeError, LsnTimeMap}; +pub use ordinal::{ + MAX_POSITIONS_PER_EPOCH, NANOS_PER_MS, OrdinalClock, calvin_txn_ordinal, ms_to_ordinal_upper, + ordinal_to_ms, +}; pub use system_time::SystemTimeScope; diff --git a/nodedb-types/src/temporal/ordinal.rs b/nodedb-types/src/temporal/ordinal.rs index 1c005746e..11786081f 100644 --- a/nodedb-types/src/temporal/ordinal.rs +++ b/nodedb-types/src/temporal/ordinal.rs @@ -86,6 +86,26 @@ pub fn ordinal_to_ms(ordinal: i64) -> i64 { ordinal / NANOS_PER_MS } +/// The most positions one sequencer epoch can hold. Each position of an +/// epoch owns one nanosecond of the epoch's millisecond, so a larger epoch +/// would spill ordinals into the next millisecond. +pub const MAX_POSITIONS_PER_EPOCH: usize = NANOS_PER_MS as usize; + +/// The system-time ordinal of the Calvin transaction at `position` of the +/// epoch created at `epoch_system_ms`. +/// +/// Every participant and every replica computes it from the replicated +/// batch alone, so all of them stamp the transaction's versions alike. +/// The sequencer mints strictly increasing epoch milliseconds and caps an +/// epoch at [`MAX_POSITIONS_PER_EPOCH`] positions, so the ordinal is +/// strictly increasing in `(epoch, position)`. +pub fn calvin_txn_ordinal(epoch_system_ms: i64, position: u32) -> i64 { + epoch_system_ms + .max(0) + .saturating_mul(NANOS_PER_MS) + .saturating_add(i64::from(position)) +} + fn wall_now_ns() -> i64 { SystemTime::now() .duration_since(UNIX_EPOCH) @@ -164,6 +184,20 @@ mod tests { assert!(next_lower > upper); } + #[test] + fn calvin_txn_ordinal_orders_by_epoch_then_position() { + let first = calvin_txn_ordinal(1_700_000_000_000, 0); + let later_position = calvin_txn_ordinal(1_700_000_000_000, 7); + let last_position = calvin_txn_ordinal( + 1_700_000_000_000, + u32::try_from(MAX_POSITIONS_PER_EPOCH - 1).expect("cap fits u32"), + ); + let next_epoch = calvin_txn_ordinal(1_700_000_000_001, 0); + assert!(first < later_position); + assert!(last_position < next_epoch); + assert_eq!(ordinal_to_ms(last_position), 1_700_000_000_000); + } + #[test] fn ms_conversions_saturate_on_overflow() { // i64::MAX ms would overflow when multiplied by 1e6; saturating op diff --git a/nodedb-types/src/trace.rs b/nodedb-types/src/trace.rs index 582054a69..e63bc4f75 100644 --- a/nodedb-types/src/trace.rs +++ b/nodedb-types/src/trace.rs @@ -50,38 +50,16 @@ impl TraceId { if parts[0] != "00" { return None; } - // trace-id: 32 lowercase hex chars → 16 bytes. - let trace_hex = parts[1]; - if trace_hex.len() != 32 { - return None; - } + // trace-id: 32 hex chars → 16 bytes. parent-id: 16 hex chars → 8 + // bytes. flags: 2 hex chars. A wrong length fails the decode. let mut trace_bytes = [0u8; 16]; - for (i, chunk) in trace_hex.as_bytes().chunks(2).enumerate() { - let hi = hex_val(chunk[0])?; - let lo = hex_val(chunk[1])?; - trace_bytes[i] = (hi << 4) | lo; - } - // parent-id: 16 lowercase hex chars → 8 bytes. - let span_hex = parts[2]; - if span_hex.len() != 16 { - return None; - } + hex::decode_to_slice(parts[1], &mut trace_bytes).ok()?; let mut span_bytes = [0u8; 8]; - for (i, chunk) in span_hex.as_bytes().chunks(2).enumerate() { - let hi = hex_val(chunk[0])?; - let lo = hex_val(chunk[1])?; - span_bytes[i] = (hi << 4) | lo; - } - // flags: exactly 2 hex chars. - let flags_hex = parts[3]; - if flags_hex.len() != 2 { - return None; - } - let fhi = hex_val(flags_hex.as_bytes()[0])?; - let flo = hex_val(flags_hex.as_bytes()[1])?; - let flags = (fhi << 4) | flo; + hex::decode_to_slice(parts[2], &mut span_bytes).ok()?; + let mut flags = [0u8; 1]; + hex::decode_to_slice(parts[3], &mut flags).ok()?; - Some((TraceId(trace_bytes), SpanId(span_bytes), flags)) + Some((TraceId(trace_bytes), SpanId(span_bytes), flags[0])) } /// Render as a W3C `traceparent` header value. @@ -92,23 +70,9 @@ impl TraceId { } } -/// Decode a single ASCII hex nibble; returns `None` for non-hex characters. -#[inline] -fn hex_val(b: u8) -> Option { - match b { - b'0'..=b'9' => Some(b - b'0'), - b'a'..=b'f' => Some(b - b'a' + 10), - b'A'..=b'F' => Some(b - b'A' + 10), - _ => None, - } -} - impl fmt::Display for TraceId { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - for byte in &self.0 { - write!(f, "{byte:02x}")?; - } - Ok(()) + f.write_str(&hex::encode(self.0)) } } @@ -127,15 +91,8 @@ impl FromStr for TraceId { type Err = TraceIdParseError; fn from_str(s: &str) -> Result { - if s.len() != 32 { - return Err(TraceIdParseError(s.to_owned())); - } let mut bytes = [0u8; 16]; - for (i, chunk) in s.as_bytes().chunks(2).enumerate() { - let hi = hex_val(chunk[0]).ok_or_else(|| TraceIdParseError(s.to_owned()))?; - let lo = hex_val(chunk[1]).ok_or_else(|| TraceIdParseError(s.to_owned()))?; - bytes[i] = (hi << 4) | lo; - } + hex::decode_to_slice(s, &mut bytes).map_err(|_| TraceIdParseError(s.to_owned()))?; Ok(TraceId(bytes)) } } @@ -163,10 +120,7 @@ impl SpanId { impl fmt::Display for SpanId { fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - for byte in &self.0 { - write!(f, "{byte:02x}")?; - } - Ok(()) + f.write_str(&hex::encode(self.0)) } } @@ -185,15 +139,8 @@ impl FromStr for SpanId { type Err = SpanIdParseError; fn from_str(s: &str) -> Result { - if s.len() != 16 { - return Err(SpanIdParseError(s.to_owned())); - } let mut bytes = [0u8; 8]; - for (i, chunk) in s.as_bytes().chunks(2).enumerate() { - let hi = hex_val(chunk[0]).ok_or_else(|| SpanIdParseError(s.to_owned()))?; - let lo = hex_val(chunk[1]).ok_or_else(|| SpanIdParseError(s.to_owned()))?; - bytes[i] = (hi << 4) | lo; - } + hex::decode_to_slice(s, &mut bytes).map_err(|_| SpanIdParseError(s.to_owned()))?; Ok(SpanId(bytes)) } } @@ -242,11 +189,11 @@ mod tests { let s = "00-4bf92f3577b34da6a3ce929d0e0e4736-00f067aa0ba902b7-01"; let (tid, sid, flags) = TraceId::from_traceparent(s).expect("valid traceparent"); // Verify trace-id bytes. - let expected_trace = hex_decode_16("4bf92f3577b34da6a3ce929d0e0e4736"); - assert_eq!(tid.0, expected_trace); + let expected_trace = hex::decode("4bf92f3577b34da6a3ce929d0e0e4736").unwrap(); + assert_eq!(tid.0.as_slice(), expected_trace.as_slice()); // Verify span-id bytes. - let expected_span = hex_decode_8("00f067aa0ba902b7"); - assert_eq!(sid.0, expected_span); + let expected_span = hex::decode("00f067aa0ba902b7").unwrap(); + assert_eq!(sid.0.as_slice(), expected_span.as_slice()); assert_eq!(flags, 0x01); } @@ -310,22 +257,4 @@ mod tests { .all(|c| c.is_ascii_hexdigit() && !c.is_uppercase()) ); } - - // ── helpers ────────────────────────────────────────────────────────────── - - fn hex_decode_16(s: &str) -> [u8; 16] { - let mut out = [0u8; 16]; - for (i, chunk) in s.as_bytes().chunks(2).enumerate() { - out[i] = u8::from_str_radix(std::str::from_utf8(chunk).unwrap(), 16).unwrap(); - } - out - } - - fn hex_decode_8(s: &str) -> [u8; 8] { - let mut out = [0u8; 8]; - for (i, chunk) in s.as_bytes().chunks(2).enumerate() { - out[i] = u8::from_str_radix(std::str::from_utf8(chunk).unwrap(), 16).unwrap(); - } - out - } } diff --git a/nodedb-types/src/value/json.rs b/nodedb-types/src/value/json.rs index 66199f273..e508efdcc 100644 --- a/nodedb-types/src/value/json.rs +++ b/nodedb-types/src/value/json.rs @@ -16,10 +16,7 @@ impl From for serde_json::Value { Value::String(s) | Value::Uuid(s) | Value::Ulid(s) | Value::Regex(s) => { serde_json::Value::String(s) } - Value::Bytes(b) => { - let hex: String = b.iter().map(|byte| format!("{byte:02x}")).collect(); - serde_json::Value::String(hex) - } + Value::Bytes(b) => serde_json::Value::String(hex::encode(b)), Value::Array(arr) | Value::Set(arr) => { serde_json::Value::Array(arr.into_iter().map(serde_json::Value::from).collect()) } diff --git a/nodedb-types/src/value/sql_literal.rs b/nodedb-types/src/value/sql_literal.rs index b68c52e2b..42a227e75 100644 --- a/nodedb-types/src/value/sql_literal.rs +++ b/nodedb-types/src/value/sql_literal.rs @@ -20,7 +20,7 @@ impl Value { | Value::Uuid(value) | Value::Ulid(value) | Value::Regex(value) => quote_literal(value), - Value::Bytes(value) => quote_literal(&format!("\\x{}", hex_encode(value))), + Value::Bytes(value) => quote_literal(&format!("\\x{}", hex::encode(value))), Value::Array(values) | Value::Set(values) => array_literal(values), Value::Object(values) => object_literal(values), Value::DateTime(value) | Value::NaiveDateTime(value) => { @@ -77,16 +77,6 @@ fn finite_float_literal(value: f64) -> String { } } -fn hex_encode(bytes: &[u8]) -> String { - const HEX: &[u8; 16] = b"0123456789abcdef"; - let mut encoded = String::with_capacity(bytes.len().saturating_mul(2)); - for byte in bytes { - encoded.push(HEX[(byte >> 4) as usize] as char); - encoded.push(HEX[(byte & 0x0f) as usize] as char); - } - encoded -} - fn array_literal(values: &[Value]) -> String { let values = values.iter().map(Value::to_sql_literal).collect::>(); format!("ARRAY[{}]", values.join(", ")) diff --git a/nodedb-types/src/vector_index_params.rs b/nodedb-types/src/vector_index_params.rs index c536353d2..934f89913 100644 --- a/nodedb-types/src/vector_index_params.rs +++ b/nodedb-types/src/vector_index_params.rs @@ -39,6 +39,11 @@ pub struct StoredVectorIndexParams { pub ivf_cells: usize, /// IVF nprobe (0 = unused). pub ivf_nprobe: usize, + /// Stamped at propose time on every put; fences a replayed delete to + /// the incarnation it targeted. + #[msgpack(default)] + #[serde(default)] + pub modification_hlc: crate::hlc::Hlc, } #[cfg(test)] @@ -60,6 +65,7 @@ mod tests { pq_m: 0, ivf_cells: 0, ivf_nprobe: 0, + modification_hlc: crate::hlc::Hlc::ZERO, }; let bytes = zerompk::to_msgpack_vec(&e).unwrap(); let back: StoredVectorIndexParams = zerompk::from_msgpack(&bytes).unwrap(); diff --git a/nodedb-types/src/wire_version.rs b/nodedb-types/src/wire_version.rs index 6236e496b..500139a20 100644 --- a/nodedb-types/src/wire_version.rs +++ b/nodedb-types/src/wire_version.rs @@ -1,19 +1,36 @@ // SPDX-License-Identifier: Apache-2.0 -//! Single source of truth for the `WIRE_FORMAT_VERSION` constant -//! shared between every crate that needs to stamp or interpret it. +//! Single source of truth for the `WIRE_FORMAT_VERSION` constant and the +//! `WIRE_BUILD_ID` build identity, shared between every crate that needs to +//! stamp or interpret them. //! -//! This is the *cluster-wide* wire format version, distinct from: +//! `WIRE_FORMAT_VERSION` is the *cluster-wide* wire format version, distinct +//! from: //! - `nodedb_cluster::wire::WIRE_VERSION` (the binary frame layout //! version of the `VShardEnvelope`), //! - the RPC frame header version in //! `nodedb_cluster::rpc_codec::header` (a private constant of that //! module). //! -//! # DO NOT BUMP THIS BEFORE 1.0 +//! # The enforced invariant before 1.0: exact build identity //! -//! It stays at `1` until the first stable release. Read this before -//! changing it — the reflex to bump on any wire-shape change is wrong here: +//! `WIRE_FORMAT_VERSION` stays at `1` until the first stable release — see +//! below for why a bump buys nothing pre-1.0. That leaves a hole: two builds +//! can change wire shapes (a new enum variant, an RPC field) without +//! touching this constant, so a version-only check lets them join the same +//! cluster and misdecode each other. `WIRE_BUILD_ID` closes it: every join +//! and every wire-version handshake also compares this value for exact +//! equality, so a cluster can only ever contain nodes running one build. +//! +//! `WIRE_BUILD_ID` is the current commit's short git hash, or +//! `CARGO_PKG_VERSION` when git is unavailable (a crates.io build) — see +//! `build.rs`. It intentionally excludes dirty-tree state: hashing local +//! edits would force a rebuild of every dependent crate on every save. +//! +//! # DO NOT BUMP `WIRE_FORMAT_VERSION` BEFORE 1.0 +//! +//! Read this before changing it — the reflex to bump on any wire-shape +//! change is wrong here: //! //! - **There is nothing to be compatible with.** Pre-1.0 there are no //! deployed clusters, so there is no older peer a new build must talk to. @@ -30,11 +47,11 @@ //! and independent. Changing the constant here therefore cannot orphan or //! corrupt anything already on disk. //! -//! So: adding a new enum variant, RPC, or payload field needs NO bump. Every -//! node in a working cluster runs the same build by construction. Ratcheting -//! this pre-1.0 only invents a stop-the-world upgrade requirement that does -//! not otherwise exist, and would leave 1.0 shipping as "wire version 20" for -//! no reason. +//! So: adding a new enum variant, RPC, or payload field needs NO bump. +//! `WIRE_BUILD_ID` already forces every node in a working cluster onto the +//! same build. Ratcheting this pre-1.0 only invents a stop-the-world upgrade +//! requirement that does not otherwise exist, and would leave 1.0 shipping +//! as "wire version 20" for no reason. //! //! After 1.0, when real deployments exist and a genuine compatibility window //! is introduced, this becomes meaningful — bump it then, deliberately, and @@ -51,8 +68,26 @@ pub const WIRE_FORMAT_VERSION: u16 = 1; /// `WIRE_FORMAT_VERSION`: floor == ceiling, no backward compat window. pub const MIN_WIRE_FORMAT_VERSION: u16 = WIRE_FORMAT_VERSION; +/// This build's exact identity: the short git commit hash, or +/// `CARGO_PKG_VERSION` when git is unavailable (a crates.io build). Set by +/// `build.rs`. +/// +/// Compared for exact equality on every join and wire-version handshake — +/// the enforced invariant before 1.0. See the module docs above. +pub const WIRE_BUILD_ID: &str = env!("NODEDB_WIRE_BUILD_ID"); + // Compile-time invariants — these constants must satisfy: // - MIN_WIRE_FORMAT_VERSION <= WIRE_FORMAT_VERSION // - WIRE_FORMAT_VERSION > 0 const _: () = assert!(MIN_WIRE_FORMAT_VERSION <= WIRE_FORMAT_VERSION); const _: () = assert!(WIRE_FORMAT_VERSION > 0); + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn wire_build_id_is_non_empty() { + assert!(!WIRE_BUILD_ID.is_empty(), "WIRE_BUILD_ID must not be empty"); + } +} diff --git a/nodedb-vector/src/collection/lifecycle_compact.rs b/nodedb-vector/src/collection/lifecycle_compact.rs index 55db42cb7..3b1a3d607 100644 --- a/nodedb-vector/src/collection/lifecycle_compact.rs +++ b/nodedb-vector/src/collection/lifecycle_compact.rs @@ -127,6 +127,16 @@ impl VectorCollection { total_removed } + /// The multi-vector documents of this index, by document surrogate, in + /// ascending order. `export_snapshot` rows carry no membership, so a + /// snapshot records it here: a one-vector document is otherwise + /// indistinguishable from a single-vector row. + pub fn multi_vector_documents(&self) -> Vec { + let mut documents: Vec = self.multi_doc_map.keys().copied().collect(); + documents.sort_unstable(); + documents + } + /// Export all live vectors for snapshot. /// /// # Errors @@ -227,4 +237,22 @@ mod tests { assert_eq!(coll.local_for_surrogate(s), Some(id)); assert_eq!(coll.live_count(), 1); } + + #[test] + fn a_one_vector_multi_vector_document_is_listed() { + let mut coll = VectorCollection::with_seal_threshold(2, HnswParams::default(), 64); + coll.insert_with_surrogate(vec![1.0, 0.0], Surrogate::new(1)) + .unwrap(); + let one: [&[f32]; 1] = [&[0.0, 1.0]]; + coll.insert_multi_vector(&one, Surrogate::new(9)).unwrap(); + let two: [&[f32]; 2] = [&[0.5, 0.5], &[0.2, 0.8]]; + coll.insert_multi_vector(&two, Surrogate::new(5)).unwrap(); + + assert_eq!( + coll.multi_vector_documents(), + vec![Surrogate::new(5), Surrogate::new(9)], + "every multi-vector document is listed, the one-vector one included; \ + the single-vector row is not" + ); + } } diff --git a/nodedb-vector/src/collection/lifecycle_insert_ops.rs b/nodedb-vector/src/collection/lifecycle_insert_ops.rs index 3acd15324..2c3713ba7 100644 --- a/nodedb-vector/src/collection/lifecycle_insert_ops.rs +++ b/nodedb-vector/src/collection/lifecycle_insert_ops.rs @@ -39,6 +39,12 @@ impl VectorCollection { /// /// A vector without the collection dimension fails with /// [`VectorError::DimensionMismatch`] before the old binding is touched. + /// + /// [`Surrogate::ZERO`] binds nothing: a headless vector has no surrogate, + /// so it never enters `surrogate_map` or `surrogate_to_local`, and no + /// lookup or delete by `ZERO` reaches it. Compaction, checkpoint restore + /// and rollback rebuild both maps from `surrogate_map` alone, so they + /// never map `ZERO` either. pub fn insert_with_surrogate( &mut self, vector: Vec, @@ -339,4 +345,30 @@ mod tests { assert_eq!(coll.vector_for_surrogate(s), None); assert_eq!(coll.vector_for_id(999), None); } + + #[test] + fn headless_vectors_bind_no_surrogate() { + let mut coll = collection(); + let first = coll + .insert_with_surrogate(vec![1.0, 0.0], Surrogate::ZERO) + .unwrap(); + let second = coll + .insert_with_surrogate(vec![0.0, 1.0], Surrogate::ZERO) + .unwrap(); + coll.insert_multi_vector(&[&[0.5, 0.5]], Surrogate::ZERO) + .unwrap(); + assert_ne!(first, second); + assert_eq!(coll.live_count(), 3, "every headless insert stays live"); + assert_eq!(coll.local_for_surrogate(Surrogate::ZERO), None); + assert_eq!(coll.get_surrogate(first), None); + assert_eq!(coll.get_surrogate(second), None); + assert!(coll.surrogate_to_local.is_empty()); + assert!(coll.multi_doc_map.is_empty()); + + assert!( + !coll.delete_by_surrogate(Surrogate::ZERO), + "a delete by ZERO reaches no vector" + ); + assert_eq!(coll.live_count(), 3); + } } diff --git a/nodedb-wal/src/error.rs b/nodedb-wal/src/error.rs index 8cedd3fe6..58c38dd1e 100644 --- a/nodedb-wal/src/error.rs +++ b/nodedb-wal/src/error.rs @@ -159,6 +159,34 @@ pub enum WalError { found_lsn: u64, }, + /// The last segment starts at or below an LSN an earlier segment already + /// holds. An interrupted roll created it while the earlier segment stayed + /// active. Resuming it would reissue LSNs, so the WAL refuses to open. + #[error( + "WAL segment {path} starts at LSN {first_lsn}, but {previous_path} already holds \ + records through LSN {previous_last_lsn}; resuming would reissue those LSNs. \ + {path} was created by an interrupted segment roll: check that it holds no \ + records, remove it, and restart" + )] + SegmentOverlapsPrevious { + path: String, + first_lsn: u64, + previous_path: String, + previous_last_lsn: u64, + }, + + /// A segment roll failed, and removing the segment file it created also + /// failed. The next open refuses that file with `SegmentOverlapsPrevious`. + #[error( + "WAL segment roll failed: {roll}; removing the unused segment {path} also failed: \ + {cleanup}" + )] + RollCleanupFailed { + path: String, + roll: Box, + cleanup: Box, + }, + /// A replay was asked for the suffix starting at `from_lsn`, but the WAL /// no longer retains it: checkpoint truncation has already deleted every /// segment below `retained_floor_lsn`. diff --git a/nodedb-wal/src/lazy_reader.rs b/nodedb-wal/src/lazy_reader.rs index bb4316480..3fb315548 100644 --- a/nodedb-wal/src/lazy_reader.rs +++ b/nodedb-wal/src/lazy_reader.rs @@ -462,6 +462,7 @@ mod tests { database_id: 0, apply_key: 0, event_source: crate::record::NO_EVENT_SOURCE, + commit_hlc: 0, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); diff --git a/nodedb-wal/src/lib.rs b/nodedb-wal/src/lib.rs index 361b071ac..8fb9fc1d9 100644 --- a/nodedb-wal/src/lib.rs +++ b/nodedb-wal/src/lib.rs @@ -60,6 +60,8 @@ pub mod segmented; #[cfg(not(target_arch = "wasm32"))] pub mod temporal_purge; #[cfg(not(target_arch = "wasm32"))] +pub mod time_anchors; +#[cfg(not(target_arch = "wasm32"))] pub mod tombstone; #[cfg(not(target_arch = "wasm32"))] pub mod torn_tail; @@ -102,6 +104,8 @@ pub use segmented::{SegmentedWal, SegmentedWalConfig}; #[cfg(not(target_arch = "wasm32"))] pub use temporal_purge::{TemporalPurgeEngine, TemporalPurgePayload}; #[cfg(not(target_arch = "wasm32"))] +pub use time_anchors::TimeAnchors; +#[cfg(not(target_arch = "wasm32"))] pub use tombstone::{CollectionTombstonePayload, MAX_COLLECTION_NAME_LEN}; #[cfg(not(target_arch = "wasm32"))] pub use torn_tail::{TailVerdict, verify_committed_prefix}; diff --git a/nodedb-wal/src/mmap_reader/reader.rs b/nodedb-wal/src/mmap_reader/reader.rs index 6a6041aa0..d60b43f57 100644 --- a/nodedb-wal/src/mmap_reader/reader.rs +++ b/nodedb-wal/src/mmap_reader/reader.rs @@ -481,6 +481,7 @@ mod tests { database_id: 0, apply_key: 0, event_source: crate::record::NO_EVENT_SOURCE, + commit_hlc: 0, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); @@ -504,6 +505,7 @@ mod tests { database_id: 0, apply_key: 0, event_source: crate::record::NO_EVENT_SOURCE, + commit_hlc: 0, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); diff --git a/nodedb-wal/src/reader.rs b/nodedb-wal/src/reader.rs index f63b03205..ce46141ab 100644 --- a/nodedb-wal/src/reader.rs +++ b/nodedb-wal/src/reader.rs @@ -473,6 +473,7 @@ mod tests { database_id: 0, apply_key: 0, event_source: crate::record::NO_EVENT_SOURCE, + commit_hlc: 0, crc32c: 0, }; std::fs::write(&path, header.to_bytes()).unwrap(); diff --git a/nodedb-wal/src/record/anchor.rs b/nodedb-wal/src/record/anchor.rs index daed8c1cf..64b2f90ba 100644 --- a/nodedb-wal/src/record/anchor.rs +++ b/nodedb-wal/src/record/anchor.rs @@ -1,61 +1,42 @@ // SPDX-License-Identifier: Apache-2.0 -//! LSN ↔ wall-clock anchor payload. +//! Time-anchor payload. //! -//! Anchors are written periodically to the WAL so that, during replay, the -//! bitemporal subsystem can reconstruct a stable `system_from_ms` for every -//! LSN via interpolation between the nearest surrounding anchors. -//! -//! Payload layout (fixed 16 bytes, little-endian): -//! -//! ```text -//! ┌─────────┬────────────┐ -//! │ lsn u64 │ wall_ms i64│ -//! └─────────┴────────────┘ -//! ``` +//! The writer appends one `TimeAnchor` record to every group-commit batch. The +//! record's header LSN is the batch's last LSN. The payload is the HLC wall +//! time the batch committed at: 8 bytes, little-endian nanoseconds since the +//! Unix epoch. use crate::error::{Result, WalError}; -/// Size of an anchor payload on disk. -pub const ANCHOR_PAYLOAD_SIZE: usize = 16; +/// Size of a time-anchor payload on disk. +pub const TIME_ANCHOR_PAYLOAD_SIZE: usize = 8; -/// LSN ↔ wall-clock milliseconds anchor. +/// Commit time of the batch a `TimeAnchor` record closes. #[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub struct LsnMsAnchorPayload { - /// WAL LSN at which this anchor was written. - pub lsn: u64, - /// Wall-clock milliseconds since Unix epoch at write time. - pub wall_ms: i64, +pub struct TimeAnchorPayload { + /// HLC wall component, in nanoseconds since the Unix epoch. + pub hlc_wall_ns: u64, } -impl LsnMsAnchorPayload { - pub const fn new(lsn: u64, wall_ms: i64) -> Self { - Self { lsn, wall_ms } +impl TimeAnchorPayload { + pub const fn new(hlc_wall_ns: u64) -> Self { + Self { hlc_wall_ns } } - pub fn to_bytes(&self) -> [u8; ANCHOR_PAYLOAD_SIZE] { - let mut buf = [0u8; ANCHOR_PAYLOAD_SIZE]; - buf[0..8].copy_from_slice(&self.lsn.to_le_bytes()); - buf[8..16].copy_from_slice(&self.wall_ms.to_le_bytes()); - buf + pub fn to_bytes(&self) -> [u8; TIME_ANCHOR_PAYLOAD_SIZE] { + self.hlc_wall_ns.to_le_bytes() } pub fn from_bytes(buf: &[u8]) -> Result { - if buf.len() != ANCHOR_PAYLOAD_SIZE { - return Err(WalError::InvalidPayload { + let bytes: [u8; TIME_ANCHOR_PAYLOAD_SIZE] = + buf.try_into().map_err(|_| WalError::InvalidPayload { detail: format!( - "LsnMsAnchor payload must be {ANCHOR_PAYLOAD_SIZE} bytes, got {}", + "TimeAnchor payload must be {TIME_ANCHOR_PAYLOAD_SIZE} bytes, got {}", buf.len() ), - }); - } - let lsn = u64::from_le_bytes([ - buf[0], buf[1], buf[2], buf[3], buf[4], buf[5], buf[6], buf[7], - ]); - let wall_ms = i64::from_le_bytes([ - buf[8], buf[9], buf[10], buf[11], buf[12], buf[13], buf[14], buf[15], - ]); - Ok(Self { lsn, wall_ms }) + })?; + Ok(Self::new(u64::from_le_bytes(bytes))) } } @@ -65,22 +46,14 @@ mod tests { #[test] fn anchor_roundtrip() { - let anchor = LsnMsAnchorPayload::new(12_345, 1_700_000_000_000); - let bytes = anchor.to_bytes(); - assert_eq!(LsnMsAnchorPayload::from_bytes(&bytes).unwrap(), anchor); - } - - #[test] - fn anchor_negative_wall_ms() { - // Wall-clock predates epoch — rare but permitted by i64 encoding. - let anchor = LsnMsAnchorPayload::new(0, -1); + let anchor = TimeAnchorPayload::new(1_700_000_000_000_000_123); let bytes = anchor.to_bytes(); - assert_eq!(LsnMsAnchorPayload::from_bytes(&bytes).unwrap(), anchor); + assert_eq!(TimeAnchorPayload::from_bytes(&bytes).unwrap(), anchor); } #[test] fn anchor_wrong_size_rejected() { - assert!(LsnMsAnchorPayload::from_bytes(&[0u8; 15]).is_err()); - assert!(LsnMsAnchorPayload::from_bytes(&[0u8; 17]).is_err()); + assert!(TimeAnchorPayload::from_bytes(&[0u8; 7]).is_err()); + assert!(TimeAnchorPayload::from_bytes(&[0u8; 9]).is_err()); } } diff --git a/nodedb-wal/src/record/fts_spatial.rs b/nodedb-wal/src/record/fts_spatial.rs index d454aebaf..599a570d3 100644 --- a/nodedb-wal/src/record/fts_spatial.rs +++ b/nodedb-wal/src/record/fts_spatial.rs @@ -17,6 +17,9 @@ //! FtsIndex additionally carries `text_len u32 + text bytes`. //! SpatialPut additionally carries `field_len u32 + field bytes + geometry_len u32 + geometry bytes`. //! SpatialDelete additionally carries `field_len u32 + field bytes`. +//! FtsDelete and SpatialDelete prefix the id with a presence tag `u8`: +//! `0` means the key's home binds no row and no id follows, `1` means the id +//! follows. //! //! These structs are net-new (no legacy records to stay compatible with). //! Field set can be extended when the handler is wired; the length-prefixed @@ -85,6 +88,40 @@ fn push_str_field(buf: &mut Vec, s: &str) -> Result<()> { Ok(()) } +/// Presence tag of an optional string field: absent. +const FIELD_ABSENT: u8 = 0; +/// Presence tag of an optional string field: present, followed by the field. +const FIELD_PRESENT: u8 = 1; + +fn push_opt_str_field(buf: &mut Vec, s: Option<&str>) -> Result<()> { + match s { + None => { + buf.push(FIELD_ABSENT); + Ok(()) + } + Some(s) => { + buf.push(FIELD_PRESENT); + push_str_field(buf, s) + } + } +} + +fn read_opt_utf8_field(buf: &[u8], offset: usize) -> Result<(Option, usize)> { + match buf.get(offset).copied() { + Some(FIELD_ABSENT) => Ok((None, offset + 1)), + Some(FIELD_PRESENT) => { + let (s, next) = read_utf8_field(buf, offset + 1)?; + Ok((Some(s), next)) + } + Some(tag) => Err(WalError::InvalidPayload { + detail: format!("optional field at offset {offset} has unknown presence tag {tag}"), + }), + None => Err(WalError::InvalidPayload { + detail: format!("truncated at offset {offset}, need a presence tag"), + }), + } +} + fn push_bytes_field(buf: &mut Vec, data: &[u8]) -> Result<()> { if data.len() > u32::MAX as usize { return Err(WalError::InvalidPayload { @@ -189,19 +226,21 @@ impl FtsIndexPayload { pub struct FtsDeletePayload { pub provenance: SyncProvenance, pub collection: String, - pub doc_id: String, + /// The deleted document's identifier. `None` when the key's home binds + /// no row: the delete removes nothing and still commits its provenance. + pub doc_id: Option, } impl FtsDeletePayload { pub fn new( provenance: SyncProvenance, collection: impl Into, - doc_id: impl Into, + doc_id: Option, ) -> Self { Self { provenance, collection: collection.into(), - doc_id: doc_id.into(), + doc_id, } } @@ -209,7 +248,7 @@ impl FtsDeletePayload { let mut buf = Vec::new(); push_provenance(&mut buf, &self.provenance); push_str_field(&mut buf, &self.collection)?; - push_str_field(&mut buf, &self.doc_id)?; + push_opt_str_field(&mut buf, self.doc_id.as_deref())?; Ok(buf) } @@ -217,7 +256,7 @@ impl FtsDeletePayload { let (provenance, mut off) = read_provenance(buf)?; let (collection, next) = read_utf8_field(buf, off)?; off = next; - let (doc_id, _) = read_utf8_field(buf, off)?; + let (doc_id, _) = read_opt_utf8_field(buf, off)?; Ok(Self { provenance, collection, @@ -295,7 +334,9 @@ pub struct SpatialDeletePayload { pub provenance: SyncProvenance, pub collection: String, pub field: String, - pub doc_id: String, + /// The deleted row's identifier. `None` when the key's home binds no + /// row: the delete removes nothing and still commits its provenance. + pub doc_id: Option, } impl SpatialDeletePayload { @@ -303,13 +344,13 @@ impl SpatialDeletePayload { provenance: SyncProvenance, collection: impl Into, field: impl Into, - doc_id: impl Into, + doc_id: Option, ) -> Self { Self { provenance, collection: collection.into(), field: field.into(), - doc_id: doc_id.into(), + doc_id, } } @@ -318,7 +359,7 @@ impl SpatialDeletePayload { push_provenance(&mut buf, &self.provenance); push_str_field(&mut buf, &self.collection)?; push_str_field(&mut buf, &self.field)?; - push_str_field(&mut buf, &self.doc_id)?; + push_opt_str_field(&mut buf, self.doc_id.as_deref())?; Ok(buf) } @@ -328,7 +369,7 @@ impl SpatialDeletePayload { off = next; let (field, next) = read_utf8_field(buf, off)?; off = next; - let (doc_id, _) = read_utf8_field(buf, off)?; + let (doc_id, _) = read_opt_utf8_field(buf, off)?; Ok(Self { provenance, collection, @@ -374,11 +415,28 @@ mod tests { #[test] fn fts_delete_roundtrip() { - let p = FtsDeletePayload::new(prov(1, 2, 3, 4), "articles", "doc-99"); + let p = FtsDeletePayload::new(prov(1, 2, 3, 4), "articles", Some("doc-99".to_string())); let bytes = p.to_bytes().unwrap(); assert_eq!(FtsDeletePayload::from_bytes(&bytes).unwrap(), p); } + #[test] + fn fts_delete_of_an_unbound_key_roundtrip() { + let p = FtsDeletePayload::new(prov(1, 2, 3, 4), "articles", None); + let bytes = p.to_bytes().unwrap(); + assert_eq!(FtsDeletePayload::from_bytes(&bytes).unwrap(), p); + } + + #[test] + fn unknown_presence_tag_rejected() { + let p = FtsDeletePayload::new(prov(1, 2, 3, 4), "articles", None); + let mut bytes = p.to_bytes().unwrap(); + if let Some(tag) = bytes.last_mut() { + *tag = 7; + } + assert!(FtsDeletePayload::from_bytes(&bytes).is_err()); + } + #[test] fn spatial_put_roundtrip() { let p = SpatialPutPayload::new( @@ -403,7 +461,19 @@ mod tests { #[test] fn spatial_delete_roundtrip() { - let p = SpatialDeletePayload::new(prov(9, 10, 11, 12), "places", "loc", "poi-1"); + let p = SpatialDeletePayload::new( + prov(9, 10, 11, 12), + "places", + "loc", + Some("poi-1".to_string()), + ); + let bytes = p.to_bytes().unwrap(); + assert_eq!(SpatialDeletePayload::from_bytes(&bytes).unwrap(), p); + } + + #[test] + fn spatial_delete_of_an_unbound_key_roundtrip() { + let p = SpatialDeletePayload::new(prov(9, 10, 11, 12), "places", "loc", None); let bytes = p.to_bytes().unwrap(); assert_eq!(SpatialDeletePayload::from_bytes(&bytes).unwrap(), p); } diff --git a/nodedb-wal/src/record/header.rs b/nodedb-wal/src/record/header.rs index 499d2cb24..5dbf29d25 100644 --- a/nodedb-wal/src/record/header.rs +++ b/nodedb-wal/src/record/header.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: Apache-2.0 -//! WAL record header: fixed 55-byte prefix + constants. +//! WAL record header: fixed 63-byte prefix + constants. use crate::error::{Result, WalError}; @@ -9,23 +9,9 @@ pub const WAL_MAGIC: u32 = 0x5359_4E57; // "SYNW" /// Current WAL format version. /// -/// v2 introduces bitemporal record layout: `LsnMsAnchor` records (type 102) -/// provide stable LSN↔wall-clock interpolation, and engine-level writers emit -/// `system_from_ms` in versioned keys. -/// -/// v3 introduces the 16-byte segment preamble (`WALP` magic) written at offset -/// 0 of every WAL segment file. -/// -/// v4 widens `record_type` u16→u32 and `vshard_id` u16→u32, adds 16 reserved -/// bytes (covered by CRC32C) before the checksum, and bumps `HEADER_SIZE` to -/// 50 bytes. Pre-release — no v1/v2/v3 readers supported. -/// -/// v1 is the initial shipped format with 54-byte headers (u64 tenant_id, -/// u16 vshard_id, u32 payload_len, u16 reserved, u32 crc32c). -/// -/// v2 adds the one-byte event source at offset 50 and grows the header to -/// 55 bytes. A v1 record does not open. -pub const WAL_FORMAT_VERSION: u16 = 2; +/// Every record header carries this version. The header layout is the one +/// [`HEADER_SIZE`] describes. A record with any other version does not open. +pub const WAL_FORMAT_VERSION: u16 = 3; /// Maximum WAL record payload size (64 MiB). Distinct from cluster RPC's limit. pub const MAX_WAL_PAYLOAD_SIZE: usize = 64 * 1024 * 1024; @@ -35,13 +21,10 @@ pub const MAX_WAL_PAYLOAD_SIZE: usize = 64 * 1024 * 1024; /// Layout (all little-endian): /// magic(4) | format_version(2) | record_type(4) | lsn(8) | tenant_id(8) /// | vshard_id(4) | payload_len(4) | database_id(8) | apply_key(8) -/// | event_source(1) | crc32c(4) +/// | event_source(1) | commit_hlc(8) | crc32c(4) /// -/// `database_id` occupies bytes 34–41 (previously part of the 16-byte reserved -/// field). `apply_key` occupies bytes 42–49. Bytes 34–41 were zero-filled in -/// prior records, so `database_id == 0` maps to `DatabaseId(0)` (the default -/// database), preserving backward compatibility without a format-version bump. -pub const HEADER_SIZE: usize = 55; +/// `database_id` occupies bytes 34–41. `apply_key` occupies bytes 42–49. +pub const HEADER_SIZE: usize = 63; /// The event source of a record that carries no row write. Replay rebuilds /// no write event from it. A write record carries the code of the source its @@ -49,15 +32,14 @@ pub const HEADER_SIZE: usize = 55; pub const NO_EVENT_SOURCE: u8 = 0; /// Bit 14 in `record_type` signals the payload is AES-256-GCM encrypted. -/// Separate from bit 15 (required flag). Both bits keep their positions; -/// the type is now u32 so the constants are widened accordingly. +/// Separate from bit 15 (required flag). pub const ENCRYPTED_FLAG: u32 = 0x0000_4000; /// Bit 15: required-flag. Records with this bit set and an unknown type /// must not be silently skipped. pub const REQUIRED_FLAG: u32 = 0x0000_8000; -/// WAL record header (fixed 55 bytes). +/// WAL record header (fixed 63 bytes). #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct RecordHeader { pub magic: u32, @@ -67,11 +49,8 @@ pub struct RecordHeader { pub tenant_id: u64, pub vshard_id: u32, pub payload_len: u32, - /// Database scope for this record. Stored as a raw `u64`; callers convert - /// to/from `DatabaseId`. Pre-Tier-2 records had zeros here, so `0` maps to - /// `DatabaseId(0)` (the default database) — fully backward compatible. - /// - /// Occupies bytes 34–41 of the on-disk header (previously part of reserved). + /// Database scope for this record, as the raw `u64` of a `DatabaseId`. + /// Decoded exactly as written. Covered by CRC32C. Occupies bytes 34–41. pub database_id: u64, /// The idempotency key of the replicated proposal whose apply appended /// this record, `0` for a record no proposal apply appended. The record @@ -83,6 +62,13 @@ pub struct RecordHeader { /// code. [`NO_EVENT_SOURCE`] for a record that carries no row write. /// Covered by CRC32C. Occupies byte 50. pub event_source: u8, + /// HLC wall time, in nanoseconds, at which the write this record belongs + /// to committed, or the proposer's stamp of the metadata entry whose + /// apply appended it. `0` for a record that belongs to neither: a cluster + /// restore judges such a record by its WAL position, never by a clock. A + /// restore keeps a record with a nonzero value only when it is below the + /// restore point's watermark. Covered by CRC32C. Occupies bytes 51–58. + pub commit_hlc: u64, pub crc32c: u32, } @@ -99,7 +85,8 @@ impl RecordHeader { buf[34..42].copy_from_slice(&self.database_id.to_le_bytes()); buf[42..50].copy_from_slice(&self.apply_key.to_le_bytes()); buf[50] = self.event_source; - buf[51..55].copy_from_slice(&self.crc32c.to_le_bytes()); + buf[51..59].copy_from_slice(&self.commit_hlc.to_le_bytes()); + buf[59..63].copy_from_slice(&self.crc32c.to_le_bytes()); buf } @@ -123,7 +110,10 @@ impl RecordHeader { buf[42], buf[43], buf[44], buf[45], buf[46], buf[47], buf[48], buf[49], ]), event_source: buf[50], - crc32c: u32::from_le_bytes([buf[51], buf[52], buf[53], buf[54]]), + commit_hlc: u64::from_le_bytes([ + buf[51], buf[52], buf[53], buf[54], buf[55], buf[56], buf[57], buf[58], + ]), + crc32c: u32::from_le_bytes([buf[59], buf[60], buf[61], buf[62]]), } } @@ -188,6 +178,7 @@ mod tests { database_id: 0, apply_key: 0, event_source: NO_EVENT_SOURCE, + commit_hlc: 0, crc32c: 0xDEAD_BEEF, } } @@ -200,11 +191,11 @@ mod tests { } #[test] - fn header_golden_55_bytes_exact_offsets() { + fn header_golden_63_bytes_exact_offsets() { // magic at 0..4, format_version at 4..6, record_type at 6..10, // lsn at 10..18, tenant_id at 18..26, vshard_id at 26..30, // payload_len at 30..34, database_id at 34..42, apply_key at 42..50, - // event_source at 50, crc32c at 51..55. + // event_source at 50, commit_hlc at 51..59, crc32c at 59..63. let header = RecordHeader { magic: WAL_MAGIC, format_version: WAL_FORMAT_VERSION, @@ -216,10 +207,11 @@ mod tests { database_id: 0xABCD_0000_1234_5678, apply_key: 0, event_source: NO_EVENT_SOURCE, + commit_hlc: 0, crc32c: 0x1234_5678, }; let b = header.to_bytes(); - assert_eq!(b.len(), 55); + assert_eq!(b.len(), 63); // magic assert_eq!(&b[0..4], &WAL_MAGIC.to_le_bytes()); // format_version @@ -240,8 +232,10 @@ mod tests { assert_eq!(&b[42..50], &[0u8; 8]); // event_source assert_eq!(b[50], NO_EVENT_SOURCE); + // commit_hlc — zero + assert_eq!(&b[51..59], &[0u8; 8]); // crc32c - assert_eq!(&b[51..55], &0x1234_5678u32.to_le_bytes()); + assert_eq!(&b[59..63], &0x1234_5678u32.to_le_bytes()); } #[test] @@ -258,6 +252,7 @@ mod tests { database_id: 7, apply_key: 0, event_source: NO_EVENT_SOURCE, + commit_hlc: 0, crc32c: 0, }; let bytes = header.to_bytes(); @@ -266,16 +261,27 @@ mod tests { } #[test] - fn pre_tier2_zero_database_id_compat() { - // A record written before Tier 2 has zeros at bytes 34..42. - // from_bytes must decode that as database_id == 0 (the default database). - let mut raw = [0u8; HEADER_SIZE]; - raw[0..4].copy_from_slice(&WAL_MAGIC.to_le_bytes()); - raw[4..6].copy_from_slice(&WAL_FORMAT_VERSION.to_le_bytes()); - raw[6..10].copy_from_slice(&1u32.to_le_bytes()); // record_type - // bytes 34..50 stay zero (pre-Tier-2 reserved field) + fn commit_hlc_roundtrip() { + let header = RecordHeader { + commit_hlc: 1_700_000_000_123_456_789, + ..make_header(1, 0) + }; + let decoded = RecordHeader::from_bytes(&header.to_bytes()); + assert_eq!(decoded.commit_hlc, 1_700_000_000_123_456_789); + assert_eq!(decoded, header); + } + + #[test] + fn database_id_decodes_from_its_own_bytes() { + // Bytes 34..42 alone decide `database_id`. The neighbouring fields + // are all ones, so a misplaced read would show. + let mut raw = [0xFFu8; HEADER_SIZE]; + raw[34..42].copy_from_slice(&0x0102_0304_0506_0708u64.to_le_bytes()); let decoded = RecordHeader::from_bytes(&raw); - assert_eq!(decoded.database_id, 0); + assert_eq!(decoded.database_id, 0x0102_0304_0506_0708); + assert_eq!(decoded.payload_len, u32::MAX); + assert_eq!(decoded.apply_key, u64::MAX); + assert_eq!(RecordHeader::from_bytes(&decoded.to_bytes()), decoded); } #[test] @@ -293,6 +299,7 @@ mod tests { database_id: 0, apply_key: 0, event_source: NO_EVENT_SOURCE, + commit_hlc: 0, crc32c: 0, }; let bytes = header.to_bytes(); @@ -334,13 +341,13 @@ mod tests { } #[test] - fn version_4_rejected() { - // Regression: bumping from v4 to v5 — a v4 header must be rejected. + fn older_version_rejected() { let mut header = make_header(0, 0); - header.format_version = 4; + header.format_version = WAL_FORMAT_VERSION - 1; assert!(matches!( header.validate(0), - Err(WalError::UnsupportedVersion { version: 4, .. }) + Err(WalError::UnsupportedVersion { version, .. }) + if version == WAL_FORMAT_VERSION - 1 )); } diff --git a/nodedb-wal/src/record/mod.rs b/nodedb-wal/src/record/mod.rs index addd1e768..280e177ea 100644 --- a/nodedb-wal/src/record/mod.rs +++ b/nodedb-wal/src/record/mod.rs @@ -12,6 +12,8 @@ pub mod header; #[cfg(not(target_arch = "wasm32"))] pub mod padding; #[cfg(not(target_arch = "wasm32"))] +pub mod restore_point; +#[cfg(not(target_arch = "wasm32"))] pub mod surrogate; #[cfg(not(target_arch = "wasm32"))] pub mod sync_seq; @@ -23,7 +25,7 @@ pub mod wal_record; #[cfg(not(target_arch = "wasm32"))] pub use aborted::{WRITE_ABORTED_PAYLOAD_SIZE, WriteAbortedPayload}; #[cfg(not(target_arch = "wasm32"))] -pub use anchor::{ANCHOR_PAYLOAD_SIZE, LsnMsAnchorPayload}; +pub use anchor::{TIME_ANCHOR_PAYLOAD_SIZE, TimeAnchorPayload}; #[cfg(not(target_arch = "wasm32"))] pub use calvin::CalvinAppliedPayload; #[cfg(not(target_arch = "wasm32"))] @@ -37,6 +39,8 @@ pub(crate) use padding::pad_buffer_to_alignment; #[cfg(not(target_arch = "wasm32"))] pub use padding::{MIN_PADDING_RECORD_SIZE, padding_record, padding_span}; #[cfg(not(target_arch = "wasm32"))] +pub use restore_point::{RESTORE_POINT_PAYLOAD_SIZE, RestorePointPayload}; +#[cfg(not(target_arch = "wasm32"))] pub use surrogate::{SURROGATE_PAYLOAD_SIZE, SurrogateAllocPayload, SurrogateBindPayload}; #[cfg(not(target_arch = "wasm32"))] pub use sync_seq::{SYNC_SEQ_ADVANCE_PAYLOAD_SIZE, SyncSeqAdvancePayload}; diff --git a/nodedb-wal/src/record/restore_point.rs b/nodedb-wal/src/record/restore_point.rs new file mode 100644 index 000000000..87726ca28 --- /dev/null +++ b/nodedb-wal/src/record/restore_point.rs @@ -0,0 +1,163 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Restore-point payload. +//! +//! A cluster restore point names one consistent instant across every Raft +//! group. Each node appends one `RestorePoint` record per group it hosts, at +//! the group's place in its log. A cluster point-in-time restore reads them to +//! restart every group at that place. +//! +//! Payload layout (little-endian): a fixed 56-byte head, then the vShards +//! the group homed at the point. +//! +//! ```text +//! ┌──────┬──────┬──────────┬───────────────┬──────┬────────────┬─────────────────┬───────┬────────────┐ +//! │ id │ hlc │ group_id │ applied_index │ term │ next_epoch │ epoch_system_ms │ count │ vshard × n │ +//! │ u64 │ u64 │ u64 │ u64 │ u64 │ u64 │ u64 │ u32 │ u32 each │ +//! └──────┴──────┴──────────┴───────────────┴──────┴────────────┴─────────────────┴───────┴────────────┘ +//! ``` + +use crate::error::{Result, WalError}; + +/// Size of a restore-point payload's fixed head on disk. +pub const RESTORE_POINT_PAYLOAD_SIZE: usize = 56; + +/// One group's place at a restore point. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct RestorePointPayload { + /// The restore point's id, shared by every group and node. + pub id: u64, + /// The point's watermark: HLC wall time in nanoseconds. A restore keeps + /// a record only when its commit HLC is below it. + pub hlc: u64, + pub group_id: u64, + /// The group's log index at the point. Every entry at or below it applied + /// before the record was appended. + pub applied_index: u64, + /// The term of the entry at `applied_index`. + pub term: u64, + /// For the Calvin sequencer group, the first epoch the sequencer may + /// propose after the point. `0` for every other group. + pub next_epoch: u64, + /// For the Calvin sequencer group, the highest epoch instant (ms) the + /// sequencer applied before the point. A restored sequencer mints above + /// it. `0` for every other group, and for a sequencer that applied no + /// epoch. + pub epoch_system_ms: u64, + /// The vShards the group homed at the point. Empty for the metadata and + /// sequencer groups. A restore places a record of these vShards that + /// carries no commit HLC before or after this record in the WAL. + pub vshards: Vec, +} + +impl RestorePointPayload { + pub fn to_bytes(&self) -> Vec { + let mut buf = Vec::with_capacity(RESTORE_POINT_PAYLOAD_SIZE + 4 + 4 * self.vshards.len()); + for value in [ + self.id, + self.hlc, + self.group_id, + self.applied_index, + self.term, + self.next_epoch, + self.epoch_system_ms, + ] { + buf.extend_from_slice(&value.to_le_bytes()); + } + let count = u32::try_from(self.vshards.len()).unwrap_or(u32::MAX); + buf.extend_from_slice(&count.to_le_bytes()); + for vshard in self.vshards.iter().take(count as usize) { + buf.extend_from_slice(&vshard.to_le_bytes()); + } + buf + } + + pub fn from_bytes(buf: &[u8]) -> Result { + let refuse = || WalError::InvalidPayload { + detail: format!( + "RestorePoint payload of {} bytes is no {RESTORE_POINT_PAYLOAD_SIZE}-byte head, \ + count and vShard list", + buf.len() + ), + }; + let head = buf.get(..RESTORE_POINT_PAYLOAD_SIZE).ok_or_else(refuse)?; + let mut fields = [0u64; 7]; + for (field, word) in fields.iter_mut().zip(head.as_chunks::<8>().0) { + *field = u64::from_le_bytes(*word); + } + let [ + id, + hlc, + group_id, + applied_index, + term, + next_epoch, + epoch_system_ms, + ] = fields; + let rest = &buf[RESTORE_POINT_PAYLOAD_SIZE..]; + let count: [u8; 4] = rest + .get(..4) + .and_then(|b| b.try_into().ok()) + .ok_or_else(refuse)?; + let count = u32::from_le_bytes(count) as usize; + let list = &rest[4..]; + if list.len() != count.checked_mul(4).ok_or_else(refuse)? { + return Err(refuse()); + } + let vshards = list + .as_chunks::<4>() + .0 + .iter() + .map(|word| u32::from_le_bytes(*word)) + .collect(); + Ok(Self { + id, + hlc, + group_id, + applied_index, + term, + next_epoch, + epoch_system_ms, + vshards, + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn restore_point_roundtrip() { + let point = RestorePointPayload { + id: 3, + hlc: 1_700_000_000_000_000_000, + group_id: 7, + applied_index: 1_234, + term: 5, + next_epoch: 0, + epoch_system_ms: 1_700_000_000_123, + vshards: vec![4, 9, 1023], + }; + assert_eq!( + RestorePointPayload::from_bytes(&point.to_bytes()).unwrap(), + point + ); + } + + #[test] + fn wrong_size_is_refused() { + assert!(RestorePointPayload::from_bytes(&[0u8; 55]).is_err()); + assert!( + RestorePointPayload::from_bytes(&[0u8; 56]).is_err(), + "no count" + ); + let mut one = vec![0u8; 60]; + one[56] = 1; + assert!( + RestorePointPayload::from_bytes(&one).is_err(), + "a count above the list" + ); + assert!(RestorePointPayload::from_bytes(&[0u8; 60]).is_ok()); + } +} diff --git a/nodedb-wal/src/record/types.rs b/nodedb-wal/src/record/types.rs index b3540cd97..19a16c25f 100644 --- a/nodedb-wal/src/record/types.rs +++ b/nodedb-wal/src/record/types.rs @@ -240,13 +240,13 @@ pub enum RecordType { /// Collection hard-delete tombstone. CollectionTombstoned = 101 | 0x8000, - /// LSN ↔ wall-clock anchor for bitemporal `system_from_ms` interpolation. - /// Emitted periodically by the WAL writer. Payload: `LsnMsAnchorPayload` - /// (fixed 16 bytes, little-endian: `[lsn: u64, wall_ms: i64]`). + /// Commit time of the group-commit batch it closes. The writer appends + /// one per batch, and its header LSN is the batch's last LSN. Payload: + /// `TimeAnchorPayload` (8 bytes, little-endian HLC wall ns). /// - /// Not required: a replay that skips these records produces a slightly - /// coarser interpolation table but does not corrupt state. - LsnMsAnchor = 102, + /// Not required: a replay that skips these records maps times to LSNs more + /// coarsely but does not corrupt state. + TimeAnchor = 102, /// Bitemporal version purge — drops one or more *superseded* row /// versions (those with finite `_ts_valid_until`) once @@ -359,6 +359,70 @@ pub enum RecordType { /// Required: skipping this record re-applies a duplicate proposal, which /// double-counts every non-idempotent effect (a timeseries append). ProposalApplied = 62 | 0x8000, + + /// Names the replicated log position of one Raft data-group entry this + /// node applied. Payload: `apply_key`, `group_id`, `log_index`, and the + /// vShard's partition epoch, each a little-endian `u64` (32 bytes). The + /// header's `apply_key` is `0`. + /// + /// Appended just before the records the entry's apply writes, which + /// carry the payload's `apply_key` in their headers. Change-data-capture + /// recovery reads it to give every replica's change events the same + /// position. Never replayed into any engine. + /// + /// Not required: skipping it loses no engine state. + ChangePosition = 63, + + /// One Raft group's place at a cluster restore point, on this node. + /// Payload: `RestorePointPayload` (restore point id, watermark HLC, group + /// id, the group's log index and term at the point, the sequencer's next + /// epoch, and the group's vShards). The header's `commit_hlc` is the + /// point's watermark. A cluster point-in-time restore reads it. Never + /// replayed into any engine. + /// + /// Not required: skipping it loses no engine state. + RestorePoint = 64, + + /// Graph engine: the edge tombstones one document delete's node cascade + /// wrote. Payload: the node and every edge it tombstoned, each with its + /// tombstone's ordinal (see `wal::redo::NodeCascadeRedo`). It is a + /// sub-record of a `WriteGroup` part, never a record of its own. Replay + /// writes exactly these tombstones at exactly these ordinals, and the + /// document delete it follows cascades nothing on replay. + /// + /// Required: skipping it leaves every cascaded edge live after replay. + GraphNodeCascade = 65 | 0x8000, + + /// A Raft snapshot install replaced the state of one data group on this + /// node with rows no WAL record carries. Payload: the group id, a + /// little-endian `u64`. A point-in-time restore to any LSN at or after this + /// record must start from a base taken after it. Never replayed into any + /// engine. + /// + /// Required: a restore that skipped it would replay the WAL across the + /// install onto a base that lacks the installed rows. + SnapshotInstalled = 66 | 0x8000, + + /// One record of a write's record group: the group's opening record, or + /// one part of the rows the write stored after apply. Payload: the group + /// descriptor and the part's engine-native sub-records (see + /// `wal::redo::WriteGroupRecord`). Replay applies the sub-records in + /// order, as it applies a `TransactionRedo` record's. A point-in-time + /// restore keeps a group only whole: a group with a part above its target, + /// or a part missing, is dropped entire. + /// + /// Required: skipping it drops the rows a write stored. + WriteGroup = 67 | 0x8000, + + /// Graph engine: one TRUNCATE share's cut of an edge collection. Payload: + /// the collection and the ordinal of the TRUNCATE's Calvin transaction + /// (see `wal::redo::EdgeCutRedo`). It is a sub-record of a + /// `TransactionRedo` record, never a record of its own. It writes no edge + /// version: every read hides the collection's versions applied below the + /// cut. + /// + /// Required: skipping it leaves every truncated edge live after replay. + GraphEdgeCut = 68 | 0x8000, } impl RecordType { @@ -402,7 +466,7 @@ impl RecordType { x if x == 42 | 0x8000 => Some(Self::ArrayFlush), x if x == 100 | 0x8000 => Some(Self::Checkpoint), x if x == 101 | 0x8000 => Some(Self::CollectionTombstoned), - 102 => Some(Self::LsnMsAnchor), + 102 => Some(Self::TimeAnchor), x if x == 103 | 0x8000 => Some(Self::TemporalPurge), x if x == 110 | 0x8000 => Some(Self::CalvinApplied), x if x == 53 | 0x8000 => Some(Self::SyncSeqAdvance), @@ -414,6 +478,12 @@ impl RecordType { x if x == 60 | 0x8000 => Some(Self::GraphNodeLabelRemove), x if x == 61 | 0x8000 => Some(Self::WriteAborted), x if x == 62 | 0x8000 => Some(Self::ProposalApplied), + 63 => Some(Self::ChangePosition), + 64 => Some(Self::RestorePoint), + x if x == 65 | 0x8000 => Some(Self::GraphNodeCascade), + x if x == 66 | 0x8000 => Some(Self::SnapshotInstalled), + x if x == 67 | 0x8000 => Some(Self::WriteGroup), + x if x == 68 | 0x8000 => Some(Self::GraphEdgeCut), _ => None, } } @@ -431,7 +501,8 @@ mod tests { assert!(!RecordType::is_required(RecordType::Noop as u32)); assert!(!RecordType::is_required(RecordType::TimeseriesBatch as u32)); assert!(!RecordType::is_required(RecordType::LogBatch as u32)); - assert!(!RecordType::is_required(RecordType::LsnMsAnchor as u32)); + assert!(!RecordType::is_required(RecordType::TimeAnchor as u32)); + assert!(!RecordType::is_required(RecordType::ChangePosition as u32)); assert!(RecordType::is_required(RecordType::TemporalPurge as u32)); assert!(RecordType::is_required(RecordType::SyncSeqAdvance as u32)); assert!(RecordType::is_required(RecordType::FtsIndex as u32)); @@ -487,7 +558,7 @@ mod tests { RecordType::SurrogateBind, RecordType::Checkpoint, RecordType::CollectionTombstoned, - RecordType::LsnMsAnchor, + RecordType::TimeAnchor, RecordType::TemporalPurge, RecordType::CalvinApplied, RecordType::SyncSeqAdvance, @@ -499,6 +570,12 @@ mod tests { RecordType::GraphNodeLabelRemove, RecordType::WriteAborted, RecordType::ProposalApplied, + RecordType::ChangePosition, + RecordType::RestorePoint, + RecordType::GraphNodeCascade, + RecordType::SnapshotInstalled, + RecordType::WriteGroup, + RecordType::GraphEdgeCut, ] { assert_eq!(RecordType::from_raw(ty as u32), Some(ty)); } diff --git a/nodedb-wal/src/record/wal_record.rs b/nodedb-wal/src/record/wal_record.rs index e34facdc4..c04140bbc 100644 --- a/nodedb-wal/src/record/wal_record.rs +++ b/nodedb-wal/src/record/wal_record.rs @@ -27,6 +27,9 @@ pub struct RecordTarget { /// The event source code of the row write the record carries. /// [`NO_EVENT_SOURCE`] for a record that carries no row write. pub event_source: u8, + /// HLC wall time, in nanoseconds, the record's write committed at. See + /// [`RecordHeader::commit_hlc`]. + pub commit_hlc: u64, } /// The header fields that tie a record to the write that appended it. @@ -37,13 +40,17 @@ pub struct RecordStamp { pub apply_key: u64, /// The event source code of the row write the record carries. pub event_source: u8, + /// HLC wall time, in nanoseconds, the record's write committed at. See + /// [`RecordHeader::commit_hlc`]. + pub commit_hlc: u64, } impl RecordStamp { - /// No proposal key and no row write. + /// No proposal key, no row write, and no commit HLC. pub const NONE: Self = Self { apply_key: 0, event_source: NO_EVENT_SOURCE, + commit_hlc: 0, }; } @@ -71,20 +78,19 @@ impl WalRecord { /// the ciphertext to its segment (preamble-swap defense). Pass `None` /// for unencrypted records (the argument is ignored in that case). /// - /// `database_id` is stored in header bytes 34-41 (previously reserved, - /// zero-filled). Pre-existing records with zeros decode to `DatabaseId(0)` - /// (the default database), preserving backward compatibility. + /// `database_id` is stored in header bytes 34-41. pub fn new(args: WalRecordArgs<'_>) -> Result { Self::new_stamped(args, RecordStamp::NONE) } - /// [`Self::new`] with the proposal key and event source of `stamp`. Both - /// ride the header, inside the CRC and the encryption AAD, so the record - /// and its stamp are durable together. + /// [`Self::new`] with the proposal key, event source and commit HLC of + /// `stamp`. They ride the header, inside the CRC and the encryption AAD, + /// so the record and its stamp are durable together. pub fn new_stamped(args: WalRecordArgs<'_>, stamp: RecordStamp) -> Result { let RecordStamp { apply_key, event_source, + commit_hlc, } = stamp; let WalRecordArgs { record_type, @@ -115,6 +121,7 @@ impl WalRecord { database_id, apply_key, event_source, + commit_hlc, crc32c: 0, }; let header_bytes = temp_header.to_bytes(); @@ -144,6 +151,7 @@ impl WalRecord { database_id, apply_key, event_source, + commit_hlc, crc32c: 0, }; @@ -299,7 +307,7 @@ impl WalRecord { /// Build the AAD buffer: `preamble_bytes || header_bytes`. /// -/// When `preamble_bytes` is `None` (no encryption or legacy path), the AAD +/// When `preamble_bytes` is `None` (no encryption), the AAD /// is just the header bytes. When present, the preamble is prepended. pub(crate) fn build_aad( preamble_bytes: Option<&[u8; PREAMBLE_SIZE]>, @@ -379,10 +387,10 @@ mod tests { #[test] fn anchor_payload_in_record() { - use super::super::anchor::LsnMsAnchorPayload; - let anchor = LsnMsAnchorPayload::new(42, 1_700_000_000_000); + use super::super::anchor::TimeAnchorPayload; + let anchor = TimeAnchorPayload::new(1_700_000_000_000_000_000); let record = WalRecord::new(WalRecordArgs { - record_type: RecordType::LsnMsAnchor as u32, + record_type: RecordType::TimeAnchor as u32, lsn: 42, tenant_id: 0, vshard_id: 0, @@ -393,8 +401,8 @@ mod tests { }) .unwrap(); record.verify_checksum().unwrap(); - assert_eq!(record.logical_record_type(), RecordType::LsnMsAnchor as u32); - let decoded = LsnMsAnchorPayload::from_bytes(&record.payload).unwrap(); + assert_eq!(record.logical_record_type(), RecordType::TimeAnchor as u32); + let decoded = TimeAnchorPayload::from_bytes(&record.payload).unwrap(); assert_eq!(decoded, anchor); } @@ -415,6 +423,7 @@ mod tests { RecordStamp { apply_key: 7, event_source: code, + commit_hlc: 11, }, ) .expect("record"); @@ -425,6 +434,7 @@ mod tests { .expect("checksum covers the source"); let decoded = RecordHeader::from_bytes(&record.header.to_bytes()); assert_eq!(decoded.event_source, code); + assert_eq!(decoded.commit_hlc, 11); } } } diff --git a/nodedb-wal/src/segment/meta.rs b/nodedb-wal/src/segment/meta.rs index e17f6c259..a6d662b49 100644 --- a/nodedb-wal/src/segment/meta.rs +++ b/nodedb-wal/src/segment/meta.rs @@ -58,10 +58,10 @@ pub fn segment_path(wal_dir: &Path, first_lsn: u64) -> PathBuf { wal_dir.join(segment_filename(first_lsn)) } -/// Parse the first_lsn from a segment filename. +/// Parse the first_lsn from a segment filename built by [`segment_filename`]. /// /// Returns `None` if the filename doesn't match the expected pattern. -pub(crate) fn parse_segment_filename(filename: &str) -> Option { +pub fn parse_segment_filename(filename: &str) -> Option { let stem = filename.strip_prefix(SEGMENT_PREFIX)?; let lsn_str = stem.strip_suffix(&format!(".{SEGMENT_EXTENSION}"))?; lsn_str.parse::().ok() @@ -90,6 +90,13 @@ mod tests { ); } + #[test] + fn parse_segment_filename_round_trips_real_names() { + for lsn in [0, 1, 999, 1 << 40, u64::MAX - 1, u64::MAX] { + assert_eq!(parse_segment_filename(&segment_filename(lsn)), Some(lsn)); + } + } + #[test] fn parse_segment_filename_invalid() { assert_eq!(parse_segment_filename("wal.log"), None); diff --git a/nodedb-wal/src/segment/mod.rs b/nodedb-wal/src/segment/mod.rs index ad36a534f..fe564b386 100644 --- a/nodedb-wal/src/segment/mod.rs +++ b/nodedb-wal/src/segment/mod.rs @@ -28,6 +28,7 @@ pub mod continuity; pub mod decrypt; pub mod discovery; pub mod meta; +pub mod orphan; pub mod retained; pub mod truncate; @@ -38,6 +39,10 @@ pub use checkpoint_frame::{read_checkpoint_framed, write_checkpoint_framed}; pub use continuity::SegmentContinuity; pub use decrypt::SegmentDecryptor; pub use discovery::discover_segments; -pub use meta::{DEFAULT_SEGMENT_TARGET_SIZE, SegmentMeta, segment_filename, segment_path}; +pub use meta::{ + DEFAULT_SEGMENT_TARGET_SIZE, SegmentMeta, parse_segment_filename, segment_filename, + segment_path, +}; +pub use orphan::check_resume_above_previous; pub use retained::check_retained_floor; pub use truncate::{TruncateResult, truncate_segments}; diff --git a/nodedb-wal/src/segment/orphan.rs b/nodedb-wal/src/segment/orphan.rs new file mode 100644 index 000000000..cdaeee83d --- /dev/null +++ b/nodedb-wal/src/segment/orphan.rs @@ -0,0 +1,166 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! Segment files a roll created but never installed. +//! +//! A roll creates the next segment file before it seals the current one. When +//! a step in between fails, the old writer stays active and goes on writing +//! the LSNs the new file's name claims. Two guards keep that file from ever +//! being resumed: +//! +//! - [`discard_unused_segment`] removes it before the roll returns its error. +//! - [`check_resume_above_previous`] refuses to open a WAL whose last segment +//! starts at or below an LSN an earlier segment already holds. That covers +//! a crash mid-roll and a removal that itself failed. + +use std::path::Path; + +use crate::error::{Result, WalError}; +use crate::recovery::recover; + +use super::atomic_io::fsync_directory; +use super::meta::SegmentMeta; + +/// Remove the segment file a failed roll created, and fsync the directory so +/// the removal survives a crash. Returns the error the caller returns: `roll_err` +/// alone, or [`WalError::RollCleanupFailed`] naming both failures. +pub(crate) fn discard_unused_segment(wal_dir: &Path, path: &Path, roll_err: WalError) -> WalError { + let removed = match std::fs::remove_file(path) { + Ok(()) => Ok(()), + // The roll may have failed before the file existed. + Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(()), + Err(e) => Err(e), + }; + let cleanup = removed + .map_err(WalError::Io) + .and_then(|()| fsync_directory(wal_dir)); + match cleanup { + Ok(()) => roll_err, + Err(cleanup_err) => WalError::RollCleanupFailed { + path: path.display().to_string(), + roll: Box::new(roll_err), + cleanup: Box::new(cleanup_err), + }, + } +} + +/// Refuse to resume `segments` (in LSN order) when the last one would reissue +/// LSNs an earlier segment already holds. +/// +/// `resumed_next_lsn` is the next LSN the writer resumed on the last segment +/// would assign. The nearest earlier segment with records must end below it. +/// +/// Refusing is safer than removing the file here. The file's name claims LSNs +/// that are already written elsewhere, but only a scan says it holds no +/// records, and a damaged header scans as empty. Deleting on that evidence +/// can destroy records. Refusing destroys nothing, and the error names the +/// file an operator removes. +pub fn check_resume_above_previous(segments: &[SegmentMeta], resumed_next_lsn: u64) -> Result<()> { + let Some((last, earlier)) = segments.split_last() else { + return Ok(()); + }; + for previous in earlier.iter().rev() { + let info = recover(&previous.path)?; + if info.record_count == 0 { + continue; + } + if info.last_lsn >= resumed_next_lsn { + return Err(WalError::SegmentOverlapsPrevious { + path: last.path.display().to_string(), + first_lsn: last.first_lsn, + previous_path: previous.path.display().to_string(), + previous_last_lsn: info.last_lsn, + }); + } + return Ok(()); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::record::RecordType; + use crate::segment::{discover_segments, segment_path}; + use crate::writer::{WalWriter, WalWriterConfig}; + + fn config() -> WalWriterConfig { + WalWriterConfig { + use_direct_io: false, + ..Default::default() + } + } + + /// Write a segment starting at `first_lsn` holding `records` records. + fn write_segment(dir: &Path, first_lsn: u64, records: u64) { + let path = segment_path(dir, first_lsn); + let mut writer = WalWriter::open_with_start_lsn(&path, config(), first_lsn).unwrap(); + for i in 0..records { + writer + .append(RecordType::Put as u32, 1, 0, 0, &i.to_le_bytes()) + .unwrap(); + } + writer.sync().unwrap(); + } + + #[test] + fn an_empty_segment_inside_the_previous_segments_lsns_is_refused() { + let dir = tempfile::tempdir().unwrap(); + // The old segment kept writing LSNs 1..=5 after a roll created 3. + write_segment(dir.path(), 1, 5); + write_segment(dir.path(), 3, 0); + let segments = discover_segments(dir.path()).unwrap(); + let err = check_resume_above_previous(&segments, 3).unwrap_err(); + assert!( + matches!( + err, + WalError::SegmentOverlapsPrevious { + first_lsn: 3, + previous_last_lsn: 5, + .. + } + ), + "{err}" + ); + } + + #[test] + fn an_empty_segment_after_the_previous_one_resumes() { + let dir = tempfile::tempdir().unwrap(); + write_segment(dir.path(), 1, 5); + write_segment(dir.path(), 6, 0); + let segments = discover_segments(dir.path()).unwrap(); + check_resume_above_previous(&segments, 6).unwrap(); + } + + #[test] + fn discard_removes_the_file_and_keeps_the_roll_error() { + let dir = tempfile::tempdir().unwrap(); + let path = segment_path(dir.path(), 9); + std::fs::write(&path, b"").unwrap(); + let err = discard_unused_segment(dir.path(), &path, WalError::Sealed); + assert!(matches!(err, WalError::Sealed), "{err}"); + assert!(!path.exists()); + } + + #[test] + fn discard_of_a_missing_file_keeps_the_roll_error() { + let dir = tempfile::tempdir().unwrap(); + let path = segment_path(dir.path(), 9); + let err = discard_unused_segment(dir.path(), &path, WalError::Sealed); + assert!(matches!(err, WalError::Sealed), "{err}"); + } + + #[test] + fn a_failed_removal_names_both_failures() { + let dir = tempfile::tempdir().unwrap(); + // A non-empty directory at the segment path cannot be removed as a file. + let path = segment_path(dir.path(), 9); + std::fs::create_dir(&path).unwrap(); + std::fs::write(path.join("x"), b"x").unwrap(); + let err = discard_unused_segment(dir.path(), &path, WalError::Sealed); + assert!(matches!(err, WalError::RollCleanupFailed { .. }), "{err}"); + let msg = err.to_string(); + assert!(msg.contains("sealed") || msg.contains("Sealed"), "{msg}"); + assert!(msg.contains(&path.display().to_string()), "{msg}"); + } +} diff --git a/nodedb-wal/src/segmented.rs b/nodedb-wal/src/segmented.rs index 4a31e212e..f4754c63b 100644 --- a/nodedb-wal/src/segmented.rs +++ b/nodedb-wal/src/segmented.rs @@ -24,9 +24,11 @@ use tracing::info; use crate::crypto::KeyRing; use crate::error::{Result, WalError}; use crate::record::{RecordTarget, WalRecord}; +use crate::segment::orphan::discard_unused_segment; use crate::segment::{ DEFAULT_SEGMENT_TARGET_SIZE, SegmentContinuity, SegmentMeta, TruncateResult, - check_retained_floor, discover_segments, segment_path, truncate_segments, + check_resume_above_previous, check_retained_floor, discover_segments, segment_path, + truncate_segments, }; use crate::writer::{WalWriter, WalWriterConfig}; @@ -123,6 +125,7 @@ impl SegmentedWal { let last = &segments[segments.len() - 1]; let writer = WalWriter::open_resuming(&last.path, config.writer_config.clone(), last.first_lsn)?; + check_resume_above_previous(&segments, writer.next_lsn())?; (writer, last.first_lsn) }; @@ -198,6 +201,7 @@ impl SegmentedWal { vshard_id, database_id, event_source: crate::record::NO_EVENT_SOURCE, + commit_hlc: 0, }, payload, 0, @@ -226,11 +230,26 @@ impl SegmentedWal { self.writer.sync() } + /// Seal the active segment and start the next one, so every record + /// appended so far sits in a sealed segment. A no-op when the active + /// segment holds no record. + pub fn seal_active_segment(&mut self) -> Result<()> { + if self.writer.next_lsn() == self.active_first_lsn { + return Ok(()); + } + self.roll_segment() + } + /// The next LSN that will be assigned. pub fn next_lsn(&self) -> u64 { self.writer.next_lsn() } + /// The largest payload one record takes (see [`WalWriter::max_payload`]). + pub fn max_payload(&self) -> usize { + self.writer.max_payload() + } + /// First LSN of the active (currently written) segment. pub fn active_segment_first_lsn(&self) -> u64 { self.active_first_lsn @@ -331,24 +350,21 @@ impl SegmentedWal { /// the install, so no record is acknowledged into a segment whose name /// might not survive a crash. fn roll_segment_with_ring(&mut self, next_ring: Option) -> Result<()> { - // Reading the next LSN does not consume it, so this is the same value - // the seal below would leave behind — nothing appends in between. + // A sync can append the batch's time anchor, so it runs before the + // new segment's first LSN is read. The seal below then has nothing + // left to append, and the LSN read here is the one it leaves behind. + self.writer.sync()?; let new_first_lsn = self.writer.next_lsn(); let new_path = segment_path(&self.wal_dir, new_first_lsn); - let mut new_writer = - WalWriter::open_with_start_lsn(&new_path, self.writer_config.clone(), new_first_lsn)?; - - crate::segment::fsync_directory(&self.wal_dir)?; - - if let Some(ref ring) = next_ring { - new_writer.set_encryption_ring(ring.clone())?; - } - - // Last fallible step. A seal is a durability barrier over the old - // segment's buffered records; if it fails those records were never - // made durable, and the writer reports that on every later call — a - // real data-loss error, not a bookkeeping state this roll created. - self.writer.seal()?; + // Every failure from here on removes the new file before returning. + // Left on disk, it names LSNs the still-active old segment goes on to + // write, and a restart would resume from it and reissue them. + let new_writer = match self.open_and_seal(&new_path, new_first_lsn, next_ring.as_ref()) { + Ok(new_writer) => new_writer, + Err(roll_err) => { + return Err(discard_unused_segment(&self.wal_dir, &new_path, roll_err)); + } + }; // Everything past here is infallible, so no failure can strand a // sealed writer that was never replaced. @@ -363,6 +379,34 @@ impl SegmentedWal { ); Ok(()) } + + /// Create the next segment at `new_path`, then seal the old writer. + fn open_and_seal( + &mut self, + new_path: &Path, + new_first_lsn: u64, + next_ring: Option<&KeyRing>, + ) -> Result { + let mut new_writer = + WalWriter::open_with_start_lsn(new_path, self.writer_config.clone(), new_first_lsn)?; + + crate::segment::fsync_directory(&self.wal_dir)?; + + if let Some(ring) = next_ring { + new_writer.set_encryption_ring(ring.clone())?; + } + + nodedb_types::fail_point_err!("wal::roll_before_seal", |detail: String| WalError::Io( + std::io::Error::other(format!("failpoint wal::roll_before_seal: {detail}")) + )); + + // Last fallible step. A seal is a durability barrier over the old + // segment's buffered records; if it fails those records were never + // made durable, and the writer reports that on every later call — a + // real data-loss error, not a bookkeeping state this roll created. + self.writer.seal()?; + Ok(new_writer) + } } /// Replay all records from all segments in a WAL directory, in LSN order. @@ -1020,4 +1064,109 @@ mod tests { let (records2, _) = wal.replay_from_limit(next_lsn, 200).unwrap(); assert_eq!(records2.len(), 15); // 20 - 5 = 15 remaining } + + /// Anchors ride every batch, keep LSNs unique across a rollover, and come + /// back from disk after a restart. + #[test] + fn time_anchors_survive_rollover_and_restart() { + use std::sync::Arc; + + use crate::time_anchors::TimeAnchors; + + let dir = tempfile::tempdir().unwrap(); + let wal_dir = dir.path().join("wal"); + let config = |anchors: &Arc| SegmentedWalConfig { + wal_dir: wal_dir.clone(), + segment_target_size: 200, + writer_config: WalWriterConfig { + use_direct_io: false, + time_anchors: Some(Arc::clone(anchors)), + ..Default::default() + }, + }; + + let live = Arc::new(TimeAnchors::new(Arc::new(nodedb_types::HlcClock::new()))); + { + let mut wal = SegmentedWal::open(config(&live)).unwrap(); + for i in 0..12u32 { + wal.append(RecordType::Put as u32, 1, 0, 0, format!("r{i}").as_bytes()) + .unwrap(); + if i % 3 == 2 { + wal.sync().unwrap(); + } + } + wal.sync().unwrap(); + assert!(wal.list_segments().unwrap().len() > 1); + } + let live_anchors = live.anchors(); + assert!(live_anchors.len() >= 4); + + let records = replay_all_segments(&wal_dir, None).unwrap(); + let lsns: Vec = records.iter().map(|r| r.header.lsn).collect(); + assert!(lsns.windows(2).all(|w| w[1] == w[0] + 1), "{lsns:?}"); + let puts = records + .iter() + .filter(|r| RecordType::from_raw(r.logical_record_type()) == Some(RecordType::Put)) + .count(); + assert_eq!(puts, 12); + + let restarted = TimeAnchors::new(Arc::new(nodedb_types::HlcClock::new())); + restarted.absorb_replayed(&records).unwrap(); + assert_eq!(restarted.anchors(), live_anchors); + } + + /// A roll that fails after creating the next segment removes it. The old + /// writer stays active, and a retried roll succeeds. + #[cfg(feature = "failpoints")] + #[test] + fn an_injected_seal_failure_leaves_no_new_segment_file() { + use nodedb_types::fail_point::FailGuard; + + let dir = tempfile::tempdir().unwrap(); + let mut wal = SegmentedWal::open(test_config(dir.path())).unwrap(); + wal.append(RecordType::Put as u32, 1, 0, 0, b"a").unwrap(); + wal.sync().unwrap(); + let before = wal.list_segments().unwrap(); + { + let _guard = FailGuard::fail("wal::roll_before_seal", "injected"); + assert!(wal.roll_segment().is_err()); + } + assert_eq!(wal.list_segments().unwrap(), before); + assert_eq!(wal.active_segment_first_lsn(), 1); + + wal.append(RecordType::Put as u32, 1, 0, 0, b"b").unwrap(); + wal.roll_segment().unwrap(); + assert_eq!(wal.list_segments().unwrap().len(), 2); + assert_eq!(wal.active_segment_first_lsn(), 3); + } + + /// A segment file left by an interrupted roll starts inside LSNs the + /// previous segment holds. Opening the WAL refuses it instead of + /// resuming there and reissuing those LSNs. + #[test] + fn open_refuses_a_segment_left_by_an_interrupted_roll() { + let dir = tempfile::tempdir().unwrap(); + { + let mut wal = SegmentedWal::open(test_config(dir.path())).unwrap(); + for i in 0..5u8 { + wal.append(RecordType::Put as u32, 1, 0, 0, &[i]).unwrap(); + } + wal.sync().unwrap(); + } + let orphan = segment_path(dir.path(), 3); + WalWriter::open_with_start_lsn(&orphan, test_config(dir.path()).writer_config, 3).unwrap(); + + match SegmentedWal::open(test_config(dir.path())) { + Err(WalError::SegmentOverlapsPrevious { + first_lsn, + previous_last_lsn, + .. + }) => { + assert_eq!(first_lsn, 3); + assert_eq!(previous_last_lsn, 5); + } + Err(other) => panic!("expected SegmentOverlapsPrevious, got {other}"), + Ok(_) => panic!("opened a WAL whose last segment reissues LSNs"), + } + } } diff --git a/nodedb-wal/src/time_anchors.rs b/nodedb-wal/src/time_anchors.rs new file mode 100644 index 000000000..cb7a474ef --- /dev/null +++ b/nodedb-wal/src/time_anchors.rs @@ -0,0 +1,226 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! The live LSN ↔ commit-time map of one WAL. +//! +//! Two paths feed it, and both go through [`TimeAnchors::record`]: +//! +//! - the writer, after the fsync that made an anchored batch durable; +//! - replay, for each `TimeAnchor` record read back at boot. +//! +//! Anchor times come from the node's HLC. The HLC never runs backwards within +//! a process, and replay folds each persisted anchor into it, so anchors stay +//! strictly increasing across restarts too. + +use std::sync::{Arc, Mutex, MutexGuard}; + +use nodedb_types::hlc::{Hlc, HlcClock}; +use nodedb_types::temporal::{LsnTimeAnchor, LsnTimeError, LsnTimeMap}; + +use crate::error::Result; +use crate::record::{RecordType, TimeAnchorPayload, WalRecord}; + +#[derive(Debug)] +struct State { + map: LsnTimeMap, + /// Last time handed to the writer. Kept apart from the map because a + /// stamped anchor enters the map only after its fsync. + last_stamp_ns: u64, +} + +/// Commit-time anchors of one WAL, plus the clock that stamps them. +#[derive(Debug)] +pub struct TimeAnchors { + clock: Arc, + state: Mutex, +} + +impl TimeAnchors { + pub fn new(clock: Arc) -> Self { + Self::with_map(clock, LsnTimeMap::new()) + } + + /// Anchors held in `map`, whose capacity bounds memory. + pub fn with_map(clock: Arc, map: LsnTimeMap) -> Self { + Self { + clock, + state: Mutex::new(State { + map, + last_stamp_ns: 0, + }), + } + } + + /// The clock anchors are stamped from. + pub fn clock(&self) -> &Arc { + &self.clock + } + + fn lock(&self) -> MutexGuard<'_, State> { + self.state.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Time for the anchor of the batch being committed now. Strictly above + /// every earlier stamp and every recorded anchor. + pub(crate) fn stamp(&self) -> u64 { + let mut state = self.lock(); + let floor = state + .map + .last() + .map_or(0, |a| a.hlc_wall_ns) + .max(state.last_stamp_ns) + .saturating_add(1); + let ns = self.clock.now().wall_ns.max(floor); + state.last_stamp_ns = ns; + ns + } + + /// Record that every record at or below `lsn` committed by `hlc_wall_ns`. + /// Returns `Ok(false)` for an anchor the map already covers. + pub fn record(&self, lsn: u64, hlc_wall_ns: u64) -> std::result::Result { + self.lock().map.push(LsnTimeAnchor::new(lsn, hlc_wall_ns)) + } + + /// Record every `TimeAnchor` in replayed `records` and fold its time into + /// the clock. + /// + /// A replayed anchor that does not advance the map's time is skipped with a + /// warning. Its records fall to the next anchor, which is later: lookups + /// get coarser there but never return an LSN committed after the target. + pub fn absorb_replayed(&self, records: &[WalRecord]) -> Result<()> { + for record in records { + if RecordType::from_raw(record.logical_record_type()) != Some(RecordType::TimeAnchor) { + continue; + } + let payload = TimeAnchorPayload::from_bytes(&record.payload)?; + self.clock.update(Hlc::new(payload.hlc_wall_ns, 0)); + if let Err(error) = self.record(record.header.lsn, payload.hlc_wall_ns) { + tracing::warn!( + lsn = record.header.lsn, + hlc_wall_ns = payload.hlc_wall_ns, + %error, + "replayed WAL time anchor skipped" + ); + } + } + Ok(()) + } + + /// Anchor every record up to `last_recovered_lsn` at the time of this call. + /// + /// A batch whose sync never finished has no anchor on disk, but its + /// records are durable by the time boot reads them back. On an empty log, + /// `last_recovered_lsn` is 0 and the anchor says nothing had committed by + /// the time the log opened. + pub fn cover_recovered(&self, last_recovered_lsn: u64) { + let covered = self + .lock() + .map + .last() + .is_some_and(|a| a.lsn >= last_recovered_lsn); + if covered { + return; + } + let ns = self.stamp(); + if let Err(error) = self.record(last_recovered_lsn, ns) { + tracing::warn!( + lsn = last_recovered_lsn, + %error, + "recovered WAL tail left without a time anchor" + ); + } + } + + /// The highest LSN committed at or before `target_ns`. + pub fn lsn_at_or_before(&self, target_ns: u64) -> std::result::Result { + self.lock().map.lsn_at_or_before(target_ns) + } + + /// The highest LSN committed at or before the end of millisecond `target_ms`. + pub fn lsn_at_or_before_ms(&self, target_ms: i64) -> std::result::Result { + self.lock().map.lsn_at_or_before_ms(target_ms) + } + + /// Commit time of the batch holding `lsn`, or `None` if no anchor covers it. + pub fn commit_ns_of(&self, lsn: u64) -> Option { + self.lock().map.commit_ns_of(lsn) + } + + /// Copy of the retained anchors, oldest first. + pub fn anchors(&self) -> Vec { + self.lock().map.anchors().to_vec() + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::record::WalRecordArgs; + + fn anchor_record(lsn: u64, ns: u64) -> WalRecord { + WalRecord::new(WalRecordArgs { + record_type: RecordType::TimeAnchor as u32, + lsn, + tenant_id: 0, + vshard_id: 0, + database_id: 0, + payload: TimeAnchorPayload::new(ns).to_bytes().to_vec(), + encryption_key: None, + preamble_bytes: None, + }) + .unwrap() + } + + #[test] + fn stamps_strictly_increase_past_recorded_anchors() { + let anchors = TimeAnchors::new(Arc::new(HlcClock::new())); + let far_future = u64::MAX / 2; + anchors.record(1, far_future).unwrap(); + let a = anchors.stamp(); + let b = anchors.stamp(); + assert!(a > far_future); + assert!(b > a); + } + + #[test] + fn replay_folds_anchor_time_into_the_clock() { + let clock = Arc::new(HlcClock::new()); + let anchors = TimeAnchors::new(Arc::clone(&clock)); + let far_future = u64::MAX / 2; + anchors + .absorb_replayed(&[ + anchor_record(3, far_future - 10), + anchor_record(7, far_future), + ]) + .unwrap(); + assert!(clock.peek().wall_ns >= far_future); + assert_eq!(anchors.lsn_at_or_before(far_future - 1).unwrap(), 3); + assert_eq!(anchors.lsn_at_or_before(far_future).unwrap(), 7); + assert!(anchors.stamp() > far_future); + } + + #[test] + fn empty_log_resolves_to_lsn_zero_after_open() { + let anchors = TimeAnchors::new(Arc::new(HlcClock::new())); + anchors.cover_recovered(0); + let opened = anchors.anchors()[0]; + assert_eq!(opened.lsn, 0); + assert_eq!(anchors.lsn_at_or_before(opened.hlc_wall_ns).unwrap(), 0); + assert_eq!( + anchors.lsn_at_or_before(opened.hlc_wall_ns - 1).unwrap(), + 0, + "nothing had committed before the log opened" + ); + } + + #[test] + fn recovered_tail_is_covered_once() { + let anchors = TimeAnchors::new(Arc::new(HlcClock::new())); + anchors.absorb_replayed(&[anchor_record(4, 1_000)]).unwrap(); + anchors.cover_recovered(9); + anchors.cover_recovered(9); + let held = anchors.anchors(); + assert_eq!(held.len(), 2); + assert_eq!(held[1].lsn, 9); + assert!(held[1].hlc_wall_ns > 1_000); + } +} diff --git a/nodedb-wal/src/writer/anchor.rs b/nodedb-wal/src/writer/anchor.rs new file mode 100644 index 000000000..743dc67b8 --- /dev/null +++ b/nodedb-wal/src/writer/anchor.rs @@ -0,0 +1,167 @@ +// SPDX-License-Identifier: Apache-2.0 + +//! One time anchor per group-commit batch. +//! +//! `sync` appends the anchor to the write buffer before the batch's single +//! `pwrite` + `fsync`, so an anchor costs no extra write and no extra fsync. +//! Every data append keeps room for it, so it never forces a flush of its own. + +use crate::crypto::AUTH_TAG_SIZE; +use crate::error::Result; +use crate::record::{ + HEADER_SIZE, NO_EVENT_SOURCE, RecordTarget, RecordType, TIME_ANCHOR_PAYLOAD_SIZE, + TimeAnchorPayload, +}; + +use super::core::WalWriter; + +impl WalWriter { + /// Buffer bytes a data append leaves free for the batch's anchor. + pub(super) fn anchor_reserve(&self) -> usize { + if self.config.time_anchors.is_none() { + return 0; + } + let tag = if self.encryption_ring().is_some() { + AUTH_TAG_SIZE + } else { + 0 + }; + HEADER_SIZE + TIME_ANCHOR_PAYLOAD_SIZE + tag + } + + /// Append the anchor closing this batch, if anchors are on and a record + /// was appended since the last one. Its LSN is the batch's last LSN. + pub(super) fn append_time_anchor(&mut self) -> Result<()> { + if !self.unanchored { + return Ok(()); + } + let Some(anchors) = self.config.time_anchors.clone() else { + return Ok(()); + }; + let hlc_wall_ns = anchors.stamp(); + let target = RecordTarget { + record_type: RecordType::TimeAnchor as u32, + tenant_id: 0, + vshard_id: 0, + database_id: 0, + event_source: NO_EVENT_SOURCE, + commit_hlc: 0, + }; + let payload = TimeAnchorPayload::new(hlc_wall_ns).to_bytes(); + let lsn = self.append_reserving(target, &payload, 0, 0)?; + self.unanchored = false; + self.pending_anchor = Some((lsn, hlc_wall_ns)); + Ok(()) + } + + /// Record the batch's anchor once its fsync succeeded. + pub(super) fn publish_time_anchor(&mut self) { + let Some((lsn, hlc_wall_ns)) = self.pending_anchor.take() else { + return; + }; + let Some(anchors) = &self.config.time_anchors else { + return; + }; + if let Err(error) = anchors.record(lsn, hlc_wall_ns) { + tracing::warn!(lsn, hlc_wall_ns, %error, "WAL time anchor not recorded"); + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_types::HlcClock; + + use crate::record::RecordType; + use crate::time_anchors::TimeAnchors; + use crate::writer::{WalWriter, WalWriterConfig}; + + fn open(path: &std::path::Path, anchors: &Arc) -> WalWriter { + WalWriter::open( + path, + WalWriterConfig { + use_direct_io: false, + time_anchors: Some(Arc::clone(anchors)), + ..Default::default() + }, + ) + .unwrap() + } + + fn put(writer: &mut WalWriter) -> u64 { + writer + .append(RecordType::Put as u32, 1, 0, 0, b"row") + .unwrap() + } + + #[test] + fn each_sync_closes_its_batch_with_one_anchor() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("a.wal"); + let anchors = Arc::new(TimeAnchors::new(Arc::new(HlcClock::new()))); + let mut writer = open(&path, &anchors); + + put(&mut writer); + put(&mut writer); + writer.sync().unwrap(); + // Nothing new: no anchor, no write. + let offset = writer.file_offset(); + writer.sync().unwrap(); + assert_eq!(writer.file_offset(), offset); + put(&mut writer); + writer.sync().unwrap(); + + let held = anchors.anchors(); + assert_eq!(held.iter().map(|a| a.lsn).collect::>(), vec![3, 5]); + assert!(held[0].hlc_wall_ns < held[1].hlc_wall_ns); + + let reader = crate::reader::WalReader::open(&path, None).unwrap(); + let types: Vec<_> = reader + .records() + .map(|r| RecordType::from_raw(r.unwrap().logical_record_type())) + .collect(); + assert_eq!( + types, + vec![ + Some(RecordType::Put), + Some(RecordType::Put), + Some(RecordType::TimeAnchor), + Some(RecordType::Put), + Some(RecordType::TimeAnchor), + ] + ); + } + + /// An anchor never adds a write: a batch that fills the buffer to the + /// last usable byte still takes exactly one `pwrite` at sync. + #[test] + fn anchor_fits_without_an_extra_flush() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("b.wal"); + let anchors = Arc::new(TimeAnchors::new(Arc::new(HlcClock::new()))); + let mut writer = WalWriter::open( + &path, + WalWriterConfig { + use_direct_io: false, + write_buffer_size: 4096, + time_anchors: Some(Arc::clone(&anchors)), + ..Default::default() + }, + ) + .unwrap(); + + let room = writer.buffer.capacity() + - writer.padding_reserve() + - writer.anchor_reserve() + - crate::record::HEADER_SIZE; + writer + .append(RecordType::Put as u32, 1, 0, 0, &vec![7u8; room]) + .unwrap(); + assert_eq!(writer.file_offset(), 0, "the data record fit unflushed"); + writer.sync().unwrap(); + assert_eq!(anchors.anchors().len(), 1); + assert_eq!(anchors.anchors()[0].lsn, 2); + } +} diff --git a/nodedb-wal/src/writer/config.rs b/nodedb-wal/src/writer/config.rs index aeff4360b..818291808 100644 --- a/nodedb-wal/src/writer/config.rs +++ b/nodedb-wal/src/writer/config.rs @@ -1,9 +1,11 @@ // SPDX-License-Identifier: Apache-2.0 use std::path::{Path, PathBuf}; +use std::sync::Arc; use crate::align::DEFAULT_ALIGNMENT; use crate::double_write::{DoubleWriteBuffer, DwbDegradation, DwbMode, DwbProtection}; +use crate::time_anchors::TimeAnchors; /// Default write buffer size: 2 MiB. /// @@ -31,6 +33,11 @@ pub struct WalWriterConfig { /// `Direct` when `use_direct_io` is true, `Buffered` otherwise. /// `Some(DwbMode::Off)` disables the DWB entirely. pub dwb_mode: Option, + + /// Commit-time anchors. When set, every sync that commits new records + /// appends one `TimeAnchor` record to its batch and, once fsynced, records + /// it here. `None` writes no anchors. + pub time_anchors: Option>, } impl Default for WalWriterConfig { @@ -40,6 +47,7 @@ impl Default for WalWriterConfig { alignment: DEFAULT_ALIGNMENT, use_direct_io: true, dwb_mode: None, + time_anchors: None, } } } diff --git a/nodedb-wal/src/writer/core.rs b/nodedb-wal/src/writer/core.rs index a311adef2..18c23c257 100644 --- a/nodedb-wal/src/writer/core.rs +++ b/nodedb-wal/src/writer/core.rs @@ -63,6 +63,13 @@ pub struct WalWriter { /// Records appended to this segment with no double-write copy behind them. pub(super) dwb_unprotected_records: u64, + + /// A record was appended after the last time anchor. + pub(super) unanchored: bool, + + /// `(lsn, hlc_wall_ns)` of an anchor in the batch whose fsync has not + /// succeeded yet. + pub(super) pending_anchor: Option<(u64, u64)>, } impl WalWriter { @@ -104,6 +111,8 @@ impl WalWriter { dwb_path: dwb.path, dwb_protection: dwb.protection, dwb_unprotected_records: 0, + unanchored: false, + pending_anchor: None, }) } @@ -205,6 +214,8 @@ impl WalWriter { dwb_path: dwb.path, dwb_protection: dwb.protection, dwb_unprotected_records: 0, + unanchored: false, + pending_anchor: None, }) } @@ -261,6 +272,7 @@ impl WalWriter { vshard_id, database_id, event_source: crate::record::NO_EVENT_SOURCE, + commit_hlc: 0, }, payload, 0, @@ -275,6 +287,18 @@ impl WalWriter { target: RecordTarget, payload: &[u8], apply_key: u64, + ) -> Result { + self.append_reserving(target, payload, apply_key, self.anchor_reserve()) + } + + /// Append a record and leave `extra_reserve` bytes free in the buffer + /// beyond the padding reserve. + pub(super) fn append_reserving( + &mut self, + target: RecordTarget, + payload: &[u8], + apply_key: u64, + extra_reserve: usize, ) -> Result { let RecordTarget { record_type, @@ -282,6 +306,7 @@ impl WalWriter { vshard_id, database_id, event_source, + commit_hlc, } = target; if self.sealed { return Err(WalError::Sealed); @@ -304,12 +329,13 @@ impl WalWriter { crate::record::RecordStamp { apply_key, event_source, + commit_hlc, }, )?; let header_bytes = record.header.to_bytes(); let total_size = HEADER_SIZE + record.payload.len(); - let reserve = self.padding_reserve(); + let reserve = self.padding_reserve() + extra_reserve; // If the record cannot fit in an empty buffer alongside its padding, // no amount of flushing will make room. @@ -341,6 +367,7 @@ impl WalWriter { self.mirror_into_dwb(lsn, &record); self.next_lsn.store(lsn + 1, Ordering::Relaxed); + self.unanchored = true; Ok(lsn) } @@ -370,6 +397,9 @@ impl WalWriter { std::io::Error::other(format!("failpoint wal::before_dwb_flush: {detail}")) )); + // The anchor rides this batch's own write and fsync. + self.append_time_anchor()?; + // Flush DWB first — records must be durable in DWB before WAL. self.flush_dwb(); self.flush_buffer()?; @@ -380,7 +410,9 @@ impl WalWriter { std::io::Error::other(format!("failpoint wal::before_wal_fsync: {detail}")) )); - fsync_and_track(&self.file, &mut self.durability) + fsync_and_track(&self.file, &mut self.durability)?; + self.publish_time_anchor(); + Ok(()) } /// Seal the WAL — no more writes will be accepted. @@ -404,6 +436,21 @@ impl WalWriter { pub fn file_offset(&self) -> u64 { self.file_offset } + + /// The largest payload one record takes: the write buffer, less the + /// padding and anchor reserves, the header and the encryption tag. + pub fn max_payload(&self) -> usize { + let tag = if self.encryption_ring().is_some() { + crate::crypto::AUTH_TAG_SIZE + } else { + 0 + }; + self.buffer + .capacity() + .saturating_sub(self.padding_reserve() + self.anchor_reserve()) + .saturating_sub(HEADER_SIZE + tag) + .min(crate::record::MAX_WAL_PAYLOAD_SIZE) + } } #[cfg(test)] diff --git a/nodedb-wal/src/writer/mod.rs b/nodedb-wal/src/writer/mod.rs index f26417c0f..ce7799ade 100644 --- a/nodedb-wal/src/writer/mod.rs +++ b/nodedb-wal/src/writer/mod.rs @@ -20,6 +20,7 @@ //! io_uring submission can be added once the bridge crate provides the TPC //! event loop integration. +mod anchor; // Crate-visible so the io_uring writer can reach `resume_offset` directly. // Re-exporting it here instead would be dead code whenever the `io-uring` // feature is off, and cfg-gating the re-export just duplicates that condition. diff --git a/nodedb-wal/tests/wal_suite/cases/durability_fsync.rs b/nodedb-wal/tests/wal_suite/cases/durability_fsync.rs index d323cd862..1887e3173 100644 --- a/nodedb-wal/tests/wal_suite/cases/durability_fsync.rs +++ b/nodedb-wal/tests/wal_suite/cases/durability_fsync.rs @@ -34,6 +34,36 @@ fn append(wal: &mut SegmentedWal, payload: &[u8]) -> WalResult { wal.append(RecordType::Put as u32, 1, 0, 0, payload) } +/// Check that `result` is a roll refused by the armed `wal::fsync_directory` +/// fail point. The fail point stays armed through the roll's cleanup, so the +/// cleanup's own directory fsync fails too: the error is `RollCleanupFailed`, +/// with the directory-fsync error as its roll cause. Returns the path of the +/// segment file the roll created. +fn expect_directory_fsync_roll_error(result: WalResult) -> std::path::PathBuf { + let is_directory_fsync = |error: &WalError| matches!(error, WalError::Io(io) if io.to_string().contains("wal::fsync_directory")); + match result { + Err(WalError::RollCleanupFailed { + path, + roll, + cleanup, + }) => { + assert!( + is_directory_fsync(&roll), + "the roll cause must be the directory fsync error, got {roll:?}" + ); + assert!( + is_directory_fsync(&cleanup), + "the cleanup cause must be the directory fsync error, got {cleanup:?}" + ); + std::path::PathBuf::from(path) + } + Err(other) => panic!("expected the directory fsync error, got {other:?}"), + Ok(lsn) => { + panic!("append at LSN {lsn} was acknowledged into a segment with no durable dirent") + } + } +} + /// A batch is flushed into the page cache and its fsync then fails. The buffer /// is already empty at that point, so a second waiter retrying `sync()` must /// not mistake emptiness for durability. @@ -111,13 +141,12 @@ fn rollover_propagates_a_failed_directory_fsync() { wal.sync().unwrap(); let _g = FailGuard::fail("wal::fsync_directory", "dirent not durable"); - match append(&mut wal, b"second") { - Err(WalError::Io(_)) => {} - Err(other) => panic!("expected the directory fsync error, got {other:?}"), - Ok(lsn) => { - panic!("append at LSN {lsn} was acknowledged into a segment with no durable dirent") - } - } + let orphan = expect_directory_fsync_roll_error(append(&mut wal, b"second")); + assert!( + !orphan.exists(), + "the failed roll removes the segment file it created: {}", + orphan.display() + ); } /// A rollover that fails must not brick the WAL. The old writer stays @@ -132,11 +161,7 @@ fn a_failed_rollover_leaves_the_wal_writable() { { let _g = FailGuard::fail("wal::fsync_directory", "dirent not durable"); - match append(&mut wal, b"rejected") { - Err(WalError::Io(_)) => {} - Err(other) => panic!("expected the directory fsync error, got {other:?}"), - Ok(lsn) => panic!("append at LSN {lsn} succeeded despite a failed roll"), - } + expect_directory_fsync_roll_error(append(&mut wal, b"rejected")); } // The fail point is disarmed; the WAL must still be usable. diff --git a/nodedb/src/bootstrap/background_loops.rs b/nodedb/src/bootstrap/background_loops.rs deleted file mode 100644 index 17d95db11..000000000 --- a/nodedb/src/bootstrap/background_loops.rs +++ /dev/null @@ -1,425 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Background loop and subsystem spawning after SharedState is ready. - -use std::sync::Arc; -use std::time::Duration; - -use tracing::info; - -use crate::ServerConfig; -use crate::control::state::SharedState; -use crate::event::bus::EventConsumerRx; -use crate::event::trigger::TriggerDlq; -use crate::event::watermark::WatermarkStore; -use crate::wal::WalManager; - -/// Event Plane components passed to [`spawn_background_loops`]. -pub struct EventPlaneComponents { - pub wal: Arc, - pub event_consumers: Vec, - pub watermark_store: Arc, - pub trigger_dlq: Arc>, -} - -/// Enumerate mirror databases that need their observer link re-established -/// after a server restart, and log the restart decisions. -/// -/// This is called once during startup, after [`SharedState`] and the catalog -/// are fully open. Databases with `MirrorStatus::Promoted` are excluded: -/// they are normal writable databases and must NOT attempt to reconnect. -/// The actual link objects are created by the cluster layer when it -/// processes each restart decision; this function only reads the catalog -/// and logs. -pub fn log_mirror_restart_decisions(shared: &Arc) { - let catalog = shared.credentials.catalog(); - match crate::control::mirror::enumerate_resumable_mirrors(catalog) { - Ok(decisions) => { - for d in &decisions { - tracing::info!( - database = %d.database_name, - resume_lsn = d.resume_from_lsn, - needs_bootstrap = d.needs_bootstrap, - "mirror restart: observer link will resume" - ); - } - } - Err(e) => { - tracing::warn!(error = %e, "mirror restart: failed to enumerate mirrors; skipping"); - } - } -} - -/// Spawn all persistent background subsystems. -/// -/// Includes: Event Plane consumers, webhook manager wiring, -/// collection GC, L2 cleanup, tenant rate/audit/memory timers, checkpoint manager, -/// usage metering flush, and cold tier task. -/// -/// Returns the [`EventPlane`] handle. The caller MUST hold this for the -/// server's lifetime — dropping it aborts every consumer task and turns -/// the per-core event ring buffers into one-way drains, which silently -/// loses every WriteEvent emitted by the Data Plane until process exit. -#[must_use = "EventPlane must be held for the server's lifetime; dropping it stops all event consumers"] -pub fn spawn_background_loops( - shared: &Arc, - shutdown_bus: crate::control::shutdown::ShutdownBus, - components: EventPlaneComponents, - config: &ServerConfig, - num_cores: usize, - shutdown_rx: tokio::sync::watch::Receiver, -) -> crate::event::EventPlane { - let EventPlaneComponents { - wal, - event_consumers, - watermark_store, - trigger_dlq, - } = components; - // Mirror restart: enumerate databases that need observer links re-established - // and log the decisions. The cluster layer processes these asynchronously - // via the mirror_link_registry once QUIC transport is available. - log_mirror_restart_decisions(shared); - - // Mirror lag monitor (5-second interval). - // Reads `_system.mirror_lag` for every active mirror and updates - // the `nodedb_database_mirror_lag_ms` metric. Also drives status - // transitions (Following → Degraded → Disconnected) and clears the - // metric when a mirror is promoted. - { - let shared_mirror = Arc::clone(shared); - crate::control::shutdown::spawn_loop( - &shared.loop_registry, - &shared.shutdown, - "mirror_lag_monitor", - crate::control::shutdown::ShutdownPhase::DrainingControlPlane, - move |mut shutdown| async move { - let mut tick = tokio::time::interval(Duration::from_secs(5)); - loop { - tokio::select! { - _ = shutdown.wait_cancelled() => break, - _ = tick.tick() => {} - } - if shutdown.is_cancelled() { - break; - } - let catalog = shared_mirror.credentials.catalog(); - let databases = match catalog.list_databases() { - Ok(d) => d, - Err(e) => { - tracing::warn!(error = %e, "mirror_lag_monitor: catalog list error"); - continue; - } - }; - for db in databases { - let origin = match db.mirror_origin.as_ref() { - Some(o) => o, - None => continue, - }; - // Promoted mirrors are normal writable databases — skip. - if matches!(origin.status, nodedb_types::MirrorStatus::Promoted) { - continue; - } - // Read the real receive timestamp from the link registry. - // `None` means no link is registered for this database (the - // cluster layer has not yet (re)established it after restart); - // `update_lag_status` falls back to the catalog's apply time - // in that case so the disconnect timer still advances. - let last_received = - shared_mirror.mirror_link_registry.last_received_ms(db.id); - crate::control::mirror::update_lag_status( - catalog, - db.id, - &db.name, - &origin.status, - last_received, - false, - &shared_mirror.database_metrics, - ); - } - } - }, - ); - info!("mirror lag monitor running"); - } - - // Wire stream delivery managers before Event Plane creation can admit - // CREATE CHANGE STREAM delivery tasks. - shared.webhook_manager.set_state(Arc::clone(shared)); - shared.kafka_manager.set_state(Arc::clone(shared)); - - // Event Plane: one consumer Tokio task per Data Plane core. - // Returned to the caller — must outlive the server, otherwise its - // Drop impl aborts every consumer and the Data Plane producers - // start dropping every WriteEvent they emit. - let event_plane = crate::event::EventPlane::spawn(crate::event::EventPlaneConfig { - consumers_rx: event_consumers, - wal: Arc::clone(&wal), - watermark_store, - shared_state: Arc::clone(shared), - trigger_dlq, - cdc_router: Arc::clone(&shared.cdc_router), - shutdown: Arc::clone(&shared.shutdown), - shutdown_bus: shutdown_bus.clone(), - }); - info!(num_cores, "event plane running"); - - // Collection hard-delete retention GC. - if let Ok(mut w) = shared.retention_settings.write() { - *w = config.retention.clone(); - } - let _collection_gc = crate::event::collection_gc::spawn_collection_gc(Arc::clone(shared)); - info!( - retention_days = config.retention.deactivated_collection_retention_days, - sweep_interval_secs = config.retention.gc_sweep_interval_secs, - "collection-gc sweeper running" - ); - - // L2 cleanup worker. - let _l2_cleanup = crate::event::collection_gc::spawn_l2_cleanup(Arc::clone(shared)); - - // Pending engine-reclaim worker: retries the redb + versioned engine - // purge for any dropped collection whose result-checked purge failed - // on this node, so a per-node hiccup can never leave engine storage - // rows behind a removed catalog row. - let _pending_reclaim = crate::event::collection_gc::spawn_pending_reclaim(Arc::clone(shared)); - - // Tenant rate counter reset (1-second timer). - let shared_rate = Arc::clone(shared); - crate::control::shutdown::spawn_loop( - &shared.loop_registry, - &shared.shutdown, - "tenant_rate_reset", - crate::control::shutdown::ShutdownPhase::DrainingControlPlane, - move |mut shutdown| async move { - let mut tick = tokio::time::interval(Duration::from_secs(1)); - loop { - tokio::select! { - _ = shutdown.wait_cancelled() => break, - _ = tick.tick() => shared_rate.reset_tenant_rate_counters(), - } - } - }, - ); - - // Data Plane core stall monitor (5-second sampling). - // Samples each core's event-loop liveness counter and publishes the set of - // cores that stopped completing iterations. Nothing else observes a core - // that wedges without panicking: the per-core panic watchdog is loop-local - // and counts panics only. - crate::bootstrap::core_stall_monitor::spawn_core_stall_monitor(shared); - info!("data plane core stall monitor running"); - - // Idle session sweep (5-second timer). - // Closes sessions whose idle timeout or OIDC token expiry has elapsed. - crate::control::security::sessions::spawn_idle_sweep_loop(shared); - info!("idle session sweep loop running"); - - // Audit log flush (10-second timer). - let shared_audit = Arc::clone(shared); - crate::control::shutdown::spawn_loop( - &shared.loop_registry, - &shared.shutdown, - "audit_log_flush", - crate::control::shutdown::ShutdownPhase::DrainingControlPlane, - move |mut shutdown| async move { - let mut tick = tokio::time::interval(Duration::from_secs(10)); - loop { - tokio::select! { - _ = shutdown.wait_cancelled() => break, - _ = tick.tick() => shared_audit.flush_audit_log(), - } - } - }, - ); - - // SIEM export delivery. Ships the audit/auth events buffered by - // `audit_record_with_db_strict` to the configured webhook. No-op (and no - // task) when no SIEM destination is configured. - crate::control::security::siem::spawn_export_loop(shared); - - // Tenant memory estimation (30-second timer). - let shared_mem = Arc::clone(shared); - crate::control::shutdown::spawn_loop( - &shared.loop_registry, - &shared.shutdown, - "tenant_memory_estimate", - crate::control::shutdown::ShutdownPhase::DrainingControlPlane, - move |mut shutdown| async move { - let mut tick = tokio::time::interval(Duration::from_secs(30)); - loop { - tokio::select! { - _ = shutdown.wait_cancelled() => break, - _ = tick.tick() => shared_mem.update_tenant_memory_estimates(), - } - } - }, - ); - - // Checkpoint manager. It is handed the Event Plane's watermark store - // because WAL truncation is bounded by the consumers' persisted progress as - // much as by the engines': a consumer recovers only from the WAL above its - // watermark, and nothing detects a gap if that suffix is deleted. - let shared_ckpt = Arc::clone(shared); - let _checkpoint_task = crate::control::checkpoint_task::spawn_checkpoint_task( - shared_ckpt, - Arc::clone(event_plane.watermark_store()), - num_cores, - config.checkpoint.to_manager_config(), - &shutdown_bus, - ); - - // Usage metering flush. - let _metering_flush = crate::control::security::metering::counter::spawn_flush_task( - Arc::clone(&shared.usage_counter), - Arc::clone(&shared.usage_store), - 60, - ); - - // Quota period rollover is lazy — see `QuotaManager::rollover_if_due`. - // Every reader/writer of quota usage rolls the scope's period over on - // access, computed exactly from `period_start` and `period_secs`, so - // there is no background sweep to spawn here and no interval to couple - // a quota's `period_secs` to. - - // Background clone materializer sweep. - // Automatically progresses cloned collections from Shadowed → Materialized - // without requiring explicit DDL. The foreground ALTER DATABASE MATERIALIZE - // and DROP DATABASE FORCE paths bypass this loop and call the blocking - // materializer directly. - { - let shared_sweep = Arc::clone(shared); - let sweep_ms = config.tuning.maintenance.clone_sweep_interval_ms; - let sweep_interval = Duration::from_millis(sweep_ms); - crate::control::shutdown::spawn_loop( - &shared.loop_registry, - &shared.shutdown, - "clone_materializer_sweep", - crate::control::shutdown::ShutdownPhase::DrainingControlPlane, - move |mut shutdown| async move { - let mut tick = tokio::time::interval(sweep_interval); - loop { - tokio::select! { - _ = shutdown.wait_cancelled() => break, - _ = tick.tick() => {} - } - if shutdown.is_cancelled() { - break; - } - let state_for_sweep = Arc::clone(&shared_sweep); - let result = tokio::task::spawn_blocking(move || { - let catalog = state_for_sweep.credentials.catalog(); - let cancel = std::sync::atomic::AtomicBool::new(false); - if let Err(e) = - crate::control::maintenance::clone_materializer::run_scheduled_sweep( - &state_for_sweep, - catalog, - &cancel, - ) - { - tracing::warn!(error = %e, "clone materializer sweep error"); - } - }) - .await; - if let Err(e) = result { - tracing::warn!(error = %e, "clone materializer sweep task panicked"); - } - } - }, - ); - info!( - interval_ms = sweep_ms, - "clone materializer background sweep running" - ); - } - - // CRDT constraint reconcile (1-second timer, one node cluster-wide). - // That node re-derives each collection's constraint set from the - // catalog and replicates it to every data-group replica's CRDT validator, - // so a collection created/altered under any leader converges everywhere. - crate::bootstrap::constraint_reconcile::spawn_constraint_reconcile( - Arc::clone(shared), - config.tuning.maintenance.constraint_reconcile_interval_ms, - ); - info!("constraint reconcile loop running"); - - // Scope grant expiry sweep. Executes each expired grant's ON EXPIRE - // action (hard revoke or downgrade to a lesser scope) through the - // replicated propose path, so the change is durable and cluster-wide. - crate::control::security::scope::expiry::spawn_expiry_task( - Arc::clone(shared), - config.tuning.maintenance.scope_expiry_interval_secs, - ); - - // Cold tier task (if configured). - if let Some(ref cold_settings) = config.cold_storage { - let shared_cold = Arc::clone(shared); - let cold_settings_clone = cold_settings.clone(); - let data_dir_clone = config.server.data_dir.clone(); - let shutdown_rx_cold = shutdown_rx.clone(); - crate::control::cold_tier::spawn_cold_tier_task( - shared_cold, - cold_settings_clone, - data_dir_clone, - shutdown_rx_cold, - ); - info!("cold tier task spawned"); - } - - event_plane -} - -/// Spawn the response poller loop (routes Data Plane responses to waiting sessions). -/// -/// The poller is a `DrainingDataPlane` participant, and it exits on the Data -/// Plane drain latch rather than on the flat shutdown signal. Two things depend -/// on that: the final checkpoint dispatched during `DrainingControlPlane` waits -/// for one response per core, and the Data Plane drain itself is finished only -/// once every in-flight response has reached its session. A poller that stopped -/// at the flat signal would strand both. -/// -/// The registry must never abort it, so it registers as no-abort. The drain -/// latch is its one safe exit boundary, and a cancellation short of that -/// strands the responses the poller exists to route. -pub fn spawn_response_poller( - shared: &Arc, - bus: &crate::control::shutdown::ShutdownBus, -) { - let shared_poller = Arc::clone(shared); - let drain_guard = bus.register_task( - crate::control::shutdown::ShutdownPhase::DrainingDataPlane, - "response_poller", - None, - ); - crate::control::shutdown::spawn_loop_no_abort( - &shared.loop_registry, - &shared.shutdown, - "response_poller", - crate::control::shutdown::ShutdownPhase::DrainingDataPlane, - move |_shutdown| async move { - let mut idle_iters: u32 = 0; - loop { - if shared_poller.data_plane_drain.is_complete() { - // One last pass so a response the drain's own final poll - // raced is still routed before the loop ends. - shared_poller.poll_and_route_responses(); - break; - } - let routed = shared_poller.poll_and_route_responses(); - if routed > 0 { - idle_iters = 0; - tokio::task::yield_now().await; - continue; - } - idle_iters = idle_iters.saturating_add(1); - if idle_iters <= 256 { - tokio::task::yield_now().await; - } else if idle_iters <= 1024 { - tokio::time::sleep(std::time::Duration::from_millis(1)).await; - } else { - tokio::time::sleep(std::time::Duration::from_millis(10)).await; - } - } - drain_guard.report_drained(); - }, - ); -} diff --git a/nodedb/src/bootstrap/background_loops/maintenance.rs b/nodedb/src/bootstrap/background_loops/maintenance.rs new file mode 100644 index 000000000..7a75d6344 --- /dev/null +++ b/nodedb/src/bootstrap/background_loops/maintenance.rs @@ -0,0 +1,167 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Maintenance loops: collection GC, owed-work retries, orphaned drain +//! sweep, and the clone materializer sweep. + +use std::sync::Arc; +use std::time::Duration; + +use tracing::info; + +use crate::ServerConfig; +use crate::control::shutdown::{ShutdownPhase, spawn_loop}; +use crate::control::state::SharedState; + +/// Period of each owed-work retry and of the orphaned drain sweep. +const OWED_WORK_PERIOD: Duration = Duration::from_secs(30); + +/// Collection hard-delete retention GC, the L2 cleanup worker, and the +/// pending engine-reclaim worker. +/// +/// The reclaim worker retries the redb and versioned engine purge for any +/// dropped collection whose result-checked purge failed on this node, so a +/// per-node error never leaves engine storage rows behind a removed +/// catalog row. +pub fn spawn_collection_gc(shared: &Arc, config: &ServerConfig) { + if let Ok(mut w) = shared.retention_settings.write() { + *w = config.retention.clone(); + } + let _collection_gc = crate::event::collection_gc::spawn_collection_gc(Arc::clone(shared)); + info!( + retention_days = config.retention.deactivated_collection_retention_days, + sweep_interval_secs = config.retention.gc_sweep_interval_secs, + "collection-gc sweeper running" + ); + let _l2_cleanup = crate::event::collection_gc::spawn_l2_cleanup(Arc::clone(shared)); + let _pending_reclaim = crate::event::collection_gc::spawn_pending_reclaim(Arc::clone(shared)); +} + +/// Owed history compaction retry: re-drives every COMPACT HISTORY whose +/// per-node fan-out failed. +pub fn spawn_pending_history_compaction(shared: &Arc) { + let shared_compaction = Arc::clone(shared); + spawn_loop( + &shared.loop_registry, + &shared.shutdown, + "pending_history_compaction", + ShutdownPhase::DrainingControlPlane, + move |mut shutdown| async move { + let mut tick = tokio::time::interval(OWED_WORK_PERIOD); + loop { + tokio::select! { + _ = shutdown.wait_cancelled() => break, + _ = tick.tick() => { + if let Err(error) = + crate::control::catalog_entry::post_apply::drain_pending_compactions( + &shared_compaction, + ) + .await + { + tracing::warn!(error = %error, "owed history compactions unreadable"); + } + } + } + } + }, + ); +} + +/// Owed leave cleanup: re-drives the lease release and drain end of every +/// node that left, until its row is gone. +pub fn spawn_pending_leave_cleanup(shared: &Arc) { + let state = Arc::clone(shared); + spawn_loop( + &shared.loop_registry, + &shared.shutdown, + "pending_leave_cleanup", + ShutdownPhase::DrainingControlPlane, + move |mut shutdown| async move { + let mut tick = tokio::time::interval(OWED_WORK_PERIOD); + loop { + tokio::select! { + _ = shutdown.wait_cancelled() => break, + _ = tick.tick() => { + if let Err(error) = + crate::control::lease::leave_cleanup::drain_pending_leave_cleanups( + &state, + ) + .await + { + tracing::warn!(error = %error, "owed leave cleanups unreadable"); + } + } + } + } + }, + ); +} + +/// Orphaned descriptor drain sweep, on the singleton worker only. Ends a +/// drain whose proposer left the topology when the Leave hook's own end +/// did not apply. +pub fn spawn_orphaned_drain_sweep(shared: &Arc) { + let state = Arc::clone(shared); + spawn_loop( + &shared.loop_registry, + &shared.shutdown, + "orphaned_drain_sweep", + ShutdownPhase::DrainingControlPlane, + move |mut shutdown| async move { + let mut tick = tokio::time::interval(OWED_WORK_PERIOD); + loop { + tokio::select! { + _ = shutdown.wait_cancelled() => break, + _ = tick.tick() => { + if state.is_singleton_worker() { + crate::control::lease::gc::end_orphaned_drains(&state).await; + } + } + } + } + }, + ); +} + +/// Background clone materializer sweep. +/// +/// Moves cloned collections from Shadowed to Materialized without explicit +/// DDL. The foreground ALTER DATABASE MATERIALIZE and DROP DATABASE FORCE +/// paths bypass this loop and call the blocking materializer directly. +pub fn spawn_clone_materializer_sweep(shared: &Arc, config: &ServerConfig) { + let shared_sweep = Arc::clone(shared); + let sweep_ms = config.tuning.maintenance.clone_sweep_interval_ms; + let sweep_interval = Duration::from_millis(sweep_ms); + spawn_loop( + &shared.loop_registry, + &shared.shutdown, + "clone_materializer_sweep", + ShutdownPhase::DrainingControlPlane, + move |mut shutdown| async move { + let mut tick = tokio::time::interval(sweep_interval); + loop { + tokio::select! { + _ = shutdown.wait_cancelled() => break, + _ = tick.tick() => {} + } + if shutdown.is_cancelled() { + break; + } + let cancel = std::sync::atomic::AtomicBool::new(false); + if let Err(e) = + crate::control::maintenance::clone_materializer::run_scheduled_sweep( + &shared_sweep, + shared_sweep.credentials.catalog(), + &cancel, + ) + .await + { + tracing::warn!(error = %e, "clone materializer sweep error"); + } + } + }, + ); + info!( + interval_ms = sweep_ms, + "clone materializer background sweep running" + ); +} diff --git a/nodedb/src/bootstrap/background_loops/mirror.rs b/nodedb/src/bootstrap/background_loops/mirror.rs new file mode 100644 index 000000000..b40f8cba9 --- /dev/null +++ b/nodedb/src/bootstrap/background_loops/mirror.rs @@ -0,0 +1,108 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Mirror database restart decisions and the mirror lag monitor. + +use std::sync::Arc; +use std::time::Duration; + +use tracing::info; + +use crate::control::shutdown::{ShutdownPhase, spawn_loop}; +use crate::control::state::SharedState; + +/// Interval between mirror lag samples. +const MIRROR_LAG_INTERVAL: Duration = Duration::from_secs(5); + +/// Enumerate mirror databases that need their observer link re-established +/// after a server restart, and log the restart decisions. +/// +/// This is called once during startup, after [`SharedState`] and the catalog +/// are fully open. Databases with `MirrorStatus::Promoted` are excluded: +/// they are normal writable databases and must NOT attempt to reconnect. +/// The actual link objects are created by the cluster layer when it +/// processes each restart decision; this function only reads the catalog +/// and logs. +pub fn log_mirror_restart_decisions(shared: &Arc) { + let catalog = shared.credentials.catalog(); + match crate::control::mirror::enumerate_resumable_mirrors(catalog) { + Ok(decisions) => { + for d in &decisions { + tracing::info!( + database = %d.database_name, + resume_lsn = d.resume_from_lsn, + needs_bootstrap = d.needs_bootstrap, + "mirror restart: observer link will resume" + ); + } + } + Err(e) => { + tracing::warn!(error = %e, "mirror restart: failed to enumerate mirrors; skipping"); + } + } +} + +/// Mirror lag monitor. +/// +/// Reads `_system.mirror_lag` for every active mirror and updates the +/// `nodedb_database_mirror_lag_ms` metric. Also drives status transitions +/// (Following → Degraded → Disconnected) and clears the metric when a +/// mirror is promoted. +pub fn spawn_mirror_lag_monitor(shared: &Arc) { + let shared_mirror = Arc::clone(shared); + spawn_loop( + &shared.loop_registry, + &shared.shutdown, + "mirror_lag_monitor", + ShutdownPhase::DrainingControlPlane, + move |mut shutdown| async move { + let mut tick = tokio::time::interval(MIRROR_LAG_INTERVAL); + loop { + tokio::select! { + _ = shutdown.wait_cancelled() => break, + _ = tick.tick() => {} + } + if shutdown.is_cancelled() { + break; + } + sample_mirror_lag(&shared_mirror); + } + }, + ); + info!("mirror lag monitor running"); +} + +/// One lag sample across every mirror database that is not promoted. +fn sample_mirror_lag(shared: &SharedState) { + let catalog = shared.credentials.catalog(); + let databases = match catalog.list_databases() { + Ok(d) => d, + Err(e) => { + tracing::warn!(error = %e, "mirror_lag_monitor: catalog list error"); + return; + } + }; + for db in databases { + let Some(origin) = db.mirror_origin.as_ref() else { + continue; + }; + // Promoted mirrors are normal writable databases — skip. + if matches!(origin.status, nodedb_types::MirrorStatus::Promoted) { + continue; + } + // Read the real receive timestamp from the link registry. `None` + // means no link is registered for this database (the cluster layer + // has not yet (re)established it after restart). `update_lag_status` + // falls back to the catalog's apply time in that case, so the + // disconnect timer still advances. + let last_received = shared.mirror_link_registry.last_received_ms(db.id); + crate::control::mirror::update_lag_status( + catalog, + db.id, + &db.name, + &origin.status, + last_received, + false, + &shared.database_metrics, + ); + } +} diff --git a/nodedb/src/bootstrap/background_loops/mod.rs b/nodedb/src/bootstrap/background_loops/mod.rs new file mode 100644 index 000000000..797924ad3 --- /dev/null +++ b/nodedb/src/bootstrap/background_loops/mod.rs @@ -0,0 +1,13 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Background loop and subsystem spawning after SharedState is ready. + +pub mod maintenance; +pub mod mirror; +pub mod response_poller; +pub mod spawn; +pub mod timers; + +pub use mirror::log_mirror_restart_decisions; +pub use response_poller::spawn_response_poller; +pub use spawn::{EventPlaneComponents, spawn_background_loops}; diff --git a/nodedb/src/bootstrap/background_loops/response_poller.rs b/nodedb/src/bootstrap/background_loops/response_poller.rs new file mode 100644 index 000000000..b9e4a7d33 --- /dev/null +++ b/nodedb/src/bootstrap/background_loops/response_poller.rs @@ -0,0 +1,65 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The response poller: routes Data Plane responses to waiting sessions. + +use std::sync::Arc; +use std::time::Duration; + +use crate::control::shutdown::{ShutdownBus, ShutdownPhase, spawn_loop_no_abort}; +use crate::control::state::SharedState; + +/// Idle passes that only yield before the poller starts sleeping. +const YIELD_IDLE_PASSES: u32 = 256; +/// Idle passes that sleep `SHORT_IDLE_SLEEP` before the longer sleep. +const SHORT_SLEEP_IDLE_PASSES: u32 = 1024; +const SHORT_IDLE_SLEEP: Duration = Duration::from_millis(1); +const LONG_IDLE_SLEEP: Duration = Duration::from_millis(10); + +/// Spawn the response poller loop (routes Data Plane responses to waiting sessions). +/// +/// The poller is a `DrainingDataPlane` participant, and it exits on the Data +/// Plane drain latch rather than on the flat shutdown signal. Two things depend +/// on that: the final checkpoint dispatched during `DrainingControlPlane` waits +/// for one response per core, and the Data Plane drain itself is finished only +/// once every in-flight response has reached its session. A poller that stops +/// at the flat signal strands both. +/// +/// The registry must never abort it, so it registers as no-abort. The drain +/// latch is its one safe exit boundary, and a cancellation short of that +/// strands the responses the poller exists to route. +pub fn spawn_response_poller(shared: &Arc, bus: &ShutdownBus) { + let shared_poller = Arc::clone(shared); + let drain_guard = bus.register_task(ShutdownPhase::DrainingDataPlane, "response_poller", None); + spawn_loop_no_abort( + &shared.loop_registry, + &shared.shutdown, + "response_poller", + ShutdownPhase::DrainingDataPlane, + move |_shutdown| async move { + let mut idle_iters: u32 = 0; + loop { + if shared_poller.data_plane_drain.is_complete() { + // One last pass so a response the drain's own final poll + // raced is still routed before the loop ends. + shared_poller.poll_and_route_responses(); + break; + } + let routed = shared_poller.poll_and_route_responses(); + if routed > 0 { + idle_iters = 0; + tokio::task::yield_now().await; + continue; + } + idle_iters = idle_iters.saturating_add(1); + if idle_iters <= YIELD_IDLE_PASSES { + tokio::task::yield_now().await; + } else if idle_iters <= SHORT_SLEEP_IDLE_PASSES { + tokio::time::sleep(SHORT_IDLE_SLEEP).await; + } else { + tokio::time::sleep(LONG_IDLE_SLEEP).await; + } + } + drain_guard.report_drained(); + }, + ); +} diff --git a/nodedb/src/bootstrap/background_loops/spawn.rs b/nodedb/src/bootstrap/background_loops/spawn.rs new file mode 100644 index 000000000..11118701d --- /dev/null +++ b/nodedb/src/bootstrap/background_loops/spawn.rs @@ -0,0 +1,195 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! [`spawn_background_loops`]: start every persistent background subsystem. + +use std::sync::Arc; + +use tracing::info; + +use super::{maintenance, mirror, timers}; +use crate::ServerConfig; +use crate::control::shutdown::ShutdownBus; +use crate::control::state::SharedState; +use crate::event::bus::EventConsumerRx; +use crate::event::trigger::TriggerDlq; +use crate::event::watermark::WatermarkStore; +use crate::wal::WalManager; + +/// Interval of the usage metering flush, in seconds. +const METERING_FLUSH_SECS: u64 = 60; + +/// Event Plane components passed to [`spawn_background_loops`]. +pub struct EventPlaneComponents { + pub wal: Arc, + pub event_consumers: Vec, + pub watermark_store: Arc, + pub trigger_dlq: Arc>, +} + +/// Spawn all persistent background subsystems. +/// +/// Includes: Event Plane consumers, webhook manager wiring, +/// collection GC, L2 cleanup, tenant rate/audit/memory timers, checkpoint manager, +/// usage metering flush, and cold tier task. +/// +/// Returns the [`crate::event::EventPlane`] handle. The caller MUST hold this +/// for the server's lifetime — dropping it aborts every consumer task and +/// turns the per-core event ring buffers into one-way drains, which silently +/// loses every WriteEvent emitted by the Data Plane until process exit. +#[must_use = "EventPlane must be held for the server's lifetime; dropping it stops all event consumers"] +pub fn spawn_background_loops( + shared: &Arc, + shutdown_bus: ShutdownBus, + components: EventPlaneComponents, + config: &ServerConfig, + num_cores: usize, + shutdown_rx: tokio::sync::watch::Receiver, + checkpoint: crate::control::checkpoint_manager::CheckpointManagerConfig, +) -> crate::event::EventPlane { + // Mirror restart: enumerate databases that need observer links + // re-established and log the decisions. The cluster layer processes + // these asynchronously via the mirror_link_registry once QUIC transport + // is available. + mirror::log_mirror_restart_decisions(shared); + mirror::spawn_mirror_lag_monitor(shared); + + let event_plane = spawn_event_plane(shared, &shutdown_bus, components, num_cores); + + maintenance::spawn_collection_gc(shared, config); + maintenance::spawn_pending_history_compaction(shared); + maintenance::spawn_pending_leave_cleanup(shared); + maintenance::spawn_orphaned_drain_sweep(shared); + timers::spawn_tenant_timers(shared); + spawn_security_loops(shared); + + // Checkpoint manager. It is handed the Event Plane's watermark store + // because WAL truncation is bounded by the consumers' persisted progress as + // much as by the engines': a consumer recovers only from the WAL above its + // watermark, and nothing detects a gap if that suffix is deleted. + let _checkpoint_task = crate::control::checkpoint_task::spawn_checkpoint_task( + Arc::clone(shared), + Arc::clone(event_plane.watermark_store()), + num_cores, + checkpoint, + &shutdown_bus, + ); + + // Usage metering flush. + let _metering_flush = crate::control::security::metering::counter::spawn_flush_task( + Arc::clone(&shared.usage_counter), + Arc::clone(&shared.usage_store), + METERING_FLUSH_SECS, + ); + + // Quota period rollover is lazy — see `QuotaManager::rollover_if_due`. + // Every reader/writer of quota usage rolls the scope's period over on + // access, computed exactly from `period_start` and `period_secs`, so + // there is no background sweep to spawn here and no interval to couple + // a quota's `period_secs` to. + + maintenance::spawn_clone_materializer_sweep(shared, config); + + // CRDT constraint reconcile (one node cluster-wide). That node + // re-derives each collection's constraint set from the catalog and + // replicates it to every data-group replica's CRDT validator, so a + // collection created/altered under any leader converges everywhere. + crate::bootstrap::constraint_reconcile::spawn_constraint_reconcile( + Arc::clone(shared), + config.tuning.maintenance.constraint_reconcile_interval_ms, + ); + info!("constraint reconcile loop running"); + + // Scope grant expiry sweep. Executes each expired grant's ON EXPIRE + // action (hard revoke or downgrade to a lesser scope) through the + // replicated propose path, so the change is durable and cluster-wide. + crate::control::security::scope::expiry::spawn_expiry_task( + Arc::clone(shared), + config.tuning.maintenance.scope_expiry_interval_secs, + ); + + spawn_storage_tiers(shared, config, shutdown_rx); + + event_plane +} + +/// Wire the stream delivery managers, then start the Event Plane: one +/// consumer Tokio task per Data Plane core. +/// +/// The managers are wired first so Event Plane creation can admit CREATE +/// CHANGE STREAM delivery tasks. The returned plane must outlive the +/// server. Its Drop impl aborts every consumer, and the Data Plane +/// producers then drop every WriteEvent they emit. +fn spawn_event_plane( + shared: &Arc, + shutdown_bus: &ShutdownBus, + components: EventPlaneComponents, + num_cores: usize, +) -> crate::event::EventPlane { + let EventPlaneComponents { + wal, + event_consumers, + watermark_store, + trigger_dlq, + } = components; + shared.webhook_manager.set_state(shared); + shared.kafka_manager.set_state(shared); + + let event_plane = crate::event::EventPlane::spawn(crate::event::EventPlaneConfig { + consumers_rx: event_consumers, + wal, + watermark_store, + shared_state: Arc::clone(shared), + trigger_dlq, + cdc_router: Arc::clone(&shared.cdc_router), + shutdown: Arc::clone(&shared.shutdown), + shutdown_bus: shutdown_bus.clone(), + }); + info!(num_cores, "event plane running"); + event_plane +} + +/// Data Plane core stall monitor, idle session sweep, and SIEM export. +fn spawn_security_loops(shared: &Arc) { + // Samples each core's event-loop liveness counter and publishes the set + // of cores that stopped completing iterations. Nothing else observes a + // core that wedges without panicking: the per-core panic watchdog is + // loop-local and counts panics only. + crate::bootstrap::core_stall_monitor::spawn_core_stall_monitor(shared); + info!("data plane core stall monitor running"); + + // Closes sessions whose idle timeout or OIDC token expiry has elapsed. + crate::control::security::sessions::spawn_idle_sweep_loop(shared); + info!("idle session sweep loop running"); + + // Ships the audit/auth events buffered by `audit_record_with_db_strict` + // to the configured webhook. No task when no SIEM destination is + // configured. + crate::control::security::siem::spawn_export_loop(shared); +} + +/// Cold tier task (when configured), PITR base snapshots, and periodic +/// cluster restore points. +fn spawn_storage_tiers( + shared: &Arc, + config: &ServerConfig, + shutdown_rx: tokio::sync::watch::Receiver, +) { + if let Some(cold_settings) = &config.cold_storage { + crate::control::cold_tier::spawn_cold_tier_task( + Arc::clone(shared), + cold_settings.clone(), + config.server.data_dir.clone(), + shutdown_rx, + ); + info!("cold tier task spawned"); + } + + // PITR base snapshots, retention, and archived WAL collection. + crate::control::pitr::spawn_base_snapshot_task(shared, &config.pitr); + if config.pitr.enabled && config.cluster.is_some() { + crate::control::pitr::restore_point::spawn_restore_point_task( + shared, + config.pitr.restore_point_interval(), + ); + } +} diff --git a/nodedb/src/bootstrap/background_loops/timers.rs b/nodedb/src/bootstrap/background_loops/timers.rs new file mode 100644 index 000000000..82b0c5a86 --- /dev/null +++ b/nodedb/src/bootstrap/background_loops/timers.rs @@ -0,0 +1,68 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Fixed-interval Control Plane timers: tenant rate reset, audit flush, +//! and tenant memory estimation. + +use std::sync::Arc; +use std::time::Duration; + +use crate::control::shutdown::{ShutdownPhase, spawn_loop}; +use crate::control::state::SharedState; + +/// Period of the tenant rate counter reset. +const TENANT_RATE_RESET_PERIOD: Duration = Duration::from_secs(1); +/// Period of the audit log flush to the catalog. +const AUDIT_FLUSH_PERIOD: Duration = Duration::from_secs(10); +/// Period of the tenant memory estimate. +const TENANT_MEMORY_PERIOD: Duration = Duration::from_secs(30); + +/// Spawn a Control Plane loop that runs `on_tick` once per `period` until +/// shutdown. The loop joins at the Control Plane drain. +pub fn spawn_tick_loop( + shared: &Arc, + name: &'static str, + period: Duration, + on_tick: F, +) where + F: Fn(&SharedState) + Send + 'static, +{ + let state = Arc::clone(shared); + spawn_loop( + &shared.loop_registry, + &shared.shutdown, + name, + ShutdownPhase::DrainingControlPlane, + move |mut shutdown| async move { + let mut tick = tokio::time::interval(period); + loop { + tokio::select! { + _ = shutdown.wait_cancelled() => break, + _ = tick.tick() => on_tick(state.as_ref()), + } + } + }, + ); +} + +/// Spawn the tenant rate reset, audit log flush, and tenant memory +/// estimate timers. +pub fn spawn_tenant_timers(shared: &Arc) { + spawn_tick_loop( + shared, + "tenant_rate_reset", + TENANT_RATE_RESET_PERIOD, + SharedState::reset_tenant_rate_counters, + ); + spawn_tick_loop( + shared, + "audit_log_flush", + AUDIT_FLUSH_PERIOD, + SharedState::flush_audit_log, + ); + spawn_tick_loop( + shared, + "tenant_memory_estimate", + TENANT_MEMORY_PERIOD, + SharedState::update_tenant_memory_estimates, + ); +} diff --git a/nodedb/src/bootstrap/cluster_ready.rs b/nodedb/src/bootstrap/cluster_ready.rs index b4e549d6a..f35bfabf0 100644 --- a/nodedb/src/bootstrap/cluster_ready.rs +++ b/nodedb/src/bootstrap/cluster_ready.rs @@ -25,13 +25,10 @@ pub struct ClusterReadyGates { /// Wait for the metadata raft group to be ready, run catalog sanity checks, /// warm the QUIC peer cache, and fire the remaining startup gates. -/// -/// In single-node mode `raft_ready_rx` is `None` and the raft-ready wait is -/// skipped. Gate fires are always performed regardless of cluster mode. pub async fn await_cluster_ready( shared: &Arc, - raft_ready_rx: Option>, - data_plane_replay_done: Vec>, + mut raft_ready_rx: tokio::sync::watch::Receiver, + data_plane_replay_done: Vec>>, gates: ClusterReadyGates, ) -> anyhow::Result<()> { let ClusterReadyGates { @@ -44,28 +41,23 @@ pub async fn await_cluster_ready( health_loop_gate, gateway_enable_gate, } = gates; - // Boot-time readiness gate: in cluster mode, wait until the - // metadata raft group has applied its first entry on this node - // before opening any client-facing listener. This eliminates the - // restart-window race where the first DDL would observe - // `metadata propose: not leader` because election had not yet - // completed. - if let Some(mut ready_rx) = raft_ready_rx { - wait_for_raft_ready( - shared, - &mut ready_rx, - &raft_gate, - RAFT_READY_STALL_TIMEOUT, - RAFT_READY_POLL_INTERVAL, - ) - .await?; - } - // Metadata raft group has applied its first entry (or we're - // in single-node mode with no raft). + // Boot-time readiness gate: wait until the metadata raft group has + // applied its first entry on this node before opening any + // client-facing listener. This eliminates the restart-window race + // where the first DDL observes `metadata propose: not leader` + // because the election has not yet completed. + wait_for_raft_ready( + shared, + &mut raft_ready_rx, + &raft_gate, + RAFT_READY_STALL_TIMEOUT, + RAFT_READY_POLL_INTERVAL, + ) + .await?; raft_gate.fire(); // Authoritatively rehydrate the Data Plane per-core schema registry - // from the durable catalog, in both single-node and cluster mode. + // from the durable catalog. // This is NOT a raft-replay side effect: it enumerates every active // stored collection and re-registers it directly, awaited and // fail-closed, so no client listener can open against a collection @@ -80,6 +72,19 @@ pub async fn await_cluster_ready( schema_gate.fail(format!("schema registry rehydration failed: {e}")); return Err(anyhow::anyhow!("schema registry rehydration failed: {e}")); } + // The metadata applier skips entries it already applied, so the + // post-apply register never re-runs at boot. + if let Err(e) = + crate::control::server::shared::ddl::neutral::continuous_agg::register_persisted_continuous_aggregates( + shared, + ) + .await + { + schema_gate.fail(format!("continuous aggregate re-registration failed: {e}")); + return Err(anyhow::anyhow!( + "continuous aggregate re-registration failed: {e}" + )); + } schema_gate.fire(); // Catalog sanity check: applied-index gate, redb @@ -108,19 +113,27 @@ pub async fn await_cluster_ready( // from the WAL on its own thread; `/healthz` must not report ready until // that is done, or a just-restarted node would serve queries against // half-rebuilt indexes (e.g. an empty vector search). A dropped sender - // means a core panicked during open/replay — fail closed, exactly as the - // raft-readiness gate does, rather than open the gateway on a broken core. + // means a core panicked during open/replay, and a signalled error means + // its replay stopped short of the WAL — fail closed on both, exactly as + // the raft-readiness gate does, rather than open the gateway on a broken + // core. const REPLAY_READY_TIMEOUT: Duration = Duration::from_secs(300); let replay_wait = async { for rx in data_plane_replay_done { - rx.await.map_err(|_| { - anyhow::anyhow!("data plane core exited before signalling WAL replay completion") - })?; + rx.await + .map_err(|_| { + anyhow::anyhow!( + "data plane core exited before signalling WAL replay completion" + ) + })? + .map_err(|e| anyhow::anyhow!("data plane core WAL replay failed: {e}"))?; } - Ok::<(), anyhow::Error>(()) + Ok::<_, anyhow::Error>(()) }; match tokio::time::timeout(REPLAY_READY_TIMEOUT, replay_wait).await { - Ok(Ok(())) => info!("all data plane cores completed WAL replay"), + Ok(Ok(())) => { + info!("all data plane cores completed WAL replay"); + } Ok(Err(e)) => { data_groups_gate.fail(format!("data plane WAL replay failed: {e}")); return Err(e); @@ -148,17 +161,54 @@ pub async fn await_cluster_ready( return Err(e); } - // A pending name-scoped reclaim must complete before the gateway opens. - // Otherwise a same-name CREATE can install a replacement that the delayed - // retry subsequently erases. Fail readiness and let the operator restart - // after the underlying storage fault is resolved. - if let Err(error) = crate::event::collection_gc::pending_reclaim::drain_once(shared).await { - data_groups_gate.fail(format!("pending collection reclaim failed: {error}")); + // Retry every owed reclaim before the gateway opens. A row that does not + // reclaim keeps the drain hold its retry took, so a same-name CREATE waits + // for the worker's retry and never opens over storage the retry erases. + // Only an unreadable queue fails the boot. + if let Err(error) = crate::event::collection_gc::pending_reclaim::drain_at_boot(shared).await { + data_groups_gate.fail(format!( + "pending collection reclaim queue unreadable: {error}" + )); return Err(anyhow::anyhow!( - "pending collection reclaim failed during startup: {error}" + "pending collection reclaim queue unreadable during startup: {error}" )); } + // A compaction applied before a crash, or whose fan-out failed, is still + // owed. A row that fails again stays queued for the retry worker: stale + // history answers old-version reads its peers refuse, and never corrupts + // current state. + match crate::control::catalog_entry::post_apply::drain_pending_compactions(shared).await { + Ok(0) => {} + Ok(still_owed) => tracing::warn!( + still_owed, + "history compactions still owed after the boot drain; the retry worker re-drives them" + ), + Err(error) => { + data_groups_gate.fail(format!("owed history compactions unreadable: {error}")); + return Err(anyhow::anyhow!( + "owed history compactions unreadable during startup: {error}" + )); + } + } + + // Cleanup owed for nodes that left: their leases and drains block DDL + // until released. A row still owed after this pass is re-driven by the + // retry worker. + match crate::control::lease::leave_cleanup::drain_pending_leave_cleanups(shared).await { + Ok(0) => {} + Ok(still_owed) => tracing::warn!( + still_owed, + "leave cleanups still owed after the boot drain; the retry worker re-drives them" + ), + Err(error) => { + data_groups_gate.fail(format!("owed leave cleanups unreadable: {error}")); + return Err(anyhow::anyhow!( + "owed leave cleanups unreadable during startup: {error}" + )); + } + } + // Grants and hierarchy edges live in collections the data groups just // finished replaying. Load them before the gateway opens, so no statement // plans against an empty permission cache. @@ -170,6 +220,14 @@ pub async fn await_cluster_ready( )); } + // Resume or compensate an interrupted MOVE TENANT before any client + // connects. Its drain ends and its re-capture both propose through the + // metadata group and read the data groups, so both must be ready first. + crate::control::server::shared::ddl::neutral::tenant::move_tenant::recovery::recover_all( + shared, + ) + .await; + data_groups_gate.fire(); transport_gate.fire(); @@ -204,20 +262,23 @@ pub async fn await_cluster_ready( warm_peers_gate.fire(); health_loop_gate.fire(); - // In a cluster, a node plans permission-checked statements only under - // an authorization lease. The renewal loop runs from Raft start, and - // every input of a grant is live by now: the Raft groups, the replayed - // data groups, the permission cache and the Event Plane. The gateway - // opens once the first lease is granted, so the first statements are - // not refused. - if let Some(timing) = shared.authorization_fence.timing() - && let Err(error) = crate::control::security::auth_lease::await_planning_admitted( - shared, - RAFT_READY_STALL_TIMEOUT, - timing.renew_every, - ) - .await - { + // A node plans permission-checked statements only under an + // authorization lease. The renewal loop runs from Raft start, and every + // input of a grant is live by now: the Raft groups, the replayed data + // groups, the permission cache and the Event Plane. The gateway opens + // once the first lease is granted, so the first statements are not + // refused. + let admitted = match crate::control::security::auth_lease::barrier::lease_timing(shared) { + Ok(_) => { + crate::control::security::auth_lease::await_planning_admitted( + shared, + RAFT_READY_STALL_TIMEOUT, + ) + .await + } + Err(error) => Err(error), + }; + if let Err(error) = admitted { gateway_enable_gate.fail(format!("authorization lease not granted: {error}")); return Err(anyhow::anyhow!( "authorization lease not granted during startup: {error}" @@ -236,19 +297,57 @@ const RAFT_READY_STALL_TIMEOUT: Duration = Duration::from_secs(30); /// How often the stall check samples the applied index while waiting. const RAFT_READY_POLL_INTERVAL: Duration = Duration::from_secs(1); +/// Wait until the metadata raft group applies its first entry, and fail +/// `raft_gate` when it does not. See [`raft_ready_progress`]. +async fn wait_for_raft_ready( + shared: &Arc, + ready_rx: &mut tokio::sync::watch::Receiver, + raft_gate: &ReadyGate, + stall_timeout: Duration, + poll_interval: Duration, +) -> anyhow::Result<()> { + match raft_ready_progress(shared, ready_rx, stall_timeout, poll_interval).await { + Ok(()) => { + info!("metadata raft group ready — opening client listeners"); + Ok(()) + } + Err(error) => { + raft_gate.fail(error.to_string()); + Err(error.into()) + } + } +} + +/// Wait until the metadata raft group that `start_raft` started applies its +/// first entry. `ready_rx` is the receiver `start_raft` returned. +/// +/// The same wait boot runs before it opens a listener. An in-process host +/// that registers no startup gate calls this before it serves requests. +pub async fn await_raft_ready( + shared: &Arc, + mut ready_rx: tokio::sync::watch::Receiver, +) -> crate::Result<()> { + raft_ready_progress( + shared, + &mut ready_rx, + RAFT_READY_STALL_TIMEOUT, + RAFT_READY_POLL_INTERVAL, + ) + .await +} + /// Wait until the metadata raft group applies its first entry. /// /// Bounds the wait on lack of PROGRESS, not on total elapsed time. A node /// replaying a large log keeps advancing `applied_index` and must be allowed /// to finish; a group that is genuinely stuck advances nothing and fails -/// after [`RAFT_READY_STALL_TIMEOUT`]. -async fn wait_for_raft_ready( +/// after `stall_timeout`. +async fn raft_ready_progress( shared: &Arc, ready_rx: &mut tokio::sync::watch::Receiver, - raft_gate: &ReadyGate, stall_timeout: Duration, poll_interval: Duration, -) -> anyhow::Result<()> { +) -> crate::Result<()> { let applied_index = || { shared .metadata_cache @@ -261,15 +360,11 @@ async fn wait_for_raft_ready( loop { match tokio::time::timeout(poll_interval, ready_rx.wait_for(|v| *v)).await { - Ok(Ok(_)) => { - info!("metadata raft group ready — opening client listeners"); - return Ok(()); - } + Ok(Ok(_)) => return Ok(()), Ok(Err(_)) => { - raft_gate.fail("raft readiness watch dropped before signalling ready"); - return Err(anyhow::anyhow!( - "raft readiness watch dropped before signalling ready" - )); + return Err(crate::Error::Internal { + detail: "raft readiness watch dropped before signalling ready".into(), + }); } // Not ready yet. Replay that is still advancing is healthy. Err(_) => { @@ -280,12 +375,13 @@ async fn wait_for_raft_ready( continue; } if last_progress.elapsed() >= stall_timeout { - let detail = format!( - "metadata group applied no entry for {stall_timeout:?} \ - (applied_index stuck at {current}) — it failed to apply its first entry" - ); - raft_gate.fail(detail.clone()); - return Err(anyhow::anyhow!(detail)); + return Err(crate::Error::Internal { + detail: format!( + "metadata group applied no entry for {stall_timeout:?} \ + (applied_index stuck at {current}) — it failed to apply its \ + first entry" + ), + }); } } } diff --git a/nodedb/src/bootstrap/constraint_reconcile.rs b/nodedb/src/bootstrap/constraint_reconcile.rs index 5feb92867..7dff168fd 100644 --- a/nodedb/src/bootstrap/constraint_reconcile.rs +++ b/nodedb/src/bootstrap/constraint_reconcile.rs @@ -2,9 +2,8 @@ //! Singleton CRDT constraint reconcile loop. //! -//! Exactly one node runs it — the metadata-group leader in a cluster, and the -//! sole node in a standalone deployment, which has no group to elect from. -//! That node periodically re-derives each collection's +//! Exactly one node runs it — the metadata-group leader. A one-node cluster +//! leads its own metadata group. That node periodically re-derives each collection's //! constraint set from the catalog and replicates it to every data-group //! replica via a `ConstraintChange` entry on the collection's vshard data //! Raft log. Each replica installs the set into its per-core CRDT validator, @@ -92,9 +91,8 @@ pub async fn reconcile_once( delivered: &mut HashMap<(TenantId, String), u64>, ) -> usize { // Only one node reconciles — every replica installing would duplicate - // proposals onto the data log for no gain. In a cluster that node is the - // metadata leader; standalone has no group to elect from, so the sole - // node does it. + // proposals onto the data log for no gain. That node is the metadata + // leader; a one-node cluster leads its own metadata group. if !shared.is_singleton_worker() { return 0; } @@ -111,10 +109,13 @@ pub async fn reconcile_once( return 0; } }; - // No proposer installed yet (still bootstrapping the Raft layer): - // skip this pass and retry next tick. - let Some(proposer) = shared.async_raft_proposer() else { - return 0; + // Boot spawns this loop after `start_raft` installed the proposer. + let proposer = match shared.async_raft_proposer() { + Ok(proposer) => proposer, + Err(error) => { + warn!(%error, "constraint reconcile: no raft proposer"); + return 0; + } }; let proposer = Arc::clone(proposer); diff --git a/nodedb/src/bootstrap/data_group_recovery.rs b/nodedb/src/bootstrap/data_group_recovery.rs index b050d30bd..950a0e87e 100644 --- a/nodedb/src/bootstrap/data_group_recovery.rs +++ b/nodedb/src/bootstrap/data_group_recovery.rs @@ -170,11 +170,10 @@ fn pending_groups(statuses: Vec) -> Vec) -> anyhow::Result<()> { let Some(status_fn) = shared.raft_status_fn.get() else { - return Ok(()); + anyhow::bail!("data group recovery: no Raft status source: start_raft has not run"); }; let status_fn = Arc::clone(status_fn); let deadline = Instant::now() + DATA_GROUP_RECOVERY_TIMEOUT; diff --git a/nodedb/src/bootstrap/data_plane.rs b/nodedb/src/bootstrap/data_plane.rs index f731db951..ed877955f 100644 --- a/nodedb/src/bootstrap/data_plane.rs +++ b/nodedb/src/bootstrap/data_plane.rs @@ -23,27 +23,20 @@ use crate::storage::quarantine::QuarantineRegistry; use crate::types::{DatabaseId, TenantId}; /// Load the persisted ND-array catalog from redb into the shared in-memory handle. +/// +/// The catalog is filled while this function owns it, and only then wrapped +/// in the shared lock. No other thread can reach it before that, so no lock +/// is taken here. pub fn load_array_catalog( config: &ServerConfig, ) -> crate::control::array_catalog::ArrayCatalogHandle { - let array_catalog = ArrayCatalog::handle(); + let mut array_catalog = ArrayCatalog::new(); let catalog_path = config.catalog_path(); match CatalogForRead::open(&catalog_path) { - Ok(Some(catalog)) => match catalog.load_all_arrays() { - Ok(entries) => { - let mut guard = array_catalog - .write() - .expect("array catalog lock poisoned at startup"); - for entry in entries { - if let Err(e) = guard.register(entry) { - tracing::warn!(error = %e, "failed to register array at startup"); - } - } - } - Err(e) => { - tracing::warn!(error = %e, "failed to load _system.arrays at startup"); - } - }, + Ok(Some(catalog)) => crate::control::array_catalog::persist::register_loaded( + &mut array_catalog, + catalog.load_all_arrays(), + ), // No catalog yet: a genuine fresh start, nothing to seed. Ok(None) => {} // A catalog EXISTS and could not be read — locked by another handle, @@ -57,7 +50,7 @@ pub fn load_array_catalog( ); } } - array_catalog + Arc::new(std::sync::RwLock::new(array_catalog)) } /// Load every active collection's `CollectionConfig` from the durable @@ -277,15 +270,16 @@ pub struct SpawnedDataPlaneCores { /// Held only to keep the Data Plane core threads alive; never read. pub handles: Vec>, /// Per-core one-shot that resolves when the core finishes `replay_all_wal` - /// (before entering its event loop). Boot awaits every one before opening - /// the client gateway. - pub replay_done: Vec>, + /// (before entering its event loop), carrying the replay's outcome. Boot + /// awaits every one before opening the client gateway. + pub replay_done: Vec>>, } /// Shared Arc resources passed to each Data Plane core at spawn time. pub struct CoreSharedResources { pub governor: Arc, pub quiesce: Arc, + pub event_interest: Arc, pub hlc: Arc, pub array_catalog: crate::control::array_catalog::ArrayCatalogHandle, pub quarantine_registry: Arc, @@ -322,6 +316,7 @@ pub fn spawn_data_plane_cores( let CoreSharedResources { governor, quiesce, + event_interest, hlc, array_catalog, quarantine_registry, @@ -361,6 +356,7 @@ pub fn spawn_data_plane_cores( compaction_config: compaction_cfg.clone(), system_metrics: Some(Arc::clone(&system_metrics)), event_producer: Some(event_producer), + event_interest: Arc::clone(&event_interest), governor: Arc::clone(&governor), quiesce: Some(Arc::clone(&quiesce)), hlc: Arc::clone(&hlc), diff --git a/nodedb/src/bootstrap/listeners.rs b/nodedb/src/bootstrap/listeners.rs index 1200fc300..512b6e656 100644 --- a/nodedb/src/bootstrap/listeners.rs +++ b/nodedb/src/bootstrap/listeners.rs @@ -54,7 +54,7 @@ pub async fn spawn_protocol_listeners( config: &ServerConfig, infra: ListenerInfra, base_acceptor: Option, - cluster_handle: &Option>, + cluster_handle: &ClusterHandle, ) { let ProtocolListeners { pg_listener, @@ -165,10 +165,12 @@ pub async fn spawn_protocol_listeners( ); // Signal readiness to systemd and cluster lifecycle. - if let Some(handle) = cluster_handle { - let nodes = handle.topology.read().map(|t| t.node_count()).unwrap_or(1); - handle.lifecycle.to_ready(nodes); - } + let nodes = cluster_handle + .topology + .read() + .map(|t| t.node_count()) + .unwrap_or(1); + cluster_handle.lifecycle.to_ready(nodes); nodedb_cluster::readiness::notify_ready(); } diff --git a/nodedb/src/bootstrap/mod.rs b/nodedb/src/bootstrap/mod.rs index 7b0ea34aa..005a9adb9 100644 --- a/nodedb/src/bootstrap/mod.rs +++ b/nodedb/src/bootstrap/mod.rs @@ -20,3 +20,4 @@ pub mod state_wiring; pub mod tls; pub mod tracing_init; pub mod wal_init; +pub mod write_group_settle; diff --git a/nodedb/src/bootstrap/panic_hook.rs b/nodedb/src/bootstrap/panic_hook.rs index 1a1fb787a..62081396a 100644 --- a/nodedb/src/bootstrap/panic_hook.rs +++ b/nodedb/src/bootstrap/panic_hook.rs @@ -1,14 +1,16 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Process-wide panic reporting that never renders application panic payloads. +//! Process-wide panic reporting that never renders application panic +//! payloads in a release build. +use std::panic::PanicHookInfo; use std::sync::OnceLock; static PANIC_HOOK: OnceLock<()> = OnceLock::new(); const PANIC_DIAGNOSTIC: &[u8] = b"nodedb: process panic intercepted\n"; #[cfg(unix)] -fn report_panic() { +fn report_fixed_diagnostic() { // SAFETY: the byte slice is valid for the duration of the call and // `STDERR_FILENO` is the conventional process stderr descriptor. `write` // is allocation-free; failures and partial writes are intentionally @@ -23,7 +25,7 @@ fn report_panic() { } #[cfg(not(unix))] -fn report_panic() { +fn report_fixed_diagnostic() { use std::io::Write as _; // The portable fallback performs no formatting and ignores every I/O @@ -31,15 +33,57 @@ fn report_panic() { let _ = std::io::stderr().write_all(PANIC_DIAGNOSTIC); } -/// Install the payload-redacting process panic hook exactly once. +/// The panic payload as text, when it is a string. +fn payload_text<'a>(info: &'a PanicHookInfo<'_>) -> Option<&'a str> { + let payload = info.payload(); + payload + .downcast_ref::<&str>() + .copied() + .or_else(|| payload.downcast_ref::().map(String::as_str)) +} + +/// Report where the panic happened, after the fixed diagnostic. +/// +/// - The source location names code, never data, so every build reports it. +/// - The payload can carry row values or keys. Only a debug build reports +/// it, so a release server never writes application data to stderr. +/// - The backtrace is captured only when `RUST_BACKTRACE` enables it. +/// +/// Every I/O error is ignored: panic reporting must never panic. +fn report_panic(info: &PanicHookInfo<'_>) { + use std::io::Write as _; + + report_fixed_diagnostic(); + let mut stderr = std::io::stderr().lock(); + if let Some(location) = info.location() { + let _ = writeln!( + stderr, + "nodedb: panic at {}:{}:{}", + location.file(), + location.line(), + location.column() + ); + } + if cfg!(debug_assertions) + && let Some(message) = payload_text(info) + { + let _ = writeln!(stderr, "nodedb: panic message: {message}"); + } + let backtrace = std::backtrace::Backtrace::capture(); + if backtrace.status() == std::backtrace::BacktraceStatus::Captured { + let _ = writeln!(stderr, "nodedb: panic backtrace:\n{backtrace}"); + } +} + +/// Install the process panic hook exactly once. /// -/// The hook intentionally neither chains the default hook nor examines the -/// panic payload or source location. Connection-level boundaries handle -/// expected wire panics; this fixed diagnostic is the last-resort report for -/// every other task. +/// The hook does not chain the default hook. Connection-level boundaries +/// handle expected wire panics. This report is the last resort for every +/// other task: the fixed diagnostic, the source location, the payload in a +/// debug build only, and the backtrace when it is enabled. pub fn install() { PANIC_HOOK.get_or_init(|| { - std::panic::set_hook(Box::new(|_| report_panic())); + std::panic::set_hook(Box::new(report_panic)); }); } diff --git a/nodedb/src/bootstrap/signal.rs b/nodedb/src/bootstrap/signal.rs index 0dfa67cb7..c1d65580c 100644 --- a/nodedb/src/bootstrap/signal.rs +++ b/nodedb/src/bootstrap/signal.rs @@ -53,7 +53,7 @@ pub fn spawn_signal_handlers( conn_semaphore: Arc, max_connections: usize, shutdown_bus: ShutdownBus, - cluster_handle: Option>, + cluster_handle: Arc, ) { let (force_stop_tx, force_stop_rx) = tokio::sync::oneshot::channel::<()>(); let sem_clone = Arc::clone(&conn_semaphore); @@ -113,23 +113,21 @@ pub fn spawn_signal_handlers( // (transitively, via the shared `MultiRaft` handle) before the // process exits. `RunningCluster::shutdown_all` consumes the // value, so it must be taken out of the handle's slot exactly - // once; a `None` here means either single-node mode or that - // shutdown already ran. - if let Some(handle) = &cluster_handle { - let running = handle - .running_cluster - .lock() - .unwrap_or_else(|p| p.into_inner()) - .take(); - if let Some(running) = running { - let errors = running - .shutdown_all(CLUSTER_SUBSYSTEM_SHUTDOWN_DEADLINE) - .await; - if errors.is_empty() { - tracing::info!("cluster subsystems stopped cleanly"); - } else { - tracing::error!(?errors, "cluster subsystem shutdown errors"); - } + // once. A `None` here means shutdown already ran, or the signal + // arrived before `start_raft` installed the running subsystems. + let running = cluster_handle + .running_cluster + .lock() + .unwrap_or_else(|p| p.into_inner()) + .take(); + if let Some(running) = running { + let errors = running + .shutdown_all(CLUSTER_SUBSYSTEM_SHUTDOWN_DEADLINE) + .await; + if errors.is_empty() { + tracing::info!("cluster subsystems stopped cleanly"); + } else { + tracing::error!(?errors, "cluster subsystem shutdown errors"); } } diff --git a/nodedb/src/bootstrap/state_wiring.rs b/nodedb/src/bootstrap/state_wiring.rs index 4c3bb1173..e14d0b31e 100644 --- a/nodedb/src/bootstrap/state_wiring.rs +++ b/nodedb/src/bootstrap/state_wiring.rs @@ -34,7 +34,7 @@ pub async fn wire_state( shared: &mut Arc, config: &ServerConfig, startup_gate: &Arc, - cluster_handle: Option<&ClusterHandle>, + cluster_handle: &ClusterHandle, components: SharedStateComponents, root_span: &tracing::Span, ) -> anyhow::Result<()> { @@ -46,13 +46,15 @@ pub async fn wire_state( // Install startup gate. `/healthz` and the HTTP startup gate read it, so // a state left on the test helpers' pre-fired gate would report ready // and open every route during boot. - Arc::get_mut(shared) - .ok_or_else(|| { - anyhow::anyhow!( - "startup gate: SharedState is already shared before the gate was installed" - ) - })? - .startup = Arc::clone(startup_gate); + // + // The data directory is installed with it, before any step below reads + // `state.data_dir`: the restore-generation seal and the PITR node life + // read their files from it. + let state = Arc::get_mut(shared).ok_or_else(|| { + anyhow::anyhow!("startup gate: SharedState is already shared before the gate was installed") + })?; + state.startup = Arc::clone(startup_gate); + state.data_dir = config.server.data_dir.clone(); // Replay surrogate WAL records. // Note: wal_records are not passed here — caller must handle surrogate replay @@ -63,39 +65,16 @@ pub async fn wire_state( state.quarantine_registry = Arc::clone(&quarantine_registry); } - // Wire cluster handles. - if let Some(handle) = cluster_handle - && let Some(state) = Arc::get_mut(shared) + // Wire cluster handles. Every server runs one: a real cluster, or the + // synthesized one-node cluster when `[cluster]` is absent. { - state.node_id = handle.node_id; - state.cluster_topology = Some(Arc::clone(&handle.topology)); - state.cluster_routing = Some(Arc::clone(&handle.routing)); - state.cluster_transport = Some(Arc::clone(&handle.transport)); - state.metadata_cache = Arc::clone(&handle.metadata_cache); - state.group_watchers = Arc::clone(&handle.group_watchers); - - // Wire the cross-shard event SENDER subsystem (cluster mode only). This - // is what lets a trigger body writing to a remote-homed collection be - // dispatched to the owning node instead of being silently mis-written - // to the local core. The dispatcher's background drain task is spawned - // later by the Event Plane (`spawn_dispatcher_task`), which injects the - // transport from `cluster_transport`; its spawn gate requires the - // dispatcher, metrics, and DLQ to be `Some` — set here. The RECEIVER - // builds its OWN HWM store at Raft group setup (rooted under the node's - // own data_dir), so the SharedState `hwm_store` stays `None`. - let cross_shard_metrics = Arc::new(crate::event::cross_shard::CrossShardMetrics::new()); - state.cross_shard_dispatcher = Some(Arc::new( - crate::event::cross_shard::CrossShardDispatcher::new( - handle.node_id, - Arc::clone(&cross_shard_metrics), - ), - )); - state.cross_shard_dlq = Some(Arc::new(std::sync::Mutex::new( - crate::event::cross_shard::CrossShardDlq::open(&config.server.data_dir)?, - ))); - state.cross_shard_metrics = Some(cross_shard_metrics); - - root_span.record("node_id", handle.node_id); + let state = Arc::get_mut(shared).ok_or_else(|| { + anyhow::anyhow!( + "cluster wiring: SharedState is already shared before the cluster handle was installed" + ) + })?; + wire_cluster_handle(state, cluster_handle, &config.server.data_dir)?; + root_span.record("node_id", cluster_handle.node_id); } // Initialise JWKS registry. @@ -113,7 +92,7 @@ pub async fn wire_state( ); } - // Initialise cold storage (L2 tiering). + // Initialise cold storage (L2 tiering and the WAL archive). if let Some(ref cold_settings) = config.cold_storage { let cold_config = cold_settings.to_cold_storage_config(); match crate::storage::cold::ColdStorage::new(cold_config) { @@ -127,11 +106,22 @@ pub async fn wire_state( ); } } + // With PITR on, truncation without an archive deletes segments + // recovery needs, so a cold store that cannot open stops the boot. + Err(e) if config.pitr.enabled => { + return Err(crate::Error::Config { + detail: format!("pitr.enabled = true but [cold_storage] failed to open: {e}"), + } + .into()); + } Err(e) => { tracing::warn!(error = %e, "cold storage init failed, tiering disabled"); } } } + if config.pitr.enabled && shared.cold_storage.is_none() { + return Err(crate::config::server::missing_cold_storage().into()); + } // Initialise snapshot storage. { @@ -157,6 +147,14 @@ pub async fn wire_state( } } + // A cluster restore's generation, sealed before any Raft group starts. + crate::control::pitr::seal_restored_generation(shared).await?; + + // PITR: this node life, and the base catalog rebuilt from snapshot storage. + if config.pitr.enabled { + crate::control::pitr::wire_pitr(shared, Arc::clone(&cluster_handle.catalog)).await?; + } + // Initialise quarantine storage. { let q_cfg = config @@ -192,6 +190,18 @@ pub async fn wire_state( state.maintenance_budget = Arc::clone(&maintenance_budget); } + // Object-store access for BACKUP / RESTORE DATABASE. + if let Some(settings) = &config.backup_storage + && let Some(state) = Arc::get_mut(shared) + { + state.backup_storage = Some(Arc::new(settings.clone())); + } + if !config.backup.schedule.is_empty() + && let Some(state) = Arc::get_mut(shared) + { + state.backup_schedules = config.backup.schedule.clone(); + } + // Load and wire backup KEK. if let Some(ref benc) = config.backup_encryption { match std::fs::read(&benc.key_path) { @@ -246,20 +256,23 @@ pub async fn wire_state( crate::control::trace_export::TraceExporter::disabled() }; state.debug_endpoints_enabled = config.observability.debug_endpoints_enabled; - state.data_dir = config.server.data_dir.clone(); state.scheduler_config = config.scheduler.clone(); } // The gateway's weak back-reference outlives this call, so every // `Arc::get_mut` install above must run BEFORE it: one placed after // it no-ops. - install_gateway(shared); + install_gateway(shared)?; // Hydrate bitemporal retention registry from array catalog. { - let guard = array_catalog - .read() - .expect("array catalog lock poisoned at startup"); + // The Data Plane cores already share this lock. A core that panicked + // while holding it leaves it poisoned. + let guard = array_catalog.read().map_err(|_| crate::Error::Internal { + detail: "array catalog lock is poisoned; array bitemporal retention \ + cannot be seeded at startup" + .into(), + })?; for entry in guard.all_entries() { if let Some(audit_ms) = entry.audit_retain_ms { if audit_ms < 0 { @@ -270,13 +283,11 @@ pub async fn wire_state( audit_retain_ms: audit_ms as u64, minimum_audit_retain_ms: entry.minimum_audit_retain_ms.unwrap_or(0), }; + // Keyed by the array's own identity, as the `PutArray` + // post-apply registers it. if let Err(e) = shared.bitemporal_retention_registry.register( - // The array catalog is global (not database-scoped), so its - // bitemporal retention is registered under the default - // database. Per-database array isolation is a separate - // initiative. - crate::types::DatabaseId::DEFAULT, - crate::types::TenantId::new(0), + entry.array_id.database_id, + entry.array_id.tenant_id, entry.name.clone(), crate::engine::bitemporal::BitemporalEngineKind::Array, retention, @@ -294,6 +305,46 @@ pub async fn wire_state( Ok(()) } +/// Wire `handle`, the node's cluster handle, into `state` before the state +/// is shared: node id, topology, routing, transport, the metadata cache, the +/// group apply watchers, and the cross-shard event sender. +/// +/// Every host runs this before `start_raft`: boot for a real cluster and for +/// the synthesized one-node cluster, and every in-process test host. +/// +/// The cross-shard event SENDER lets a trigger body writing to a +/// remote-homed collection reach the owning node instead of being +/// mis-written to the local core. The Event Plane spawns the dispatcher's +/// drain task (`spawn_dispatcher_task`), which injects the transport from +/// `cluster_transport`. Its spawn gate requires the dispatcher, metrics, and +/// DLQ this sets. Raft group setup opens the dedup store under `data_dir`. +pub fn wire_cluster_handle( + state: &mut SharedState, + handle: &ClusterHandle, + data_dir: &std::path::Path, +) -> crate::Result<()> { + state.node_id = handle.node_id; + state.cluster_topology = Some(Arc::clone(&handle.topology)); + state.cluster_routing = Some(Arc::clone(&handle.routing)); + state.cluster_transport = Some(Arc::clone(&handle.transport)); + state.metadata_cache = Arc::clone(&handle.metadata_cache); + state.group_watchers = Arc::clone(&handle.group_watchers); + state.migration_tracker = Some(Arc::clone(&handle.migration_tracker)); + + let cross_shard_metrics = Arc::new(crate::event::cross_shard::CrossShardMetrics::new()); + state.cross_shard_dispatcher = Some(Arc::new( + crate::event::cross_shard::CrossShardDispatcher::new( + handle.node_id, + Arc::clone(&cross_shard_metrics), + ), + )); + state.cross_shard_dlq = Some(Arc::new(std::sync::Mutex::new( + crate::event::cross_shard::CrossShardDlq::open(data_dir)?, + ))); + state.cross_shard_metrics = Some(cross_shard_metrics); + Ok(()) +} + /// Construct and install the gateway and the DDL plan-cache invalidator. /// /// `Gateway` holds a `Weak` back-reference to its own @@ -303,11 +354,23 @@ pub async fn wire_state( /// /// Every `Arc::get_mut` install on `shared` must run before this call. /// A later `get_mut` sees the weak reference and no-ops. -pub fn install_gateway(shared: &Arc) { +/// +/// A second call is a wiring bug and returns `Error::Internal`. +pub fn install_gateway(shared: &Arc) -> crate::Result<()> { let gateway = Arc::new(crate::control::gateway::Gateway::new(Arc::clone(shared))); let invalidator = Arc::new(crate::control::gateway::PlanCacheInvalidator::new( &gateway.plan_cache, )); - let _ = shared.gateway.set(gateway); - let _ = shared.gateway_invalidator.set(invalidator); + let already = |what: &str| crate::Error::Internal { + detail: format!("{what} is already installed; install_gateway ran twice"), + }; + shared + .gateway + .set(gateway) + .map_err(|_| already("gateway"))?; + shared + .gateway_invalidator + .set(invalidator) + .map_err(|_| already("gateway plan-cache invalidator"))?; + Ok(()) } diff --git a/nodedb/src/bootstrap/wal_init.rs b/nodedb/src/bootstrap/wal_init.rs index 6a745b2d5..7c1d79dbf 100644 --- a/nodedb/src/bootstrap/wal_init.rs +++ b/nodedb/src/bootstrap/wal_init.rs @@ -59,6 +59,34 @@ pub fn init_wal( std::process::exit(1); } }; + // Replay recorded the persisted time anchors. Records past the last one + // are durable now, so they take the boot time. + wal.anchor_recovered_tail(); + + // Settle every record group a crash left broken before any replay, then + // read the stream the settle completed. + let wal_records = match crate::bootstrap::write_group_settle::settle_write_groups( + &wal, + &config.server.data_dir, + &wal_records, + ) + .and_then(|appended| { + if appended { + wal.replay() + .map(|records| Arc::from(records.into_boxed_slice())) + } else { + Ok(wal_records) + } + }) { + Ok(records) => records, + Err(e) => { + tracing::error!( + error = %e, + "StartupError: settling the write groups a crash left broken failed" + ); + std::process::exit(1); + } + }; tracing::warn!( catalog = %config.catalog_path().display(), diff --git a/nodedb/src/bootstrap/write_group_settle/mod.rs b/nodedb/src/bootstrap/write_group_settle/mod.rs new file mode 100644 index 000000000..e33ab18f1 --- /dev/null +++ b/nodedb/src/bootstrap/write_group_settle/mod.rs @@ -0,0 +1,9 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod scan; +pub mod settle; +pub mod stores; + +pub use scan::{GroupScan, WalGroups}; +pub use settle::settle_write_groups; +pub use stores::StoredWriteSets; diff --git a/nodedb/src/bootstrap/write_group_settle/scan.rs b/nodedb/src/bootstrap/write_group_settle/scan.rs new file mode 100644 index 000000000..e9f619cbd --- /dev/null +++ b/nodedb/src/bootstrap/write_group_settle/scan.rs @@ -0,0 +1,205 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The record groups a recovered WAL shows, and which of them are whole. + +use std::collections::{BTreeMap, HashMap, HashSet}; + +use nodedb_wal::WalRecord; +use nodedb_wal::record::RecordType; + +use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::wal::WriteGroupRecord; + +/// Where a record sits and whose write it is: its tenant, vShard and +/// database, and the apply key, event source and commit HLC its header +/// carries. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct RecordHome { + pub tenant_id: TenantId, + pub vshard_id: VShardId, + pub database_id: DatabaseId, + pub apply_key: u64, + /// The WAL code of the write's event source. + pub event_source: u8, + pub commit_hlc: u64, +} + +/// One record group as the WAL shows it. +#[derive(Debug, Default)] +pub struct GroupScan { + /// The part count the parts announce. + pub parts: Option, + /// LSN and home of every part seen, by part number. + pub seen: BTreeMap, + /// Home of the record that starts the group, when the WAL holds it. + pub start: Option, + /// LSN and home of the group's announcements and continuations. + pub members: Vec<(u64, RecordHome)>, + /// On a committed record whole at append: the continuation count, and + /// whether the last continuation is present. + pub closed: Option<(u32, bool)>, +} + +impl GroupScan { + /// Whether every announced part is present, or the record is whole at + /// append and its last continuation is present. + pub fn is_whole(&self) -> bool { + if let Some((_, last_seen)) = self.closed { + return last_seen; + } + self.parts + .is_some_and(|parts| (1..=parts).all(|part| self.seen.contains_key(&part))) + } +} + +/// Every record group of a recovered WAL, and the LSNs it holds. +#[derive(Debug, Default)] +pub struct WalGroups { + pub groups: BTreeMap, + /// Home of every record replay applies, by LSN. A record a marker + /// cancelled is absent. + pub present: HashMap, + /// LSNs of the committed transaction records. Such a record is never + /// cancelled: replay applies it whether or not its group is whole. + pub committed: HashSet, +} + +impl WalGroups { + /// Scan `records`, the stream replay reads. + pub fn scan(records: &[WalRecord]) -> crate::Result { + let mut scanned = Self::default(); + for record in records { + let home = RecordHome { + tenant_id: TenantId::new(record.header.tenant_id), + vshard_id: VShardId::new(record.header.vshard_id), + database_id: DatabaseId::new(record.header.database_id), + apply_key: record.header.apply_key, + event_source: record.header.event_source, + commit_hlc: record.header.commit_hlc, + }; + let lsn = record.header.lsn; + scanned.present.insert(lsn, home); + match RecordType::from_raw(record.logical_record_type()) { + Some(RecordType::WriteGroup) => {} + Some(RecordType::TransactionRedo) => { + scanned.committed.insert(lsn); + continue; + } + _ => continue, + } + let group = WriteGroupRecord::from_bytes(&record.payload)?.group; + let origin = group.origin_at(lsn); + let scan = scanned.groups.entry(origin).or_default(); + if let Some((index, count)) = group.closed_continuation() { + let last_seen = scan.closed.is_some_and(|(_, seen)| seen); + scan.closed = Some((count, last_seen || index == count)); + } + if group.opens() { + scan.start = Some(home); + if lsn != origin { + scan.members.push((lsn, home)); + } + } else { + scan.parts = Some(scan.parts.map_or(group.parts, |n| n.max(group.parts))); + scan.seen.insert(group.part, (lsn, home)); + } + } + Ok(scanned) + } + + /// The part numbers of the group at `origin` the WAL holds. + pub fn parts_present(&self, origin: u64) -> HashSet { + self.groups + .get(&origin) + .map(|scan| scan.seen.keys().copied().collect()) + .unwrap_or_default() + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::wal::WriteGroup; + use crate::wal::manager::{NO_APPLY_KEY, WalManager}; + + #[test] + fn a_scan_sees_announcements_parts_and_wholeness() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = WalManager::open_for_testing(&dir.path().join("w.wal")).expect("wal"); + // Row-write records name their write's source, as the funnel's do. + let appender = wal + .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User); + let (t, v, d) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); + let forward = appender.append_put(t, v, d, b"f").expect("forward"); + let group = |group: WriteGroup| WriteGroupRecord { + group, + ops: Vec::new(), + redo: None, + }; + appender + .append_write_group(t, v, d, &group(WriteGroup::announcing(forward.as_u64()))) + .expect("announce"); + let opening = appender + .append_write_group(t, v, d, &group(WriteGroup::OPENING)) + .expect("open"); + appender + .append_write_group(t, v, d, &group(WriteGroup::part_of(opening.as_u64(), 1, 1))) + .expect("part"); + wal.sync().expect("sync"); + let scanned = WalGroups::scan(&wal.replay().expect("replay")).expect("scan"); + + let announced = &scanned.groups[&forward.as_u64()]; + assert!(announced.start.is_some() && !announced.is_whole()); + let opened = &scanned.groups[&opening.as_u64()]; + assert!(opened.is_whole()); + assert_eq!(scanned.parts_present(opening.as_u64()), HashSet::from([1])); + assert!(scanned.present.contains_key(&forward.as_u64())); + } + + /// A committed record whole at append and split over the record limit + /// is whole once its last continuation is present, with no part. Before + /// the last one, it is not. + #[test] + fn a_record_whole_at_append_is_whole_without_parts() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = WalManager::open_for_testing(&dir.path().join("w.wal")).expect("wal"); + let appender = wal + .appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User); + let (t, v, d) = (TenantId::new(1), VShardId::new(0), DatabaseId::DEFAULT); + let record = crate::wal::RedoRecord { + version: 1, + ops: Vec::new(), + calvin_stamp: None, + cross_shard_applied: None, + row_sources: Vec::new(), + publishes: Vec::new(), + row_changes: Vec::new(), + }; + let origin = appender + .append_whole_transaction_redo(t, v, d, &record) + .expect("append"); + let continuation = |index: u32| WriteGroupRecord { + group: WriteGroup::continuing_closed(origin.as_u64(), index, 2), + ops: Vec::new(), + redo: None, + }; + appender + .append_write_group(t, v, d, &continuation(1)) + .expect("first continuation"); + wal.sync().expect("sync"); + let cut = WalGroups::scan(&wal.replay().expect("replay")).expect("scan"); + assert!(!cut.groups[&origin.as_u64()].is_whole()); + + appender + .append_write_group(t, v, d, &continuation(2)) + .expect("last continuation"); + wal.sync().expect("sync"); + let scanned = WalGroups::scan(&wal.replay().expect("replay")).expect("scan"); + let group = &scanned.groups[&origin.as_u64()]; + assert!(group.is_whole()); + assert!(group.seen.is_empty(), "no part follows the record"); + assert!(scanned.committed.contains(&origin.as_u64())); + } +} diff --git a/nodedb/src/bootstrap/write_group_settle/settle.rs b/nodedb/src/bootstrap/write_group_settle/settle.rs new file mode 100644 index 000000000..87420f5da --- /dev/null +++ b/nodedb/src/bootstrap/write_group_settle/settle.rs @@ -0,0 +1,284 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Settle every record group a crash left broken, before any replay. +//! +//! A grouped write stores its write set beside its effects (see +//! `data::executor::core_loop::write_set_journal`). Boot compares the stored +//! write sets with the recovered WAL: +//! +//! - A stored write set whose origin the WAL holds gets the parts the WAL +//! lacks, and the marker that cancels its origin when the write set says +//! so and the WAL lacks it. +//! - A stored write set whose origin the crash cut journals the origin again +//! from the stored append inputs, then every part. The effects are +//! durable, so the WAL must name them. +//! - A broken group with no stored write set had no durable effect. Its +//! origin, announcements, continuations and every part the WAL holds are +//! cancelled. A committed +//! transaction record is the exception: it always applies, so it stays, +//! its group closes with one part with no rows, and replay runs its +//! install. +//! - A committed Calvin record is whole at append and takes no part. Its +//! stored write set holds no row, so boot journals nothing for it. +//! +//! Restart replay, a point-in-time restore of the archived WAL, and the event +//! stream rebuilt from it then agree with the stores. Boot makes the appended +//! records durable before it drops the stored write sets. + +use nodedb_wal::WalRecord; + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::wal_dispatch::{ + GroupOrigin, WalAppendRequest, WriteSetTarget, append_group_origin, append_group_parts, + append_planned_parts, plan_group_parts, +}; +use crate::event::EventSource; +use crate::event::cdc::position::{ChangePositionMarker, ReplicatedPosition}; +use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; +use crate::wal::manager::{WalAppender, WalManager}; +use crate::wal::{OriginAppend, WriteSetCapture}; + +use super::scan::WalGroups; +use super::stores::StoredWriteSets; + +/// Settle every broken group of `records`, the stream replay read from +/// `wal`, against the write sets stored under `data_dir`. Returns whether it +/// appended a record, in which case the caller reads the stream again. +pub fn settle_write_groups( + wal: &WalManager, + data_dir: &std::path::Path, + records: &[WalRecord], +) -> crate::Result { + let stored = StoredWriteSets::read(data_dir)?; + let appended = settle_against(wal, records, &stored.captures)?; + if appended { + wal.sync()?; + } + stored.clear()?; + Ok(appended) +} + +/// The settle itself, over write sets already read. +pub(crate) fn settle_against( + wal: &WalManager, + records: &[WalRecord], + captures: &[WriteSetCapture], +) -> crate::Result { + let scanned = WalGroups::scan(records)?; + let wal_end = wal.next_lsn().as_u64(); + let mut appended = false; + let mut captured = std::collections::HashSet::new(); + for capture in captures { + captured.insert(capture.origin); + appended |= if capture.origin >= wal_end { + journal_cut_write(wal, capture)? + } else { + complete_group(wal, &scanned, capture)? + }; + } + for (&origin, scan) in &scanned.groups { + if scan.is_whole() || captured.contains(&origin) { + continue; + } + if scanned.committed.contains(&origin) { + appended |= close_committed_group(wal, &scanned, origin, scan)?; + continue; + } + let mut cancelled: Vec<(u64, super::scan::RecordHome)> = scan + .seen + .values() + .copied() + .chain(scan.members.iter().copied()) + .collect(); + if let Some(home) = scanned.present.get(&origin) { + cancelled.push((origin, *home)); + } + for (lsn, home) in cancelled { + wal.appender(crate::wal::manager::NO_APPLY_KEY) + .append_write_aborted( + home.tenant_id, + home.vshard_id, + home.database_id, + Lsn::new(lsn), + )?; + appended = true; + } + } + Ok(appended) +} + +/// Close the group of the committed transaction record at `origin` with one +/// part with no rows. The record's install never became durable, so replay +/// runs it for the first time. The edge tombstones of its deletes are its +/// own `EdgeDelete` sub-records, which replay writes at their ordinals. +fn close_committed_group( + wal: &WalManager, + scanned: &WalGroups, + origin: u64, + scan: &super::scan::GroupScan, +) -> crate::Result { + let Some(home) = scanned.present.get(&origin) else { + return Ok(false); + }; + // A record whole at append takes no part. Replay applies the + // continuations the WAL holds. + if !scan.seen.is_empty() || scan.closed.is_some() { + return Ok(false); + } + // The part carries the apply key, event source and commit instant of + // the record it closes, as every part of a write does. + let source = EventSource::from_wal_code(home.event_source).ok_or(crate::Error::Internal { + detail: format!( + "committed record at lsn {origin} names unknown event source code {}", + home.event_source + ), + })?; + wal.appender(home.apply_key) + .with_event_source(source) + .with_commit_hlc(home.commit_hlc) + .append_write_group( + home.tenant_id, + home.vshard_id, + home.database_id, + &crate::wal::WriteGroupRecord { + group: crate::wal::WriteGroup::part_of(origin, 1, 1), + ops: Vec::new(), + redo: None, + }, + )?; + Ok(true) +} + +/// Append the parts of `capture`'s group the WAL lacks, and the marker that +/// cancels its origin when the write set cancels it and the WAL still +/// applies the origin. +fn complete_group( + wal: &WalManager, + scanned: &WalGroups, + capture: &WriteSetCapture, +) -> crate::Result { + // A Calvin flush's origin is a committed record whole at append: no part + // follows it. + if is_calvin_flush(&stored_plan(&capture.origin_append)?) { + return Ok(false); + } + let appender = part_appender(wal, &capture.origin_append)?; + let target = target(capture, Lsn::new(capture.origin)); + let write_set = capture.write_set(); + let parts = plan_group_parts(&target, &write_set, wal.max_payload())?; + let cancels_origin = parts.cancels_origin; + let present = scanned.parts_present(capture.origin); + let mut appended = + append_planned_parts(appender, &target, parts, |part| present.contains(&part))?.is_some(); + if cancels_origin && scanned.present.contains_key(&capture.origin) { + appender.append_write_aborted( + target.tenant_id, + target.vshard_id, + target.database_id, + target.origin.lsn, + )?; + appended = true; + } + Ok(appended) +} + +/// Journal a write whose origin the crash cut: its change position marker, +/// its origin, then every part. +fn journal_cut_write(wal: &WalManager, capture: &WriteSetCapture) -> crate::Result { + let inputs = &capture.origin_append; + let appender = part_appender(wal, inputs)?; + let plan = stored_plan(inputs)?; + // A Calvin flush's origin is the transaction's committed record. The + // sequencer holds the transaction, and its recovery runs it again. + if is_calvin_flush(&plan) { + tracing::warn!( + origin = capture.origin, + "boot found a Calvin flush whose committed record the crash cut; the \ + sequencer runs the transaction again" + ); + return Ok(false); + } + let home = target(capture, Lsn::ZERO); + if let Some((epoch, group_id, log_index)) = inputs.change_position { + let marker = ChangePositionMarker { + apply_key: inputs.apply_key, + position: ReplicatedPosition { + epoch, + group_id, + log_index, + }, + }; + // The marker's header names no apply key, as the funnel's does. + let marker_appender = wal.appender(crate::wal::manager::NO_APPLY_KEY); + let marker_appender = match inputs.commit_hlc { + Some(hlc) => marker_appender.with_commit_hlc(hlc), + None => marker_appender, + }; + marker_appender.append_change_position( + home.tenant_id, + home.vshard_id, + home.database_id, + &marker.to_bytes(), + )?; + } + let origin = append_group_origin(WalAppendRequest { + wal: appender, + event_source: event_source(inputs)?, + tenant_id: home.tenant_id, + vshard_id: home.vshard_id, + database_id: home.database_id, + plan: &plan, + credentials: None, + now_override: inputs.resolved_now_ms, + })? + .origin; + append_group_parts(appender, target(capture, origin.lsn), &capture.write_set())?; + Ok(true) +} + +/// The plan a stored write set journals the origin of. +fn stored_plan(inputs: &OriginAppend) -> crate::Result { + zerompk::from_msgpack(&inputs.plan).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("stored write set plan decode: {e}"), + }) +} + +/// Whether `plan` is a Calvin flush, whose origin is whole at append. +fn is_calvin_flush(plan: &PhysicalPlan) -> bool { + matches!( + plan, + PhysicalPlan::Meta(nodedb_physical::physical_plan::MetaOp::CalvinFlush { .. }) + ) +} + +/// The appender every record of the write carried. +fn part_appender<'a>(wal: &'a WalManager, inputs: &OriginAppend) -> crate::Result> { + let appender = wal + .appender(inputs.apply_key) + .with_event_source(event_source(inputs)?); + Ok(match inputs.commit_hlc { + Some(hlc) => appender.with_commit_hlc(hlc), + None => appender, + }) +} + +fn event_source(inputs: &OriginAppend) -> crate::Result { + EventSource::from_wal_code(inputs.event_source).ok_or(crate::Error::Serialization { + format: "msgpack".into(), + detail: format!( + "stored write set names unknown event source code {}", + inputs.event_source + ), + }) +} + +fn target(capture: &WriteSetCapture, origin: Lsn) -> WriteSetTarget<'_> { + WriteSetTarget { + tenant_id: TenantId::new(capture.tenant_id), + vshard_id: VShardId::new(capture.vshard_id), + database_id: DatabaseId::new(capture.database_id), + collection: &capture.collection, + origin: GroupOrigin { lsn: origin }, + } +} diff --git a/nodedb/src/bootstrap/write_group_settle/stores.rs b/nodedb/src/bootstrap/write_group_settle/stores.rs new file mode 100644 index 000000000..7c89c04af --- /dev/null +++ b/nodedb/src/bootstrap/write_group_settle/stores.rs @@ -0,0 +1,69 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The write sets every core's sparse store holds at boot. +//! +//! Boot reads them before any core opens its store, and drops each store's +//! handle before the cores start. + +use std::path::{Path, PathBuf}; + +use crate::engine::sparse::btree::SparseEngine; +use crate::wal::WriteSetCapture; + +/// The stored write sets of every core, and the stores that hold them. +pub struct StoredWriteSets { + stores: Vec, + /// Every stored write set, in origin order. + pub captures: Vec, +} + +impl StoredWriteSets { + /// Read every core's sparse store under `data_dir`. + pub fn read(data_dir: &Path) -> crate::Result { + let mut stores = Vec::new(); + let mut captures = Vec::new(); + for path in sparse_stores(data_dir)? { + let store = SparseEngine::open(&path)?; + for (_, bytes) in store.load_write_set_captures()? { + captures.push(WriteSetCapture::from_bytes(&bytes)?); + } + stores.push(store); + } + captures.sort_by_key(|capture| capture.origin); + Ok(Self { stores, captures }) + } + + /// Drop every stored write set, durably. + pub fn clear(&self) -> crate::Result<()> { + for store in &self.stores { + store.clear_write_set_captures()?; + } + Ok(()) + } +} + +/// Every core's sparse store file under `data_dir`. +fn sparse_stores(data_dir: &Path) -> crate::Result> { + let first = crate::data::executor::snapshot::layout::sparse_store_path(data_dir, 0); + let Some(dir) = first.parent() else { + return Ok(Vec::new()); + }; + let entries = match std::fs::read_dir(dir) { + Ok(entries) => entries, + Err(error) if error.kind() == std::io::ErrorKind::NotFound => return Ok(Vec::new()), + Err(error) => return Err(error.into()), + }; + let mut paths = Vec::new(); + for entry in entries { + let path = entry?.path(); + let is_core_store = path + .file_name() + .and_then(|name| name.to_str()) + .is_some_and(|name| name.starts_with("core-") && name.ends_with(".redb")); + if is_core_store { + paths.push(path); + } + } + paths.sort(); + Ok(paths) +} diff --git a/nodedb/src/bridge/admission_chokepoint.rs b/nodedb/src/bridge/admission_chokepoint.rs index 4927c17b9..e1d775ef3 100644 --- a/nodedb/src/bridge/admission_chokepoint.rs +++ b/nodedb/src/bridge/admission_chokepoint.rs @@ -105,7 +105,7 @@ mod tests { key: b"k".to_vec(), value: b"v".to_vec(), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), returning: None, rls_filters: Vec::new(), provenance: None, @@ -117,7 +117,7 @@ mod tests { PhysicalPlan::Document(DocumentOp::PointGet { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), document_id: "d".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: None, pk_bytes: Vec::new(), rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, @@ -144,6 +144,7 @@ mod tests { txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission, } } @@ -232,7 +233,7 @@ mod tests { dest_collection: QualifiedCollection::new(DatabaseId::DEFAULT, "dst"), item_key: b"item".to_vec(), dest_key: b"item".to_vec(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), source_rls_write_check: source, dest_rls_write_check: dest, }) diff --git a/nodedb/src/bridge/dispatch/core_channel.rs b/nodedb/src/bridge/dispatch/core_channel.rs index e093d978b..4f2175c31 100644 --- a/nodedb/src/bridge/dispatch/core_channel.rs +++ b/nodedb/src/bridge/dispatch/core_channel.rs @@ -14,6 +14,7 @@ use crate::data::eventfd::EventFdNotifier; use crate::types::Lsn; use super::dispatcher::{BridgeRequest, BridgeResponse}; +use super::journal::{JournalGroup, WriteSetJournal}; /// A pair of SPSC channels for one Data Plane core, augmented with a /// weighted-fair staging queue that enforces per-database fairness before @@ -66,6 +67,14 @@ pub struct CoreChannel { /// the core whose set holds its id; no separate request→core map is /// needed. pub outstanding: HashSet, + + /// The record group of each queued request that journals one, by request + /// id. The push that hands the request to the core takes its entry. + pub journals: HashMap, + + /// Origins on this core whose parts are durable. The next push hands + /// them to the core, which drops their stored write sets. + pub settled_write_sets: Vec, } impl CoreChannel { @@ -103,9 +112,14 @@ impl CoreChannel { }; let db_id = req.database_id.as_u64(); let req_id = req.request_id.as_u64(); + let journal = WriteSetJournal { + group: self.journals.remove(&req_id), + settled: std::mem::take(&mut self.settled_write_sets), + }; match self.request_tx.try_push(BridgeRequest { inner: req, outcome_floor, + journal, }) { Ok(()) => { flushed += 1; diff --git a/nodedb/src/bridge/dispatch/dispatcher.rs b/nodedb/src/bridge/dispatch/dispatcher.rs index 35981fd38..d69296ce3 100644 --- a/nodedb/src/bridge/dispatch/dispatcher.rs +++ b/nodedb/src/bridge/dispatch/dispatcher.rs @@ -22,6 +22,7 @@ use crate::types::Lsn; use super::core_channel::{CoreChannel, CoreChannelDataSide}; use super::dispatched_lsns::DispatchedLsns; +use super::journal::WriteSetJournal; use super::outcome_floor::OutcomeFloor; /// Per-core request queue capacity of the server's bridge dispatcher. @@ -42,15 +43,18 @@ pub struct BridgeRequest { /// The outcome floor when this request entered the ring: every record at /// or below it that any core receives has a final outcome. pub outcome_floor: Lsn, + /// The request's record group and the groups settled since the last push. + pub journal: WriteSetJournal, } impl BridgeRequest { - /// A request that carries no outcome floor. A core that reads it learns - /// nothing about the floor. + /// A request that carries no outcome floor and no journal. A core that + /// reads it learns nothing about either. pub fn unfloored(inner: envelope::Request) -> Self { Self { inner, outcome_floor: Lsn::ZERO, + journal: WriteSetJournal::default(), } } } @@ -163,6 +167,8 @@ impl Dispatcher { db_pressure: HashMap::new(), wake_notifier: None, outstanding: HashSet::new(), + journals: HashMap::new(), + settled_write_sets: Vec::new(), }); data_sides.push(CoreChannelDataSide { diff --git a/nodedb/src/bridge/dispatch/enqueue.rs b/nodedb/src/bridge/dispatch/enqueue.rs index ada1ef612..ade60a51f 100644 --- a/nodedb/src/bridge/dispatch/enqueue.rs +++ b/nodedb/src/bridge/dispatch/enqueue.rs @@ -13,9 +13,10 @@ use tracing::warn; use crate::DispatchCapacityScope; use crate::bridge::admission_chokepoint::{assert_write_admitted, reject_uninjected_write}; use crate::bridge::envelope; -use crate::types::Lsn; +use crate::types::{Lsn, VShardId}; use super::dispatcher::Dispatcher; +use super::journal::JournalGroup; use super::refusal::DispatchRefusal; impl Dispatcher { @@ -33,6 +34,17 @@ impl Dispatcher { /// A caller that must retry a capacity refusal re-sends the returned /// request. The dispatcher tracks nothing for a refused request. pub fn try_dispatch(&mut self, request: envelope::Request) -> Result<(), Box> { + self.try_dispatch_journalled(request, None) + } + + /// Dispatch like [`Self::try_dispatch`]. `journal` is the record group + /// the write journals its write set into, which the push hands to the + /// core with the request. + pub fn try_dispatch_journalled( + &mut self, + request: envelope::Request, + journal: Option, + ) -> Result<(), Box> { if let Err(error) = reject_uninjected_write(&request) { return Err(DispatchRefusal::boxed(error, request)); } @@ -97,11 +109,24 @@ impl Dispatcher { request, )); } + if let Some(journal) = journal { + channel.journals.insert(req_id, journal); + } self.commit_enqueued(core_id, database_id, tenant_id, req_id, wal_lsn); Ok(()) } + /// Note that the parts of the group at `origin`, a write of `vshard_id`, + /// are durable. The next push to the write's core hands the note over. + pub fn note_write_set_settled(&mut self, vshard_id: VShardId, origin: Lsn) { + if let Some(core_id) = self.router.resolve(vshard_id) + && let Some(channel) = self.cores.get_mut(core_id) + { + channel.settled_write_sets.push(origin); + } + } + /// Dispatch a request directly to a specific core by index. /// /// Bypasses vShard routing. Used by the checkpoint manager to send diff --git a/nodedb/src/bridge/dispatch/journal.rs b/nodedb/src/bridge/dispatch/journal.rs new file mode 100644 index 000000000..343c7f47d --- /dev/null +++ b/nodedb/src/bridge/dispatch/journal.rs @@ -0,0 +1,41 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! What a core learns about the write-set journal of the requests it runs. +//! +//! A write that journals its write set after apply names its record group +//! here. The core then stores the write set beside the write's effects, +//! together with what boot needs to journal the write's origin again when a +//! crash cut it from the WAL (see `crate::bootstrap::write_group_settle`). +//! The Control Plane reports each group whose parts are durable, and the core +//! drops its stored write set. +//! +//! The Control Plane keeps the journal of a queued request beside the request +//! and hands both over in one ring push. + +use crate::event::cdc::position::ReplicatedPosition; +use crate::types::Lsn; + +/// The record group a dispatched write journals its write set into. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct JournalGroup { + /// LSN of the group's origin record. + pub origin: Lsn, + /// The collection of every write-set entry without its own. + pub collection: String, + /// The idempotency key every record of the write carries. + pub apply_key: u64, + /// The commit instant every record of the write carries, when it was + /// known before the append. + pub commit_hlc: Option, + /// The replicated position whose marker precedes the origin. + pub change_position: Option, +} + +/// The journal part of one ring push. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +pub struct WriteSetJournal { + /// The group of the pushed request, when it journals one. + pub group: Option, + /// Origins whose parts are durable since the last push to this core. + pub settled: Vec, +} diff --git a/nodedb/src/bridge/dispatch/mod.rs b/nodedb/src/bridge/dispatch/mod.rs index 2bd5862d2..dbfa33695 100644 --- a/nodedb/src/bridge/dispatch/mod.rs +++ b/nodedb/src/bridge/dispatch/mod.rs @@ -6,6 +6,7 @@ mod dispatched_lsns; mod dispatcher; mod drain; mod enqueue; +mod journal; mod outcome_floor; mod refusal; mod response_poll; @@ -18,5 +19,6 @@ pub use dispatcher::{ DefaultPriorityResolver, Dispatcher, }; pub use drain::CorePending; +pub use journal::{JournalGroup, WriteSetJournal}; pub use outcome_floor::{OutcomeFloor, ResendRefusal, StuckFloor, WriteWindow}; pub use refusal::DispatchRefusal; diff --git a/nodedb/src/bridge/dispatch/test_requests.rs b/nodedb/src/bridge/dispatch/test_requests.rs index 442aae21a..efb02d80d 100644 --- a/nodedb/src/bridge/dispatch/test_requests.rs +++ b/nodedb/src/bridge/dispatch/test_requests.rs @@ -20,7 +20,7 @@ pub(super) fn make_request(vshard: u32) -> envelope::Request { plan: PhysicalPlan::Document(DocumentOp::PointGet { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "users"), document_id: "u1".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: None, pk_bytes: Vec::new(), rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, @@ -38,6 +38,7 @@ pub(super) fn make_request(vshard: u32) -> envelope::Request { txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: Admission::Exempt(ExemptReason::Read), } } @@ -52,7 +53,7 @@ pub(super) fn make_request_for_db(vshard: u32, db: u64, req_id: u64) -> envelope plan: PhysicalPlan::Document(DocumentOp::PointGet { collection: nodedb_types::QualifiedCollection::new(DatabaseId::new(db), "c"), document_id: "d".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: None, pk_bytes: Vec::new(), rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, @@ -70,6 +71,7 @@ pub(super) fn make_request_for_db(vshard: u32, db: u64, req_id: u64) -> envelope txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: Admission::Exempt(ExemptReason::Read), } } diff --git a/nodedb/src/bridge/envelope/error_code.rs b/nodedb/src/bridge/envelope/error_code.rs index 33f7ff573..24c1167f7 100644 --- a/nodedb/src/bridge/envelope/error_code.rs +++ b/nodedb/src/bridge/envelope/error_code.rs @@ -57,8 +57,6 @@ pub enum ErrorCode { expected: [u8; 32], actual: [u8; 32], }, - /// Fan-out limit exceeded for graph/scatter queries. - FanOutExceeded, /// Memory budget exhausted — DataFusion should spill. ResourcesExhausted, /// Edge creation rejected: source or destination node does not exist. @@ -201,6 +199,16 @@ impl From for ErrorCode { } } +/// A KV write that binds a row to `Surrogate::ZERO` fails prevalidation, the +/// same refusal the dispatch-time check returns. +impl From for ErrorCode { + fn from(e: crate::engine::kv::UnboundKvWrite) -> Self { + Self::RejectedPrevalidation { + reason: e.to_string(), + } + } +} + impl From for ErrorCode { fn from(e: crate::Error) -> Self { match e { @@ -216,11 +224,10 @@ impl From for ErrorCode { | crate::Error::CollectionDeactivated { .. } | crate::Error::DocumentNotFound { .. } => Self::NotFound, crate::Error::RejectedAuthz { resource, .. } => Self::RejectedAuthz { resource }, - // The Control Plane gives all three `40001` (serialization_failure). - crate::Error::ConflictRetry { .. } - | crate::Error::CalvinSerializationConflict - | crate::Error::SourceFrozen { .. } => Self::ConflictRetry, - crate::Error::FanOutExceeded { .. } => Self::FanOutExceeded, + // The Control Plane gives both `40001` (serialization_failure). + crate::Error::ConflictRetry { .. } | crate::Error::CalvinSerializationConflict => { + Self::ConflictRetry + } crate::Error::MemoryExhausted { .. } => Self::ResourcesExhausted, crate::Error::Backpressure { .. } => Self::ResourcesExhausted, crate::Error::AppendOnlyViolation { collection, .. } => { @@ -363,7 +370,9 @@ impl From for ErrorCode { e @ (crate::Error::NoLeader { .. } | crate::Error::GroupQuorumUnavailable { .. } | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::BackupCaptureMoved { .. } | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::LinearizableReadRefused { .. } | crate::Error::StaleReadNotLeader { .. }) => Self::RetryableRefusal { reason: e.to_string(), }, @@ -428,6 +437,8 @@ impl From for ErrorCode { // Data-Plane twin, and none is raised on the Data Plane. e @ (crate::Error::MaterializedSumResolutionMissing { .. } | crate::Error::RetryableLeaderChange { .. } + | crate::Error::CommittedResultUnavailable { .. } + | crate::Error::ProposalOutcomeUnknown { .. } | crate::Error::MetadataLeaderUnavailable | crate::Error::Wal(_) | crate::Error::Dispatch { .. } @@ -442,12 +453,15 @@ impl From for ErrorCode { | crate::Error::Encryption { .. } | crate::Error::Bridge { .. } | crate::Error::VersionCompat { .. } + | crate::Error::RestoreTargetNotEmpty { .. } + | crate::Error::RestoreVerificationFailed { .. } | crate::Error::Internal { .. } | crate::Error::Shaping(_) | crate::Error::RemoteTyped { .. } | crate::Error::Ddl(_) | crate::Error::DescriptorVersionAnomaly { .. } | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CollectionUnstamped { .. } | crate::Error::CatalogIntegrityViolation { .. } | crate::Error::CascadeCycle { .. }) => Self::Internal { detail: e.to_string(), diff --git a/nodedb/src/bridge/envelope/mod.rs b/nodedb/src/bridge/envelope/mod.rs index f09bb0a47..91adf9576 100644 --- a/nodedb/src/bridge/envelope/mod.rs +++ b/nodedb/src/bridge/envelope/mod.rs @@ -14,6 +14,6 @@ pub use nodedb_physical::kv_atomic::CounterFault; pub use nodedb_physical::physical_plan::PhysicalPlan; pub use payload::Payload; pub use request::{Admission, ExemptReason, Request}; -pub use response::{Response, WriteSetEntry}; +pub use response::{EdgeImage, Response, RowEffect, RowVersion, WriteSetEntry}; pub use status::{Priority, Status}; pub use sync_hold::SyncHold; diff --git a/nodedb/src/bridge/envelope/request.rs b/nodedb/src/bridge/envelope/request.rs index cfc21210e..2c2c29e8d 100644 --- a/nodedb/src/bridge/envelope/request.rs +++ b/nodedb/src/bridge/envelope/request.rs @@ -102,6 +102,14 @@ pub struct Request { /// before this field existed. pub resolved_now_ms: Option, + /// HLC wall time, in nanoseconds, at which the write committed, when the + /// Control Plane fixed it before dispatch. The Data Plane stamps it on + /// every event the write emits, so the event's time is the commit's on + /// every path, with or without a WAL record of its own. `None` for reads + /// and control ops, and for a write whose events take the commit instant + /// of the WAL record they reproduce. + pub commit_hlc: Option, + /// Write-admission decision for this request. /// /// Every write-class [`PhysicalPlan`] MUST pass the neutral write-admission @@ -205,7 +213,7 @@ mod tests { plan: PhysicalPlan::Document(DocumentOp::PointGet { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "users"), document_id: "doc-1".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: None, pk_bytes: Vec::new(), rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, @@ -223,6 +231,7 @@ mod tests { txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: Admission::Exempt(ExemptReason::Read), } } @@ -277,6 +286,7 @@ mod tests { txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: Admission::Exempt(ExemptReason::Read), }; match req.plan { diff --git a/nodedb/src/bridge/envelope/response.rs b/nodedb/src/bridge/envelope/response.rs index cafd7fbce..53fff8144 100644 --- a/nodedb/src/bridge/envelope/response.rs +++ b/nodedb/src/bridge/envelope/response.rs @@ -8,14 +8,16 @@ use super::status::Status; use crate::types::{Lsn, RequestId}; use nodedb_types::RowIdentity; -/// One row-level effect of an applied write, carried back from the Data Plane -/// so the Control Plane can mint a durable redo record *after* apply. +/// One row-level effect of an applied document write, carried back from the +/// Data Plane so the Control Plane journals it *after* apply. /// -/// Populated only by write handlers whose autocommit path mints no WAL redo of -/// its own but whose effect must still survive a WAL-only restart — today, a -/// `PointUpdate` on a document collection carrying a secondary vector (HNSW) -/// index (see `data::executor::handlers::point::update`). `value` is the -/// post-image body for a put; empty and ignored when `is_delete`. +/// Every document and edge write handler reports every row it stores or +/// removes that its pre-dispatch WAL record does not carry exactly: the +/// post-image of an update, a bulk or batch write, a derived +/// materialized-sum target row, the stamped image of a versioned row, and an +/// edge version or tombstone at its decided ordinal. The Control +/// Plane journals the entries, in entry order, as the parts of the write's +/// record group, and WAL replay applies them in LSN order. #[derive(Debug, Clone)] pub struct WriteSetEntry { /// The row's stable global surrogate. @@ -25,11 +27,8 @@ pub struct WriteSetEntry { /// The redo record journals this text as its `document_id`, so a WAL /// replay names the same row a live event names. pub identity: RowIdentity, - /// `true` for a delete effect (no body), `false` for a put (post-image in - /// `value`). - pub is_delete: bool, - /// Post-image body for a put; empty for a delete. - pub value: Vec, + /// What the write did to the row. + pub effect: RowEffect, /// Collection this entry's row belongs to. /// /// `None` means the statement's own collection, which is every entry a @@ -40,6 +39,136 @@ pub struct WriteSetEntry { pub collection: Option, } +/// The effect one [`WriteSetEntry`] reports. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum RowEffect { + /// The row holds `value` now. `value` is the MessagePack body the write + /// path encodes into storage, never a strict Binary Tuple: replay hands + /// it to that same encoder. + Put { + value: Vec, + /// The version key of a row on a `bitemporal=true` collection. + version: Option, + }, + /// The row is gone. + Delete { + /// The system time of the tombstone of a row on a `bitemporal=true` + /// collection. + system_from_ms: Option, + }, + /// Replay must not apply the write's own pre-dispatch record: the apply + /// wrote nothing, or another entry of this write set carries the row's + /// exact image. The Control Plane cancels that record with a + /// `WriteAborted` marker once every image of the write set is appended. + CancelForward, + /// A graph edge version the apply wrote, at the ordinal it decided. The + /// entry's collection is `None`: an edge record homes to the write's own + /// vShard. + Edge(EdgeImage), +} + +/// The graph edge versions an apply wrote, in the payload shape their WAL +/// sub-record carries. Every ordinal is the one the apply decided. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum EdgeImage { + /// One edge version. `system_from` is always `Some`. + Put(crate::wal::EdgePutRedo), + /// One edge tombstone. `system_from` is always `Some`. + Delete(crate::wal::EdgeDeleteRedo), +} + +/// The version key a `bitemporal=true` row landed at. Replay installs the row +/// at exactly this key. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct RowVersion { + pub sys_from_ms: i64, + pub valid_from_ms: i64, + pub valid_until_ms: i64, +} + +impl RowVersion { + /// The key of a version valid for all time, written at `sys_from_ms`. + pub fn open(sys_from_ms: i64) -> Self { + Self { + sys_from_ms, + valid_from_ms: i64::MIN, + valid_until_ms: i64::MAX, + } + } +} + +impl WriteSetEntry { + /// The row identified by `surrogate` now holds `value`. + pub fn put(surrogate: u32, identity: RowIdentity, value: Vec) -> Self { + Self { + surrogate, + identity, + effect: RowEffect::Put { + value, + version: None, + }, + collection: None, + } + } + + /// The row identified by `surrogate` is gone. + pub fn delete(surrogate: u32, identity: RowIdentity) -> Self { + Self { + surrogate, + identity, + effect: RowEffect::Delete { + system_from_ms: None, + }, + collection: None, + } + } + + /// The write's pre-dispatch record for the row identified by `surrogate` + /// must not replay. + pub fn cancel_forward(surrogate: u32, identity: RowIdentity) -> Self { + Self { + surrogate, + identity, + effect: RowEffect::CancelForward, + collection: None, + } + } + + /// The edge version an apply wrote. The entry names the edge by its + /// source endpoint. + pub fn edge(image: EdgeImage) -> Self { + let (src_surrogate, src_id) = match &image { + EdgeImage::Put(put) => (put.src_surrogate, put.src_id.as_str()), + EdgeImage::Delete(delete) => (delete.src_surrogate, delete.src_id.as_str()), + }; + Self { + surrogate: src_surrogate, + identity: RowIdentity::from_user_key(src_id), + effect: RowEffect::Edge(image), + collection: None, + } + } + + /// This entry, at the version a `bitemporal=true` collection wrote it at. + /// `None` leaves the entry unversioned. + pub fn versioned(mut self, version: Option) -> Self { + match &mut self.effect { + RowEffect::Put { version: slot, .. } => *slot = version, + RowEffect::Delete { system_from_ms } => { + *system_from_ms = version.map(|v| v.sys_from_ms); + } + RowEffect::CancelForward | RowEffect::Edge(_) => {} + } + self + } + + /// This entry, naming a row of `collection` rather than the statement's. + pub fn in_collection(mut self, collection: String) -> Self { + self.collection = Some(collection); + self + } +} + /// Response envelope: Data Plane -> Control Plane. /// /// Every field is mandatory. @@ -90,10 +219,9 @@ pub struct Response { /// buffer to base on `Some(true)` and drops it on `Some(false)`. pub read_set_valid: Option, - /// Row-level effects the Control Plane must turn into durable redo records - /// *after* the Data Plane applied them. Empty for every response that owns - /// its durability on the pre-dispatch WAL path (the common case); non-empty - /// only for post-apply-redo writes (see [`WriteSetEntry`]). + /// Row-level effects the Control Plane turns into durable redo records + /// *after* the Data Plane applied them (see [`WriteSetEntry`]). Empty for + /// a response whose pre-dispatch WAL record carries its whole effect. pub write_set: Vec, } diff --git a/nodedb/src/bridge/quiesce/drain.rs b/nodedb/src/bridge/quiesce/drain.rs index 24fbe33a6..e6c7d065f 100644 --- a/nodedb/src/bridge/quiesce/drain.rs +++ b/nodedb/src/bridge/quiesce/drain.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Drain coordination: `begin_drain`, `wait_until_drained`, `clear_drain`. +//! Drain coordination: `wait_until_drained`. use std::future::Future; use std::pin::Pin; @@ -9,150 +9,7 @@ use std::task::{Context, Poll}; use super::refcount::CollectionQuiesce; -/// Exclusive per-name lifecycle hold used by no-Raft DDL. -/// -/// Dropping the guard releases one drain hold and wakes CREATE waiters. Call -/// [`disarm`](LifecycleDrainGuard::disarm) after ownership is transferred to a -/// durable pending-reclaim record. -pub struct LifecycleDrainGuard { - registry: Arc, - database_id: u64, - tenant_id: u64, - collection: String, - active: bool, -} - -impl LifecycleDrainGuard { - pub fn disarm(mut self) { - self.active = false; - } -} - -impl Drop for LifecycleDrainGuard { - fn drop(&mut self) { - if self.active { - self.registry - .clear_drain(self.database_id, self.tenant_id, &self.collection); - } - } -} - impl CollectionQuiesce { - /// Acquire one lifecycle drain hold for `(database, tenant, collection)`. - /// New scans and same-name CREATE operations remain blocked until every - /// matching hold is released by `clear_drain` or `forget`. - pub fn begin_drain(&self, database_id: u64, tenant_id: u64, collection: &str) { - let mut inner = self.inner_mut(); - let entry = inner - .states - .entry((database_id, tenant_id, collection.to_string())) - .or_default(); - entry.drain_holders = entry.drain_holders.saturating_add(1); - } - - /// Stop the drain marker, allowing new scans again. Only called when - /// the purge is aborted or a recreate happens — on a normal purge - /// the collection metadata is gone so new scans naturally return - /// `collection_not_found` from that point on, and the drain entry - /// is garbage-collected via `forget`. - pub fn clear_drain(&self, database_id: u64, tenant_id: u64, collection: &str) { - let mut inner = self.inner_mut(); - let key = (database_id, tenant_id, collection.to_string()); - let remove = if let Some(state) = inner.states.get_mut(&key) { - state.drain_holders = state.drain_holders.saturating_sub(1); - state.drain_holders == 0 && state.open_scans == 0 - } else { - false - }; - if remove { - inner.states.remove(&key); - } - drop(inner); - self.notify.notify_waiters(); - } - - /// Drop the entry entirely once reclaim has completed. After this, - /// `is_draining` returns false and `open_scans` is 0. Called by - /// the purge handler right before emitting the reclaim ack. - pub fn forget(&self, database_id: u64, tenant_id: u64, collection: &str) { - let mut inner = self.inner_mut(); - let key = (database_id, tenant_id, collection.to_string()); - let remove = if let Some(state) = inner.states.get_mut(&key) { - state.drain_holders = state.drain_holders.saturating_sub(1); - state.drain_holders == 0 - } else { - false - }; - if remove { - inner.states.remove(&key); - } - drop(inner); - self.notify.notify_waiters(); - } - - /// Try to acquire the exclusive lifecycle drain without waiting. - /// Synchronous DDL uses this form because it may run on a current-thread - /// Tokio runtime where blocking on an async waiter would panic. - pub fn try_acquire_lifecycle( - self: &Arc, - database_id: u64, - tenant_id: u64, - collection: &str, - ) -> Option { - let mut inner = self.inner_mut(); - let entry = inner - .states - .entry((database_id, tenant_id, collection.to_string())) - .or_default(); - if entry.drain_holders > 0 { - return None; - } - entry.drain_holders = 1; - Some(LifecycleDrainGuard { - registry: Arc::clone(self), - database_id, - tenant_id, - collection: collection.to_string(), - active: true, - }) - } - - /// Exclusively acquire the lifecycle drain for one collection name. - /// - /// The check-and-acquire is performed under the registry mutex, so two - /// concurrent local DDL operations cannot both enter the destructive - /// lifecycle section. - pub async fn acquire_lifecycle( - self: &Arc, - database_id: u64, - tenant_id: u64, - collection: &str, - ) -> LifecycleDrainGuard { - loop { - let notified = self.notify.notified(); - tokio::pin!(notified); - notified.as_mut().enable(); - { - let mut inner = self.inner_mut(); - let entry = inner - .states - .entry((database_id, tenant_id, collection.to_string())) - .or_default(); - if entry.drain_holders == 0 { - entry.drain_holders = 1; - return LifecycleDrainGuard { - registry: Arc::clone(self), - database_id, - tenant_id, - collection: collection.to_string(), - active: true, - }; - } - } - notified.await; - } - } - /// Returns a future that resolves once every open scan against /// `(tenant_id, collection)` has completed. Safe to await from the /// Control Plane (tokio) — internally uses [`tokio::sync::Notify`] @@ -175,10 +32,6 @@ impl CollectionQuiesce { notified: None, } } - - fn inner_mut(&self) -> std::sync::MutexGuard<'_, super::refcount::Inner> { - self.inner.lock().expect("CollectionQuiesce mutex poisoned") - } } /// Future returned by [`CollectionQuiesce::wait_until_drained`]. @@ -215,16 +68,16 @@ impl Future for WaitDrain { // Arm a notification, then re-check. If a release fires // between the check and the arm we handled it on the next // iteration (open_scans would be 0 then). - if self.notified.is_none() { - let notify: &tokio::sync::Notify = &self.registry.notify; - // SAFETY: we hold an Arc; the Notify - // inside it outlives `self`. + let registry = Arc::clone(&self.registry); + let fut = self.notified.get_or_insert_with(|| { + let notify: &tokio::sync::Notify = ®istry.notify; + // SAFETY: `self` holds an Arc for its + // whole life; the Notify inside it outlives `self`. let notified: tokio::sync::futures::Notified<'_> = notify.notified(); let notified: tokio::sync::futures::Notified<'static> = unsafe { std::mem::transmute(notified) }; - self.notified = Some(Box::pin(notified)); - } - let fut = self.notified.as_mut().expect("just set"); + Box::pin(notified) + }); match fut.as_mut().poll(cx) { Poll::Ready(()) => { self.notified = None; @@ -258,7 +111,7 @@ mod tests { #[tokio::test] async fn drain_resolves_immediately_when_no_open_scans() { let q = CollectionQuiesce::new(); - q.begin_drain(DB, 1, "c"); + let _hold = q.begin_drain(DB, 1, "c"); q.wait_until_drained(DB, 1, "c").await; } @@ -267,7 +120,7 @@ mod tests { let q = CollectionQuiesce::new(); let g1 = q.try_start_scan(DB, 1, "c").unwrap(); let g2 = q.try_start_scan(DB, 1, "c").unwrap(); - q.begin_drain(DB, 1, "c"); + let _hold = q.begin_drain(DB, 1, "c"); let q_clone = Arc::clone(&q); let drain_task = tokio::spawn(async move { @@ -293,52 +146,27 @@ mod tests { } #[tokio::test] - async fn single_lifecycle_holder_clears_on_forget() { + async fn a_released_hold_clears_the_drain() { let q = CollectionQuiesce::new(); - q.begin_drain(DB, 1, "c"); + let hold = q.begin_drain(DB, 1, "c"); assert!(q.is_draining(DB, 1, "c")); - q.forget(DB, 1, "c"); + hold.release(); assert!(!q.is_draining(DB, 1, "c")); } #[tokio::test] - async fn lifecycle_acquisition_is_exclusive() { + async fn is_draining_until_every_holder_releases() { let q = CollectionQuiesce::new(); - let first = q.acquire_lifecycle(DB, 1, "c").await; - - let q_clone = Arc::clone(&q); - let second = tokio::spawn(async move { q_clone.acquire_lifecycle(DB, 1, "c").await }); - tokio::task::yield_now().await; - assert!(!second.is_finished()); - - drop(first); - let second = second.await.unwrap(); - drop(second); - assert!(!q.is_draining(DB, 1, "c")); - } - - #[tokio::test] - async fn is_draining_until_every_lifecycle_holder_forgets() { - let q = CollectionQuiesce::new(); - q.begin_drain(DB, 1, "c"); - q.begin_drain(DB, 1, "c"); + let first = q.begin_drain(DB, 1, "c"); + let second = q.begin_drain(DB, 1, "c"); // One holder released — still draining while the other holds. - q.forget(DB, 1, "c"); + drop(first); assert!(q.is_draining(DB, 1, "c")); // Last holder released — drain clears. - q.forget(DB, 1, "c"); - assert!(!q.is_draining(DB, 1, "c")); - } - - #[tokio::test] - async fn forget_clears_state() { - let q = CollectionQuiesce::new(); - q.begin_drain(DB, 1, "c"); - q.wait_until_drained(DB, 1, "c").await; - q.forget(DB, 1, "c"); + drop(second); assert!(!q.is_draining(DB, 1, "c")); assert!(q.try_start_scan(DB, 1, "c").is_ok()); } diff --git a/nodedb/src/bridge/quiesce/hold.rs b/nodedb/src/bridge/quiesce/hold.rs new file mode 100644 index 000000000..b6b352763 --- /dev/null +++ b/nodedb/src/bridge/quiesce/hold.rs @@ -0,0 +1,252 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Owned drain holds. +//! +//! Each hold carries an id, and a release removes exactly that id. A release +//! therefore never frees another holder's drain. +//! +//! A `_system.pending_reclaim` row owns at most one hold, keyed by the row's +//! own key, [`ReclaimOwner`]. Only a path that removed that row releases it. +//! A second hold handed to the same row is released at once. Row-owned holds +//! live in memory only, so after a restart a row owns none until its retry +//! takes one. + +use std::sync::{Arc, MutexGuard}; + +use super::refcount::{CollectionQuiesce, Inner}; + +/// Key of one collection's quiesce state: `(database, tenant, name)`. +type Key = (u64, u64, String); + +/// The owner of a row-owned hold: the `_system.pending_reclaim` row, keyed +/// `(database, tenant, name)` like the row. +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub struct ReclaimOwner { + pub database_id: u64, + pub tenant_id: u64, + pub name: String, +} + +impl ReclaimOwner { + pub fn new(database_id: u64, tenant_id: u64, name: &str) -> Self { + Self { + database_id, + tenant_id, + name: name.to_string(), + } + } + + fn key(&self) -> Key { + (self.database_id, self.tenant_id, self.name.clone()) + } +} + +/// One drain hold. New scans and same-name CREATE stay blocked while it +/// lives. Dropping it releases the hold. +#[must_use = "the drain is released when the hold drops"] +pub struct DrainHold { + registry: Arc, + key: Key, + id: u64, + active: bool, +} + +impl DrainHold { + /// Release the hold now. Equivalent to dropping it. + pub fn release(self) { + drop(self); + } + + /// Pass the hold to the `_system.pending_reclaim` row of the same + /// collection. A second hold handed to that row is released: the row + /// already holds the name. + pub fn hand_to_reclaim(mut self) { + self.active = false; + let (database_id, tenant_id, name) = self.key.clone(); + self.registry.park_reclaim_hold( + ReclaimOwner { + database_id, + tenant_id, + name, + }, + self.id, + ); + } +} + +impl Drop for DrainHold { + fn drop(&mut self) { + if self.active { + self.registry.release_hold(&self.key, self.id); + } + } +} + +impl std::fmt::Debug for DrainHold { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("DrainHold") + .field("key", &self.key) + .field("id", &self.id) + .field("active", &self.active) + .finish() + } +} + +impl CollectionQuiesce { + fn locked(&self) -> MutexGuard<'_, Inner> { + self.inner.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Record a new hold on `key` and return its id. The caller holds the + /// registry lock. + fn add_hold(inner: &mut Inner, key: &Key) -> u64 { + let id = inner.next_hold_id; + inner.next_hold_id = inner.next_hold_id.wrapping_add(1); + inner + .states + .entry(key.clone()) + .or_default() + .drain_holders + .insert(id); + id + } + + fn wrap(self: &Arc, key: Key, id: u64) -> DrainHold { + DrainHold { + registry: Arc::clone(self), + key, + id, + active: true, + } + } + + /// Stop new scans on the collection. Wait for open scans with + /// [`Self::wait_until_drained`]. The drain lasts until every hold on the + /// collection is released. + pub fn begin_drain( + self: &Arc, + database_id: u64, + tenant_id: u64, + collection: &str, + ) -> DrainHold { + let key = (database_id, tenant_id, collection.to_string()); + let id = Self::add_hold(&mut self.locked(), &key); + self.wrap(key, id) + } + + /// Give `owner` a hold on its collection, unless it already owns one. + pub fn ensure_reclaim_hold(&self, owner: &ReclaimOwner) { + let mut inner = self.locked(); + if inner.reclaim_holds.contains_key(owner) { + return; + } + let id = Self::add_hold(&mut inner, &owner.key()); + inner.reclaim_holds.insert(owner.clone(), id); + } + + /// Whether `owner` holds its collection. + pub fn has_reclaim_hold(&self, owner: &ReclaimOwner) -> bool { + self.locked().reclaim_holds.contains_key(owner) + } + + /// Release the hold `owner` owns, if any. Call it only after the row is + /// removed. No other holder's drain is touched. + pub fn release_reclaim_hold(&self, owner: &ReclaimOwner) { + let owned = self.locked().reclaim_holds.remove(owner); + if let Some(id) = owned { + self.release_hold(&owner.key(), id); + } + } + + fn park_reclaim_hold(&self, owner: ReclaimOwner, id: u64) { + let duplicate = { + let mut inner = self.locked(); + if inner.reclaim_holds.contains_key(&owner) { + true + } else { + inner.reclaim_holds.insert(owner.clone(), id); + false + } + }; + if duplicate { + self.release_hold(&owner.key(), id); + } + } + + /// Remove hold `id` from `key` and wake CREATE and drain waiters. + fn release_hold(&self, key: &Key, id: u64) { + { + let mut inner = self.locked(); + let remove = match inner.states.get_mut(key) { + Some(state) => { + state.drain_holders.remove(&id); + state.drain_holders.is_empty() && state.open_scans == 0 + } + None => false, + }; + if remove { + inner.states.remove(key); + } + } + self.notify.notify_waiters(); + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const DB: u64 = 0; + + /// Releasing the reclaim path's hold never frees another holder, and a + /// release with no owned hold, as after a restart, touches nothing. + #[test] + fn a_reclaim_release_frees_only_its_own_hold() { + let q = CollectionQuiesce::new(); + let row = ReclaimOwner::new(DB, 1, "c"); + let other = q.begin_drain(DB, 1, "c"); + + q.release_reclaim_hold(&row); + assert!(q.is_draining(DB, 1, "c"), "no owned hold, nothing released"); + + q.ensure_reclaim_hold(&row); + q.ensure_reclaim_hold(&row); + q.release_reclaim_hold(&row); + assert!( + q.is_draining(DB, 1, "c"), + "the other holder's drain survives the reclaim release" + ); + + drop(other); + assert!(!q.is_draining(DB, 1, "c")); + } + + /// A second hold handed to the reclaim path is released at once, so + /// one release frees the name. + #[test] + fn a_second_handed_hold_is_released() { + let q = CollectionQuiesce::new(); + q.begin_drain(DB, 1, "c").hand_to_reclaim(); + q.begin_drain(DB, 1, "c").hand_to_reclaim(); + assert!(q.is_draining(DB, 1, "c")); + + q.release_reclaim_hold(&ReclaimOwner::new(DB, 1, "c")); + assert!(!q.is_draining(DB, 1, "c")); + } + + /// Rows of two collections own separate holds: releasing one row's hold + /// leaves the other's drain in place. + #[test] + fn row_owned_holds_are_keyed_by_row() { + let q = CollectionQuiesce::new(); + let first = ReclaimOwner::new(DB, 1, "a"); + let second = ReclaimOwner::new(DB, 1, "b"); + q.ensure_reclaim_hold(&first); + q.ensure_reclaim_hold(&second); + + q.release_reclaim_hold(&first); + assert!(!q.is_draining(DB, 1, "a")); + assert!(q.is_draining(DB, 1, "b")); + assert!(q.has_reclaim_hold(&second)); + } +} diff --git a/nodedb/src/bridge/quiesce/mod.rs b/nodedb/src/bridge/quiesce/mod.rs index e9d8f3830..f98897894 100644 --- a/nodedb/src/bridge/quiesce/mod.rs +++ b/nodedb/src/bridge/quiesce/mod.rs @@ -11,15 +11,17 @@ //! Split: //! - [`refcount`] — per-`(tenant, collection)` scan refcount, `ScanGuard` //! RAII type, `try_start_scan` entry point. -//! - [`drain`] — `begin_drain` / `wait_until_drained` / `clear_drain` -//! async drain coordination. +//! - [`drain`] — `wait_until_drained`. +//! - [`hold`] — owned drain holds and the pending-reclaim hold. //! //! The backing [`CollectionQuiesce`] is shared (Arc). Calls are rare //! (one bump per scan, one drain per purge), so the internal `Mutex` //! is not a hot-path bottleneck. pub mod drain; +pub mod hold; pub mod refcount; pub use drain::WaitDrain; +pub use hold::{DrainHold, ReclaimOwner}; pub use refcount::{CollectionQuiesce, ScanGuard, ScanStartError}; diff --git a/nodedb/src/bridge/quiesce/refcount.rs b/nodedb/src/bridge/quiesce/refcount.rs index 94df7716f..eda744d53 100644 --- a/nodedb/src/bridge/quiesce/refcount.rs +++ b/nodedb/src/bridge/quiesce/refcount.rs @@ -2,7 +2,7 @@ //! Scan refcount + `ScanGuard` RAII wrapper. -use std::collections::HashMap; +use std::collections::{BTreeSet, HashMap}; use std::sync::{Arc, Mutex}; use tokio::sync::Notify; @@ -12,9 +12,10 @@ use tokio::sync::Notify; pub(super) struct CollectionState { /// Number of scans currently open against this collection. pub(super) open_scans: usize, - /// Number of lifecycle operations currently holding the collection drain. - /// CREATE remains blocked until every holder completes. - pub(super) drain_holders: usize, + /// Ids of the drain holds on this collection. New scans and CREATE stay + /// blocked until every hold is released. Each hold releases only its own + /// id, so a release never frees another holder. + pub(super) drain_holders: BTreeSet, } /// Why `try_start_scan` refused a new scan. @@ -51,6 +52,11 @@ pub struct CollectionQuiesce { #[derive(Debug, Default)] pub(super) struct Inner { pub(super) states: HashMap<(u64, u64, String), CollectionState>, + /// Id the next drain hold receives. + pub(super) next_hold_id: u64, + /// The hold each `_system.pending_reclaim` row owns. It outlives the + /// operation that took it and is released once the row is removed. + pub(super) reclaim_holds: HashMap, } impl CollectionQuiesce { @@ -67,12 +73,12 @@ impl CollectionQuiesce { tenant_id: u64, collection: &str, ) -> Result { - let mut inner = self.inner.lock().expect("CollectionQuiesce mutex poisoned"); + let mut inner = self.inner.lock().unwrap_or_else(|p| p.into_inner()); let entry = inner .states .entry((database_id, tenant_id, collection.to_string())) .or_default(); - if entry.drain_holders > 0 { + if !entry.drain_holders.is_empty() { return Err(ScanStartError::Draining); } entry.open_scans += 1; @@ -87,7 +93,7 @@ impl CollectionQuiesce { /// Current open-scan count. For tests / metrics. pub fn open_scans(&self, database_id: u64, tenant_id: u64, collection: &str) -> usize { - let inner = self.inner.lock().expect("CollectionQuiesce mutex poisoned"); + let inner = self.inner.lock().unwrap_or_else(|p| p.into_inner()); inner .states .get(&(database_id, tenant_id, collection.to_string())) @@ -96,15 +102,15 @@ impl CollectionQuiesce { /// Whether a drain is currently in progress for this collection. pub fn is_draining(&self, database_id: u64, tenant_id: u64, collection: &str) -> bool { - let inner = self.inner.lock().expect("CollectionQuiesce mutex poisoned"); + let inner = self.inner.lock().unwrap_or_else(|p| p.into_inner()); inner .states .get(&(database_id, tenant_id, collection.to_string())) - .is_some_and(|s| s.drain_holders > 0) + .is_some_and(|s| !s.drain_holders.is_empty()) } pub(super) fn release_scan(&self, database_id: u64, tenant_id: u64, collection: &str) { - let mut inner = self.inner.lock().expect("CollectionQuiesce mutex poisoned"); + let mut inner = self.inner.lock().unwrap_or_else(|p| p.into_inner()); if let Some(state) = inner .states .get_mut(&(database_id, tenant_id, collection.to_string())) @@ -192,7 +198,7 @@ mod tests { #[test] fn drain_rejects_new_scans() { let q = CollectionQuiesce::new(); - q.begin_drain(DB, 1, "c"); + let _hold = q.begin_drain(DB, 1, "c"); let err = q.try_start_scan(DB, 1, "c").unwrap_err(); assert_eq!(err, ScanStartError::Draining); assert!(q.is_draining(DB, 1, "c")); @@ -201,7 +207,7 @@ mod tests { #[test] fn drain_does_not_affect_other_scopes() { let q = CollectionQuiesce::new(); - q.begin_drain(DB, 1, "c"); + let _hold = q.begin_drain(DB, 1, "c"); assert!(q.try_start_scan(DB, 1, "other").is_ok()); assert!(q.try_start_scan(DB, 2, "c").is_ok()); assert!(q.try_start_scan(1, 1, "c").is_ok()); diff --git a/nodedb/src/config/server/backup.rs b/nodedb/src/config/server/backup.rs new file mode 100644 index 000000000..5f89de847 --- /dev/null +++ b/nodedb/src/config/server/backup.rs @@ -0,0 +1,267 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use std::collections::HashSet; + +use serde::{Deserialize, Serialize}; + +use super::ServerConfig; + +/// Scheduled logical backups. +/// +/// Each entry runs `BACKUP DATABASE ` on its cron schedule. A run +/// writes one envelope named `-.ndbb` under `target`, where +/// `` is the scheduled minute. It then deletes the oldest envelopes +/// of that database under `target` beyond `keep`. Other objects under +/// `target` are never touched. +/// +/// Example TOML: +/// ```toml +/// [[backup.schedule]] +/// database = "sales" +/// target = "s3://my-backups/nightly/sales" +/// cron = "0 3 * * *" +/// keep = 7 +/// +/// [backup_encryption] +/// key_path = "/etc/nodedb/keys/backup.key" +/// ``` +/// +/// `target` resolves like a `BACKUP DATABASE ... TO` URI, against +/// `[backup_storage]`. A `file://` target lies inside `local_root`. +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct BackupSettings { + #[serde(default)] + pub schedule: Vec, +} + +/// One scheduled backup. +#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct BackupScheduleSettings { + /// Database to back up. + pub database: String, + /// URI prefix the envelopes go under: `file:///` or + /// `s3:///`. + pub target: String, + /// 5-field cron expression, evaluated in `[scheduler] cron_timezone`. + pub cron: String, + /// Envelopes of this database kept under `target`. Must be positive. + pub keep: u64, +} + +impl BackupScheduleSettings { + /// `target` without trailing slashes. + pub fn target_prefix(&self) -> &str { + self.target.trim_end_matches('/') + } + + /// The job name the scheduler records history under. + pub fn job_name(&self) -> String { + format!("backup:{}:{}", self.database, self.target_prefix()) + } + + /// The config incarnation: a fingerprint of the database, target and + /// cron. Every node computes the same value for the same entry. A + /// changed entry is a new incarnation, with its own schedule mark. + pub fn incarnation(&self) -> u64 { + use sha2::{Digest, Sha256}; + let mut hasher = Sha256::new(); + for part in [ + self.database.as_str(), + self.target_prefix(), + self.cron.trim(), + ] { + hasher.update((part.len() as u64).to_le_bytes()); + hasher.update(part.as_bytes()); + } + let digest = hasher.finalize(); + let mut head = [0u8; 8]; + head.copy_from_slice(&digest[..8]); + u64::from_le_bytes(head) + } +} + +/// Refuse a schedule that can never run: a bad cron, a zero `keep`, a target +/// no backup URI accepts, no backup key, or two entries sharing one database +/// and target, whose `keep` retention deletes each other's envelopes. +pub(super) fn validate_backup(config: &ServerConfig) -> crate::Result<()> { + let schedules = &config.backup.schedule; + if schedules.is_empty() { + return Ok(()); + } + if config.backup_encryption.is_none() { + return Err(invalid( + "backup.schedule needs a [backup_encryption] section: every backup envelope is \ + encrypted. Add [backup_encryption] key_path or remove backup.schedule" + .into(), + )); + } + let mut seen = HashSet::new(); + for (index, schedule) in schedules.iter().enumerate() { + let entry = format!("backup.schedule[{index}]"); + if schedule.database.trim().is_empty() { + return Err(invalid(format!("{entry}.database is empty"))); + } + if schedule.keep == 0 { + return Err(invalid(format!( + "{entry}.keep is 0: expected a positive number of envelopes to keep" + ))); + } + crate::event::scheduler::cron::CronExpr::parse(&schedule.cron) + .map_err(|e| invalid(format!("{entry}.cron '{}': {e}", schedule.cron)))?; + check_target(&entry, schedule, config)?; + if !seen.insert((schedule.database.as_str(), schedule.target_prefix())) { + return Err(invalid(format!( + "{entry} repeats database '{}' with target '{}'; each pair runs once", + schedule.database, schedule.target + ))); + } + } + Ok(()) +} + +fn check_target( + entry: &str, + schedule: &BackupScheduleSettings, + config: &ServerConfig, +) -> crate::Result<()> { + let target = schedule.target_prefix(); + if let Some(path) = target.strip_prefix("file://") { + let root = config + .backup_storage + .as_ref() + .and_then(|storage| storage.local_root.as_deref()) + .ok_or_else(|| { + invalid(format!( + "{entry}.target '{target}' is a file:// URI, which needs \ + [backup_storage] local_root" + )) + })?; + let inside = std::path::Path::new(path) + .strip_prefix(root) + .is_ok_and(|rest| !rest.as_os_str().is_empty()); + if !inside { + return Err(invalid(format!( + "{entry}.target '{target}' must name a directory inside [backup_storage] \ + local_root '{}'", + root.display() + ))); + } + return Ok(()); + } + if let Some(rest) = target.strip_prefix("s3://") { + let has_prefix = rest + .split_once('/') + .is_some_and(|(bucket, prefix)| !bucket.is_empty() && !prefix.is_empty()); + if has_prefix { + return Ok(()); + } + return Err(invalid(format!( + "{entry}.target '{target}': expected s3:///" + ))); + } + Err(invalid(format!( + "{entry}.target '{target}': expected file:/// or s3:///" + ))) +} + +fn invalid(detail: String) -> crate::Error { + crate::Error::Config { detail } +} + +#[cfg(test)] +mod tests { + use super::*; + + const KEY: &str = "\n[backup_encryption]\nkey_path = \"/k\"\n"; + const ROOT: &str = "\n[backup_storage]\nlocal_root = \"/srv/backups\"\n"; + + fn parse(raw: &str) -> ServerConfig { + toml::from_str(raw).expect("deserialize") + } + + fn entry(target: &str, cron: &str, keep: u64) -> String { + format!( + "[[backup.schedule]]\ndatabase = \"sales\"\ntarget = \"{target}\"\n\ + cron = \"{cron}\"\nkeep = {keep}\n" + ) + } + + fn error(raw: &str) -> String { + match parse(raw).validate() { + Err(crate::Error::Config { detail }) => detail, + other => panic!("expected a config error, got {other:?}"), + } + } + + #[test] + fn a_valid_schedule_parses() { + let raw = format!("{}{KEY}{ROOT}", entry("s3://b/nightly/", "0 3 * * *", 7)); + let cfg = parse(&raw); + cfg.validate().expect("valid schedule"); + let schedule = &cfg.backup.schedule[0]; + assert_eq!(schedule.keep, 7); + assert_eq!(schedule.target_prefix(), "s3://b/nightly"); + assert_eq!(schedule.job_name(), "backup:sales:s3://b/nightly"); + let local = format!( + "{}{KEY}{ROOT}", + entry("file:///srv/backups/sales", "* * * * *", 1) + ); + parse(&local) + .validate() + .expect("file target inside the root"); + } + + #[test] + fn the_incarnation_follows_database_target_and_cron_only() { + let base = BackupScheduleSettings { + database: "sales".into(), + target: "s3://b/p".into(), + cron: "0 3 * * *".into(), + keep: 2, + }; + let same = BackupScheduleSettings { + target: "s3://b/p/".into(), + keep: 9, + ..base.clone() + }; + assert_eq!(base.incarnation(), same.incarnation()); + let recron = BackupScheduleSettings { + cron: "0 4 * * *".into(), + ..base.clone() + }; + assert_ne!(base.incarnation(), recron.incarnation()); + } + + #[test] + fn no_schedule_needs_nothing() { + ServerConfig::default().validate().expect("no schedule"); + } + + #[test] + fn invalid_schedules_are_config_errors() { + let s3 = "s3://b/p"; + assert!(error(&entry(s3, "0 3 * * *", 1)).contains("[backup_encryption]")); + assert!(error(&format!("{}{KEY}", entry(s3, "0 3 * * *", 0))).contains("keep")); + assert!(error(&format!("{}{KEY}", entry(s3, "bad", 1))).contains("cron")); + assert!(error(&format!("{}{KEY}", entry("s3://b", "0 3 * * *", 1))).contains("target")); + assert!(error(&format!("{}{KEY}", entry("ftp://x/y", "0 3 * * *", 1))).contains("target")); + let no_root = format!("{}{KEY}", entry("file:///srv/backups/x", "0 3 * * *", 1)); + assert!(error(&no_root).contains("local_root")); + let outside = format!("{}{KEY}{ROOT}", entry("file:///etc/x", "0 3 * * *", 1)); + assert!(error(&outside).contains("inside")); + let twice = format!( + "{}{}{KEY}", + entry(s3, "0 3 * * *", 1), + entry("s3://b/p/", "0 4 * * *", 2) + ); + assert!(error(&twice).contains("repeats")); + } + + #[test] + fn unknown_schedule_field_rejected() { + let raw = format!("{}retain = 3\n", entry("s3://b/p", "0 3 * * *", 1)); + assert!(toml::from_str::(&raw).is_err()); + } +} diff --git a/nodedb/src/config/server/backup_storage.rs b/nodedb/src/config/server/backup_storage.rs new file mode 100644 index 000000000..5e828c09c --- /dev/null +++ b/nodedb/src/config/server/backup_storage.rs @@ -0,0 +1,53 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use std::path::PathBuf; + +use serde::{Deserialize, Serialize}; + +/// Where `BACKUP DATABASE ... TO ''` writes and `RESTORE DATABASE ... +/// FROM ''` reads. Credentials come from here, never from the SQL text. +/// +/// Example TOML: +/// ```toml +/// [backup_storage] +/// local_root = "/srv/nodedb/backups" +/// endpoint = "http://localhost:9000" +/// access_key = "..." +/// secret_key = "..." +/// region = "us-east-1" +/// ``` +/// +/// * A `file://` URI names a path inside `local_root`. Without `local_root`, +/// every `file://` URI is refused: SQL never names an arbitrary server path. +/// * An `s3:///` URI uses `endpoint` (empty = AWS), `region` and +/// the keys (empty = IAM role or instance credentials). +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct BackupStorageSettings { + #[serde(default)] + pub local_root: Option, + #[serde(default)] + pub endpoint: String, + #[serde(default)] + pub access_key: String, + #[serde(default)] + pub secret_key: String, + #[serde(default = "default_region")] + pub region: String, +} + +fn default_region() -> String { + "us-east-1".into() +} + +impl Default for BackupStorageSettings { + fn default() -> Self { + Self { + local_root: None, + endpoint: String::new(), + access_key: String::new(), + secret_key: String::new(), + region: default_region(), + } + } +} diff --git a/nodedb/src/config/server/checkpoint.rs b/nodedb/src/config/server/checkpoint.rs index 394bef981..57beeda85 100644 --- a/nodedb/src/config/server/checkpoint.rs +++ b/nodedb/src/config/server/checkpoint.rs @@ -13,6 +13,7 @@ use serde::{Deserialize, Serialize}; /// interval_secs = 300 /// core_timeout_secs = 30 /// wal_segment_target_mb = 64 +/// wal_archive_interval_secs = 10 /// ``` #[derive(Debug, Clone, Serialize, Deserialize)] pub struct CheckpointSettings { @@ -39,6 +40,13 @@ pub struct CheckpointSettings { #[serde(default = "default_wal_segment_target_mb")] pub wal_segment_target_mb: u64, + /// How often sealed WAL segments are uploaded to cold storage (seconds). + /// Runs only when `[cold_storage]` is configured. A sealed segment + /// reaches the archive within one interval. + /// Default: 10. + #[serde(default = "default_wal_archive_interval")] + pub wal_archive_interval_secs: u64, + /// How often each Data Plane core runs automatic compaction (seconds). /// Compaction removes tombstoned vectors from HNSW indexes, compacts /// CSR write buffers, and sweeps dangling edges. @@ -60,6 +68,7 @@ impl Default for CheckpointSettings { interval_secs: default_checkpoint_interval(), core_timeout_secs: default_core_timeout(), wal_segment_target_mb: default_wal_segment_target_mb(), + wal_archive_interval_secs: default_wal_archive_interval(), compaction_interval_secs: default_compaction_interval(), compaction_tombstone_threshold: default_compaction_tombstone_threshold(), } @@ -68,11 +77,17 @@ impl Default for CheckpointSettings { impl CheckpointSettings { /// Convert to the checkpoint manager config used by the Control Plane. - pub fn to_manager_config(&self) -> crate::control::checkpoint_manager::CheckpointManagerConfig { - crate::control::checkpoint_manager::CheckpointManagerConfig { - interval: std::time::Duration::from_secs(self.interval_secs), - core_timeout: std::time::Duration::from_secs(self.core_timeout_secs), - } + /// + /// Fails with `Error::Config` when `interval_secs` or + /// `wal_archive_interval_secs` is zero, whatever path set it. + pub fn to_manager_config( + &self, + ) -> crate::Result { + crate::control::checkpoint_manager::CheckpointManagerConfig::new( + std::time::Duration::from_secs(self.interval_secs), + std::time::Duration::from_secs(self.core_timeout_secs), + std::time::Duration::from_secs(self.wal_archive_interval_secs), + ) } /// WAL segment target size in bytes. @@ -98,6 +113,10 @@ fn default_wal_segment_target_mb() -> u64 { 64 } +fn default_wal_archive_interval() -> u64 { + 10 +} + fn default_compaction_interval() -> u64 { 600 } diff --git a/nodedb/src/config/server/cluster.rs b/nodedb/src/config/server/cluster.rs index e3ada31fd..a72feb69c 100644 --- a/nodedb/src/config/server/cluster.rs +++ b/nodedb/src/config/server/cluster.rs @@ -24,6 +24,14 @@ pub struct ClusterSettings { /// Address to bind the Raft RPC QUIC listener. pub listen: SocketAddr, + /// UDP address for the SWIM failure detector. + /// + /// Default: the `listen` IP with port `listen` port + 1. The node + /// advertises the bound address to peers, so each node can override it + /// independently. Startup fails if the address cannot be bound. + #[serde(default)] + pub swim_listen: Option, + /// Seed node addresses for cluster formation or joining. /// On first startup, the first reachable seed bootstraps the cluster. /// Subsequent nodes join by contacting any seed. @@ -241,6 +249,7 @@ mod tests { log_compaction_threshold: None, join_retry_max_attempts: 8, join_retry_max_backoff_secs: 32, + swim_listen: None, } } diff --git a/nodedb/src/config/server/config.rs b/nodedb/src/config/server/config.rs index aac4351e9..2656ba073 100644 --- a/nodedb/src/config/server/config.rs +++ b/nodedb/src/config/server/config.rs @@ -8,10 +8,13 @@ use std::path::PathBuf; use nodedb_types::config::TuningConfig; use serde::{Deserialize, Serialize}; +use super::backup::BackupSettings; +use super::backup_storage::BackupStorageSettings; use super::checkpoint::CheckpointSettings; use super::cluster::ClusterSettings; use super::cold_storage::ColdStorageSettings; use super::observability::ObservabilityConfig; +use super::pitr::PitrSettings; use super::retention::RetentionSettings; use super::scheduler::SchedulerConfig; use super::section::ServerSection; @@ -71,6 +74,16 @@ pub struct ServerConfig { #[serde(default)] pub backup_encryption: Option, + /// Object-store access for `BACKUP DATABASE` and `RESTORE DATABASE`. + /// Absent = every `file://` URI is refused, and `s3://` URIs use IAM + /// credentials against AWS. + #[serde(default)] + pub backup_storage: Option, + + /// Scheduled logical backups: `[[backup.schedule]]` entries. + #[serde(default)] + pub backup: BackupSettings, + /// Checkpoint and WAL management settings. #[serde(default)] pub checkpoint: CheckpointSettings, @@ -83,7 +96,9 @@ pub struct ServerConfig { /// Cluster mode settings. When present, the node participates in a /// distributed cluster via Multi-Raft consensus over QUIC transport. - /// When absent, runs in single-node mode (default). + /// When absent (default), the node synthesizes a one-node cluster and runs + /// the single-node Calvin sequencer, so cross-core transactions commit + /// atomically. #[serde(default)] pub cluster: Option, @@ -92,6 +107,10 @@ pub struct ServerConfig { #[serde(default)] pub cold_storage: Option, + /// Point-in-time recovery. `enabled = true` requires `cold_storage`. + #[serde(default)] + pub pitr: PitrSettings, + /// Snapshot storage configuration. /// Controls where warm-tier snapshots are persisted. When absent, defaults /// to local filesystem at `{data_dir}/snapshots`. diff --git a/nodedb/src/config/server/domain.rs b/nodedb/src/config/server/domain.rs index a3d7abea2..bb2cd2950 100644 --- a/nodedb/src/config/server/domain.rs +++ b/nodedb/src/config/server/domain.rs @@ -31,7 +31,7 @@ fn reject(field: &str, value: impl std::fmt::Display, expected: &str) -> crate:: } } -fn positive_u64(value: u64, field: &str) -> crate::Result<()> { +pub(super) fn positive_u64(value: u64, field: &str) -> crate::Result<()> { if value == 0 { return Err(reject(field, value, "a positive integer")); } @@ -65,6 +65,12 @@ pub(super) fn validate_domain(config: &ServerConfig) -> crate::Result<()> { config.checkpoint.wal_segment_target_mb, "checkpoint.wal_segment_target_mb", )?; + positive_u64( + config.checkpoint.wal_archive_interval_secs, + "checkpoint.wal_archive_interval_secs", + )?; + super::pitr::validate_pitr(config)?; + super::backup::validate_backup(config)?; let ts = &config.tuning.timeseries; positive_usize( diff --git a/nodedb/src/config/server/env/rows/cluster.rs b/nodedb/src/config/server/env/rows/cluster.rs index a53f82d3d..5e64e8032 100644 --- a/nodedb/src/config/server/env/rows/cluster.rs +++ b/nodedb/src/config/server/env/rows/cluster.rs @@ -1,14 +1,15 @@ // SPDX-License-Identifier: BUSL-1.1 -//! `NODEDB_NODE_ID` / `NODEDB_SEED_NODES` / `NODEDB_JOIN_RETRY_MAX_ATTEMPTS` -//! / `NODEDB_JOIN_RETRY_MAX_BACKOFF_SECS` overrides. +//! `NODEDB_NODE_ID` / `NODEDB_SEED_NODES` / `NODEDB_SWIM_LISTEN` / +//! `NODEDB_JOIN_RETRY_MAX_ATTEMPTS` / `NODEDB_JOIN_RETRY_MAX_BACKOFF_SECS` +//! overrides. //! //! Every row here needs a `[cluster]` section already in the loaded config. //! The process cannot invent a cluster identity for itself. use crate::config::server::ServerConfig; -use super::super::parse::{parse_u32_positive, parse_u64_positive}; +use super::super::parse::{parse_socket_addr, parse_u32_positive, parse_u64_positive}; use super::super::seed_nodes::parse_seed_nodes; use super::super::table::EnvRow; @@ -28,6 +29,13 @@ fn apply_seed_nodes(config: &mut ServerConfig, raw: &str) -> Result<(), &'static Ok(()) } +fn apply_swim_listen(config: &mut ServerConfig, raw: &str) -> Result<(), &'static str> { + let addr = parse_socket_addr(raw, "a UDP socket address such as 10.0.0.1:9401")?; + let cluster = config.cluster.as_mut().ok_or(NO_CLUSTER)?; + cluster.swim_listen = Some(addr); + Ok(()) +} + fn apply_join_retry_max_attempts(config: &mut ServerConfig, raw: &str) -> Result<(), &'static str> { let attempts = parse_u32_positive(raw)?; let cluster = config.cluster.as_mut().ok_or(NO_CLUSTER)?; @@ -56,6 +64,11 @@ pub(in super::super) const ROWS: &[EnvRow] = &[ apply: apply_seed_nodes, redact: false, }, + EnvRow { + name: "NODEDB_SWIM_LISTEN", + apply: apply_swim_listen, + redact: false, + }, EnvRow { name: "NODEDB_JOIN_RETRY_MAX_ATTEMPTS", apply: apply_join_retry_max_attempts, @@ -90,6 +103,7 @@ mod tests { log_compaction_threshold: None, join_retry_max_attempts: 8, join_retry_max_backoff_secs: 32, + swim_listen: None, } } @@ -98,6 +112,7 @@ mod tests { unsafe { std::env::set_var("NODEDB_NODE_ID", "42"); std::env::set_var("NODEDB_SEED_NODES", "10.0.0.1:9400,10.0.0.2:9400"); + std::env::set_var("NODEDB_SWIM_LISTEN", "10.0.0.1:9501"); } let mut cfg = ServerConfig { cluster: Some(make_cluster(1)), @@ -116,9 +131,15 @@ mod tests { ); assert_eq!(cluster.seed_nodes[0].to_string(), "10.0.0.1:9400"); assert_eq!(cluster.seed_nodes[1].to_string(), "10.0.0.2:9400"); + assert_eq!( + cluster.swim_listen.map(|a| a.to_string()).as_deref(), + Some("10.0.0.1:9501"), + "NODEDB_SWIM_LISTEN must override swim_listen" + ); unsafe { std::env::remove_var("NODEDB_NODE_ID"); std::env::remove_var("NODEDB_SEED_NODES"); + std::env::remove_var("NODEDB_SWIM_LISTEN"); } } } diff --git a/nodedb/src/config/server/mod.rs b/nodedb/src/config/server/mod.rs index a5bac6b5e..6b0fe00a6 100644 --- a/nodedb/src/config/server/mod.rs +++ b/nodedb/src/config/server/mod.rs @@ -1,5 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 +mod backup; +mod backup_storage; mod checkpoint; mod cluster; mod cold_storage; @@ -10,6 +12,7 @@ mod env_expand; mod log_format; mod observability; mod paths; +mod pitr; mod ports; mod retention; pub mod scheduler; @@ -19,6 +22,8 @@ mod snapshot_storage; mod test_support; mod tls; +pub use backup::{BackupScheduleSettings, BackupSettings}; +pub use backup_storage::BackupStorageSettings; pub use checkpoint::CheckpointSettings; pub use cluster::{ClusterSettings, TlsPaths}; pub use cold_storage::ColdStorageSettings; @@ -29,6 +34,7 @@ pub use observability::{ ObservabilityConfig, OtlpConfig, OtlpExportConfig, OtlpReceiverConfig, PromqlConfig, validate_feature_availability, }; +pub use pitr::{PitrSettings, missing_cold_storage}; pub use ports::{DEFAULT_SYNC_PORT, PortsConfig}; pub use retention::RetentionSettings; pub use scheduler::{CronTimezone, SchedulerConfig}; diff --git a/nodedb/src/config/server/pitr.rs b/nodedb/src/config/server/pitr.rs new file mode 100644 index 000000000..5b5698333 --- /dev/null +++ b/nodedb/src/config/server/pitr.rs @@ -0,0 +1,229 @@ +// SPDX-License-Identifier: BUSL-1.1 + +use std::num::NonZeroUsize; +use std::time::Duration; + +use serde::{Deserialize, Serialize}; + +use super::ServerConfig; + +/// Default seconds between two base snapshots: one day. +const DEFAULT_BASE_SNAPSHOT_INTERVAL_SECS: u64 = 86_400; + +/// Default number of base snapshots retention keeps. +const DEFAULT_BASE_SNAPSHOT_RETENTION: u64 = 7; + +/// Default seconds between two periodic cluster restore points: none. +const DEFAULT_RESTORE_POINT_INTERVAL_SECS: u64 = 0; + +/// Point-in-time recovery configuration. +/// +/// PITR needs every WAL segment in the archive before truncation deletes it. +/// Without `[cold_storage]` there is no archive, and truncation deletes +/// segments that were never archived. That is correct only when PITR is off. +/// +/// Base snapshots are encrypted with the WAL key, so PITR also needs +/// `[encryption]`. +/// +/// Example TOML: +/// ```toml +/// [pitr] +/// enabled = true +/// base_snapshot_interval_secs = 86400 +/// base_snapshot_retention = 7 +/// restore_point_interval_secs = 3600 +/// +/// [cold_storage] +/// bucket = "my-nodedb-cold" +/// +/// [encryption] +/// key_path = "/etc/nodedb/keys/wal.key" +/// ``` +#[derive(Debug, Clone, Serialize, Deserialize)] +#[serde(deny_unknown_fields)] +pub struct PitrSettings { + /// Require a continuous WAL archive and take periodic base snapshots. + /// Boot fails without `[cold_storage]` and `[encryption]`. Default: false. + #[serde(default)] + pub enabled: bool, + /// Seconds between two base snapshots. It also bounds how long one + /// snapshot can take. Must be positive. Default: 86400. + #[serde(default = "default_base_snapshot_interval_secs")] + pub base_snapshot_interval_secs: u64, + /// Base snapshots kept. Older bases are deleted, then archived WAL that + /// only they needed. Must be positive. Default: 7. + #[serde(default = "default_base_snapshot_retention")] + pub base_snapshot_retention: u64, + /// Seconds between two periodic cluster restore points. `0` takes none; + /// `CREATE RESTORE POINT` still takes one on demand. Only a cluster takes + /// restore points. Default: 0. + #[serde(default = "default_restore_point_interval_secs")] + pub restore_point_interval_secs: u64, +} + +impl Default for PitrSettings { + fn default() -> Self { + Self { + enabled: false, + base_snapshot_interval_secs: DEFAULT_BASE_SNAPSHOT_INTERVAL_SECS, + base_snapshot_retention: DEFAULT_BASE_SNAPSHOT_RETENTION, + restore_point_interval_secs: DEFAULT_RESTORE_POINT_INTERVAL_SECS, + } + } +} + +impl PitrSettings { + pub fn base_snapshot_interval(&self) -> Duration { + Duration::from_secs(self.base_snapshot_interval_secs) + } + + /// Interval between periodic restore points, `None` when off. + pub fn restore_point_interval(&self) -> Option { + (self.restore_point_interval_secs > 0) + .then(|| Duration::from_secs(self.restore_point_interval_secs)) + } + + /// Base snapshots retention keeps. A zero value is a config error. + pub fn retention(&self) -> crate::Result { + super::domain::positive_u64(self.base_snapshot_retention, "pitr.base_snapshot_retention")?; + Ok(usize::try_from(self.base_snapshot_retention) + .ok() + .and_then(NonZeroUsize::new) + .unwrap_or(NonZeroUsize::MAX)) + } +} + +fn default_base_snapshot_interval_secs() -> u64 { + DEFAULT_BASE_SNAPSHOT_INTERVAL_SECS +} + +fn default_base_snapshot_retention() -> u64 { + DEFAULT_BASE_SNAPSHOT_RETENTION +} + +fn default_restore_point_interval_secs() -> u64 { + DEFAULT_RESTORE_POINT_INTERVAL_SECS +} + +/// Refuse a config that enables PITR with no archive to recover from, with no +/// key to write base snapshots, or with a zero interval or retention. +pub(super) fn validate_pitr(config: &ServerConfig) -> crate::Result<()> { + super::domain::positive_u64( + config.pitr.base_snapshot_interval_secs, + "pitr.base_snapshot_interval_secs", + )?; + super::domain::positive_u64( + config.pitr.base_snapshot_retention, + "pitr.base_snapshot_retention", + )?; + if !config.pitr.enabled { + return Ok(()); + } + if config.cold_storage.is_none() { + return Err(missing_cold_storage()); + } + if config.encryption.is_none() { + return Err(crate::Error::Config { + detail: "pitr.enabled = true requires an [encryption] section: base snapshots \ + are encrypted with the WAL key. Add [encryption] key_path or set \ + pitr.enabled = false" + .into(), + }); + } + Ok(()) +} + +/// The error for `pitr.enabled = true` with no usable `[cold_storage]`. +pub fn missing_cold_storage() -> crate::Error { + crate::Error::Config { + detail: "pitr.enabled = true requires a [cold_storage] section: the WAL archive is \ + the only copy of a segment once truncation deletes it. Add [cold_storage] \ + or set pitr.enabled = false" + .into(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + const COLD: &str = "\n[cold_storage]\nbucket = \"b\"\n"; + const KEY: &str = "\n[encryption]\nkey_path = \"/k\"\n"; + + fn parse(raw: &str) -> ServerConfig { + toml::from_str(raw).expect("deserialize") + } + + #[test] + fn pitr_is_off_by_default() { + let cfg = ServerConfig::default(); + assert!(!cfg.pitr.enabled); + assert_eq!(cfg.pitr.base_snapshot_interval_secs, 86_400); + assert_eq!(cfg.pitr.base_snapshot_retention, 7); + cfg.validate().expect("PITR off needs no cold storage"); + } + + #[test] + fn pitr_without_cold_storage_refuses_to_boot() { + let cfg = parse(&format!("[pitr]\nenabled = true\n{KEY}")); + let msg = cfg.validate().unwrap_err().to_string(); + assert!(msg.contains("pitr.enabled"), "{msg}"); + assert!(msg.contains("[cold_storage]"), "{msg}"); + } + + #[test] + fn pitr_without_encryption_refuses_to_boot() { + let cfg = parse(&format!("[pitr]\nenabled = true\n{COLD}")); + let msg = cfg.validate().unwrap_err().to_string(); + assert!(msg.contains("[encryption]"), "{msg}"); + } + + #[test] + fn pitr_with_cold_storage_and_a_key_boots() { + let cfg = parse(&format!("[pitr]\nenabled = true\n{COLD}{KEY}")); + cfg.validate() + .expect("PITR with cold storage and a key is valid"); + } + + #[test] + fn a_zero_base_snapshot_interval_is_a_config_error() { + let cfg = parse(&format!( + "[pitr]\nenabled = true\nbase_snapshot_interval_secs = 0\n{COLD}{KEY}" + )); + match cfg.validate() { + Err(crate::Error::Config { detail }) => { + assert!( + detail.contains("pitr.base_snapshot_interval_secs"), + "{detail}" + ); + } + other => panic!("expected a config error, got {other:?}"), + } + } + + #[test] + fn a_zero_base_snapshot_retention_is_a_config_error() { + let cfg = parse("[pitr]\nbase_snapshot_retention = 0\n"); + match cfg.validate() { + Err(crate::Error::Config { detail }) => { + assert!(detail.contains("pitr.base_snapshot_retention"), "{detail}"); + } + other => panic!("expected a config error, got {other:?}"), + } + } + + #[test] + fn from_file_applies_the_pitr_gate() { + let path = std::env::temp_dir().join("nodedb-pitr-gate.toml"); + std::fs::write(&path, "[pitr]\nenabled = true\n").expect("write temp config"); + let err = ServerConfig::from_file(&path).unwrap_err(); + std::fs::remove_file(&path).ok(); + assert!(err.to_string().contains("[cold_storage]"), "{err}"); + } + + #[test] + fn unknown_pitr_field_rejected() { + let result: Result = toml::from_str("[pitr]\nenable = true\n"); + assert!(result.is_err(), "a misspelled pitr field must be rejected"); + } +} diff --git a/nodedb/src/config/server/section.rs b/nodedb/src/config/server/section.rs index 0bf7ab602..03a46b20c 100644 --- a/nodedb/src/config/server/section.rs +++ b/nodedb/src/config/server/section.rs @@ -79,26 +79,6 @@ pub struct ServerSection { #[serde(default)] pub tls: Option, - /// Enable the Calvin transaction stack on a standalone (non-cluster) - /// server. Default `false`. - /// - /// When `true` and `[cluster]` is absent, the server synthesizes a - /// one-node cluster (this node as its own sole seed, replication - /// factor 1) and drives the same cluster startup a real deployment - /// uses: the sequencer Raft group and per-vShard schedulers come up, - /// its QUIC transport binds to a loopback port that never dials a - /// peer, and a cross-core (cross-vShard) transaction traverses the - /// deterministic Calvin path exactly as in cluster mode. - /// - /// On by default: a standalone node stands up the single-node Calvin - /// stack so cross-core (cross-vShard) transactions commit atomically - /// through the deterministic sequencer path instead of being rejected. - /// Set to `false` to force the legacy single-node path (no Calvin stack; - /// cross-shard interactive transactions are rejected) — e.g. for a - /// minimal deployment that never issues cross-core transactions. - #[serde(default = "default_single_node_calvin")] - pub single_node_calvin: bool, - /// Capture an out-of-process minidump when the server dies to a native /// fault (SIGSEGV, abort, stack overflow) rather than a Rust panic. /// Default `false`. @@ -127,19 +107,11 @@ impl Default for ServerSection { max_connections: default_max_connections(), log_format: LogFormat::Text, tls: None, - single_node_calvin: default_single_node_calvin(), native_crash_dumps: false, } } } -/// Default for [`ServerSection::single_node_calvin`]: on, so a standalone -/// node supports cross-core transactions through the single-node Calvin path -/// out of the box. -fn default_single_node_calvin() -> bool { - true -} - fn default_host() -> IpAddr { IpAddr::V4(Ipv4Addr::LOCALHOST) } @@ -234,4 +206,10 @@ mod tests { let result: Result = toml::from_str("frobnicate = true\n"); assert!(result.is_err()); } + + #[test] + fn single_node_calvin_key_rejected() { + let result: Result = toml::from_str("single_node_calvin = false\n"); + assert!(result.is_err()); + } } diff --git a/nodedb/src/control/array_catalog/cell_route.rs b/nodedb/src/control/array_catalog/cell_route.rs new file mode 100644 index 000000000..9cd3521e9 --- /dev/null +++ b/nodedb/src/control/array_catalog/cell_route.rs @@ -0,0 +1,178 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Route a replicated array cell write to the array incarnation it targets. +//! +//! A data-group replica can apply a cell write after its metadata group +//! already applied a MOVE TENANT or a DROP of the array. The proposer stamps +//! the incarnation it wrote against. The replica then applies the write under +//! its own key, routes it to the database the incarnation moved to, or +//! refuses it as superseded when the incarnation no longer exists. +//! +//! The incarnation's gate (`control::write_gate`) orders the two sides: +//! a replica routes under it shared, and the array delete's post-apply moves +//! or drops the array under it exclusive. + +use nodedb_types::Hlc; + +use crate::control::array_catalog::ArrayCatalog; +use crate::control::state::SharedState; +use crate::control::wal_replication::{ReplicatedEntry, ReplicatedWrite}; +use crate::types::{DatabaseId, TenantId}; + +/// Where a cell write for `(tenant, database, name)` at an incarnation lands. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum CellRoute { + /// The key the write names still holds its incarnation. + Here, + /// The incarnation moved to this database. + Moved(DatabaseId), + /// The incarnation no longer exists: the write has nothing to mutate. + Superseded, +} + +/// Decide the route against the in-memory array mirror. An unstamped side +/// (`Hlc::ZERO`) matches by key alone. +pub(crate) fn route( + mirror: &ArrayCatalog, + tenant_id: TenantId, + database_id: DatabaseId, + name: &str, + incarnation: Hlc, +) -> CellRoute { + if let Some(entry) = mirror.lookup_by_name_in_database(tenant_id, database_id, name) + && (incarnation == Hlc::ZERO + || entry.incarnation == Hlc::ZERO + || entry.incarnation == incarnation) + { + return CellRoute::Here; + } + if incarnation == Hlc::ZERO { + return CellRoute::Superseded; + } + mirror + .all_entries() + .into_iter() + .find(|entry| { + entry.array_id.tenant_id == tenant_id + && entry.name == name + && entry.incarnation == incarnation + }) + .map_or(CellRoute::Superseded, |entry| { + CellRoute::Moved(entry.array_id.database_id) + }) +} + +/// The incarnation whose gate a write for `(tenant, database, name)` at +/// `incarnation` holds: the stamped one, or for an unstamped write the one +/// its key holds now. `None` when an unstamped write names no array. +pub(crate) fn gate_incarnation( + mirror: &ArrayCatalog, + tenant_id: TenantId, + database_id: DatabaseId, + name: &str, + incarnation: Hlc, +) -> Option { + if incarnation != Hlc::ZERO { + return Some(incarnation); + } + mirror + .lookup_by_name_in_database(tenant_id, database_id, name) + .map(|entry| entry.incarnation) +} + +/// Stamp the incarnation of the array a cell write names, as this proposer's +/// mirror holds it. Every other entry passes through. +pub(crate) fn stamp_incarnation(state: &SharedState, entry: &mut ReplicatedEntry) { + let (ReplicatedWrite::ArrayCellPut { + array, incarnation, .. + } + | ReplicatedWrite::ArrayCellDelete { + array, incarnation, .. + } + | ReplicatedWrite::ArrayOp { + array, incarnation, .. + }) = &mut entry.write + else { + return; + }; + let mirror = match state.array_catalog.read() { + Ok(mirror) => mirror, + Err(poisoned) => poisoned.into_inner(), + }; + if let Some(row) = mirror.lookup_by_name_in_database( + TenantId::new(entry.tenant_id), + DatabaseId::new(entry.database_id), + array, + ) { + *incarnation = row.incarnation; + } +} + +#[cfg(test)] +mod tests { + use nodedb_array::types::ArrayId; + + use super::*; + use crate::control::array_catalog::ArrayCatalogEntry; + + fn row(db: u64, name: &str, incarnation: Hlc) -> ArrayCatalogEntry { + ArrayCatalogEntry { + array_id: ArrayId::in_database(TenantId::new(1), DatabaseId::new(db), name), + name: name.to_string(), + schema_msgpack: vec![0x90], + schema_hash: 7, + created_at_ms: 0, + prefix_bits: 8, + audit_retain_ms: None, + minimum_audit_retain_ms: None, + modification_hlc: incarnation, + incarnation, + } + } + + fn at(mirror: &ArrayCatalog, db: u64, incarnation: Hlc) -> CellRoute { + route( + mirror, + TenantId::new(1), + DatabaseId::new(db), + "grid", + incarnation, + ) + } + + #[test] + fn a_write_follows_its_moved_incarnation() { + let mut mirror = ArrayCatalog::new(); + mirror + .register(row(4, "grid", Hlc::new(10, 0))) + .expect("register"); + assert_eq!(at(&mirror, 4, Hlc::new(10, 0)), CellRoute::Here); + assert_eq!( + at(&mirror, 3, Hlc::new(10, 0)), + CellRoute::Moved(DatabaseId::new(4)) + ); + } + + /// A write for a dropped incarnation is superseded, even when a later + /// incarnation of the same name holds the key. + #[test] + fn a_write_for_a_dropped_incarnation_is_superseded() { + let mut mirror = ArrayCatalog::new(); + assert_eq!(at(&mirror, 3, Hlc::new(10, 0)), CellRoute::Superseded); + mirror + .register(row(3, "grid", Hlc::new(20, 0))) + .expect("register"); + assert_eq!(at(&mirror, 3, Hlc::new(10, 0)), CellRoute::Superseded); + assert_eq!(at(&mirror, 3, Hlc::new(20, 0)), CellRoute::Here); + } + + #[test] + fn an_unstamped_write_matches_by_key() { + let mut mirror = ArrayCatalog::new(); + mirror + .register(row(3, "grid", Hlc::new(20, 0))) + .expect("register"); + assert_eq!(at(&mirror, 3, Hlc::ZERO), CellRoute::Here); + assert_eq!(at(&mirror, 5, Hlc::ZERO), CellRoute::Superseded); + } +} diff --git a/nodedb/src/control/array_catalog/ddl.rs b/nodedb/src/control/array_catalog/ddl.rs index dad6f908f..fd3c8e08c 100644 --- a/nodedb/src/control/array_catalog/ddl.rs +++ b/nodedb/src/control/array_catalog/ddl.rs @@ -1,229 +1,133 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Authorized Array DDL catalog transitions. +//! Array DDL through the replicated catalog. //! -//! SQL conversion is deliberately read-only. This module is called only by -//! the Control-Plane write funnel after a task owns an authorization capability -//! and immediately before the task is handed to the Data Plane. +//! `CREATE`, `ALTER`, and `DROP ARRAY` each propose one catalog entry through +//! the metadata group, built from committed state under the DDL preparation +//! lease. Every node applies it: the `_system.arrays` row, the in-memory +//! mirror, and the per-core open or drop in post-apply. With no metadata +//! group this node applies the same entry through the same apply path. +//! +//! SQL conversion only validates and emits the DDL task. The front doors hand +//! the authorized task here instead of dispatching it to a core. use nodedb_physical::physical_plan::{ArrayOp, MetaOp}; -use nodedb_types::config::retention::BitemporalRetention; +use nodedb_physical::physical_task::PhysicalTask; +use nodedb_types::Hlc; -use crate::bridge::envelope::{PhysicalPlan, Response}; +use crate::bridge::envelope::{Payload, PhysicalPlan, Response, Status}; use crate::control::array_catalog::ArrayCatalogEntry; -use crate::engine::bitemporal::BitemporalEngineKind; -use crate::engine::bitemporal::registry::Entry as RetentionEntry; -use crate::types::TraceId; -use crate::types::{DatabaseId, TenantId}; - -/// The exact state replaced by an authorized Array DDL transition. -/// -/// A token is created only after authorization and admission. The caller must -/// roll it back if dispatch cannot prove the Data Plane applied the task, and -/// finalize it after a successful response. DROP deliberately leaves its -/// durable row and surrogate bindings in place until finalization: recreating -/// bindings from a failed broadcast would otherwise be impossible. -pub(crate) struct AuthorizedDdlTransition { - kind: TransitionKind, - tenant_id: TenantId, - database_id: DatabaseId, - durable: Option, - in_memory: Option, - retention: Option, - /// The post-transition entry for CREATE/ALTER; needed to undo CREATE - /// without deriving identity from any mutable mirror. - transition_entry: Option, -} +use crate::control::catalog_entry::CatalogEntry; +use crate::control::metadata_proposer::propose_catalog_batch_async; +use crate::control::security::catalog::SystemCatalog; +use crate::control::server::shared::authorization::AuthorizedTask; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, Lsn, RequestId, TenantId}; -enum TransitionKind { - None, - Create, - Alter, - Drop { - array_id: nodedb_array::types::ArrayId, - }, +/// Whether `plan` is array DDL, which runs through [`run_authorized_array_ddl`] +/// and never reaches a core as a task. +pub(crate) fn is_array_ddl(plan: &PhysicalPlan) -> bool { + matches!( + plan, + PhysicalPlan::Array(ArrayOp::OpenArray { .. } | ArrayOp::DropArray { .. }) + | PhysicalPlan::Meta(MetaOp::AlterArray { .. }) + ) } -impl AuthorizedDdlTransition { - /// Restore all mirrors to their exact pre-transition state. This is - /// idempotent so a caller can safely invoke it for any failed response. - pub(crate) fn rollback(&self, state: &crate::control::state::SharedState) -> crate::Result<()> { - match &self.durable { - Some(entry) => super::persist::persist(state.credentials.catalog(), entry), - None => match &self.kind { - TransitionKind::Create => { - let entry = - self.transition_entry - .as_ref() - .ok_or_else(|| crate::Error::PlanError { - detail: "CREATE ARRAY rollback lost created entry".into(), - })?; - super::persist::remove(state.credentials.catalog(), &entry.array_id) - } - _ => Ok(()), - }, - } - .map_err(|e| crate::Error::PlanError { - detail: format!("array DDL rollback catalog: {e}"), - })?; - - let name = self - .in_memory - .as_ref() - .or(self.durable.as_ref()) - .or(self.transition_entry.as_ref()) - .map(|entry| entry.name.as_str()) - .or(match &self.kind { - TransitionKind::Drop { array_id } => Some(array_id.name.as_str()), - _ => None, - }); - if let Some(name) = name { - let mut catalog = state - .array_catalog - .write() - .map_err(|_| crate::Error::PlanError { - detail: "array catalog lock poisoned".into(), - })?; - catalog.unregister_in_database(self.tenant_id, self.database_id, name); - if let Some(entry) = &self.in_memory { - catalog - .register(entry.clone()) - .map_err(|e| crate::Error::PlanError { - detail: format!("array DDL rollback catalog mirror: {e}"), - })?; - } - } - - if let Some(name) = name { - state - .bitemporal_retention_registry - .unregister(self.database_id, self.tenant_id, name); - } - if let Some(retention) = &self.retention { - state - .bitemporal_retention_registry - .register( - retention.database_id, - retention.tenant_id, - retention.collection.clone(), - retention.engine, - retention.retention, - ) - .map_err(|e| crate::Error::PlanError { - detail: format!("array DDL rollback retention: {e}"), - })?; - } - Ok(()) - } - - /// Whether an enqueued CREATE/ALTER must be preserved if its response is - /// ambiguous. Their catalog transition may already match Data-Plane state, - /// so rolling it back would leave an opened ghost engine. - pub(crate) fn preserves_on_ambiguous_apply(&self) -> bool { - matches!(self.kind, TransitionKind::Create | TransitionKind::Alter) - } - - /// Complete irreversible work only after the Data Plane confirmed success. - pub(crate) fn finalize(&self, state: &crate::control::state::SharedState) -> crate::Result<()> { - if let TransitionKind::Drop { array_id } = &self.kind { - super::persist::remove_with_surrogates(state.credentials.catalog(), array_id).map_err( - |e| crate::Error::PlanError { - detail: format!( - "DROP ARRAY {}: catalog/surrogate delete: {e}", - array_id.name - ), - }, - )?; - } - Ok(()) - } +/// Propose the catalog entry an authorized array DDL task names, and answer +/// with the status the statement reports. The front doors call this with the +/// authorization they hold, so none of them unwraps it. +pub(crate) async fn run_authorized_array_ddl( + state: &SharedState, + authorized: AuthorizedTask, +) -> crate::Result { + run_trusted_array_ddl(state, authorized.into_physical_task()).await } -/// Apply the reversible catalog transition required by an authorized Array DDL -/// task. In-memory mirrors change only after their durable update commits. -/// Execute the all-core reversible DROP protocol after authorization. -/// -/// Catalog deletion is deferred until every core has staged its directory. On -/// any stage or finalize failure, compensation is broadcast to every core -/// before catalog mirrors are restored. Once deletion is durable, failed purge -/// is returned to the caller without recreating mirrors; tombstones then fence -/// later CREATE attempts until an operator/retry completes the purge. -pub(crate) async fn run_authorized_drop( - state: &crate::control::state::SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - plan: PhysicalPlan, - trace_id: TraceId, +/// [`run_authorized_array_ddl`] for a task whose authority comes from an +/// already admitted operation. +pub(crate) async fn run_trusted_array_ddl( + state: &SharedState, + task: PhysicalTask, ) -> crate::Result { - let array_id = match &plan { - PhysicalPlan::Array(ArrayOp::DropArray { array_id }) => array_id.clone(), - _ => { - return Err(crate::Error::PlanError { - detail: "array drop protocol received non-drop plan".into(), - }); - } - }; - let transition = apply_authorized_ddl(state, tenant_id, database_id, &plan)?; - let restore = PhysicalPlan::Array(ArrayOp::RestoreArrayDrop { - array_id: array_id.clone(), - }); - let stage = crate::control::server::broadcast::broadcast_count_to_all_cores( - state, + let PhysicalTask { tenant_id, database_id, plan, - trace_id, - "dropped", - ) - .await; - let response = match stage { - Ok(response) => response, - Err(error) => { - crate::control::server::broadcast::broadcast_count_to_all_cores( - state, - tenant_id, - database_id, - restore, - trace_id, - "restored", - ) - .await?; - transition.rollback(state)?; - return Err(error); + .. + } = task; + let (statement, key, payload) = match &plan { + PhysicalPlan::Array(ArrayOp::OpenArray { .. }) => ("CREATE ARRAY", "opened", None), + PhysicalPlan::Array(ArrayOp::DropArray { .. }) => ("DROP ARRAY", "dropped", None), + PhysicalPlan::Meta(MetaOp::AlterArray { + audit_retain_ms, .. + }) => { + // The acknowledgement a core answered ALTER ARRAY with: the new + // retention, or 0 when it is cleared. + let ack = audit_retain_ms + .flatten() + .and_then(|ms| u64::try_from(ms).ok()) + .unwrap_or(0); + ("ALTER ARRAY", "altered", Some(ack.to_le_bytes().to_vec())) + } + _ => { + return Err(crate::Error::PlanError { + detail: "array DDL path received a non-DDL plan".into(), + }); } }; - if let Err(error) = transition.finalize(state) { - crate::control::server::broadcast::broadcast_count_to_all_cores( - state, - tenant_id, - database_id, - restore, - trace_id, - "restored", - ) - .await?; - transition.rollback(state)?; - return Err(error); + // No catalog overlay replays uncommitted array DDL, so a transaction + // cannot read the array it created. + if crate::control::server::shared::session::ddl_buffer::is_active() { + return Err(crate::Error::NotInTransactionBlock { + statement: statement.into(), + }); } - crate::control::server::broadcast::broadcast_count_to_all_cores( - state, - tenant_id, - database_id, - PhysicalPlan::Array(ArrayOp::PurgeArrayDrop { array_id }), - trace_id, - "purged", - ) + propose_array_entries(state, |catalog| { + Ok(vec![entry_for(catalog, tenant_id, database_id, &plan)?]) + }) .await?; - Ok(response) + let payload = match payload { + Some(bytes) => bytes, + None => crate::data::executor::response_codec::encode_count(key, 1)?, + }; + Ok(Response { + request_id: RequestId::new(0), + status: Status::Ok, + attempt: 1, + partial: false, + payload: Payload::from_vec(payload), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }) +} + +/// Propose the entries `plan` builds from the committed catalog. +/// +/// `plan` reads committed state under the DDL preparation lease, and the +/// apply follows it under the same guard. Two statements never both find an identity absent and both +/// create it. Every core opens or drops the array before this returns. +pub(crate) async fn propose_array_entries( + state: &SharedState, + plan: impl FnOnce(&SystemCatalog) -> crate::Result>, +) -> crate::Result<()> { + propose_catalog_batch_async(state, plan).await?; + Ok(()) } -pub(crate) fn apply_authorized_ddl( - state: &crate::control::state::SharedState, +/// The entry `plan` makes of the committed catalog. CREATE requires the +/// identity absent, ALTER and DROP require it present. +fn entry_for( + catalog: &SystemCatalog, tenant_id: TenantId, database_id: DatabaseId, plan: &PhysicalPlan, -) -> crate::Result { - let (kind, target) = match plan { +) -> crate::Result { + let committed = |name: &str| catalog.get_array_in_database(tenant_id, database_id, name); + match plan { PhysicalPlan::Array(ArrayOp::OpenArray { array_id, schema_msgpack, @@ -231,9 +135,13 @@ pub(crate) fn apply_authorized_ddl( prefix_bits, audit_retain_ms, minimum_audit_retain_ms, - }) => ( - TransitionKind::Create, - Some(ArrayCatalogEntry { + }) => { + if committed(&array_id.name)?.is_some() { + return Err(crate::Error::PlanError { + detail: format!("CREATE ARRAY {}: already exists", array_id.name), + }); + } + Ok(CatalogEntry::PutArray(Box::new(ArrayCatalogEntry { array_id: array_id.clone(), name: array_id.name.clone(), schema_msgpack: schema_msgpack.clone(), @@ -242,259 +150,137 @@ pub(crate) fn apply_authorized_ddl( prefix_bits: *prefix_bits, audit_retain_ms: *audit_retain_ms, minimum_audit_retain_ms: *minimum_audit_retain_ms, - }), - ), - PhysicalPlan::Array(ArrayOp::DropArray { array_id }) => ( - TransitionKind::Drop { - array_id: array_id.clone(), - }, - None, - ), + // Frozen by the proposer's stamp. + modification_hlc: Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, + }))) + } PhysicalPlan::Meta(MetaOp::AlterArray { array_id, audit_retain_ms, minimum_audit_retain_ms, }) => { - let current = lookup_memory(state, tenant_id, database_id, array_id)?; - let updated = ArrayCatalogEntry { + let current = committed(array_id)?.ok_or_else(|| crate::Error::PlanError { + detail: format!("ALTER ARRAY {array_id}: not found"), + })?; + Ok(CatalogEntry::PutArray(Box::new(ArrayCatalogEntry { audit_retain_ms: audit_retain_ms.unwrap_or(current.audit_retain_ms), minimum_audit_retain_ms: minimum_audit_retain_ms .unwrap_or(current.minimum_audit_retain_ms), ..current - }; - (TransitionKind::Alter, Some(updated)) - } - _ => (TransitionKind::None, None), - }; - if matches!(&kind, TransitionKind::None) { - return Ok(AuthorizedDdlTransition { - kind, - tenant_id, - database_id, - durable: None, - in_memory: None, - retention: None, - transition_entry: None, - }); - } - - let identity = target - .as_ref() - .map(|entry| entry.array_id.clone()) - .or_else(|| match &kind { - TransitionKind::Drop { array_id } => Some(array_id.clone()), - _ => None, - }) - .ok_or_else(|| crate::Error::PlanError { - detail: "array DDL transition missing identity".into(), - })?; - let durable = state - .credentials - .catalog() - .get_array_in_database(tenant_id, database_id, &identity.name) - .map_err(|e| crate::Error::PlanError { - detail: format!("array DDL catalog read: {e}"), - })?; - let in_memory = state - .array_catalog - .read() - .map_err(|_| crate::Error::PlanError { - detail: "array catalog lock poisoned".into(), - })? - .lookup_by_id(&identity); - let retention = state - .bitemporal_retention_registry - .snapshot() - .into_iter() - .find(|entry| { - entry.database_id == database_id - && entry.tenant_id == tenant_id - && entry.collection == identity.name - }); - let token = AuthorizedDdlTransition { - kind, - tenant_id, - database_id, - durable, - in_memory, - retention, - transition_entry: target.clone(), - }; - - let apply = match &token.kind { - TransitionKind::Create => { - let entry = target.as_ref().ok_or_else(|| crate::Error::PlanError { - detail: "CREATE ARRAY transition missing entry".into(), - })?; - // CREATE is authorized only from complete catalog absence. Checking - // both copies prevents an incomplete DROP rollback (or a stale - // mirror) from reaching OpenArray and purging its reversible - // tombstone as if it were a finalized prior DROP. - if token.durable.is_some() || token.in_memory.is_some() { - return Err(crate::Error::PlanError { - detail: format!("CREATE ARRAY {}: already exists", entry.name), - }); - } - super::persist::persist(state.credentials.catalog(), entry).map_err(|e| { - crate::Error::PlanError { - detail: format!("CREATE ARRAY {}: catalog persist: {e}", entry.name), - } - })?; - install_memory(state, tenant_id, database_id, entry) - .and_then(|_| register_retention(state, database_id, entry)) - } - TransitionKind::Alter => { - let entry = target.as_ref().ok_or_else(|| crate::Error::PlanError { - detail: "ALTER ARRAY transition missing entry".into(), - })?; - if token.in_memory.is_none() { - return Err(crate::Error::PlanError { - detail: format!("ALTER ARRAY {}: not found", entry.name), - }); - } - super::persist::persist(state.credentials.catalog(), entry).map_err(|e| { - crate::Error::PlanError { - detail: format!("ALTER ARRAY {}: catalog persist: {e}", entry.name), - } - })?; - install_memory(state, tenant_id, database_id, entry) - .and_then(|_| register_retention(state, database_id, entry)) + }))) } - TransitionKind::Drop { array_id } => { - if token.in_memory.is_none() { + PhysicalPlan::Array(ArrayOp::DropArray { array_id }) => { + if committed(&array_id.name)?.is_none() { return Err(crate::Error::PlanError { detail: format!("DROP ARRAY {}: not found", array_id.name), }); } - remove_memory(state, tenant_id, database_id, &array_id.name)?; - state - .bitemporal_retention_registry - .unregister(database_id, tenant_id, &array_id.name); - Ok(()) - } - TransitionKind::None => Ok(()), - }; - if let Err(error) = apply { - let _ = token.rollback(state); - return Err(error); - } - Ok(token) -} - -fn lookup_memory( - state: &crate::control::state::SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - name: &str, -) -> crate::Result { - state - .array_catalog - .read() - .map_err(|_| crate::Error::PlanError { - detail: "array catalog lock poisoned".into(), - })? - .lookup_by_name_in_database(tenant_id, database_id, name) - .ok_or_else(|| crate::Error::PlanError { - detail: format!("ALTER ARRAY {name}: not found"), - }) -} - -fn install_memory( - state: &crate::control::state::SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - entry: &ArrayCatalogEntry, -) -> crate::Result<()> { - let mut catalog = state - .array_catalog - .write() - .map_err(|_| crate::Error::PlanError { - detail: "array catalog lock poisoned".into(), - })?; - catalog.unregister_in_database(tenant_id, database_id, &entry.name); - catalog - .register(entry.clone()) - .map_err(|e| crate::Error::PlanError { - detail: format!("array catalog register: {e}"), - }) -} - -fn remove_memory( - state: &crate::control::state::SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - name: &str, -) -> crate::Result<()> { - let mut catalog = state - .array_catalog - .write() - .map_err(|_| crate::Error::PlanError { - detail: "array catalog lock poisoned".into(), - })?; - catalog.unregister_in_database(tenant_id, database_id, name); - Ok(()) -} - -fn register_retention( - state: &crate::control::state::SharedState, - database_id: DatabaseId, - entry: &ArrayCatalogEntry, -) -> crate::Result<()> { - match entry.audit_retain_ms { - Some(audit_retain_ms) => state - .bitemporal_retention_registry - .register( - database_id, - entry.array_id.tenant_id, - &entry.name, - BitemporalEngineKind::Array, - BitemporalRetention { - data_retain_ms: 0, - audit_retain_ms: audit_retain_ms as u64, - minimum_audit_retain_ms: entry.minimum_audit_retain_ms.unwrap_or(0), - }, - ) - .map_err(|e| crate::Error::PlanError { - detail: format!("array retention register: {e}"), - }), - None => { - state.bitemporal_retention_registry.unregister( - database_id, - entry.array_id.tenant_id, - &entry.name, - ); - Ok(()) + Ok(CatalogEntry::DeleteArray { + database_id: database_id.as_u64(), + tenant_id: tenant_id.as_u64(), + name: array_id.name.clone(), + // Frozen by the proposer's stamp. + target_hlc: Hlc::ZERO, + moved_to: None, + }) } + _ => Err(crate::Error::PlanError { + detail: "array DDL path received a non-DDL plan".into(), + }), } } fn now_epoch_ms() -> i64 { std::time::SystemTime::now() .duration_since(std::time::UNIX_EPOCH) - .map(|duration| duration.as_millis() as i64) + .map(|duration| i64::try_from(duration.as_millis()).unwrap_or(i64::MAX)) .unwrap_or(0) } #[cfg(test)] mod tests { + use std::sync::Arc; + + use nodedb_array::types::ArrayId; + use super::*; + use crate::control::security::credential::CredentialStore; - fn token(kind: TransitionKind) -> AuthorizedDdlTransition { - AuthorizedDdlTransition { - kind, - tenant_id: TenantId::new(1), - database_id: DatabaseId::DEFAULT, - durable: None, - in_memory: None, - retention: None, - transition_entry: None, - } + fn catalog() -> (Arc, tempfile::TempDir) { + let tmp = tempfile::tempdir().expect("tmpdir"); + let store = Arc::new(CredentialStore::open(&tmp.path().join("system.redb")).expect("open")); + (store, tmp) + } + + fn open_plan(name: &str) -> PhysicalPlan { + PhysicalPlan::Array(ArrayOp::OpenArray { + array_id: ArrayId::in_database(TenantId::new(1), DatabaseId::DEFAULT, name), + schema_msgpack: vec![0x90], + schema_hash: 7, + prefix_bits: 8, + audit_retain_ms: None, + minimum_audit_retain_ms: None, + }) + } + + fn entry(catalog: &SystemCatalog, plan: &PhysicalPlan) -> crate::Result { + entry_for(catalog, TenantId::new(1), DatabaseId::DEFAULT, plan) } #[test] - fn only_create_and_alter_preserve_catalog_on_ambiguous_apply() { - assert!(token(TransitionKind::Create).preserves_on_ambiguous_apply()); - assert!(token(TransitionKind::Alter).preserves_on_ambiguous_apply()); - assert!(!token(TransitionKind::None).preserves_on_ambiguous_apply()); + fn create_requires_absence_and_drop_requires_presence() { + let (store, _tmp) = catalog(); + let catalog = store.catalog(); + let drop = PhysicalPlan::Array(ArrayOp::DropArray { + array_id: ArrayId::in_database(TenantId::new(1), DatabaseId::DEFAULT, "grid"), + }); + assert!(entry(catalog, &drop).is_err()); + + let CatalogEntry::PutArray(created) = entry(catalog, &open_plan("grid")).expect("create") + else { + unreachable!("CREATE ARRAY builds a put"); + }; + catalog.put_array(&created).expect("apply the create"); + assert!(entry(catalog, &open_plan("grid")).is_err()); + assert!(matches!( + entry(catalog, &drop).expect("drop"), + CatalogEntry::DeleteArray { moved_to: None, .. } + )); + } + + /// ALTER rewrites only the retention fields it names. + #[test] + fn alter_keeps_every_field_it_does_not_name() { + let (store, _tmp) = catalog(); + let catalog = store.catalog(); + let CatalogEntry::PutArray(created) = entry(catalog, &open_plan("grid")).expect("create") + else { + unreachable!("CREATE ARRAY builds a put"); + }; + catalog.put_array(&created).expect("apply the create"); + + let alter = PhysicalPlan::Meta(MetaOp::AlterArray { + array_id: "grid".to_string(), + audit_retain_ms: Some(Some(60_000)), + minimum_audit_retain_ms: None, + }); + let CatalogEntry::PutArray(altered) = entry(catalog, &alter).expect("alter") else { + unreachable!("ALTER ARRAY builds a put"); + }; + assert_eq!(altered.audit_retain_ms, Some(60_000)); + assert_eq!(altered.minimum_audit_retain_ms, None); + assert_eq!(altered.schema_hash, created.schema_hash); + assert_eq!(altered.array_id, created.array_id); + } + + #[test] + fn only_array_ddl_takes_the_catalog_path() { + assert!(is_array_ddl(&open_plan("grid"))); + assert!(!is_array_ddl(&PhysicalPlan::Array( + ArrayOp::PurgeArrayDrop { + array_id: ArrayId::in_database(TenantId::new(1), DatabaseId::DEFAULT, "grid"), + } + ))); } } diff --git a/nodedb/src/control/array_catalog/entry.rs b/nodedb/src/control/array_catalog/entry.rs index 9fb796b2a..b3fe8ce10 100644 --- a/nodedb/src/control/array_catalog/entry.rs +++ b/nodedb/src/control/array_catalog/entry.rs @@ -39,7 +39,7 @@ pub struct ArrayCatalogEntry { /// Existing rows without this field deserialize to 8. #[serde(default = "default_prefix_bits")] pub prefix_bits: u8, - /// Bitemporal audit retention window in milliseconds. Compaction may + /// Bitemporal audit retention window in milliseconds. Compaction can /// discard superseded tile versions older than `now - audit_retain_ms`. /// `None` means retain all versions forever (default, non-bitemporal). #[serde(default)] @@ -49,4 +49,22 @@ pub struct ArrayCatalogEntry { /// value. `None` means no floor (default). #[serde(default)] pub minimum_audit_retain_ms: Option, + /// Stamped at propose time on every replicated put. Fences a replayed + /// delete to the incarnation it targeted. + #[serde(default)] + pub modification_hlc: nodedb_types::Hlc, + /// The `modification_hlc` of the `PutArray` that created the array. ALTER + /// and MOVE TENANT keep it, so it names one array wherever the array + /// lives. `Hlc::ZERO` on a row no proposer stamped. + #[serde(default)] + pub incarnation: nodedb_types::Hlc, +} + +/// The target of a MOVE TENANT array rekey. +#[derive(Debug, Clone, Copy, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct ArrayMove { + /// The database the array moves to. + pub target_db_id: u64, + /// The moved tenant. With the source database it names the move's drain. + pub mover_tenant_id: u64, } diff --git a/nodedb/src/control/array_catalog/mod.rs b/nodedb/src/control/array_catalog/mod.rs index 5ec3b0c05..02124108b 100644 --- a/nodedb/src/control/array_catalog/mod.rs +++ b/nodedb/src/control/array_catalog/mod.rs @@ -7,10 +7,11 @@ //! The Data Plane never touches this directly — dispatch handlers //! consult the in-memory registry via the shared [`ArrayCatalogHandle`]. +pub(crate) mod cell_route; pub(crate) mod ddl; pub mod entry; pub mod persist; pub mod registry; -pub use entry::ArrayCatalogEntry; +pub use entry::{ArrayCatalogEntry, ArrayMove}; pub use registry::{ArrayCatalog, ArrayCatalogHandle}; diff --git a/nodedb/src/control/array_catalog/persist.rs b/nodedb/src/control/array_catalog/persist.rs index 4c5689505..5531ff00b 100644 --- a/nodedb/src/control/array_catalog/persist.rs +++ b/nodedb/src/control/array_catalog/persist.rs @@ -5,8 +5,8 @@ //! Mirrors the `trigger` / `sequence` registry pattern: the //! [`SystemCatalog`] owns the redb table and exposes typed read/write //! helpers (see `control/security/catalog/arrays.rs`). This module -//! provides the bulk `load_all` entry used at server startup and the -//! `persist` / `remove` wrappers used by DDL handlers. +//! provides the bulk `load_all` entry and the `persist` / `remove` +//! wrappers the `PutArray` / `DeleteArray` apply writes through. use nodedb_types::NodeDbError; @@ -26,6 +26,26 @@ pub fn load_all(catalog: &SystemCatalog) -> Result { Ok(reg) } +/// Register every entry of a boot-time `load_all_arrays` read into `reg`. +/// +/// Boot fails open: a read error or an entry that does not register logs a +/// warning and the rest boots. `register` refuses an id already present, so +/// a second load never duplicates an entry. +pub fn register_loaded(reg: &mut ArrayCatalog, loaded: crate::Result>) { + match loaded { + Ok(entries) => { + for entry in entries { + if let Err(e) = reg.register(entry) { + tracing::warn!(error = %e, "failed to register array at startup"); + } + } + } + Err(e) => { + tracing::warn!(error = %e, "failed to load _system.arrays at startup"); + } + } +} + /// Persist (or overwrite) a single entry. pub fn persist(catalog: &SystemCatalog, entry: &ArrayCatalogEntry) -> Result<(), NodeDbError> { catalog.put_array(entry).map_err(NodeDbError::from) @@ -42,6 +62,24 @@ pub fn remove( .map_err(NodeDbError::from) } +/// Remove the array under `array_id` and move every surrogate mapping to +/// `to`, in one durable transaction. +pub fn move_with_surrogates( + catalog: &SystemCatalog, + array_id: &nodedb_array::types::ArrayId, + to: nodedb_types::DatabaseId, +) -> Result<(), NodeDbError> { + catalog + .move_array_surrogates_in_database( + array_id.tenant_id, + array_id.database_id, + to, + &array_id.name, + ) + .map(|_existed| ()) + .map_err(NodeDbError::from) +} + /// Remove an array and every surrogate mapping in one durable transaction. pub fn remove_with_surrogates( catalog: &SystemCatalog, @@ -75,6 +113,8 @@ mod tests { prefix_bits: 8, audit_retain_ms: None, minimum_audit_retain_ms: None, + modification_hlc: nodedb_types::Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/array_catalog/registry.rs b/nodedb/src/control/array_catalog/registry.rs index fe1f424ad..56b9fd538 100644 --- a/nodedb/src/control/array_catalog/registry.rs +++ b/nodedb/src/control/array_catalog/registry.rs @@ -121,6 +121,8 @@ mod tests { prefix_bits: 8, audit_retain_ms: None, minimum_audit_retain_ms: None, + modification_hlc: nodedb_types::Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/array_sync/ack_registry.rs b/nodedb/src/control/array_sync/ack_registry.rs index 14c8e5d92..e552f1633 100644 --- a/nodedb/src/control/array_sync/ack_registry.rs +++ b/nodedb/src/control/array_sync/ack_registry.rs @@ -64,6 +64,12 @@ pub struct ArrayAckRegistry { cache: std::sync::RwLock>, } +impl crate::storage::RedbBacked for ArrayAckRegistry { + fn redb_database(&self) -> &redb::Database { + &self.db + } +} + impl ArrayAckRegistry { /// Open or create the ack registry database at `{data_dir}/array_sync/acks.redb`. pub fn open(data_dir: &Path) -> crate::Result> { diff --git a/nodedb/src/control/array_sync/catalog_register.rs b/nodedb/src/control/array_sync/catalog_register.rs index 0f6f2a4ab..0c912dfd8 100644 --- a/nodedb/src/control/array_sync/catalog_register.rs +++ b/nodedb/src/control/array_sync/catalog_register.rs @@ -1,66 +1,110 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Shared `array_catalog` registration helper. +//! Register a Lite-synced array in the replicated array catalog. //! -//! Split out of `raft_apply.rs` (which was pushing the 500-line file-size -//! limit) so both array-schema-import codepaths — the Raft-apply path -//! ([`crate::control::array_sync::raft_apply::apply_array_schema`], run on -//! every replica after Raft commit) and the single-node direct-import path -//! (`OriginArrayInbound::handle_schema`'s no-cluster branch, which never -//! goes through Raft) — converge on one registration routine instead of -//! duplicating (and risking drift between) the catalog-entry construction. +//! A synced schema lands in `array_sync_schemas` on the replicas of its data +//! group. Its catalog row goes through `PutArray` on the metadata group, the +//! entry SQL DDL uses, so every node can open the array and a SQL DROP or +//! re-CREATE of the same identity orders against it by the same incarnation +//! fence. A row that already exists stays: a re-synced schema never replaces +//! a definition. -use std::sync::Arc; +use nodedb_array::schema::ArraySchema; +use nodedb_array::sync::SchemaDoc; +use nodedb_array::sync::hlc::Hlc as SchemaHlc; +use nodedb_array::sync::replica_id::ReplicaId; +use nodedb_array::types::ArrayId; +use nodedb_types::sync::wire::array::{ArrayRejectMsg, ArrayRejectReason, ArraySchemaSyncMsg}; +use tracing::warn; +use super::inbound::OriginArrayInbound; +use super::reject::build_reject; use crate::control::array_catalog::entry::ArrayCatalogEntry; +use crate::control::catalog_entry::CatalogEntry; use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId}; -/// Register (or no-op if already present) an [`ArrayCatalogEntry`] for -/// `array` by reading back its just-imported schema from -/// `state.array_sync_schemas`, and persist it to the system catalog. -/// -/// Without this call, a synced array's schema lands in `array_sync_schemas` -/// but the array never becomes openable by the Data Plane -/// (`ensure_array_open` looks it up in `array_catalog`) and never becomes -/// visible to system-catalog introspection (`SHOW COLLECTIONS` merges in -/// `array_catalog::all_entries()`). -/// -/// The persist is not optional. `ArrayCatalog::register` mutates an in-memory -/// registry that is rebuilt at boot from the system catalog alone -/// (`array_catalog::persist::load_all`), so an entry that is only registered -/// vanishes on restart — taking with it the Data Plane's ability to open the -/// array, and therefore WAL replay's ability to restore its cells. Both the -/// Raft-apply and the single-node direct-import path report success to a caller -/// that treats it as durable, so both must make it durable here. +impl OriginArrayInbound { + /// Put the synced array in the replicated catalog, so every node can + /// open it. A failure reaches the sync sender: reporting + /// `SchemaImported` for an array no node can open hides the error. + pub(super) async fn register_in_catalog( + &self, + msg: &ArraySchemaSyncMsg, + remote_hlc: SchemaHlc, + ) -> Result<(), Option> { + register_array_catalog_entry( + self.shared(), + self.tenant_id(), + self.database_id(), + &msg.array, + &msg.snapshot_payload, + remote_hlc, + ) + .await + .map_err(|e| { + warn!(array = %msg.array, error = %e, "array_inbound: catalog registration failed"); + Some(build_reject( + &msg.array, + remote_hlc, + ArrayRejectReason::EngineRejected, + format!("catalog registration error: {e}"), + )) + }) + } +} + +/// Put the array a synced schema snapshot defines in the replicated catalog, +/// unless the catalog already holds the identity. /// -/// Returns `Ok(())` when an entry already exists or was freshly registered. -/// Every successful return guarantees that the entry is persisted, so a retry -/// repairs a row that was lost after an earlier in-memory registration. -/// Returns `Err` on a genuine registration failure (schema not readable back, -/// encode failure, or catalog write error). -pub(crate) fn register_array_catalog_entry( - state: &Arc, - tenant_id: crate::types::TenantId, - database_id: crate::types::DatabaseId, +/// Returns only once this node applied the entry, so the Data Plane here can +/// open the array for the cell ops that follow. +async fn register_array_catalog_entry( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, array: &str, + snapshot: &[u8], + schema_hlc: SchemaHlc, ) -> crate::Result<()> { - let schema = state - .array_sync_schemas - .to_array_schema_in_database(database_id, tenant_id.as_u64(), array) - .ok_or_else(|| { - crate::Error::Internal { - detail: format!( - "register_array_catalog_entry: to_array_schema returned None for '{array}' after import" - ), + let catalog = state.credentials.catalog(); + if catalog + .get_array_in_database(tenant_id, database_id, array)? + .is_some() + { + return Ok(()); + } + let entry = catalog_entry_of(tenant_id, database_id, array, snapshot, schema_hlc)?; + crate::control::array_catalog::ddl::propose_array_entries(state, |catalog| { + // Read again under the serialization `propose_array_entries` holds: + // a concurrent CREATE ARRAY or sync of the same identity wins. + if catalog + .get_array_in_database(tenant_id, database_id, array)? + .is_some() + { + return Ok(Vec::new()); } - })?; + Ok(vec![CatalogEntry::PutArray(Box::new(entry))]) + }) + .await +} + +/// The catalog row a schema snapshot defines. The snapshot decodes on its +/// own, so a node that does not replicate the array's data group builds the +/// same row. +fn catalog_entry_of( + tenant_id: TenantId, + database_id: DatabaseId, + array: &str, + snapshot: &[u8], + schema_hlc: SchemaHlc, +) -> crate::Result { + let schema = schema_of_snapshot(array, snapshot, schema_hlc)?; let schema_msgpack = zerompk::to_msgpack_vec(&schema).map_err(|e| crate::Error::Internal { - detail: format!("register_array_catalog_entry: schema_msgpack encode failed: {e}"), + detail: format!("synced array '{array}': schema encode: {e}"), })?; - - let array_id = nodedb_array::types::ArrayId::in_database(tenant_id, database_id, array); - let entry = ArrayCatalogEntry { - array_id: array_id.clone(), + Ok(ArrayCatalogEntry { + array_id: ArrayId::in_database(tenant_id, database_id, array), name: array.to_string(), schema_msgpack, schema_hash: 0, @@ -68,78 +112,77 @@ pub(crate) fn register_array_catalog_entry( prefix_bits: 8, audit_retain_ms: None, minimum_audit_retain_ms: None, - }; - // Persist first, including the duplicate/retry case. A previous attempt - // may have registered this entry in memory but failed before its durable - // write; treating that state as a no-op would make the array disappear on - // restart. Keep the lock through the write so concurrent callers cannot - // observe a durable/in-memory split. - let mut cat = state - .array_catalog - .write() - .unwrap_or_else(|p| p.into_inner()); - persist_then_register_if_missing(&mut cat, entry, |entry| { - crate::control::array_catalog::persist::persist(state.credentials.catalog(), entry).map_err( - |e| crate::Error::Internal { - detail: format!("register_array_catalog_entry: catalog persist failed: {e}"), - }, - ) + // Frozen by the proposer's stamp. + modification_hlc: nodedb_types::Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, }) } -fn persist_then_register_if_missing( - catalog: &mut crate::control::array_catalog::ArrayCatalog, - entry: ArrayCatalogEntry, - persist: F, -) -> crate::Result<()> -where - F: FnOnce(&ArrayCatalogEntry) -> crate::Result<()>, -{ - persist(&entry)?; - if catalog.lookup_by_id(&entry.array_id).is_none() { - catalog - .register(entry) - .map_err(|e| crate::Error::Internal { - detail: format!("register_array_catalog_entry: catalog register failed: {e}"), - })?; - } - Ok(()) +fn schema_of_snapshot( + array: &str, + snapshot: &[u8], + schema_hlc: SchemaHlc, +) -> crate::Result { + let mut doc = SchemaDoc::new(ReplicaId(0)); + doc.import_snapshot_replicated(snapshot, schema_hlc) + .and_then(|()| doc.to_schema()) + .map_err(|e| crate::Error::Internal { + detail: format!("synced array '{array}': schema snapshot decode: {e}"), + }) } #[cfg(test)] mod tests { use super::*; - use nodedb_array::types::ArrayId; - use nodedb_types::{DatabaseId, TenantId}; + use nodedb_array::schema::ArraySchemaBuilder; + use nodedb_array::schema::attr_spec::{AttrSpec, AttrType}; + use nodedb_array::schema::dim_spec::{DimSpec, DimType}; + use nodedb_array::types::domain::{Domain, DomainBound}; - fn entry() -> ArrayCatalogEntry { - ArrayCatalogEntry { - array_id: ArrayId::in_database(TenantId::new(9), DatabaseId::new(4), "synced"), - name: "synced".into(), - schema_msgpack: vec![0x80], - schema_hash: 0, - created_at_ms: 0, - prefix_bits: 8, - audit_retain_ms: None, - minimum_audit_retain_ms: None, - } + fn snapshot() -> (Vec, SchemaHlc) { + let schema = ArraySchemaBuilder::new("synced") + .dim(DimSpec::new( + "x", + DimType::Int64, + Domain::new(DomainBound::Int64(0), DomainBound::Int64(15)), + )) + .attr(AttrSpec::new("v", AttrType::Int64, true)) + .tile_extents(vec![4]) + .build() + .expect("schema"); + let hlc_gen = nodedb_array::sync::hlc::HlcGenerator::new(ReplicaId(7)); + let doc = SchemaDoc::from_schema(ReplicaId(7), &schema, &hlc_gen).expect("doc"); + (doc.export_snapshot().expect("snapshot"), doc.schema_hlc()) } + /// Two nodes that decode the same snapshot build the same row, so the + /// entry a sync proposes does not depend on which node received it. #[test] - fn retry_repairs_missing_durable_entry_already_in_memory() { - let entry = entry(); - let mut catalog = crate::control::array_catalog::ArrayCatalog::new(); - catalog.register(entry.clone()).unwrap(); - let mut persisted = false; - assert!(!persisted, "fixture begins with the durable row missing"); - - persist_then_register_if_missing(&mut catalog, entry.clone(), |_| { - persisted = true; - Ok(()) - }) - .unwrap(); + fn a_snapshot_decodes_to_one_catalog_row() { + let (bytes, hlc) = snapshot(); + let first = catalog_entry_of(TenantId::new(9), DatabaseId::new(4), "synced", &bytes, hlc) + .expect("decode"); + let second = catalog_entry_of(TenantId::new(9), DatabaseId::new(4), "synced", &bytes, hlc) + .expect("decode"); + assert_eq!(first, second); + assert_eq!( + first.array_id, + ArrayId::in_database(TenantId::new(9), DatabaseId::new(4), "synced") + ); + } - assert!(persisted); - assert_eq!(catalog.lookup_by_id(&entry.array_id), Some(entry)); + #[test] + fn a_corrupt_snapshot_is_refused() { + let (_, hlc) = snapshot(); + assert!( + catalog_entry_of( + TenantId::new(9), + DatabaseId::new(4), + "synced", + b"garbage", + hlc + ) + .is_err() + ); } } diff --git a/nodedb/src/control/array_sync/inbound.rs b/nodedb/src/control/array_sync/inbound.rs index d7fb00307..38bdbc208 100644 --- a/nodedb/src/control/array_sync/inbound.rs +++ b/nodedb/src/control/array_sync/inbound.rs @@ -14,9 +14,9 @@ //! 3. Schema-gate: reject if the array is unknown or the op's schema HLC is //! ahead of the local registry. //! 4. Build `ReplicatedWrite::ArrayOp` and serialize to a `ReplicatedEntry`. -//! 5. `raft_proposer(vshard_id, bytes)` — propose to the Raft group that -//! owns the destination vShard. -//! 6. Await Raft commit via `ProposeTracker`. +//! 5. The async Raft proposer proposes to the Raft group that owns the +//! destination vShard, forwarding to its leader. +//! 6. Await the Raft commit and this node's apply. //! 7. On commit: `distributed_applier` decodes the entry, dispatches it to //! the Data Plane, calls `record_applied`. //! @@ -70,7 +70,7 @@ pub enum InboundOutcome { Applied, /// The op was already present; no state was changed (idempotent replay). Idempotent, - /// The op was rejected; the caller should send `ArrayRejectMsg` back. + /// The op was rejected; the caller sends `ArrayRejectMsg` back. Rejected(ApplyRejection), /// A snapshot chunk was buffered; more chunks are expected. SnapshotPartial { received: u32, total: u32 }, @@ -80,7 +80,7 @@ pub enum InboundOutcome { SchemaImported, /// An ack was recorded into the ack-vector (GC frontier tracking). AckRecorded, - /// A catchup request was received and logged (serving deferred to Phase H). + /// A catchup request was received and logged (serving is not part of this outcome). CatchupRequested, } @@ -126,7 +126,7 @@ impl OriginArrayInbound { /// The tenant this inbound engine is bound to. `pub(crate)` so the sync /// session builder's guard test can assert the session tenant was threaded - /// through (see `session_handler::array`), not just the sibling propose + /// through (see `session_handler::array`), not only the sibling propose /// path. pub(crate) fn tenant_id(&self) -> TenantId { self.tenant_id @@ -139,10 +139,6 @@ impl OriginArrayInbound { self.database_id } - pub(super) fn identity(&self) -> &crate::control::security::identity::AuthenticatedIdentity { - &self.identity - } - pub(super) fn authorize_array( &self, array: &str, @@ -173,17 +169,9 @@ impl OriginArrayInbound { }) } - pub(super) fn engine(&self) -> &Arc { - &self.engine - } - pub(super) fn schemas(&self) -> &Arc { &self.schemas } - - pub(super) fn apply_observer(&self) -> Option<&Arc> { - self.apply_observer.as_ref() - } } impl OriginArrayInbound { @@ -307,7 +295,8 @@ impl OriginArrayInbound { /// Import an array schema CRDT snapshot from a Lite peer. /// /// Proposes the schema through Raft so it is applied atomically on all - /// replicas. Returns `SchemaImported` on successful commit. + /// replicas of its data group, then puts the array in the replicated + /// catalog. Returns `SchemaImported` once both committed. pub async fn handle_schema( &self, msg: &ArraySchemaSyncMsg, @@ -320,52 +309,6 @@ impl OriginArrayInbound { crate::control::security::identity::Permission::Write, )?; - // In single-node mode (no raft_proposer) fall back to direct import. - if self.shared.raft_proposer.get().is_none() { - let _authorized_scope = authorization.into_scope(); - if let Err(e) = self.schemas.import_snapshot_in_database( - self.database_id, - self.tenant_id.as_u64(), - &msg.array, - &msg.snapshot_payload, - remote_hlc, - ) { - warn!(array = %msg.array, error = %e, "array_inbound: schema import failed"); - return Err(Some(build_reject( - &msg.array, - remote_hlc, - ArrayRejectReason::EngineRejected, - format!("schema import error: {e}"), - ))); - } - // Single-node has no Raft applier to register the array_catalog - // entry, so this direct-import path must do it itself — mirrors - // `raft_apply::apply_array_schema`'s post-import registration. - // Without this, the array is importable but never openable by - // the Data Plane and never visible to `SHOW COLLECTIONS`. - // - // Unlike the Raft-apply path, this path's `Result` is still live - // and reaches the sync sender, so a registration failure is - // propagated rather than swallowed: reporting `SchemaImported` - // while the array stays unregistered would be a silent - // catalog-visibility inconsistency. - if let Err(e) = super::catalog_register::register_array_catalog_entry( - &self.shared, - self.tenant_id, - self.database_id, - &msg.array, - ) { - warn!(array = %msg.array, error = %e, "array_inbound: catalog registration failed"); - return Err(Some(build_reject( - &msg.array, - remote_hlc, - ArrayRejectReason::EngineRejected, - format!("catalog registration error: {e}"), - ))); - } - return Ok(InboundOutcome::SchemaImported); - } - let vshard_id = VShardId::new(array_vshard_for_name(&msg.array)); let write = ReplicatedWrite::ArraySchema { array: msg.array.clone(), @@ -379,20 +322,15 @@ impl OriginArrayInbound { write, ); - match self - .propose_and_await(entry, &msg.array, remote_hlc, authorization) - .await - { - Ok(()) => Ok(InboundOutcome::SchemaImported), - Err(Some(r)) => Err(Some(r)), - Err(None) => Err(None), - } + self.propose_and_await(entry, &msg.array, remote_hlc, authorization) + .await?; + self.register_in_catalog(msg, remote_hlc).await?; + Ok(InboundOutcome::SchemaImported) } // ─── Internal helpers ───────────────────────────────────────────────────── - /// Validate a decoded op, then route it through Raft (or directly to the - /// Data Plane in single-node mode) before returning. + /// Validate a decoded op, then route it through Raft before returning. /// /// # Fast-path idempotency /// @@ -462,23 +400,21 @@ impl OriginArrayInbound { return Ok(InboundOutcome::Idempotent); } - // 4. In single-node mode (no raft_proposer): apply directly to the - // Data Plane, matching the pre-Raft behaviour. This path is only - // exercised when the cluster stack has not been started (development, - // single-node Origin, unit tests without a raft setup). - if self.shared.raft_proposer.get().is_none() { - return self.apply_op_direct(op, provenance, authorization).await; - } - - // 5. Multi-node path: propose through Raft. + // 4. Propose through Raft. A put's cell surrogate is + // bound here, at the array's home, and every replica binds the same + // one when the entry applies. + let cell_surrogate = self.cell_surrogate(&op).await?; let hlc_bytes = op.header.hlc.to_bytes(); let write = ReplicatedWrite::ArrayOp { array: op.header.array.clone(), op_bytes: raw_op_bytes.to_vec(), + cell_surrogate: cell_surrogate.map(nodedb_types::Surrogate::as_u32), schema_hlc_bytes: hlc_bytes, provenance: provenance .as_ref() .and_then(|p| zerompk::to_msgpack_vec(p).ok()), + // Stamped with the floor, below. + incarnation: nodedb_types::Hlc::ZERO, }; let vshard = self.vshard_for_op(&op); let entry = ReplicatedEntry::new( diff --git a/nodedb/src/control/array_sync/inbound_advisory.rs b/nodedb/src/control/array_sync/inbound_advisory.rs index eb39170d5..a2264c137 100644 --- a/nodedb/src/control/array_sync/inbound_advisory.rs +++ b/nodedb/src/control/array_sync/inbound_advisory.rs @@ -121,7 +121,7 @@ mod tests { shared .permissions .grant( - &format!("collection:{TENANT}:{ARRAY}"), + &format!("collection:0:{TENANT}:{ARRAY}"), "user:advisory-user", permission, "test", diff --git a/nodedb/src/control/array_sync/inbound_propose.rs b/nodedb/src/control/array_sync/inbound_propose.rs index 1b741faac..a11379cdc 100644 --- a/nodedb/src/control/array_sync/inbound_propose.rs +++ b/nodedb/src/control/array_sync/inbound_propose.rs @@ -1,13 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Raft propose / single-node dispatch helpers for [`OriginArrayInbound`]. +//! Raft propose helpers for [`OriginArrayInbound`]. //! -//! These methods handle the multi-node Raft path (`propose_and_await`) and -//! the single-node fallback (`apply_op_direct`), plus the small helpers for -//! computing the destination vShard and converting an `ArrayOp` into a -//! Data Plane plan. - -use std::time::Duration; +//! These methods handle the Raft path (`propose_and_await`), plus the small +//! helpers for binding a put's cell surrogate and computing the destination +//! vShard. use nodedb_array::sync::hlc::Hlc; use nodedb_array::sync::op::ArrayOp; @@ -17,9 +14,8 @@ use tracing::{error, warn}; use crate::control::wal_replication::ReplicatedEntry; use crate::types::{TraceId, VShardId}; -use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; -use super::inbound::{InboundOutcome, OriginArrayInbound}; +use super::inbound::OriginArrayInbound; use super::reject::build_reject; impl OriginArrayInbound { @@ -51,7 +47,7 @@ impl OriginArrayInbound { /// distributed applier. Returns a reject on proposal or timeout failure. pub(super) async fn propose_and_await( &self, - entry: ReplicatedEntry, + mut entry: ReplicatedEntry, array: &str, hlc: Hlc, authorization: crate::control::server::shared::authorization::AuthorizedCollection, @@ -67,220 +63,86 @@ impl OriginArrayInbound { "array replicated entry scope mismatch".to_string(), ))); } + // Both proposers below take prebuilt bytes, so the floor is stamped + // here: replicas hold the write until they applied the array's DDL. + // The commit instant is stamped once, so every replica dates the + // write alike. + entry.write_hlc = self.shared().hlc_clock.now().wall_ns; + crate::control::wal_replication::stamp_metadata_floor(self.shared(), &mut entry); + crate::control::array_catalog::cell_route::stamp_incarnation(self.shared(), &mut entry); let vshard_id = entry.vshard_id; let idempotency_key = entry.idempotency_key; - let data = entry.to_bytes(); - - // Use the async proposer (with transparent leader forwarding + apply - // wait) when available. It returns the apply payload directly. - if let Some(async_proposer) = self.shared().async_raft_proposer().map(|a| a.as_ref()) { - let deadline = tokio::time::Instant::now() - + std::time::Duration::from_secs( - self.shared().tuning.network.default_deadline_secs, - ); - return match async_proposer(vshard_id, idempotency_key, data, deadline).await { - Ok((_payload, _committed_version)) => Ok(()), - Err(e) => { - warn!(array = %array, error = %e, "array_inbound: raft propose+apply failed"); - Err(Some(build_reject( - array, - hlc, - ArrayRejectReason::EngineRejected, - format!("raft propose failed: {e}"), - ))) - } - }; - } - - // Sync proposer fallback: register a tracker waiter after proposing. - // Used only in single-node mode where there is no forwarding race. - let tracker = match self.shared().propose_tracker.get().map(|a| a.as_ref()) { - Some(t) => t, - None => { - return Err(Some(build_reject( - array, - hlc, - ArrayRejectReason::EngineRejected, - "raft proposer not available".to_string(), - ))); - } - }; - - let proposer = match self.shared().raft_proposer.get().map(|a| a.as_ref()) { - Some(p) => p, - None => { - return Err(Some(build_reject( + let data = entry.encode().map_err(|e| { + error!(array = %array, error = %e, "array_inbound: replicated entry encode failed"); + Some(build_reject( + array, + hlc, + ArrayRejectReason::EngineRejected, + format!("replicated entry encode failed: {e}"), + )) + })?; + + // The async proposer forwards to the group leader and waits for the + // apply. It returns the apply payload directly. + let async_proposer = self + .shared() + .async_raft_proposer() + .map_err(|e| { + Some(build_reject( array, hlc, ArrayRejectReason::EngineRejected, - "raft proposer not available".to_string(), - ))); - } - }; - - let (group_id, log_index) = match proposer(vshard_id, data) { - Ok(pair) => pair, + format!("raft proposer not available: {e}"), + )) + })? + .as_ref(); + let deadline = tokio::time::Instant::now() + + std::time::Duration::from_secs(self.shared().tuning.network.default_deadline_secs); + match async_proposer(vshard_id, idempotency_key, data, deadline).await { + Ok((_payload, _committed_version)) => Ok(()), Err(e) => { - warn!(array = %array, error = %e, "array_inbound: raft propose failed"); - return Err(Some(build_reject( - array, - hlc, - ArrayRejectReason::EngineRejected, - format!("raft propose failed: {e}"), - ))); - } - }; - - let rx = tracker.register(group_id, log_index, idempotency_key); - let timeout_secs = self.shared().tuning.network.default_deadline_secs; - - match tokio::time::timeout(Duration::from_secs(timeout_secs), rx).await { - Ok(Ok(Ok(_payload))) => Ok(()), - Ok(Ok(Err(e))) => { - warn!(array = %array, error = %e, "array_inbound: raft commit apply error"); - Err(Some(build_reject( - array, - hlc, - ArrayRejectReason::EngineRejected, - format!("apply error: {e}"), - ))) - } - Ok(Err(_)) => Err(Some(build_reject( - array, - hlc, - ArrayRejectReason::EngineRejected, - "propose waiter channel closed".to_string(), - ))), - Err(_) => { - warn!(array = %array, "array_inbound: raft commit timeout"); + warn!(array = %array, error = %e, "array_inbound: raft propose+apply failed"); Err(Some(build_reject( array, hlc, ArrayRejectReason::EngineRejected, - format!("raft commit timeout (group={group_id} index={log_index})"), + format!("raft propose failed: {e}"), ))) } } } - /// Single-node fallback: dispatch directly to the Data Plane, bypassing - /// Raft. Used when `raft_proposer` is absent (development / unit tests). - pub(super) async fn apply_op_direct( + /// The bound surrogate of a put's cell, assigned at the array's home under + /// the `(array, zerompk(coord))` key the SQL insert path binds. A delete + /// or erasure names a coordinate, not a row, and carries none. + pub(super) async fn cell_surrogate( &self, - op: ArrayOp, - provenance: Option, - authorization: crate::control::server::shared::authorization::AuthorizedCollection, - ) -> Result> { - self.consume_array_write_authorization(authorization, &op.header.array, op.header.hlc)?; - let data_plane_op = self.op_to_data_plane_plan(&op, provenance)?; - let vshard = self.vshard_for_op(&op); - let task = PhysicalTask { - tenant_id: self.tenant_id(), - database_id: self.database_id(), - vshard_id: vshard, - plan: data_plane_op, - post_set_op: PostSetOp::None, - txn_id: None, - }; - let tenant_id = task.tenant_id; - let emitter = crate::control::security::audit::ArcAuditEmitter(std::sync::Arc::clone( - &self.shared().audit, - )); - let checked = match crate::control::server::shared::clone_write::intercept_and_authorize( - crate::control::server::shared::clone_write::InterceptAndAuthorizeParams { - state: self.shared(), - task, - identity: self.identity(), - tenant_id, - permissions: &self.shared().permissions, - roles: &self.shared().roles, - emitter: &emitter, - }, - ) - .await - .map_err(|error| { + op: &ArrayOp, + ) -> Result, Option> { + use nodedb_array::sync::op::ArrayOpKind; + if !matches!(op.kind, ArrayOpKind::Put) { + return Ok(None); + } + let reject = |detail: String| { Some(build_reject( &op.header.array, op.header.hlc, ArrayRejectReason::EngineRejected, - error.to_string(), + detail, )) - })? { - crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed(checked) => { - checked - } - // An array write is never a clone-write shape. - crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(_) => { - return Err(Some(build_reject( - &op.header.array, - op.header.hlc, - ArrayRejectReason::EngineRejected, - "unexpected clone-write interception of an array task".to_string(), - ))); - } }; - - // This is the only durability this op will ever get: nothing appended a - // redo for it upstream (no Raft entry on this path), and the reply below - // tells the peer the op is applied — after which it never re-sends. The - // write must therefore own its record: the funnel appends it under the - // write-admission guard, stamps the minted LSN into the plan so the tile - // version matches what replay will stamp from the record header, and - // holds the ack behind the durable-at-ack barrier. - let response = - match crate::control::server::dispatch_utils::dispatch_authorized_autocommit_write_with_source( - self.shared(), - checked, - TraceId::ZERO, - crate::event::EventSource::CrdtSync, - ) - .await - { - Ok(r) => r, - Err(e) => { - warn!( - array = %op.header.array, - error = %e, - "array_inbound: Data Plane dispatch failed" - ); - return Err(Some(build_reject( - &op.header.array, - op.header.hlc, - ArrayRejectReason::EngineRejected, - format!("dispatch error: {e}"), - ))); - } - }; - - // Surface fenced / error responses as rejects so the caller sends - // ArrayRejectMsg rather than a silent success. - if response.status == crate::bridge::envelope::Status::Error { - return Err(Some(build_reject( - &op.header.array, - op.header.hlc, - ArrayRejectReason::EngineRejected, - "Data Plane rejected (fenced or error)".to_string(), - ))); - } - - if let Err(e) = self.engine().record_applied_in_database( - self.database_id(), - self.tenant_id().as_u64(), - &op, - ) { - error!( - array = %op.header.array, - hlc = ?op.header.hlc, - error = %e, - "array_inbound: op applied but op-log append failed (replay may re-apply)" - ); - } - - if let Some(observer) = self.apply_observer() { - observer.on_op_applied(&op); - } - - Ok(InboundOutcome::Applied) + let pk = zerompk::to_msgpack_vec(&op.coord) + .map_err(|e| reject(format!("array cell coord encode: {e}")))?; + let surrogate = crate::control::server::surrogate_exchange::assign_surrogate_routed( + self.shared(), + nodedb_types::CollectionKey::from_bare(self.database_id(), &op.header.array), + self.tenant_id(), + &pk, + TraceId::ZERO, + ) + .await + .map_err(|e| reject(format!("array cell surrogate assign: {e}")))?; + Ok(Some(surrogate)) } /// Compute the vShard that owns this op's tile. @@ -321,71 +183,4 @@ impl OriginArrayInbound { &tile_extents, )) } - - /// Convert a decoded `ArrayOp` (from sync) into a `PhysicalPlan::Array` variant. - /// - /// `provenance` is `Some` on the single-node sync path; `None` for snapshot - /// replay and any non-sync caller. The value is threaded into - /// `DataArrayOp::Put/Delete.provenance` so the Data Plane can run epoch - /// fencing and advance the HWM. - pub(super) fn op_to_data_plane_plan( - &self, - op: &ArrayOp, - provenance: Option, - ) -> Result> { - use nodedb_array::sync::op::ArrayOpKind; - use nodedb_physical::physical_plan::ArrayOp as DataArrayOp; - - let array_id = nodedb_array::types::ArrayId::in_database( - self.tenant_id(), - self.database_id(), - &op.header.array, - ); - - let data_op = match op.kind { - ArrayOpKind::Put => { - let cells = vec![crate::engine::array::wal::ArrayPutCell { - coord: op.coord.clone(), - attrs: op.attrs.clone().unwrap_or_default(), - surrogate: nodedb_types::Surrogate::ZERO, - system_from_ms: op.header.system_from_ms, - valid_from_ms: op.header.valid_from_ms, - valid_until_ms: op.header.valid_until_ms, - }]; - let cells_msgpack = zerompk::to_msgpack_vec(&cells).map_err(|e| { - Some(build_reject( - &op.header.array, - op.header.hlc, - ArrayRejectReason::ShapeInvalid, - format!("cells encode: {e}"), - )) - })?; - DataArrayOp::Put { - array_id, - cells_msgpack, - wal_lsn: 0, - provenance, - } - } - ArrayOpKind::Delete | ArrayOpKind::Erase => { - let coords = vec![op.coord.clone()]; - let coords_msgpack = zerompk::to_msgpack_vec(&coords).map_err(|e| { - Some(build_reject( - &op.header.array, - op.header.hlc, - ArrayRejectReason::ShapeInvalid, - format!("coords encode: {e}"), - )) - })?; - DataArrayOp::Delete { - array_id, - coords_msgpack, - wal_lsn: 0, - provenance, - } - } - }; - - Ok(crate::bridge::envelope::PhysicalPlan::Array(data_op)) - } } diff --git a/nodedb/src/control/array_sync/op_log.rs b/nodedb/src/control/array_sync/op_log.rs index 3e0a852b6..1fbac4adb 100644 --- a/nodedb/src/control/array_sync/op_log.rs +++ b/nodedb/src/control/array_sync/op_log.rs @@ -66,6 +66,12 @@ pub struct OriginOpLog { db: Arc, } +impl crate::storage::RedbBacked for OriginOpLog { + fn redb_database(&self) -> &redb::Database { + &self.db + } +} + impl OriginOpLog { /// Open or create the op-log database at `{data_dir}/array_sync/op_log.redb`. pub fn open(data_dir: &Path) -> crate::Result { diff --git a/nodedb/src/control/array_sync/outbound/subscriber_state.rs b/nodedb/src/control/array_sync/outbound/subscriber_state.rs index e7dbe96a3..eb3409c85 100644 --- a/nodedb/src/control/array_sync/outbound/subscriber_state.rs +++ b/nodedb/src/control/array_sync/outbound/subscriber_state.rs @@ -96,6 +96,12 @@ pub struct SubscriberMap { store: Arc, } +impl crate::storage::RedbBacked for SubscriberMap { + fn redb_database(&self) -> &redb::Database { + &self.store.db + } +} + impl SubscriberMap { /// Construct from a pre-loaded backing store. pub fn new(store: Arc) -> Self { @@ -107,7 +113,7 @@ impl SubscriberMap { /// Register a new subscriber (or restore an existing one from the store). /// - /// Returns the current `ArraySubscriberState` (may have a non-ZERO + /// Returns the current `ArraySubscriberState` (can have a non-ZERO /// `last_pushed_hlc` if the subscriber previously connected). pub fn register( &self, diff --git a/nodedb/src/control/array_sync/raft_apply/cell.rs b/nodedb/src/control/array_sync/raft_apply/cell.rs index cb9e0f1a4..623f7c7ab 100644 --- a/nodedb/src/control/array_sync/raft_apply/cell.rs +++ b/nodedb/src/control/array_sync/raft_apply/cell.rs @@ -48,6 +48,9 @@ pub(crate) struct ArrayCellTarget { /// forwarded so this replica's redo record carries it verbatim rather than /// re-reading a clock that has since moved. pub resolved_now_ms: Option, + /// The source the proposer stamped on the entry: `User` for a client + /// write, `Restore` for a cell a RESTORE re-issued. + pub event_source: crate::event::EventSource, } /// Apply a decoded array cell write plan (`PhysicalPlan::Array(Put | Delete)`) @@ -75,6 +78,7 @@ pub(crate) async fn apply_array_cell_write( database_id, vshard, resolved_now_ms, + event_source, } = target; // The caller (the distributed apply loop) only routes decoded @@ -115,13 +119,16 @@ pub(crate) async fn apply_array_cell_write( vshard, plan, // Cluster mode has exactly one write-apply path — this one — so a - // committed user write keeps the `User` source its proposer had, - // exactly as the generic committed-write branch does. - event_source: crate::event::EventSource::User, + // committed write keeps the source its proposer stamped, exactly + // as the generic committed-write branch does. A restored cell + // stays `Restore` and raises no user write mark. + event_source, resolved_now_ms, apply_key: applied_key, commit_hlc, op_label: "array cell write", + group_id, + log_index, }, ) .await; diff --git a/nodedb/src/control/array_sync/raft_apply/common.rs b/nodedb/src/control/array_sync/raft_apply/common.rs index 1b945ab24..4e9635db1 100644 --- a/nodedb/src/control/array_sync/raft_apply/common.rs +++ b/nodedb/src/control/array_sync/raft_apply/common.rs @@ -41,6 +41,16 @@ impl AppliedPosition { pub(crate) fn carried_commit_hlc(&self) -> Option { (self.commit_hlc != 0).then_some(self.commit_hlc) } + + /// The entry's Raft log position for a write of `vshard`, which + /// positions its change events. + pub(crate) fn change_position( + &self, + state: &SharedState, + vshard: u32, + ) -> crate::event::cdc::position::ReplicatedPosition { + crate::event::cdc::position::entry_position(state, vshard, self.group_id, self.log_index) + } } /// One committed array write, ready for the Control-Plane write funnel. @@ -61,6 +71,10 @@ pub(super) struct ArrayWriteSubmit { pub commit_hlc: Option, /// Contextual label for the error surfaced to the propose waiter. pub op_label: &'static str, + /// The committed entry the write applies. Its change events stage under + /// it until the apply loop settles it. + pub group_id: u64, + pub log_index: u64, } /// Submit a committed array write through the shared Control-Plane write funnel @@ -91,6 +105,8 @@ pub(super) async fn submit_array_write( apply_key, commit_hlc, op_label, + group_id, + log_index, } = params; let outcome = submit_write( @@ -110,21 +126,23 @@ pub(super) async fn submit_array_write( // committed plan: the proposer's LSN is deliberately not carried on // the wire, and the array engine's tile state has no other // durability path than this record's replay. + // An array write emits no Data-Plane change event to position. durability: WalDurability::AppendHere { now_override: resolved_now_ms, apply_key, commit_hlc, + change_position: None, }, // Raft committed this entry at a fixed log index and every replica // applies it in that order; re-entering the write-admission gate - // would re-decide an ordering that is already final. + // re-decides an ordering that is already final. ordering: WriteOrdering::AlreadyOrdered, - // An `ArrayOp::Put` / `Delete` does yield change metadata, but this - // apply path runs on EVERY replica of the committed entry — the - // node that proposed it owns the single publish. Emitting here - // would give each subscriber one copy per replica plus a NOTIFY - // fan-out from each. See [`ChangeFeedOwner`]. - change_feed: ChangeFeedOwner::Unowned, + // Every replica stages the write's change events under the entry, + // and publishes them at its log position once the entry settles. + change_feed: ChangeFeedOwner::Replicated { + group_id, + log_index, + }, }, ) .await?; @@ -277,6 +295,7 @@ pub(super) fn build_array_request( txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: crate::bridge::envelope::Admission::Exempt( crate::bridge::envelope::ExemptReason::AlreadyOrdered, ), diff --git a/nodedb/src/control/array_sync/raft_apply/op.rs b/nodedb/src/control/array_sync/raft_apply/op.rs index 500f0380c..b01487b49 100644 --- a/nodedb/src/control/array_sync/raft_apply/op.rs +++ b/nodedb/src/control/array_sync/raft_apply/op.rs @@ -28,6 +28,7 @@ pub(crate) async fn apply_array_op( pos: AppliedPosition, target: ArrayOpTarget<'_>, op_bytes: &[u8], + cell_surrogate: Option, provenance_bytes: Option<&[u8]>, ) -> bool { let ArrayOpTarget { @@ -131,10 +132,30 @@ pub(crate) async fn apply_array_op( let data_op = match op.kind { ArrayOpKind::Put => { + // A put carries the cell surrogate its proposer bound at the + // array's home. An entry without one names no row and is refused. + let Some(surrogate) = cell_surrogate.map(nodedb_types::Surrogate::new) else { + warn!( + group_id, index = log_index, array = %array, + "apply_array_op: put carries no cell surrogate" + ); + tracker.complete( + group_id, + log_index, + applied_key, + Err(crate::Error::Internal { + detail: format!( + "array put into '{array}' carries no cell surrogate; every put \ + carries the one its proposer bound" + ), + }), + ); + return false; + }; let cells = vec![crate::engine::array::wal::ArrayPutCell { coord: op.coord.clone(), attrs: op.attrs.clone().unwrap_or_default(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate, system_from_ms: op.header.system_from_ms, valid_from_ms: op.header.valid_from_ms, valid_until_ms: op.header.valid_until_ms, @@ -162,6 +183,7 @@ pub(crate) async fn apply_array_op( // the one replay reconstructs from the record header. wal_lsn: 0, provenance: provenance.clone(), + vshard_id: vshard.as_u32(), } } ArrayOpKind::Delete | ArrayOpKind::Erase => { @@ -187,11 +209,27 @@ pub(crate) async fn apply_array_op( // Minted and stamped by the funnel — see the `Put` arm above. wal_lsn: 0, provenance, + vshard_id: vshard.as_u32(), } } }; - let plan = crate::bridge::envelope::PhysicalPlan::Array(data_op); + let mut plan = crate::bridge::envelope::PhysicalPlan::Array(data_op); + // Install the put's carried cell surrogate in this replica's catalog, + // first-wins, so a later SQL read or write of the cell resolves it. + if let Err(e) = crate::control::surrogate::bind_plan_identities( + &state.surrogate_assigner, + database_id, + tenant_id, + &mut plan, + ) { + warn!( + group_id, index = log_index, array = %array, error = %e, + "apply_array_op: cell surrogate bind failed" + ); + tracker.complete(group_id, log_index, applied_key, Err(e)); + return false; + } let result = submit_array_write( state, ArrayWriteSubmit { @@ -206,6 +244,8 @@ pub(crate) async fn apply_array_op( apply_key: applied_key, commit_hlc, op_label: "array op", + group_id, + log_index, }, ) .await; diff --git a/nodedb/src/control/array_sync/raft_apply/schema.rs b/nodedb/src/control/array_sync/raft_apply/schema.rs index f5db9f3f8..e05a6d2f2 100644 --- a/nodedb/src/control/array_sync/raft_apply/schema.rs +++ b/nodedb/src/control/array_sync/raft_apply/schema.rs @@ -20,20 +20,19 @@ pub(crate) struct ArraySchemaPayload<'a> { pub schema_hlc_bytes: [u8; 18], } -/// Apply a committed `ArraySchema` entry on the local node. +/// Apply a committed `ArraySchema` entry on the local node: import the Loro +/// snapshot into the local `OriginSchemaRegistry`. /// -/// 1. Imports the Loro snapshot into the local `OriginSchemaRegistry`. -/// 2. Decodes the `ArraySchema` and registers + persists an `ArrayCatalogEntry` -/// so the Data Plane can open the array when a subsequent `ArrayOp` arrives. -/// This is the canonical DDL propagation path for followers: the Raft -/// `ArraySchema` entry is the single source of truth — no out-of-band -/// catalog registration is needed. +/// The array's catalog row is not written here. The receiving node proposes +/// it as a `PutArray` on the metadata group, so every node holds it, not only +/// the replicas of this data group. /// -/// This is the one apply path that mints no WAL redo record and needs none: both -/// steps land in fsync-committed redb transactions before it returns, which is -/// the same fact the durable applied floor asserts for every other branch. +/// This is the one apply path that mints no WAL redo record and needs none: +/// the import lands in a fsync-committed redb transaction before it returns, +/// which is the same fact the durable applied floor asserts for every other +/// branch. /// -/// Returns `true` when BOTH committed durably, `false` otherwise. The caller +/// Returns `true` when the import committed durably, `false` otherwise. The caller /// uses this to gate the durable applied floor and Raft log compaction. pub(crate) fn apply_array_schema( state: &Arc, @@ -86,42 +85,6 @@ pub(crate) fn apply_array_schema( return false; } - // Decode the ArraySchema from the just-imported Loro document, register it - // in the array catalog, and persist it, so the Data Plane can open the array - // on this node now and after a restart. Shared with the single-node - // direct-import path in `inbound.rs` via - // `catalog_register::register_array_catalog_entry` so both codepaths - // converge on the same catalog-visibility guarantee. - // - // A failure here fails the whole apply. The caller advances this group's - // DURABLE applied floor on our `true`, and the next boot resumes Raft - // delivery above that floor — so reporting success without the catalog entry - // would leave the array permanently unopenable, with the one entry that - // could repair it excluded from redelivery. Returning `false` leaves the - // floor behind and keeps the entry replayable; both steps are idempotent - // (the import re-imports the same committed HLC, the register no-ops on an - // existing entry), so a redelivery converges. - if let Err(e) = crate::control::array_sync::catalog_register::register_array_catalog_entry( - state, - tenant_id, - database_id, - array, - ) { - warn!( - group_id, index = log_index, array = %array, error = %e, - "apply_array_schema: register_array_catalog_entry failed" - ); - tracker.complete( - group_id, - log_index, - applied_key, - Err(crate::Error::Internal { - detail: format!("array catalog register: {e}"), - }), - ); - return false; - } - // A schema import touches the registries, not a Data-Plane collection, so it // publishes no per-collection write-version. tracker.complete( diff --git a/nodedb/src/control/array_sync/schema_registry.rs b/nodedb/src/control/array_sync/schema_registry.rs index 324ffd010..95e218deb 100644 --- a/nodedb/src/control/array_sync/schema_registry.rs +++ b/nodedb/src/control/array_sync/schema_registry.rs @@ -42,6 +42,12 @@ pub struct OriginSchemaRegistry { docs: Mutex>, } +impl crate::storage::RedbBacked for OriginSchemaRegistry { + fn redb_database(&self) -> &redb::Database { + &self.db + } +} + impl OriginSchemaRegistry { /// Open or create the schema registry table in `db` and cold-load its /// entries. diff --git a/nodedb/src/control/array_sync/snapshot_store.rs b/nodedb/src/control/array_sync/snapshot_store.rs index 5d0410e0f..da98fb367 100644 --- a/nodedb/src/control/array_sync/snapshot_store.rs +++ b/nodedb/src/control/array_sync/snapshot_store.rs @@ -71,6 +71,12 @@ pub struct OriginSnapshotStore { db: Arc, } +impl crate::storage::RedbBacked for OriginSnapshotStore { + fn redb_database(&self) -> &redb::Database { + &self.db + } +} + impl OriginSnapshotStore { /// Open or create the snapshot database at `{data_dir}/array_sync/snapshots.redb`. pub fn open(data_dir: &Path) -> crate::Result> { diff --git a/nodedb/src/control/backup/bind_capture.rs b/nodedb/src/control/backup/bind_capture.rs new file mode 100644 index 000000000..238806f8b --- /dev/null +++ b/nodedb/src/control/backup/bind_capture.rs @@ -0,0 +1,225 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Capture a tenant's PK→surrogate binds from the nodes that hold them. +//! +//! A bind lives on its collection's home vShard and, in a collection that +//! holds edges, on the key vShard of its key (see [`RecordHomes`]). With more +//! nodes than the replication factor no one node holds every bind. The capture +//! asks the source node of each vShard, the node a backup reads that vShard's +//! data from, for the binds with a home among its vShards, over the same +//! `ExecuteRequest` path as the per-node data snapshot. A bind two sources +//! return is kept once. + +use std::collections::{BTreeMap, HashSet}; + +use futures::future::join_all; +use nodedb_physical::physical_plan::ClusterEventOp; +use nodedb_types::{CollectionKey, DatabaseId}; + +use crate::Error; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::state::SharedState; +use crate::types::{HomedRecord, RecordHomes, SurrogateBindEntry, TenantId}; + +use super::node_snapshot::{is_self, snapshot_remote}; +use super::orchestrator::source_assignment; + +/// This node's binds of `tenant_id`'s `collections` in `database_id` with a +/// home among `vshards`. +pub(crate) fn local_binds( + state: &SharedState, + tenant_id: u64, + database_id: DatabaseId, + vshards: &HashSet, + collections: &[String], +) -> Result, Error> { + let catalog = state.credentials.catalog(); + let mut binds = Vec::new(); + for name in collections { + let holds_edges = catalog + .get_collection(database_id, tenant_id, name)? + .is_some_and(|coll| coll.has_implicit_edges); + let key = CollectionKey::from_bare(database_id, name); + // Without edges every bind homes on the collection's vShard. + if !holds_edges && !vshards.contains(&key.vshard().as_u32()) { + continue; + } + for (pk, surrogate) in + catalog.scan_surrogates_for_collection(key, TenantId::new(tenant_id))? + { + let homes = RecordHomes::of(HomedRecord::Bind { + collection: key, + key: &pk, + holds_edges, + }); + if homes.intersects(vshards) { + binds.push(SurrogateBindEntry { + database_id: database_id.as_u64(), + tenant_id, + collection: name.clone(), + pk, + surrogate: surrogate.as_u32(), + }); + } + } + } + Ok(binds) +} + +/// Every bind of `tenant_id`'s `collections` in `database_id`, each from a +/// source node that holds it. Any node error fails the capture. +pub(crate) async fn capture_binds( + state: &SharedState, + tenant_id: u64, + database_id: DatabaseId, + collections: &[String], +) -> Result, Error> { + let answers = join_all(source_assignment(state)?.into_iter().map( + |(node_id, vshards)| async move { + let binds = if is_self(state, node_id) { + local_binds(state, tenant_id, database_id, &vshards, collections)? + } else { + remote_binds( + state, + node_id, + tenant_id, + database_id, + &vshards, + collections, + ) + .await? + }; + Ok::<_, Error>((vshards, binds)) + }, + )) + .await; + let mut parts = Vec::with_capacity(answers.len()); + for answer in answers { + parts.push(answer?); + } + Ok(merge_binds(parts)) +} + +/// Ask `node_id` for its binds with a home among `vshards`. +async fn remote_binds( + state: &SharedState, + node_id: u64, + tenant_id: u64, + database_id: DatabaseId, + vshards: &HashSet, + collections: &[String], +) -> Result, Error> { + let mut vshards: Vec = vshards.iter().copied().collect(); + vshards.sort_unstable(); + let plan = PhysicalPlan::ClusterEvent(ClusterEventOp::SurrogateBinds { + tenant_id, + database_id, + vshards, + collections: collections.to_vec(), + }); + let body = snapshot_remote(state, node_id, tenant_id, database_id, &plan).await?; + decode_binds(&body, node_id) +} + +/// Encode binds for the wire. +pub(crate) fn encode_binds(binds: Vec) -> Result, Error> { + zerompk::to_msgpack_vec(&binds).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("surrogate bind capture: encode: {e}"), + }) +} + +fn decode_binds(bytes: &[u8], node_id: u64) -> Result, Error> { + zerompk::from_msgpack(bytes).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("surrogate bind capture: decode the answer of node {node_id}: {e}"), + }) +} + +/// Merge the binds each source returned, keeping one per key. +/// +/// A source whose vShards hold the bind's collection home wins: a row's +/// identity is bound there. A different surrogate from another source is a +/// split identity, logged and dropped. +fn merge_binds(parts: Vec<(HashSet, Vec)>) -> Vec { + type BindKey = (u64, u64, String, Vec); + let mut merged: BTreeMap = BTreeMap::new(); + let mut deferred = Vec::new(); + for (vshards, binds) in parts { + for bind in binds { + let home = + CollectionKey::from_bare(DatabaseId::new(bind.database_id), &bind.collection) + .vshard() + .as_u32(); + if vshards.contains(&home) { + let key = bind_key(&bind); + merged.insert(key, bind); + } else { + deferred.push(bind); + } + } + } + for bind in deferred { + let key = bind_key(&bind); + match merged.get(&key) { + Some(kept) if kept.surrogate != bind.surrogate => { + tracing::warn!( + collection = %bind.collection, + kept = kept.surrogate, + dropped = bind.surrogate, + "surrogate bind capture: two sources bind one key to different surrogates" + ); + } + Some(_) => {} + None => { + merged.insert(key, bind); + } + } + } + merged.into_values().collect() +} + +fn bind_key(bind: &SurrogateBindEntry) -> (u64, u64, String, Vec) { + ( + bind.database_id, + bind.tenant_id, + bind.collection.clone(), + bind.pk.clone(), + ) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn bind(collection: &str, pk: &str, surrogate: u32) -> SurrogateBindEntry { + SurrogateBindEntry { + database_id: 0, + tenant_id: 1, + collection: collection.into(), + pk: pk.as_bytes().to_vec(), + surrogate, + } + } + + #[test] + fn a_bind_two_sources_return_is_kept_once() { + let home = CollectionKey::from_bare(DatabaseId::DEFAULT, "g") + .vshard() + .as_u32(); + let owner: HashSet = [home].into_iter().collect(); + let other: HashSet = [(home + 1) % 1024].into_iter().collect(); + let merged = merge_binds(vec![ + (other.clone(), vec![bind("g", "a", 9), bind("g", "b", 5)]), + (owner, vec![bind("g", "a", 7)]), + ]); + assert_eq!(merged, vec![bind("g", "a", 7), bind("g", "b", 5)]); + } + + #[test] + fn binds_round_trip_the_wire() { + let binds = vec![bind("g", "a", 7)]; + let bytes = encode_binds(binds.clone()).expect("encode"); + assert_eq!(decode_binds(&bytes, 2).expect("decode"), binds); + } +} diff --git a/nodedb/src/control/backup/capture.rs b/nodedb/src/control/backup/capture.rs new file mode 100644 index 000000000..d8d58ed4e --- /dev/null +++ b/nodedb/src/control/backup/capture.rs @@ -0,0 +1,277 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Capture named collections of one tenant in one database from the whole +//! cluster. +//! +//! The capture takes the same consistent cut a backup takes. Each vShard is +//! read from exactly one source node: the leader of its group. Each record is +//! kept by the source of its owner home, so a graph edge comes from the leader +//! of its `from_key(src)` vShard. The result is one merged +//! `TenantDataSnapshot`, in the shape the restore re-issue reads. + +use std::collections::{BTreeSet, HashSet}; + +use futures::future::join_all; +use nodedb_physical::physical_plan::MetaOp; +use nodedb_types::DatabaseId; + +use crate::Error; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::state::SharedState; +use crate::types::{RecordHomes, TenantDataSnapshot}; + +use super::node_snapshot::{is_self, snapshot_remote, snapshot_self}; +use super::orchestrator::source_assignment; +use super::restore::sections::append_snapshot; +use super::snapshot_keys::{ + StoredRecord, extract_db_tenant_scoped_collection, homes_of_stored, + retain_tenant_data_for_vshards, stored_collection_key, +}; + +/// Capture `collections` (bare names) of `tenant_id` in `database_id` from +/// every node that is the source of a vShard. `arrays` captures the arrays +/// among the names too. Any node error fails the whole capture. +pub async fn capture_collections( + state: &SharedState, + tenant_id: u64, + database_id: DatabaseId, + collections: &BTreeSet, + arrays: bool, +) -> Result { + let assignment = source_assignment(state)?; + let watermark = super::cut::consistent_cut(state, tenant_id).await?; + + let per_node = join_all( + assignment + .into_iter() + .map(|(node_id, source_vshards)| async move { + let body = if is_self(state, node_id) { + snapshot_self(state, tenant_id, database_id, arrays).await? + } else { + // The remote node takes the cut at the same watermark + // before it snapshots. + let plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id, + cut_watermark: Some(watermark), + cut_capture: None, + arrays, + }); + snapshot_remote(state, node_id, tenant_id, database_id, &plan).await? + }; + decode_owned( + &body, + node_id, + tenant_id, + database_id, + &source_vshards, + collections, + ) + }), + ) + .await; + + let mut merged = TenantDataSnapshot::default(); + for snap in per_node { + append_snapshot(&mut merged, snap?); + } + // The Data Plane snapshot carries no surrogate binds. They live in the + // catalog of each node that holds their homes. + let names: Vec = collections.iter().cloned().collect(); + merged + .surrogate_pk + .extend(super::bind_capture::capture_binds(state, tenant_id, database_id, &names).await?); + Ok(merged) +} + +/// Decode one node's snapshot and keep only the records of the captured +/// collections whose owner home that node is the source for. +fn decode_owned( + body: &[u8], + node_id: u64, + tenant_id: u64, + database_id: DatabaseId, + source_vshards: &HashSet, + collections: &BTreeSet, +) -> Result { + let mut snap: TenantDataSnapshot = + zerompk::from_msgpack(body).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!( + "capture: decode the tenant snapshot of node {node_id}, database {}: {e}", + database_id.as_u64() + ), + })?; + let homes_of = |record: StoredRecord<'_>| homes_if_captured(database_id, collections, record); + retain_tenant_data_for_vshards(&mut snap, tenant_id, source_vshards, homes_of); + // The vector build parameters are keyed like `vectors`. The shared filter + // leaves them alone, and a re-issue configures a collection the + // capture does not name. + let owned = |key: &str| { + extract_db_tenant_scoped_collection(key, tenant_id).is_some_and(|collection| { + homes_of(StoredRecord::Row { collection }) + .is_some_and(|homes| homes.owned_by(source_vshards)) + }) + }; + snap.vector_params.retain(|(key, _)| owned(key)); + snap.index_configs.retain(|(key, _)| owned(key)); + Ok(snap) +} + +/// The homes of `record`, or `None` when its collection is not one of +/// `collections`. +fn homes_if_captured( + database_id: DatabaseId, + collections: &BTreeSet, + record: StoredRecord<'_>, +) -> Option { + let key = stored_collection_key(database_id, record.collection()); + collections + .contains(key.name()) + .then(|| homes_of_stored(database_id, record)) +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::{CollectionKey, QualifiedCollection}; + + const DB: DatabaseId = DatabaseId::new(1025); + + fn captured(names: &[&str]) -> BTreeSet { + names.iter().map(|n| (*n).to_string()).collect() + } + + #[test] + fn a_captured_collection_keeps_its_home_vshard() { + let names = captured(&["orders"]); + let stored = QualifiedCollection::new(DB, "orders"); + let expected = CollectionKey::from_bare(DB, "orders").vshard(); + for collection in [stored.as_str(), "orders"] { + let homes = homes_if_captured(DB, &names, StoredRecord::Row { collection }) + .expect("captured collection"); + assert_eq!(homes.owner(), expected); + assert!(homes.is_single()); + } + } + + #[test] + fn an_uncaptured_collection_has_no_homes() { + let names = captured(&["orders"]); + let stored = QualifiedCollection::new(DB, "invoices"); + let row = StoredRecord::Row { + collection: stored.as_str(), + }; + assert!(homes_if_captured(DB, &names, row).is_none()); + let edge = StoredRecord::Edge { + collection: stored.as_str(), + src: "a", + dst: "b", + }; + assert!(homes_if_captured(DB, &names, edge).is_none()); + } + + #[test] + fn a_captured_edge_is_homed_on_its_endpoints() { + let names = captured(&["follows"]); + let stored = QualifiedCollection::new(DB, "follows"); + let edge = StoredRecord::Edge { + collection: stored.as_str(), + src: "a", + dst: "b", + }; + assert_eq!( + homes_if_captured(DB, &names, edge), + Some(RecordHomes::edge("a", "b")) + ); + } + + /// Four nodes, four RF=3 groups. Each node holds the edges homed on the + /// groups it replicates and is the source of the group it leads. Every + /// edge of a captured collection is captured exactly once, and no edge of + /// an uncaptured collection is captured. + #[test] + fn every_edge_is_captured_once_when_nodes_outnumber_rf() { + const NODES: u32 = 4; + const RF: u32 = 3; + const TID: u64 = 7; + let group_of = |vshard: crate::types::VShardId| vshard.as_u32() % NODES; + let replicates = |node: u32, group: u32| (0..RF).any(|i| (group + i) % NODES == node); + + let names = captured(&["follows"]); + let follows = QualifiedCollection::new(DB, "follows"); + let other = QualifiedCollection::new(DB, "likes"); + let key = |collection: &str, i: u32| { + crate::engine::graph::edge_store::versioned_edge_key( + collection, + &format!("u{i}"), + "L", + &format!("v{}", i * 7 + 3), + 1, + ) + .expect("edge key") + }; + let edges: Vec = (0..256).map(|i| key(follows.as_str(), i)).collect(); + let uncaptured: Vec = (0..16).map(|i| key(other.as_str(), i)).collect(); + let cross = edges + .iter() + .filter_map(|k| StoredRecord::from_edge_key(k)) + .filter(|r| !homes_of_stored(DB, *r).is_single()) + .count(); + assert!(cross > 0, "the fixture holds cross-shard edges"); + + let mut seen: std::collections::BTreeMap = Default::default(); + for node in 0..NODES { + let held: Vec<(String, Vec)> = edges + .iter() + .chain(&uncaptured) + .filter(|k| { + StoredRecord::from_edge_key(k).is_some_and(|r| { + homes_of_stored(DB, r) + .iter() + .any(|home| replicates(node, group_of(home))) + }) + }) + .map(|k| (k.clone(), vec![])) + .collect(); + let body = zerompk::to_msgpack_vec(&TenantDataSnapshot { + edges: held, + ..Default::default() + }) + .expect("encode"); + let source: HashSet = (0..nodedb_cluster::routing::VSHARD_COUNT) + .filter(|v| v % NODES == node) + .collect(); + let kept = + decode_owned(&body, u64::from(node), TID, DB, &source, &names).expect("decode"); + for (k, _) in kept.edges { + *seen.entry(k).or_default() += 1; + } + } + for k in &edges { + assert_eq!(seen.get(k), Some(&1), "edge {k:?} captured once"); + } + assert_eq!(seen.len(), edges.len(), "no uncaptured edge is kept"); + } + + #[test] + fn the_filter_drops_every_entry_of_an_uncaptured_collection() { + let names = captured(&["orders"]); + let all: HashSet = (0..nodedb_cluster::routing::VSHARD_COUNT).collect(); + let kept = QualifiedCollection::new(DB, "orders"); + let dropped = QualifiedCollection::new(DB, "invoices"); + let mut snap = TenantDataSnapshot { + kv_tables: vec![ + (format!("1025:7:{}", kept.as_str()), vec![1]), + (format!("1025:7:{}", dropped.as_str()), vec![2]), + ], + ..Default::default() + }; + retain_tenant_data_for_vshards(&mut snap, 7, &all, |record| { + homes_if_captured(DB, &names, record) + }); + assert_eq!( + snap.kv_tables, + vec![(format!("1025:7:{}", kept.as_str()), vec![1])] + ); + } +} diff --git a/nodedb/src/control/backup/cut.rs b/nodedb/src/control/backup/cut.rs index 897ae8863..02cb48bde 100644 --- a/nodedb/src/control/backup/cut.rs +++ b/nodedb/src/control/backup/cut.rs @@ -17,7 +17,9 @@ //! [`ReplicatedWrite::CutBarrier`] carrying `W` into every data group this //! node hosts and waits for this node's apply of it. Every entry before the //! barrier applied first. Every entry after it records a commit HLC above -//! `W`, however early its proposer stamped it. +//! `W`, however early its proposer stamped it. A database backup's barrier +//! also captures its tenants at that log position (see +//! [`super::cut_capture`]). //! - **Calvin transactions.** A Calvin transaction commits at its place in //! the sequencer log. The cut proposes a `CutMarker` carrying `W` into the //! sequencer log and waits until every Calvin scheduler on this node passed @@ -27,9 +29,6 @@ //! after its record is minted inside an outcome-floor window. The cut reads //! the highest LSN any window minted, after it picks `W`, and waits for the //! outcome floor to reach it. A record minted later stamps itself above `W`. -//! On a server with no Raft groups a write stamps itself before it mints, so -//! its mark is durable first. The cut first waits for every such stamp at or -//! below `W` to mint, then reads the highest LSN. //! //! Two stamps can share a wall time. Once it picks `W`, the cut moves this //! node's clock past `W`, so every later stamp here reads above it. @@ -39,12 +38,12 @@ //! `W`, and its Control Plane runs [`cut_at`] first. use std::collections::BTreeMap; -use std::sync::Arc; use std::time::Duration; use nodedb_cluster::METADATA_GROUP_ID; use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; use nodedb_cluster::calvin::SequencerEntry; +use nodedb_physical::physical_plan::CutCaptureRequest; use nodedb_types::Hlc; use crate::Error; @@ -55,7 +54,7 @@ use crate::types::{DatabaseId, VShardId}; /// Pick the envelope watermark and wait until every write committed below it /// has a final outcome on this node. Returns the watermark. -pub(super) async fn consistent_cut(state: &Arc, tenant_id: u64) -> Result { +pub(super) async fn consistent_cut(state: &SharedState, tenant_id: u64) -> Result { let watermark = state.hlc_clock.now().wall_ns; cut_at(state, tenant_id, watermark).await?; Ok(watermark) @@ -65,38 +64,56 @@ pub(super) async fn consistent_cut(state: &Arc, tenant_id: u64) -> /// write committed below it has a final outcome here. A remote source node /// runs it for the watermark the backup's coordinator picked. pub(crate) async fn cut_at( - state: &Arc, + state: &SharedState, tenant_id: u64, watermark: u64, +) -> Result<(), Error> { + cut_at_point(state, tenant_id, watermark, 0).await +} + +/// [`cut_at`] for the cluster restore point `restore_point`, `0` for a +/// backup's cut. The barriers and the Calvin marker carry the point's id, so +/// every replica records its group's place at the point. +pub(crate) async fn cut_at_point( + state: &SharedState, + tenant_id: u64, + watermark: u64, + restore_point: u64, +) -> Result<(), Error> { + cut_barriers(state, tenant_id, watermark, restore_point, None).await +} + +/// [`cut_at`] whose barriers carry a database backup's capture `request`. +/// The leader of each group this node hosts captures the request's tenants +/// when it applies the request's first barrier in that group. +pub(crate) async fn cut_with_capture( + state: &SharedState, + tenant_id: u64, + watermark: u64, + request: &CutCaptureRequest, +) -> Result<(), Error> { + cut_barriers(state, tenant_id, watermark, 0, Some(request)).await +} + +async fn cut_barriers( + state: &SharedState, + tenant_id: u64, + watermark: u64, + restore_point: u64, + capture: Option<&CutCaptureRequest>, ) -> Result<(), Error> { state .hlc_clock .update(Hlc::new(watermark.saturating_add(1), 0)); let timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); let deadline = tokio::time::Instant::now() + timeout; - // A local write stamped at or below the watermark has not always minted - // its record yet. Every stamp taken from here on reads above it. - if !state - .tenant_marks - .await_local_stamps_minted(watermark, deadline) - .await - { - return Err(Error::Internal { - detail: format!( - "backup: local writes stamped at or below watermark {watermark} did not mint \ - their WAL records within {}s, so the backup cannot take a consistent cut. \ - Retry the backup", - timeout.as_secs(), - ), - }); - } // Read after the watermark: a record minted after this read stamps its // write above the watermark. let target = state.outcome_floor.max_noted(); let (groups, calvin) = tokio::join!( - cut_data_groups(state, tenant_id, watermark), - cut_calvin(state, watermark, deadline), + cut_data_groups(state, tenant_id, watermark, restore_point, capture), + cut_calvin_point(state, watermark, restore_point, deadline), ); groups?; calvin?; @@ -120,13 +137,13 @@ pub(crate) async fn cut_at( /// Propose a cut barrier carrying `watermark` into every data group this node /// hosts, and wait for this node's apply of each. async fn cut_data_groups( - state: &Arc, + state: &SharedState, tenant_id: u64, watermark: u64, + restore_point: u64, + capture: Option<&CutCaptureRequest>, ) -> Result<(), Error> { - let Some(proposer) = state.async_raft_proposer() else { - return Ok(()); - }; + let proposer = state.async_raft_proposer()?; let barriers = futures::future::join_all(barrier_vshards(state).into_iter().map( |(group_id, vshard_id)| { // The barrier orders every entry of its group, whatever database @@ -136,7 +153,11 @@ async fn cut_data_groups( tenant_id, DatabaseId::DEFAULT.as_u64(), vshard_id, - ReplicatedWrite::CutBarrier { hlc: watermark }, + ReplicatedWrite::CutBarrier { + hlc: watermark, + restore_point, + capture: capture.cloned(), + }, ); async move { propose_replicated_entry(state, proposer, entry) @@ -180,9 +201,20 @@ const CUT_MARKER_RETRY: Duration = Duration::from_secs(1); /// Propose a Calvin cut marker carrying `watermark`, and wait until every /// Calvin scheduler on this node passed it, or `deadline`. pub(crate) async fn cut_calvin( - state: &Arc, + state: &SharedState, watermark: u64, deadline: tokio::time::Instant, +) -> Result<(), Error> { + cut_calvin_point(state, watermark, 0, deadline).await +} + +/// [`cut_calvin`] with a marker that carries the cluster restore point +/// `restore_point`, `0` for none. +async fn cut_calvin_point( + state: &SharedState, + watermark: u64, + restore_point: u64, + deadline: tokio::time::Instant, ) -> Result<(), Error> { let cuts = &state.calvin.cuts; if cuts.is_empty() { @@ -199,11 +231,13 @@ pub(crate) async fn cut_calvin( once the cluster finished starting" .into(), })?; - let marker = zerompk::to_msgpack_vec(&SequencerEntry::CutMarker { hlc: watermark }).map_err( - |error| Error::Internal { - detail: format!("backup: encode the Calvin cut marker: {error}"), - }, - )?; + let marker = zerompk::to_msgpack_vec(&SequencerEntry::CutMarker { + hlc: watermark, + restore_point, + }) + .map_err(|error| Error::Internal { + detail: format!("backup: encode the Calvin cut marker: {error}"), + })?; let mut last_refusal = None; loop { if let Err(error) = proposer.propose(marker.clone()) { diff --git a/nodedb/src/control/backup/cut_capture/apply.rs b/nodedb/src/control/backup/cut_capture/apply.rs new file mode 100644 index 000000000..10f09491b --- /dev/null +++ b/nodedb/src/control/backup/cut_capture/apply.rs @@ -0,0 +1,136 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The capture a cut barrier takes in its group's apply path. +//! +//! The apply loop runs a capturing barrier with nothing else of its group in +//! flight: every earlier entry of the group finished on its core, and no later +//! entry starts until the capture returns. The snapshot therefore holds every +//! entry at or below the barrier and none above it. The snapshot reaches the +//! cores over the SPSC bridge, like every other Control-Plane request. + +use std::collections::HashSet; +use std::time::Duration; + +use nodedb_cluster::routing::VSHARD_COUNT; +use nodedb_physical::physical_plan::CutCaptureRequest; + +use crate::Error; +use crate::control::backup::snapshot_keys::{ + StoredRecord, extract_db_tenant_scoped_collection, homes_of_stored, + retain_tenant_data_for_vshards, +}; +use crate::control::security::auth_fence::cluster::group_of_vshard; +use crate::control::server::exchange::snapshot_tenant_on_local_cores; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantDataSnapshot, TenantId}; + +use super::registry::GroupCapture; + +/// How long one tenant's snapshot on this node's cores can take. +const TENANT_SNAPSHOT_TIMEOUT: Duration = Duration::from_secs(120); + +/// Capture `request`'s tenants at a barrier of `group_id` this node applies. +/// +/// Only the first barrier of the request in the group captures, and only on +/// the group's leader at that apply. The capture, or the reason it failed, is +/// parked for the backup to collect. +pub(crate) async fn capture_at_barrier( + state: &SharedState, + group_id: u64, + request: &CutCaptureRequest, +) { + if !state + .cut_captures + .first_barrier(request.request_id, group_id) + { + return; + } + if !leads_group(state, group_id) { + return; + } + let capture = match capture_group(state, group_id, request).await { + Ok(tenants) => GroupCapture::Taken(tenants), + Err(error) => GroupCapture::Failed(error.to_string()), + }; + state + .cut_captures + .park(request.request_id, group_id, capture); +} + +/// Whether this node leads `group_id` now. A node with no Raft status leads +/// every group it applies. +fn leads_group(state: &SharedState, group_id: u64) -> bool { + match state.raft_status_fn.get() { + Some(status) => status() + .iter() + .any(|group| group.group_id == group_id && group.leader_id == state.node_id), + None => true, + } +} + +/// Every vShard `group_id` homes. With no routing table this node's one +/// group homes them all. +pub(crate) fn group_vshards(state: &SharedState, group_id: u64) -> HashSet { + if state.cluster_routing.is_none() { + return (0..VSHARD_COUNT).collect(); + } + (0..VSHARD_COUNT) + .filter(|vshard| group_of_vshard(state, *vshard).ok() == Some(group_id)) + .collect() +} + +/// Snapshot every tenant of `request` on this node's cores, and keep the +/// records `group_id` homes. +async fn capture_group( + state: &SharedState, + group_id: u64, + request: &CutCaptureRequest, +) -> Result)>, Error> { + let vshards = group_vshards(state, group_id); + let database_id = DatabaseId::new(request.database_id); + let mut tenants = Vec::with_capacity(request.tenants.len()); + for &tenant_id in &request.tenants { + let body = snapshot_tenant_on_local_cores( + state, + TenantId::new(tenant_id), + database_id, + TENANT_SNAPSHOT_TIMEOUT, + true, + ) + .await?; + let snapshot = keep_group_records(&body, tenant_id, database_id, &vshards)?; + tenants.push((tenant_id, snapshot)); + } + Ok(tenants) +} + +/// Decode one tenant's snapshot and keep the records whose owner home is in +/// `vshards`. +pub(crate) fn keep_group_records( + body: &[u8], + tenant_id: u64, + database_id: DatabaseId, + vshards: &HashSet, +) -> Result, Error> { + let mut snap: TenantDataSnapshot = + zerompk::from_msgpack(body).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("cut capture: decode the tenant {tenant_id} snapshot: {e}"), + })?; + retain_tenant_data_for_vshards(&mut snap, tenant_id, vshards, |record| { + Some(homes_of_stored(database_id, record)) + }); + // The vector build parameters are keyed like `vectors`. The shared filter + // leaves them alone, and every group's capture carries them all. + let owned = |key: &str| { + extract_db_tenant_scoped_collection(key, tenant_id).is_some_and(|collection| { + homes_of_stored(database_id, StoredRecord::Row { collection }).owned_by(vshards) + }) + }; + snap.vector_params.retain(|(key, _)| owned(key)); + snap.index_configs.retain(|(key, _)| owned(key)); + zerompk::to_msgpack_vec(&snap).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("cut capture: encode the tenant {tenant_id} snapshot: {e}"), + }) +} diff --git a/nodedb/src/control/backup/cut_capture/collect.rs b/nodedb/src/control/backup/cut_capture/collect.rs new file mode 100644 index 000000000..f37668f4d --- /dev/null +++ b/nodedb/src/control/backup/cut_capture/collect.rs @@ -0,0 +1,192 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A database backup's capture of every tenant at one cut. +//! +//! The coordinator picks the cut watermark and a request id, then takes the +//! cut with capturing barriers on its own Raft data groups, a one-node +//! cluster's included. Every other source node receives the request, takes +//! the same cut on its groups, and answers with the captures it parked. Each +//! group's capture comes from its leader at the apply of the request's first +//! barrier. A group with no capture fails the backup with a retryable error +//! that names it: its leadership moved before the capture was collected. +//! The backup never falls back to a live read. + +use std::collections::{BTreeMap, BTreeSet}; +use std::sync::Arc; + +use nodedb_physical::physical_plan::{CutCaptureRequest, MetaOp}; + +use crate::Error; +use crate::bridge::envelope::PhysicalPlan as BridgePlan; +use crate::control::backup::node_snapshot::{is_self, snapshot_remote}; +use crate::control::backup::orchestrator::source_assignment; +use crate::control::backup::restore::sections::append_snapshot; +use crate::control::security::auth_fence::cluster::group_of_vshard; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantDataSnapshot}; + +use super::registry::GroupCapture; + +/// One group's capture on the wire. +#[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub(crate) struct GroupReply { + pub group_id: u64, + /// `(tenant_id, TenantDataSnapshot msgpack)`. + pub tenants: Vec<(u64, Vec)>, + /// The reason the capture failed, when it did. + pub failure: Option, +} + +/// Every tenant of a database, captured at one cut. +pub(crate) struct DatabaseCapture { + /// The cut's HLC wall time, in nanoseconds. + pub cut: u64, + /// Each tenant's rows, merged over every group. + pub tenants: BTreeMap, +} + +/// Capture every tenant of `tenants` in `database_id` at one cut. +pub(crate) async fn capture_database( + state: &Arc, + database_id: DatabaseId, + tenants: &BTreeSet, +) -> Result { + if state.cluster_routing.is_none() { + return Err(Error::Internal { + detail: "database backup: this node holds no routing table, so it cannot place \ + a capturing cut in every data group" + .into(), + }); + } + let cut = state.hlc_clock.now().wall_ns; + let request = CutCaptureRequest { + // The watermark is unique on this node. The node id keeps two + // coordinators apart. + request_id: (cut & !0xFFFF) | (state.node_id & 0xFFFF), + database_id: database_id.as_u64(), + tenants: tenants.iter().copied().collect(), + }; + // The barrier names a tenant only to frame its Raft entry. + let framing = tenants.first().copied().unwrap_or(0); + + super::super::cut::cut_with_capture(state, framing, cut, &request).await?; + let mut replies = take_replies(state, request.request_id); + let remote: Vec = source_assignment(state)? + .into_iter() + .map(|(node_id, _)| node_id) + .filter(|node_id| !is_self(state, *node_id)) + .collect(); + let plan = BridgePlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id: framing, + cut_watermark: Some(cut), + cut_capture: Some(request.clone()), + arrays: true, + }); + let answers = futures::future::join_all(remote.iter().map(|&node_id| { + let plan = &plan; + async move { snapshot_remote(state, node_id, framing, database_id, plan).await } + })) + .await; + for answer in answers { + let decoded: Vec = + zerompk::from_msgpack(&answer?).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("database backup: decode a node's cut captures: {e}"), + })?; + replies.extend(decoded); + } + // Ask again locally: a remote node's barrier can be the first one of a + // group this node leads. + replies.extend(take_replies(state, request.request_id)); + // A test parks the backup here: every capture is taken, so a write from + // now on lies above the cut. + #[cfg(feature = "failpoints")] + crate::control::fail_gate::wait("backup::database::after_cut").await; + assemble(state, cut, replies) +} + +/// The captures of `request` this node parked, as wire replies. +pub(crate) fn take_replies(state: &SharedState, request: u64) -> Vec { + state + .cut_captures + .take_all(request) + .into_iter() + .map(|(group_id, capture)| match capture { + GroupCapture::Taken(tenants) => GroupReply { + group_id, + tenants, + failure: None, + }, + GroupCapture::Failed(reason) => GroupReply { + group_id, + tenants: Vec::new(), + failure: Some(reason), + }, + }) + .collect() +} + +/// Take the cut on this node for a coordinator's `request`, and answer with +/// the encoded captures this node parked. +pub(crate) async fn cut_and_reply( + state: &SharedState, + framing: u64, + watermark: u64, + request: &CutCaptureRequest, +) -> Result, Error> { + super::super::cut::cut_with_capture(state, framing, watermark, request).await?; + zerompk::to_msgpack_vec(&take_replies(state, request.request_id)).map_err(|e| { + Error::Serialization { + format: "msgpack".into(), + detail: format!("cut capture: encode this node's captures: {e}"), + } + }) +} + +/// Every data group that homes a vShard, from this node's routing table. +fn data_groups(state: &SharedState) -> BTreeSet { + (0..nodedb_cluster::routing::VSHARD_COUNT) + .filter_map(|vshard| group_of_vshard(state, vshard).ok()) + .collect() +} + +/// Check that every data group has one capture, and merge each tenant's rows +/// over the groups. +fn assemble( + state: &SharedState, + cut: u64, + replies: Vec, +) -> Result { + let mut by_group: BTreeMap = BTreeMap::new(); + for reply in replies { + // Two leaders of one term cannot both apply a group's first barrier, + // so two replies for a group hold the same capture. + by_group.entry(reply.group_id).or_insert(reply); + } + let mut tenants: BTreeMap = BTreeMap::new(); + for group_id in data_groups(state) { + let reply = by_group + .remove(&group_id) + .ok_or(Error::BackupCaptureMoved { group_id })?; + if let Some(reason) = reply.failure { + return Err(Error::Internal { + detail: format!( + "database backup: the capture of raft group {group_id} at the cut failed: \ + {reason}" + ), + }); + } + for (tenant_id, body) in reply.tenants { + let snap: TenantDataSnapshot = + zerompk::from_msgpack(&body).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!( + "database backup: decode the tenant {tenant_id} capture of raft group \ + {group_id}: {e}" + ), + })?; + append_snapshot(tenants.entry(tenant_id).or_default(), snap); + } + } + Ok(DatabaseCapture { cut, tenants }) +} diff --git a/nodedb/src/control/backup/cut_capture/mod.rs b/nodedb/src/control/backup/cut_capture/mod.rs new file mode 100644 index 000000000..e0c6a3403 --- /dev/null +++ b/nodedb/src/control/backup/cut_capture/mod.rs @@ -0,0 +1,7 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod apply; +pub mod collect; +pub mod registry; + +pub use registry::CutCaptures; diff --git a/nodedb/src/control/backup/cut_capture/registry.rs b/nodedb/src/control/backup/cut_capture/registry.rs new file mode 100644 index 000000000..1d4e17055 --- /dev/null +++ b/nodedb/src/control/backup/cut_capture/registry.rs @@ -0,0 +1,126 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Where a node parks the tenant captures its cut barriers took, until the +//! database backup that asked for them collects them. +//! +//! Every replica of a group applies the same log, so every replica sees the +//! same first barrier of a request in that group. Only that first barrier +//! captures. A later barrier of the same request, proposed by another node, +//! sits above entries the first barrier already cut away, and never captures. + +use std::collections::HashMap; +use std::sync::Mutex; +use std::time::{Duration, Instant}; + +/// How long an uncollected capture, or the note of a request's first +/// barrier, stays on this node. A backup collects its captures within its +/// statement deadline, far below this. +const CAPTURE_TTL: Duration = Duration::from_secs(30 * 60); + +/// One group's capture of a request's tenants. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) enum GroupCapture { + /// `(tenant_id, TenantDataSnapshot msgpack)` per tenant, filtered to the + /// group's vShards. + Taken(Vec<(u64, Vec)>), + /// The snapshot failed. The backup fails with this reason. + Failed(String), +} + +#[derive(Default)] +struct Registry { + /// `(request, group)` whose first barrier this node applied. + seen: HashMap<(u64, u64), Instant>, + parked: HashMap<(u64, u64), (Instant, GroupCapture)>, +} + +impl Registry { + fn sweep(&mut self, now: Instant) { + self.seen + .retain(|_, at| now.saturating_duration_since(*at) < CAPTURE_TTL); + self.parked + .retain(|_, (at, _)| now.saturating_duration_since(*at) < CAPTURE_TTL); + } +} + +/// Every capture this node parked, by `(request, group)`. +#[derive(Default)] +pub struct CutCaptures { + inner: Mutex, +} + +impl CutCaptures { + pub fn new() -> Self { + Self::default() + } + + /// Note a barrier of `request` in `group`. `true` for the first one only. + pub(crate) fn first_barrier(&self, request: u64, group: u64) -> bool { + let now = Instant::now(); + let mut registry = self.inner.lock().unwrap_or_else(|p| p.into_inner()); + registry.sweep(now); + registry.seen.insert((request, group), now).is_none() + } + + /// Park `capture` for `request` in `group`. + pub(crate) fn park(&self, request: u64, group: u64, capture: GroupCapture) { + let mut registry = self.inner.lock().unwrap_or_else(|p| p.into_inner()); + registry + .parked + .insert((request, group), (Instant::now(), capture)); + } + + /// Remove and return every capture of `request` this node parked, by + /// group. + pub(crate) fn take_all(&self, request: u64) -> Vec<(u64, GroupCapture)> { + let mut registry = self.inner.lock().unwrap_or_else(|p| p.into_inner()); + let groups: Vec = registry + .parked + .keys() + .filter(|(r, _)| *r == request) + .map(|(_, group)| *group) + .collect(); + groups + .into_iter() + .filter_map(|group| { + registry + .parked + .remove(&(request, group)) + .map(|(_, capture)| (group, capture)) + }) + .collect() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn only_the_first_barrier_of_a_request_captures() { + let captures = CutCaptures::new(); + assert!(captures.first_barrier(7, 1)); + assert!(!captures.first_barrier(7, 1)); + assert!(captures.first_barrier(7, 2)); + assert!(captures.first_barrier(8, 1)); + } + + #[test] + fn take_all_returns_one_request_and_removes_it() { + let captures = CutCaptures::new(); + captures.park(7, 1, GroupCapture::Taken(vec![(1, vec![1])])); + captures.park(7, 2, GroupCapture::Failed("x".into())); + captures.park(8, 1, GroupCapture::Taken(Vec::new())); + let mut taken = captures.take_all(7); + taken.sort_by_key(|(group, _)| *group); + assert_eq!( + taken, + vec![ + (1, GroupCapture::Taken(vec![(1, vec![1])])), + (2, GroupCapture::Failed("x".into())), + ] + ); + assert!(captures.take_all(7).is_empty()); + assert_eq!(captures.take_all(8).len(), 1); + } +} diff --git a/nodedb/src/control/backup/database.rs b/nodedb/src/control/backup/database.rs new file mode 100644 index 000000000..cd017590d --- /dev/null +++ b/nodedb/src/control/backup/database.rs @@ -0,0 +1,251 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `BACKUP DATABASE` and `RESTORE DATABASE`: every tenant of one database. +//! +//! The backup captures every tenant with a collection or an array in the +//! database at one consistent cut (see [`super::cut_capture`]): no row +//! written after the cut is in the backup. It frames one tenant envelope per +//! tenant, scoped to that database, and packs them into one encrypted +//! database envelope (see `nodedb_types::backup_envelope::database`) whose +//! manifest records the cut. The restore checks every tenant envelope as a +//! DRY RUN first ([`check_database`]), so a refused tenant stops the restore +//! before any write, then restores each tenant through the tenant restore, +//! verification included ([`apply_database`]). + +use std::collections::BTreeSet; +use std::sync::Arc; + +use nodedb_cluster::routing::VSHARD_COUNT; +use nodedb_types::backup_envelope::{ + DATABASE_BACKUP_TENANT, DEFAULT_MAX_TOTAL_BYTES, DatabaseBackupManifest, DatabaseDataSection, + EnvelopeMeta, EnvelopeWriter, SECTION_ORIGIN_DATABASE_MANIFEST, parse_encrypted, +}; + +use crate::Error; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantDataSnapshot}; + +use super::RestoreStats; +use super::cut_capture::collect::capture_database; + +fn backup_kek(state: &SharedState) -> Result<&[u8; 32], Error> { + state.backup_kek.as_deref().ok_or_else(|| Error::Internal { + detail: "database backup: no [backup_encryption] KEK configured; set \ + [backup_encryption] in the server config" + .into(), + }) +} + +fn envelope_error(what: &str, e: impl std::fmt::Display) -> Error { + Error::Internal { + detail: format!("database backup envelope ({what}): {e}"), + } +} + +/// Every tenant with a collection or an array in `database_id`: the tenants +/// a backup of it captures. +pub fn database_tenants( + state: &SharedState, + database_id: DatabaseId, +) -> Result, Error> { + let catalog = state.credentials.catalog(); + let mut tenants: BTreeSet = catalog + .load_all_collections(database_id)? + .iter() + .map(|coll| coll.tenant_id) + .collect(); + tenants.extend( + catalog + .load_all_arrays()? + .iter() + .filter(|array| array.array_id.database_id == database_id) + .map(|array| array.array_id.tenant_id.as_u64()), + ); + Ok(tenants) +} + +/// The rows of `tenants`, from [`database_tenants`], in `database_id`, named +/// `name`, as one encrypted database envelope. +pub async fn backup_database( + state: &Arc, + database_id: DatabaseId, + name: &str, + tenants: &BTreeSet, +) -> Result, Error> { + let kek = *backup_kek(state)?; + let mut captured = capture_database(state, database_id, tenants).await?; + + let manifest = DatabaseBackupManifest { + database: name.to_string(), + cut_hlc: captured.cut, + tenants: tenants.iter().copied().collect(), + }; + let mut writer = EnvelopeWriter::new(EnvelopeMeta { + tenant_id: DATABASE_BACKUP_TENANT, + source_vshard_count: VSHARD_COUNT as u16, + hash_seed: 0, + snapshot_watermark: captured.cut, + }); + let body = zerompk::to_msgpack_vec(&manifest).map_err(|e| envelope_error("manifest", e))?; + writer + .push_section(SECTION_ORIGIN_DATABASE_MANIFEST, body) + .map_err(|e| envelope_error("manifest", e))?; + for &tenant_id in tenants { + let snapshot = captured.tenants.remove(&tenant_id).unwrap_or_default(); + let envelope = + tenant_envelope(state, tenant_id, database_id, captured.cut, snapshot).await?; + writer + .push_section(tenant_id, envelope.to_vec()) + .map_err(|e| envelope_error("tenant section", e))?; + } + writer + .finalize_encrypted(&kek) + .map_err(|e| envelope_error("encryption", e)) +} + +/// The envelope of `tenant_id` scoped to `database_id`, holding `snapshot`, +/// its rows captured at the cut `cut`. +async fn tenant_envelope( + state: &Arc, + tenant_id: u64, + database_id: DatabaseId, + cut: u64, + snapshot: TenantDataSnapshot, +) -> Result { + let mut databases = super::metadata::tenant_databases(state, tenant_id)?; + databases.retain(|database| database.id() == database_id); + let section = DatabaseDataSection { + database_id: database_id.as_u64(), + snapshot: super::metadata::encode_section_part("tenant capture", &snapshot)?, + }; + let body = super::metadata::encode_section_part("data section", §ion)?; + super::orchestrator::assemble_envelope( + state, + tenant_id, + cut, + &databases, + vec![(state.node_id, body)], + ) + .await +} + +/// A decrypted database backup whose manifest names the database to restore. +pub struct OpenedBackup { + manifest: DatabaseBackupManifest, + /// `(tenant_id, tenant envelope)` in manifest order. + tenants: Vec<(u64, Vec)>, +} + +impl OpenedBackup { + /// Every tenant the backup restores, in manifest order. + pub fn tenant_ids(&self) -> &[u64] { + &self.manifest.tenants + } + + /// HLC wall time, in nanoseconds, of the backup's consistent cut. + pub fn cut_hlc(&self) -> u64 { + self.manifest.cut_hlc + } +} + +/// What a database restore did, in total and per tenant. +#[derive(Debug, Default)] +pub struct DatabaseRestore { + pub total: RestoreStats, + /// `(tenant_id, that tenant's stats)` in manifest order. + pub tenants: Vec<(u64, RestoreStats)>, +} + +/// Decrypt the database backup `bytes` and check that it holds the database +/// `name`. Writes nothing. +pub fn open_database_backup( + state: &SharedState, + name: &str, + bytes: &[u8], +) -> Result { + let kek = *backup_kek(state)?; + let env = parse_encrypted(bytes, DEFAULT_MAX_TOTAL_BYTES, &kek)?; + if env.meta.tenant_id != DATABASE_BACKUP_TENANT { + return Err(Error::BadRequest { + detail: "the backup object is a tenant backup, not a database backup; restore it \ + with COPY tenant_restore() FROM STDIN" + .into(), + }); + } + let mut sections = env.sections.into_iter(); + let manifest_section = sections + .next() + .filter(|s| s.origin_node_id == SECTION_ORIGIN_DATABASE_MANIFEST) + .ok_or_else(|| Error::Internal { + detail: "invalid backup format: the database backup has no manifest".into(), + })?; + let manifest: DatabaseBackupManifest = + zerompk::from_msgpack(&manifest_section.body).map_err(|_| Error::Internal { + detail: "invalid backup format: the database backup manifest is not decodable".into(), + })?; + if manifest.database != name { + return Err(Error::BadRequest { + detail: format!( + "the backup holds database '{}', not '{name}'; restore it with RESTORE \ + DATABASE {} FROM ...", + manifest.database, manifest.database + ), + }); + } + let tenants: Vec<(u64, Vec)> = sections.map(|s| (s.origin_node_id, s.body)).collect(); + if !tenants + .iter() + .map(|(id, _)| *id) + .eq(manifest.tenants.iter().copied()) + { + return Err(Error::Internal { + detail: "invalid backup format: the database backup's tenant sections do not \ + match its manifest" + .into(), + }); + } + Ok(OpenedBackup { manifest, tenants }) +} + +/// Check every tenant envelope of `backup` as a DRY RUN, verification of the +/// envelope included. Writes nothing. The stats list every collection's rows, +/// which a restore admits its write quota against. +pub async fn check_database( + state: &Arc, + backup: &OpenedBackup, + force: bool, +) -> Result { + restore_each(state, backup, true, force).await +} + +/// Restore every tenant of `backup` through the tenant restore. Run +/// [`check_database`] first: a tenant refused there stops the restore before +/// any tenant writes. +pub async fn apply_database( + state: &Arc, + backup: &OpenedBackup, + force: bool, +) -> Result { + restore_each(state, backup, false, force).await +} + +async fn restore_each( + state: &Arc, + backup: &OpenedBackup, + dry_run: bool, + force: bool, +) -> Result { + let mut result = DatabaseRestore { + total: RestoreStats { + dry_run, + ..Default::default() + }, + tenants: Vec::with_capacity(backup.tenants.len()), + }; + for (tenant_id, envelope) in &backup.tenants { + let stats = super::restore_tenant(state, *tenant_id, envelope, dry_run, force).await?; + result.total.absorb(&stats); + result.tenants.push((*tenant_id, stats)); + } + Ok(result) +} diff --git a/nodedb/src/control/backup/metadata.rs b/nodedb/src/control/backup/metadata.rs index 1d821085f..c51abc48e 100644 --- a/nodedb/src/control/backup/metadata.rs +++ b/nodedb/src/control/backup/metadata.rs @@ -13,7 +13,9 @@ //! - the PK-to-surrogate binds of those collections //! (`SECTION_ORIGIN_SURROGATE_PK`); //! - the WAL tombstones of the tenant's purged collections in it -//! (`SECTION_ORIGIN_SOURCE_TOMBSTONES`). +//! (`SECTION_ORIGIN_SOURCE_TOMBSTONES`); +//! - the catalog row of each of the tenant's arrays in it +//! (`SECTION_ORIGIN_ARRAY_CATALOG`). //! //! Every entry names its database by the source id. Restore maps each source //! id to a destination database by name. @@ -21,12 +23,13 @@ use std::collections::{BTreeMap, BTreeSet}; use nodedb_types::backup_envelope::{ - DatabaseBlob, EnvelopeWriter, SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_DATABASES, - SECTION_ORIGIN_SOURCE_TOMBSTONES, SECTION_ORIGIN_SURROGATE_PK, SourceTombstoneEntry, - StoredCollectionBlob, SurrogateBindBlob, + ArrayCatalogBlob, DatabaseBlob, EnvelopeWriter, SECTION_ORIGIN_ARRAY_CATALOG, + SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_DATABASES, SECTION_ORIGIN_SOURCE_TOMBSTONES, + SECTION_ORIGIN_SURROGATE_PK, SourceTombstoneEntry, StoredCollectionBlob, SurrogateBindBlob, }; use crate::Error; +use crate::control::array_catalog::ArrayCatalogEntry; use crate::control::security::catalog::{DatabaseDescriptor, StoredCollection, SystemCatalog}; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId}; @@ -38,6 +41,8 @@ pub struct TenantDatabase { /// included: UNDROP works after a restore of a backup taken during the /// retention window. pub collections: Vec, + /// Every array of the tenant in this database. + pub arrays: Vec, } impl TenantDatabase { @@ -46,28 +51,50 @@ impl TenantDatabase { } } -/// Every database `tenant_id` has a collection in, in database-id order. +/// Every database `tenant_id` has a collection or an array in, in +/// database-id order. /// /// A collection whose database has no catalog entry fails the backup: the -/// restore could not recreate that database, and its rows would be lost. +/// restore cannot recreate that database, and its rows are lost. +/// +/// A collection still delegating reads to a clone source fails the backup, as +/// [`refuse_unmaterialized_clone`] explains. pub fn tenant_databases(state: &SharedState, tenant_id: u64) -> Result, Error> { - let catalog = state.credentials.catalog(); - let mut by_database: BTreeMap> = BTreeMap::new(); + tenant_databases_in(state.credentials.catalog(), tenant_id) +} + +fn tenant_databases_in( + catalog: &SystemCatalog, + tenant_id: u64, +) -> Result, Error> { + let mut by_database: BTreeMap, Vec)> = + BTreeMap::new(); for coll in catalog.load_all_collections_across_databases()? { if coll.tenant_id == tenant_id { + refuse_unmaterialized_clone(&coll)?; by_database .entry(coll.database_id.as_u64()) .or_default() + .0 .push(coll); } } + for array in catalog.load_all_arrays()? { + if array.array_id.tenant_id.as_u64() == tenant_id { + by_database + .entry(array.array_id.database_id.as_u64()) + .or_default() + .1 + .push(array); + } + } let mut databases = Vec::with_capacity(by_database.len()); - for (raw_id, collections) in by_database { + for (raw_id, (collections, arrays)) in by_database { let descriptor = catalog .get_database(DatabaseId::new(raw_id))? .ok_or_else(|| Error::Internal { detail: format!( - "backup: tenant {tenant_id} has collections in database {raw_id}, \ + "backup: tenant {tenant_id} has collections or arrays in database {raw_id}, \ but the catalog has no entry for that database. Restore the \ database entry, then retry the backup" ), @@ -75,18 +102,44 @@ pub fn tenant_databases(state: &SharedState, tenant_id: u64) -> Result Result<(), Error> { + let Some(origin) = &coll.cloned_from else { + return Ok(()); + }; + Err(Error::BadRequest { + detail: format!( + "collection '{}' in database {} is an unmaterialized clone of '{}' in database {}; \ + run ALTER DATABASE ... MATERIALIZE on its database, then retry", + coll.name, + coll.database_id.as_u64(), + origin.source_collection, + origin.source_database.as_u64() + ), + }) +} + +/// Push the metadata sections of `databases`, with `binds` from +/// [`surrogate_binds`]. A catalog read or encode error fails the backup: an +/// envelope without these sections restores rows into no database, rows a +/// point lookup cannot find, or a purged collection. pub fn push_metadata_sections( state: &SharedState, tenant_id: u64, databases: &[TenantDatabase], + binds: &[SurrogateBindBlob], writer: &mut EnvelopeWriter, ) -> Result<(), Error> { let catalog = state.credentials.catalog(); @@ -116,12 +169,28 @@ pub fn push_metadata_sections( } push_nonempty(writer, SECTION_ORIGIN_CATALOG_ROWS, "catalog rows", &rows)?; + let mut arrays = Vec::new(); + for database in databases { + for array in &database.arrays { + arrays.push(ArrayCatalogBlob { + database_id: database.id().as_u64(), + name: array.name.clone(), + bytes: encode_section_part("array catalog row", array)?, + }); + } + } + push_nonempty( + writer, + SECTION_ORIGIN_ARRAY_CATALOG, + "array catalog rows", + &arrays, + )?; + // PK→surrogate identity map. This is DATA-derived per-node state that the // per-node engine sections do NOT carry (the Data-Plane snapshot handler // has no catalog access). Without it a restored node has documents but // cannot resolve PK point-lookups (`WHERE id=`). - let binds = surrogate_binds(catalog, tenant, databases)?; - push_nonempty(writer, SECTION_ORIGIN_SURROGATE_PK, "surrogate pk", &binds)?; + push_nonempty(writer, SECTION_ORIGIN_SURROGATE_PK, "surrogate pk", binds)?; let backed_up: BTreeSet = databases.iter().map(|d| d.id().as_u64()).collect(); let mut tombs = Vec::new(); @@ -142,29 +211,31 @@ pub fn push_metadata_sections( ) } -/// Every PK→surrogate bind of the tenant's collections in `databases`. -fn surrogate_binds( - catalog: &SystemCatalog, - tenant: TenantId, +/// Every PK→surrogate bind of the tenant's collections in `databases`, each +/// from a source node that holds it (see +/// [`super::bind_capture::capture_binds`]). No one node holds every bind when +/// nodes outnumber the replication factor. +pub async fn surrogate_binds( + state: &SharedState, + tenant_id: u64, databases: &[TenantDatabase], ) -> Result, Error> { let mut binds = Vec::new(); for database in databases { - for coll in &database.collections { - let rows = catalog.scan_surrogates_for_collection( - nodedb_types::CollectionKey::from_bare(database.id(), &coll.name), - tenant, - )?; - for (pk, surrogate) in rows { - binds.push(SurrogateBindBlob { - database_id: database.id().as_u64(), - tenant_id: tenant.as_u64(), - collection: coll.name.clone(), - pk, - surrogate: surrogate.as_u32(), - }); - } - } + let names: Vec = database + .collections + .iter() + .map(|coll| coll.name.clone()) + .collect(); + let captured = + super::bind_capture::capture_binds(state, tenant_id, database.id(), &names).await?; + binds.extend(captured.into_iter().map(|bind| SurrogateBindBlob { + database_id: bind.database_id, + tenant_id: bind.tenant_id, + collection: bind.collection, + pk: bind.pk, + surrogate: bind.surrogate, + })); } Ok(binds) } @@ -197,3 +268,40 @@ fn push_nonempty( detail: format!("backup envelope ({what}): {e}"), }) } + +#[cfg(test)] +mod tests { + use super::*; + + fn open_catalog() -> (tempfile::TempDir, SystemCatalog) { + let dir = tempfile::tempdir().unwrap(); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).unwrap(); + (dir, catalog) + } + + /// A backup refuses a clone that still delegates to its source, and takes + /// the same collection once it is materialized. + #[test] + fn backup_refuses_an_unmaterialized_clone() { + let (_dir, catalog) = open_catalog(); + let mut coll = StoredCollection::stamped_for_test(1, "orders", "admin"); + coll.cloned_from = Some(nodedb_types::CloneOrigin { + source_database: DatabaseId::new(1024), + source_collection: "orders".into(), + as_of_lsn: nodedb_types::Lsn::new(10), + clone_created_at: nodedb_types::Lsn::new(11), + kv_surrogate_ceiling: None, + }); + catalog.put_collection(DatabaseId::DEFAULT, &coll).unwrap(); + + let err = tenant_databases_in(&catalog, 1) + .err() + .expect("an unmaterialized clone fails the backup"); + assert!(matches!(err, Error::BadRequest { .. }), "{err}"); + + coll.cloned_from = None; + coll.clone_status = nodedb_types::CloneStatus::Materialized; + catalog.put_collection(DatabaseId::DEFAULT, &coll).unwrap(); + assert_eq!(tenant_databases_in(&catalog, 1).unwrap().len(), 1); + } +} diff --git a/nodedb/src/control/backup/mod.rs b/nodedb/src/control/backup/mod.rs index f0c34fd1e..d266eaab2 100644 --- a/nodedb/src/control/backup/mod.rs +++ b/nodedb/src/control/backup/mod.rs @@ -1,15 +1,23 @@ // SPDX-License-Identifier: BUSL-1.1 +pub mod bind_capture; +pub mod capture; pub mod cut; +pub mod cut_capture; +pub mod database; pub mod detect; pub mod metadata; pub mod node_snapshot; pub mod orchestrator; pub mod restore; +pub mod schedule; pub mod snapshot_keys; pub mod state; +pub mod store; +pub mod store_local; +pub mod verify; pub use detect::{CopyIntent, detect}; pub use orchestrator::backup_tenant; -pub use restore::{RestoreStats, restore_tenant}; +pub use restore::{CollectionRows, RestoreStats, restore_tenant}; pub use state::{RestorePending, RestoreState}; diff --git a/nodedb/src/control/backup/node_snapshot.rs b/nodedb/src/control/backup/node_snapshot.rs index e1c3ea647..c5fa42bc0 100644 --- a/nodedb/src/control/backup/node_snapshot.rs +++ b/nodedb/src/control/backup/node_snapshot.rs @@ -6,7 +6,6 @@ //! plan in a `RaftRpc::ExecuteRequest`. Either way the request names the //! database, and the Data Plane snapshot covers that database only. -use std::sync::Arc; use std::time::Duration; use nodedb_cluster::rpc_codec::{ExecuteRequest, ExecuteResponse, RaftRpc, TypedClusterError}; @@ -23,30 +22,33 @@ const NODE_SNAPSHOT_TIMEOUT: Duration = Duration::from_secs(120); /// Whether `node_id` names this node. pub(super) fn is_self(state: &SharedState, node_id: u64) -> bool { - node_id == state.node_id || node_id == 0 || state.cluster_transport.is_none() + node_id == state.node_id || node_id == 0 } -/// Snapshot `tenant_id` in `database_id` on every core of this node. +/// Snapshot `tenant_id` in `database_id` on every core of this node, with +/// every array cell version when `arrays` is set. /// /// This node already took the backup's cut, so the local snapshot carries no /// cut request. pub(super) async fn snapshot_self( - state: &Arc, + state: &SharedState, tenant_id: u64, database_id: DatabaseId, + arrays: bool, ) -> Result, Error> { snapshot_tenant_on_local_cores( state, TenantId::new(tenant_id), database_id, NODE_SNAPSHOT_TIMEOUT, + arrays, ) .await } /// Snapshot `tenant_id` in `database_id` on the remote node `node_id`. pub(super) async fn snapshot_remote( - state: &Arc, + state: &SharedState, node_id: u64, tenant_id: u64, database_id: DatabaseId, @@ -71,6 +73,8 @@ pub(super) async fn snapshot_remote( descriptor_versions: Vec::new(), // Backup snapshot dispatch is not session-transaction-scoped. txn_id: None, + vshard_id: None, + read_groups: Vec::new(), }); let resp = transport @@ -140,5 +144,7 @@ fn map_typed_error(err: TypedClusterError, node_id: u64) -> Error { constraint, detail, }, + // A Calvin abort keeps the error a local submit returns. + aborted @ TypedClusterError::CalvinAborted { .. } => Error::from(aborted), } } diff --git a/nodedb/src/control/backup/orchestrator.rs b/nodedb/src/control/backup/orchestrator.rs index 64962f836..1ebd1d5db 100644 --- a/nodedb/src/control/backup/orchestrator.rs +++ b/nodedb/src/control/backup/orchestrator.rs @@ -5,21 +5,25 @@ //! Discovers every node that holds a vShard owning some of the //! tenant's data, dispatches `MetaOp::CreateTenantSnapshot` to each //! (local SPSC for self, `RaftRpc::ExecuteRequest` for remotes), -//! and packs the gathered per-node snapshots into a `BackupEnvelope`. +//! and packs the gathered per-node snapshots into a `BackupEnvelope`, +//! with a verification section that records every collection's row count +//! and digest. //! -//! The backup covers every database the tenant has a collection in. Each -//! source node snapshots each of those databases after one consistent cut, -//! and each snapshot becomes one data section that names its database. +//! The backup covers every database the tenant has a collection or an array +//! in. Each source node snapshots each of those databases after one +//! consistent cut, and each snapshot becomes one data section that names its +//! database. //! -//! Single-node mode is the degenerate case: routing table absent -//! (or 1 node) → one source, origin = self. +//! A one-node cluster is the degenerate case: one source, origin = self. use std::collections::{BTreeMap, HashSet}; use std::sync::Arc; use bytes::Bytes; use nodedb_cluster::routing::VSHARD_COUNT; -use nodedb_types::backup_envelope::{DatabaseDataSection, EnvelopeMeta, EnvelopeWriter}; +use nodedb_types::backup_envelope::{ + DatabaseDataSection, EnvelopeMeta, EnvelopeWriter, SECTION_ORIGIN_VERIFICATION, +}; use crate::Error; use crate::bridge::envelope::PhysicalPlan; @@ -33,25 +37,25 @@ use super::node_snapshot::{is_self, snapshot_remote, snapshot_self}; /// Build a complete tenant backup envelope by fanning out across the /// cluster, gathering each node's slice, and framing the result. /// -/// Single-node and cluster paths converge here — a single-node server -/// produces one data section per database with origin = self. +/// A one-node cluster produces one data section per database with +/// origin = self. pub async fn backup_tenant(state: &Arc, tenant_id: u64) -> Result { // Assign every vshard to exactly ONE source node (the leader of its Raft // group, or — when no leader is elected yet — the lowest-id member), and // gather only from those source nodes. Under RF>1 every replica holds the - // full vshard data, so gathering from all members and merging would - // MULTIPLY append-style engine rows (columnar / timeseries) by the + // full vshard data, so gathering from all members and merging + // MULTIPLIES append-style engine rows (columnar / timeseries) by the // replication factor. Filtering each source node's snapshot to the vshards // it owns makes the union cover each vshard exactly once. - let assignment = source_assignment(state); + let assignment = source_assignment(state)?; // The envelope watermark is the consistent cut: every user write // committed below it has applied before the snapshots below, and every // write committed at or above it refuses a restore of this envelope. let snapshot_watermark = super::cut::consistent_cut(state, tenant_id).await?; - // Every database the tenant has a collection in. Read after the cut, so - // a collection created before the cut is in the list. + // Every database the tenant has a collection or an array in. Read after + // the cut, so one created before the cut is in the list. let databases = super::metadata::tenant_databases(state, tenant_id)?; // Each source node snapshots its databases in order. The nodes run @@ -74,7 +78,31 @@ pub async fn backup_tenant(state: &Arc, tenant_id: u64) -> Result, + tenant_id: u64, + snapshot_watermark: u64, + databases: &[TenantDatabase], + data_sections: Vec<(u64, Vec)>, +) -> Result { let meta = EnvelopeMeta { tenant_id, source_vshard_count: VSHARD_COUNT as u16, @@ -83,16 +111,6 @@ pub async fn backup_tenant(state: &Arc, tenant_id: u64) -> Result, tenant_id: u64) -> Result Vec<(u64, HashSet)> { - let Some(routing) = state.cluster_routing.as_ref() else { - return vec![(state.node_id, (0..VSHARD_COUNT).collect())]; - }; +/// A one-node cluster leads every group, so it returns +/// `[(self.node_id, ALL vshards)]`. Refuses when the cluster is not wired. +pub(super) fn source_assignment(state: &SharedState) -> Result)>, Error> { + let routing = state.cluster_routing.as_ref().ok_or(Error::Internal { + detail: "backup source assignment: no routing table is wired on this node".to_owned(), + })?; let table = routing.read().unwrap_or_else(|p| p.into_inner()); let mut by_node: BTreeMap> = BTreeMap::new(); @@ -208,19 +249,18 @@ fn source_assignment(state: &SharedState) -> Vec<(u64, HashSet)> { } if by_node.is_empty() { - return vec![(state.node_id, (0..VSHARD_COUNT).collect())]; + return Ok(vec![(state.node_id, (0..VSHARD_COUNT).collect())]); } - by_node.into_iter().collect() + Ok(by_node.into_iter().collect()) } /// Decode a gathered per-node `TenantDataSnapshot` of `database_id`, filter /// it in place to the vshards this node is the assigned source for, and /// re-encode it. /// -/// The per-section vshard classification is shared with the Raft snapshot SEND -/// builder via `snapshot_keys::retain_tenant_data_for_vshards`. The vshard-of -/// closure routes each stored name in `database_id`, matching the snapshot -/// builder. +/// Each record is kept by the source of its owner home, the rule the Raft +/// snapshot SEND builder and MOVE TENANT capture share through +/// `snapshot_keys::homes_of_stored`. fn filter_node_snapshot( body: Vec, tenant_id: u64, @@ -235,7 +275,7 @@ fn filter_node_snapshot( &mut snap, tenant_id, source_vshards, - |collection| super::snapshot_keys::vshard_of_stored(database_id, collection), + |record| Some(super::snapshot_keys::homes_of_stored(database_id, record)), ); zerompk::to_msgpack_vec(&snap).map_err(|e| Error::Internal { detail: format!("backup: re-encode filtered snapshot: {e}"), diff --git a/nodedb/src/control/backup/restore/array_reissue.rs b/nodedb/src/control/backup/restore/array_reissue.rs new file mode 100644 index 000000000..4700a445c --- /dev/null +++ b/nodedb/src/control/backup/restore/array_reissue.rs @@ -0,0 +1,328 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Restore of arrays: each array's catalog row, then every cell version. +//! +//! The catalog row goes through `PutArray` like `CREATE ARRAY`, so every node +//! opens the array. A destination array of the same name, schema and routing +//! keeps its incarnation and takes the backup's settings: retention, +//! `audit_retain_ms`, and every other field of the row. One of another schema +//! or routing refuses the envelope before any write. +//! +//! Each cell version re-issues as a durable array write, in system-time +//! order, with its own system time and valid time: a restored array keeps +//! its history. A live cell takes the destination's surrogate for its +//! coordinate. Each write goes to the vShard its cells route to. + +use std::collections::{BTreeMap, BTreeSet}; + +use nodedb_array::types::ArrayId; +use nodedb_physical::physical_plan::ArrayOp; +use nodedb_types::backup_envelope::{ + ArrayCatalogBlob, DatabaseBlob, Envelope, SECTION_ORIGIN_ARRAY_CATALOG, +}; +use nodedb_types::{CollectionKey, Hlc}; + +use crate::Error; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::array_catalog::ArrayCatalogEntry; +use crate::control::array_catalog::ddl::propose_array_entries; +use crate::control::catalog_entry::CatalogEntry; +use crate::control::security::catalog::SystemCatalog; +use crate::control::state::SharedState; +use crate::engine::array::export::ArrayCellVersion; +use crate::engine::array::wal::{ArrayDeleteCell, ArrayPutCell}; +use crate::event::EventSource; +use crate::types::{ArrayCellsBlob, DatabaseId, TenantDataSnapshot, TenantId, VShardId}; + +use super::databases::DatabaseMap; +use super::target::DatabaseTarget; + +/// Cells one re-issued write carries at most. +const CELLS_PER_WRITE: usize = 1024; + +/// Whether the destination's `existing` array takes the backup's `row` in +/// place: the same schema, and cells that route the same way. +fn compatible(existing: &ArrayCatalogEntry, row: &ArrayCatalogEntry) -> bool { + existing.schema_hash == row.schema_hash && existing.prefix_bits == row.prefix_bits +} + +fn another_schema(array: &str, database: &str) -> Error { + Error::BadRequest { + detail: format!( + "array '{array}' exists in database '{database}' with another schema or routing; \ + drop it, then retry the restore" + ), + } +} + +fn malformed(detail: String) -> Error { + Error::Internal { + detail: format!("invalid backup format: {detail}"), + } +} + +fn codec(array: &str, e: impl std::fmt::Display) -> Error { + Error::Serialization { + format: "msgpack".into(), + detail: format!("restore: array '{array}': {e}"), + } +} + +/// Every array catalog row of `env`, with its source database id. +fn array_rows(env: &Envelope) -> Result, Error> { + let mut rows = Vec::new(); + for section in env + .sections + .iter() + .filter(|s| s.origin_node_id == SECTION_ORIGIN_ARRAY_CATALOG) + { + let blobs: Vec = zerompk::from_msgpack(§ion.body) + .map_err(|_| malformed("array catalog section is not decodable".into()))?; + for blob in blobs { + let entry: ArrayCatalogEntry = zerompk::from_msgpack(&blob.bytes).map_err(|_| { + malformed(format!( + "catalog row of array '{}' is not decodable", + blob.name + )) + })?; + rows.push((blob.database_id, entry)); + } + } + Ok(rows) +} + +/// Every array row names a listed database, and every array a data section +/// carries cells of has a row. A destination array of the same name must +/// have the same schema. +pub(super) fn validate_array_rows( + catalog: &SystemCatalog, + tenant: TenantId, + env: &Envelope, + databases: &[DatabaseBlob], + merged: &BTreeMap, +) -> Result<(), Error> { + let names: BTreeMap = databases + .iter() + .map(|blob| (blob.database_id, blob.name.as_str())) + .collect(); + let mut rowed: BTreeSet<(u64, String)> = BTreeSet::new(); + for (source, entry) in array_rows(env)? { + let Some(database) = names.get(&source) else { + return Err(malformed(format!( + "the row of array '{}' names database {source}, which the backup's database \ + section does not list", + entry.name + ))); + }; + if let Some(dest) = catalog.get_database_id_by_name(database)? + && let Some(existing) = catalog.get_array_in_database(tenant, dest, &entry.name)? + && !compatible(&existing, &entry) + { + return Err(another_schema(&entry.name, database)); + } + rowed.insert((source, entry.name)); + } + for (source, snap) in merged { + if let Some(blob) = snap + .arrays + .iter() + .find(|blob| !rowed.contains(&(*source, blob.array.clone()))) + { + return Err(malformed(format!( + "cells of array '{}' in database {source} have no catalog row", + blob.array + ))); + } + } + Ok(()) +} + +/// Create each backed-up array in its destination database, or apply the +/// backup's row to the destination's compatible array of the same name. +/// Returns the rows restored. +pub(super) async fn restore_array_rows( + state: &SharedState, + tenant_id: u64, + env: &Envelope, + databases: &DatabaseMap, +) -> Result { + let tenant = TenantId::new(tenant_id); + let rows = array_rows(env)?; + for (source, row) in &rows { + let dest = databases.target(*source)?.dest; + let entry = ArrayCatalogEntry { + array_id: ArrayId::in_database(tenant, dest, &row.name), + // A new array takes a fresh incarnation from the proposer's stamp. + modification_hlc: Hlc::ZERO, + incarnation: Hlc::ZERO, + ..row.clone() + }; + propose_array_entries(state, |catalog| { + let entry = match catalog.get_array_in_database(tenant, dest, &entry.name)? { + // The array keeps its incarnation, so its in-flight cell + // writes still route to it. + Some(existing) if compatible(&existing, &entry) => ArrayCatalogEntry { + incarnation: existing.incarnation, + ..entry.clone() + }, + Some(_) => { + return Err(another_schema(&entry.name, &dest.as_u64().to_string())); + } + None => entry.clone(), + }; + Ok(vec![CatalogEntry::PutArray(Box::new(entry))]) + }) + .await?; + } + Ok(rows.len()) +} + +/// Re-issue every cell version of `blobs` into `target.dest`. Returns the +/// versions re-issued. +pub(in crate::control::backup::restore) async fn reissue_array_cells( + state: &SharedState, + tenant_id: u64, + target: DatabaseTarget, + blobs: Vec, +) -> Result { + let tenant = TenantId::new(tenant_id); + let mut reissued = 0usize; + for blob in blobs { + let mut versions: Vec = + zerompk::from_msgpack(&blob.cells).map_err(|e| codec(&blob.array, e))?; + // Every version of one coordinate is in this blob: its cells route to + // one vShard. System-time order rebuilds each coordinate's history. + versions.sort_by_key(|version| version.system_from_ms); + super::durable::log_reissue_step( + state, + "array", + &blob.array, + VShardId::new(blob.vshard), + versions.len(), + ); + let write = CellWrite { + tenant, + target, + array: &blob.array, + vshard: blob.vshard, + }; + let mut run: Vec = Vec::new(); + for version in versions { + let kind_changes = run + .last() + .is_some_and(|last| last.payload.is_some() != version.payload.is_some()); + if kind_changes || run.len() == CELLS_PER_WRITE { + reissued += run.len(); + write.run(state, std::mem::take(&mut run)).await?; + } + run.push(version); + } + reissued += run.len(); + write.run(state, run).await?; + } + Ok(reissued) +} + +/// Where one blob's cells re-issue. +struct CellWrite<'a> { + tenant: TenantId, + target: DatabaseTarget, + array: &'a str, + vshard: u32, +} + +impl CellWrite<'_> { + /// Re-issue `run`: all puts or all deletes. + async fn run(&self, state: &SharedState, run: Vec) -> Result<(), Error> { + let Some(first) = run.first() else { + return Ok(()); + }; + let puts = first.payload.is_some(); + let array_id = ArrayId::in_database(self.tenant, self.target.dest, self.array); + let key = CollectionKey::from_bare(self.target.dest, self.array); + let op = if puts { + let mut staged = Vec::with_capacity(run.len()); + for version in run { + let Some(payload) = version.payload else { + continue; + }; + let pk = + zerompk::to_msgpack_vec(&version.coord).map_err(|e| codec(self.array, e))?; + staged.push((pk, version.coord, version.system_from_ms, payload)); + } + // Every cell's surrogate in one batch at the array's home. + let pks: Vec<&[u8]> = staged.iter().map(|(pk, ..)| pk.as_slice()).collect(); + let surrogates = crate::control::server::surrogate_exchange::assign_surrogates_routed( + state, + key, + self.tenant, + &pks, + crate::types::TraceId::ZERO, + ) + .await?; + let cells: Vec = staged + .into_iter() + .zip(surrogates) + .map( + |((_, coord, system_from_ms, payload), surrogate)| ArrayPutCell { + surrogate, + coord, + attrs: payload.attrs, + system_from_ms, + valid_from_ms: payload.valid_from_ms, + valid_until_ms: payload.valid_until_ms, + }, + ) + .collect(); + ArrayOp::Put { + array_id, + cells_msgpack: zerompk::to_msgpack_vec(&cells).map_err(|e| codec(self.array, e))?, + wal_lsn: 0, + provenance: None, + vshard_id: self.vshard, + } + } else { + let cells: Vec = run + .into_iter() + .map(|version| ArrayDeleteCell { + coord: version.coord, + system_from_ms: version.system_from_ms, + erasure: version.erased, + }) + .collect(); + ArrayOp::Delete { + array_id, + coords_msgpack: zerompk::to_msgpack_vec(&cells) + .map_err(|e| codec(self.array, e))?, + wal_lsn: 0, + provenance: None, + vshard_id: self.vshard, + } + }; + self.write(state, PhysicalPlan::Array(op)).await + } + + /// Write `plan` durably: propose it to the group of the vShard its cells + /// route to. + async fn write(&self, state: &SharedState, plan: PhysicalPlan) -> Result<(), Error> { + let dest: DatabaseId = self.target.dest; + let proposer = state.async_raft_proposer()?; + let entry = crate::control::wal_replication::to_replicated_entry( + self.tenant, + dest, + VShardId::new(self.vshard), + &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, + )? + .ok_or_else(|| Error::Internal { + detail: format!( + "restore reissue: the cells restored into array '{}' did not map to a \ + replicated write", + self.array + ), + })? + .with_event_source(EventSource::Restore) + .with_restore_id(self.target.restore_id); + crate::control::wal_replication::propose_replicated_entry(state, proposer, entry).await?; + Ok(()) + } +} diff --git a/nodedb/src/control/backup/restore/bind_conflicts.rs b/nodedb/src/control/backup/restore/bind_conflicts.rs new file mode 100644 index 000000000..976624407 --- /dev/null +++ b/nodedb/src/control/backup/restore/bind_conflicts.rs @@ -0,0 +1,259 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Refuse a re-issue that binds a carried surrogate another key holds. +//! +//! Binds are first-wins per key, and nothing else stops two keys of one +//! collection from naming one surrogate: the second key's row installs +//! over the first's. Before any bind, the re-issue asks this node and the home +//! leader of each carried surrogate which key holds it. A different key fails +//! the re-issue with a constraint error naming the collection, the key and the +//! surrogate. Nothing is bound when the check fails. + +use std::collections::{BTreeMap, HashMap}; + +use futures::future::join_all; +use nodedb_physical::physical_plan::ClusterEventOp; +use nodedb_types::{CollectionKey, DatabaseId, Surrogate}; + +use crate::Error; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::state::SharedState; +use crate::types::{HomedRecord, RecordHomes, TenantId}; + +use super::super::node_snapshot::snapshot_remote; + +/// The constraint a surrogate conflict names. +pub(crate) const SURROGATE_IDENTITY: &str = "surrogate_identity"; + +/// A surrogate a re-issue binds, with the key it names. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct CarriedBind { + /// Bare catalog name of the collection. + pub collection: String, + pub pk: Vec, + pub surrogate: u32, +} + +/// `(collection, surrogate, pk)`: the key a node binds a surrogate to. +type Holder = (String, u32, Vec); + +/// The key this node binds each `(collection, surrogate)` of `entries` to, for +/// the entries it binds at all. +pub(crate) fn local_holders( + state: &SharedState, + tenant_id: u64, + database_id: DatabaseId, + entries: &[(String, u32)], +) -> Result, Error> { + let catalog = state.credentials.catalog(); + let mut holders = Vec::new(); + for (collection, surrogate) in entries { + if let Some(pk) = catalog.get_pk_for_surrogate( + CollectionKey::from_bare(database_id, collection), + TenantId::new(tenant_id), + Surrogate::new(*surrogate), + )? { + holders.push((collection.clone(), *surrogate, pk)); + } + } + Ok(holders) +} + +/// Fail when any carried surrogate is bound to a key other than the one it +/// carries, on this node or on the leader of one of its homes, or when the +/// re-issue itself carries it for two keys. +/// +/// `carried` is an owned `Vec`, never a generic iterator: an iterator of +/// borrowing closures in this future's type makes the caller's future fail +/// its `Send` check. +pub(crate) async fn check_carried_binds( + state: &SharedState, + tenant_id: u64, + database_id: DatabaseId, + carried: Vec, +) -> Result<(), Error> { + let mut expected: BTreeMap<(String, u32), Vec> = BTreeMap::new(); + for bind in carried { + let slot = (bind.collection.clone(), bind.surrogate); + match expected.get(&slot) { + Some(pk) if *pk != bind.pk => return Err(conflict(&bind, pk)), + Some(_) => {} + None => { + expected.insert(slot, bind.pk); + } + } + } + if expected.is_empty() { + return Ok(()); + } + + let by_node = asked_nodes(state, tenant_id, database_id, &expected)?; + let answers = join_all(by_node.into_iter().map(|(node_id, entries)| async move { + if node_id == state.node_id { + local_holders(state, tenant_id, database_id, &entries) + } else { + remote_holders(state, node_id, tenant_id, database_id, entries).await + } + })) + .await; + for answer in answers { + for (collection, surrogate, holder) in answer? { + if let Some(pk) = expected.get(&(collection.clone(), surrogate)) + && *pk != holder + { + let bind = CarriedBind { + collection, + pk: pk.clone(), + surrogate, + }; + return Err(conflict(&bind, &holder)); + } + } + } + Ok(()) +} + +/// The entries each node answers for: this node answers for all of them, and +/// the leader of each home of a carried bind answers for that bind. +fn asked_nodes( + state: &SharedState, + tenant_id: u64, + database_id: DatabaseId, + expected: &BTreeMap<(String, u32), Vec>, +) -> Result>, Error> { + let all: Vec<(String, u32)> = expected.keys().cloned().collect(); + let mut by_node: BTreeMap> = BTreeMap::new(); + by_node.insert(state.node_id, all); + let Some(routing) = state.cluster_routing.as_ref() else { + return Ok(by_node); + }; + let catalog = state.credentials.catalog(); + let mut holds_edges: HashMap = HashMap::new(); + for ((collection, surrogate), pk) in expected { + let edges = match holds_edges.get(collection) { + Some(edges) => *edges, + None => { + let edges = catalog + .get_collection(database_id, tenant_id, collection)? + .is_some_and(|coll| coll.has_implicit_edges); + holds_edges.insert(collection.clone(), edges); + edges + } + }; + let homes = RecordHomes::of(HomedRecord::Bind { + collection: CollectionKey::from_bare(database_id, collection), + key: pk, + holds_edges: edges, + }); + let table = routing.read().unwrap_or_else(|p| p.into_inner()); + for home in homes.iter() { + let leader = table + .group_for_vshard(home.as_u32()) + .ok() + .and_then(|group| table.group_info(group)) + .and_then(|info| { + std::iter::once(info.leader) + .chain(info.members.iter().copied()) + .find(|node| *node != 0) + }); + if let Some(node) = leader + && node != state.node_id + { + by_node + .entry(node) + .or_default() + .push((collection.clone(), *surrogate)); + } + } + } + Ok(by_node) +} + +/// Ask `node_id` which key it binds each entry to. +async fn remote_holders( + state: &SharedState, + node_id: u64, + tenant_id: u64, + database_id: DatabaseId, + entries: Vec<(String, u32)>, +) -> Result, Error> { + let plan = PhysicalPlan::ClusterEvent(ClusterEventOp::SurrogateHolders { + tenant_id, + database_id, + entries, + }); + let body = snapshot_remote(state, node_id, tenant_id, database_id, &plan).await?; + decode_holders(&body, node_id) +} + +/// Encode holders for the wire. +pub(crate) fn encode_holders(holders: Vec) -> Result, Error> { + zerompk::to_msgpack_vec(&holders).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("surrogate holders: encode: {e}"), + }) +} + +fn decode_holders(bytes: &[u8], node_id: u64) -> Result, Error> { + zerompk::from_msgpack(bytes).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("surrogate holders: decode the answer of node {node_id}: {e}"), + }) +} + +/// The conflict error of `bind`, whose surrogate `holder` already holds. The +/// detail is the text a client reads, so it names the collection too. +fn conflict(bind: &CarriedBind, holder: &[u8]) -> Error { + Error::RejectedConstraint { + collection: bind.collection.clone(), + constraint: SURROGATE_IDENTITY.into(), + detail: format!( + "{SURROGATE_IDENTITY}: restored key '{}' of collection '{}' carries surrogate {}, \ + which key '{}' already holds; nothing was re-issued", + String::from_utf8_lossy(&bind.pk), + bind.collection, + bind.surrogate, + String::from_utf8_lossy(holder) + ), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn bind(pk: &str, surrogate: u32) -> CarriedBind { + CarriedBind { + collection: "docs".into(), + pk: pk.as_bytes().to_vec(), + surrogate, + } + } + + #[test] + fn a_conflict_names_the_collection_key_and_surrogate() { + let err = conflict(&bind("a", 7), b"b"); + let Error::RejectedConstraint { + collection, + constraint, + detail, + } = err + else { + panic!("a surrogate conflict is a constraint error"); + }; + assert_eq!(collection, "docs"); + assert_eq!(constraint, SURROGATE_IDENTITY); + assert!(detail.contains("'a'") && detail.contains("7") && detail.contains("'b'")); + assert!( + detail.contains("'docs'"), + "the client-visible detail names the collection: {detail}" + ); + } + + #[test] + fn holders_round_trip_the_wire() { + let holders = vec![("docs".to_string(), 7u32, b"a".to_vec())]; + let bytes = encode_holders(holders.clone()).expect("encode"); + assert_eq!(decode_holders(&bytes, 2).expect("decode"), holders); + } +} diff --git a/nodedb/src/control/backup/restore/crdt_reissue.rs b/nodedb/src/control/backup/restore/crdt_reissue.rs index 437742509..70cff33a7 100644 --- a/nodedb/src/control/backup/restore/crdt_reissue.rs +++ b/nodedb/src/control/backup/restore/crdt_reissue.rs @@ -5,16 +5,11 @@ //! Direct-dispatch snapshot install (`RestoreTenantSnapshot` → //! `import_snapshot_bytes`) is race-prone on a freshly spawned cluster (a //! leaderless group is skipped) and not durable across restart. RESTORE -//! instead re-issues each collection's Loro snapshot through Raft (cluster) -//! or WAL + live dispatch (single-node), routed to the vshard that owns it. - -use std::time::Duration; - -use nodedb_types::id::DatabaseId; +//! instead re-issues each collection's Loro snapshot through Raft, routed to +//! the vshard that owns it. use crate::Error; -use crate::bridge::envelope::{PhysicalPlan, Status}; -use crate::control::server::dispatch_utils::{AutocommitWrite, dispatch_autocommit_write}; +use crate::bridge::envelope::PhysicalPlan; use crate::control::state::SharedState; use crate::event::EventSource; use crate::types::TenantId; @@ -22,23 +17,21 @@ use nodedb_physical::physical_plan::CrdtOp; use super::target::{DatabaseTarget, RestoredName}; -/// Per-import dispatch timeout. Generous: a collection's Loro snapshot may be -/// large. -const REISSUE_TIMEOUT: Duration = Duration::from_secs(120); - /// Re-issue one collection's snapshot import to the data group owning its /// vshard. /// -/// Branches identically to a normal write (and to `durable::reissue_plan_durably`): -/// - Cluster: `to_replicated_entry` + `propose_replicated_entry`. -/// - Single-node: the autocommit funnel appends the redo and installs it. +/// Proposes the import as a normal replicated write does (and as +/// `durable::reissue_plan_durably` does): `to_replicated_entry` + +/// `propose_replicated_entry`. The entry carries `target.restore_id`, so +/// every replica marks the write under that restore. async fn reissue_crdt_collection( state: &SharedState, tenant_id: TenantId, - database_id: DatabaseId, + target: DatabaseTarget, name: RestoredName, bytes: Vec, ) -> crate::Result<()> { + let database_id = target.dest; let vshard = name.key(database_id).vshard(); let plan = PhysicalPlan::Crdt(CrdtOp::ImportSnapshot { tenant_id: tenant_id.as_u64(), @@ -46,62 +39,20 @@ async fn reissue_crdt_collection( bytes, }); - if let Some(proposer) = state.async_raft_proposer() { - let entry = crate::control::wal_replication::to_replicated_entry( - tenant_id, - database_id, - vshard, - &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, - )? - .ok_or_else(|| Error::Internal { - detail: "restore reissue: crdt import did not map to a replicated write".into(), - })? - .with_event_source(EventSource::Restore); - crate::control::wal_replication::propose_replicated_entry(state, proposer, entry).await?; - return Ok(()); - } - - // Single-node: hold the frontier slot across WAL append, live import, and - // the durable-at-ack fsync barrier. The clustered branch above is already - // sequenced by its public proposer. - state - .vshard_admission_sequencer - .run(vshard, || async { - let response = tokio::time::timeout( - REISSUE_TIMEOUT, - dispatch_autocommit_write( - state, - AutocommitWrite { - tenant_id, - database_id, - vshard_id: vshard, - plan, - trace_id: crate::types::TraceId::ZERO, - event_source: EventSource::Restore, - txn_id: None, - }, - ), - ) - .await - .map_err(|_| Error::Internal { - detail: format!( - "restore reissue: CRDT import timed out after {}ms", - REISSUE_TIMEOUT.as_millis() - ), - })??; - if response.status != Status::Ok { - return Err(response - .error_code - .as_deref() - .cloned() - .map(Error::DataPlane) - .unwrap_or_else(|| Error::Internal { - detail: "restore reissue: CRDT import failed without an error code".into(), - })); - } - Ok(()) - }) - .await + let proposer = state.async_raft_proposer()?; + let entry = crate::control::wal_replication::to_replicated_entry( + tenant_id, + database_id, + vshard, + &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, + )? + .ok_or_else(|| Error::Internal { + detail: "restore reissue: crdt import did not map to a replicated write".into(), + })? + .with_event_source(EventSource::Restore) + .with_restore_id(target.restore_id); + crate::control::wal_replication::propose_replicated_entry(state, proposer, entry).await?; + Ok(()) } /// Durably re-issue every restored CRDT collection snapshot of one database. @@ -128,7 +79,7 @@ pub(crate) async fn reissue_crdt_snapshots( }); } let name = target.resolve(&collection)?; - reissue_crdt_collection(state, TenantId::new(tid), target.dest, name, bytes).await?; + reissue_crdt_collection(state, TenantId::new(tid), target, name, bytes).await?; imported += 1; } diff --git a/nodedb/src/control/backup/restore/databases.rs b/nodedb/src/control/backup/restore/databases.rs index 39e643b98..3e57fa6c9 100644 --- a/nodedb/src/control/backup/restore/databases.rs +++ b/nodedb/src/control/backup/restore/databases.rs @@ -19,8 +19,7 @@ use nodedb_types::backup_envelope::{DatabaseBlob, Envelope, SECTION_ORIGIN_DATAB use crate::Error; use crate::control::catalog_entry::CatalogEntry; -use crate::control::catalog_entry::post_apply::quota as quota_apply; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::catalog::{DatabaseDescriptor, DatabaseStatus}; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId}; @@ -77,12 +76,13 @@ pub(super) fn decode_databases(env: &Envelope) -> Result, Erro /// Map every database of `blobs` to its destination. Unless `dry_run`, /// create each database the destination lacks and restore the tenant's -/// quota in each database that has none. -pub(super) fn resolve_databases( +/// quota in each database that has none. Every target carries `restore_id`. +pub(super) async fn resolve_databases( state: &SharedState, tenant_id: u64, blobs: &[DatabaseBlob], dry_run: bool, + restore_id: u64, ) -> Result { let catalog = state.credentials.catalog(); let mut map = DatabaseMap::default(); @@ -95,17 +95,18 @@ pub(super) fn resolve_databases( None if dry_run => continue, None => { map.created += 1; - create_database(state, blob)? + create_database(state, blob).await? } }; if !dry_run && let Some(record) = &blob.tenant_quota { - restore_tenant_quota(state, dest, TenantId::new(tenant_id), record)?; + restore_tenant_quota(state, dest, TenantId::new(tenant_id), record).await?; } map.targets.insert( blob.database_id, DatabaseTarget { source: DatabaseId::new(blob.database_id), dest, + restore_id, }, ); } @@ -113,7 +114,11 @@ pub(super) fn resolve_databases( } /// A restore writes rows, so the destination database must take writes. -fn require_writable(state: &SharedState, id: DatabaseId, name: &str) -> Result<(), Error> { +pub(super) fn require_writable( + state: &SharedState, + id: DatabaseId, + name: &str, +) -> Result<(), Error> { let status = state .credentials .catalog() @@ -133,7 +138,7 @@ fn require_writable(state: &SharedState, id: DatabaseId, name: &str) -> Result<( /// Create the database `blob` describes under a fresh id, with its settings /// and its quota. Returns the new id. -fn create_database(state: &SharedState, blob: &DatabaseBlob) -> Result { +async fn create_database(state: &SharedState, blob: &DatabaseBlob) -> Result { let source: DatabaseDescriptor = zerompk::from_msgpack(&blob.descriptor).map_err(|_| Error::Internal { detail: format!( @@ -141,8 +146,7 @@ fn create_database(state: &SharedState, blob: &DatabaseBlob) -> Result Result Result crate::Result<()> { - let vshard = nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(); + let database_id = target.dest; + let vshard = RecordHomes::of(HomedRecord::Row(nodedb_types::CollectionKey::from_bare( + database_id, + collection, + ))) + .owner(); - if let Some(proposer) = state.async_raft_proposer() { - let entry = crate::control::wal_replication::to_replicated_entry( + // The entry logs resolved rows: a timeseries ingest resolves here, before + // its entry exists. + let resolved = crate::control::write_resolve::resolve_for_log( + state, + crate::control::write_resolve::WriteResolveContext { tenant_id, database_id, - vshard, - &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, - )? - .ok_or_else(|| Error::Internal { - detail: format!( - "restore reissue: the plan restored into '{collection}' did not map to a \ - replicated write" - ), - })? - // Every replica applies the write as restored: AFTER triggers do not - // fire again for it. - .with_event_source(crate::event::EventSource::Restore); - let (_, write_version) = - crate::control::wal_replication::propose_replicated_entry(state, proposer, entry) - .await?; - tracing::debug!( - collection, - vshard_id = vshard.as_u32(), - write_version = write_version.as_u64(), - "restore: re-issued write applied on this node" - ); - return Ok(()); - } - - // Single-node: WAL first (durable for restart replay), then install live. - // The record's outcome-floor window opens before the append and closes - // from the install's outcome. - let owner = RecordOwner { - tenant_id, - database_id, - vshard_id: vshard, - }; - let minted = MintedRecords::open(&state.outcome_floor); - if let Err(error) = minted.append_plan( - &state.wal, - owner, + }, + vshard, &plan, - sync_dispatch::SystemReason::BackupRestore.event_source(), - ) { - // Any record appended before the error never reaches a core. - minted.cancel(&state.wal, owner, 0).await?; - return Err(error); - } - sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::BackupRestore, - tenant_id, - nodedb_types::CollectionKey::from_bare(database_id, collection), - plan, - ) - .with_minted(minted), - REISSUE_TIMEOUT, ) .await?; + let plan = resolved.unwrap_or(plan); + + let proposer = state.async_raft_proposer()?; + let entry = crate::control::wal_replication::to_replicated_entry( + tenant_id, + database_id, + vshard, + &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, + )? + .ok_or_else(|| Error::Internal { + detail: format!( + "restore reissue: the plan restored into '{collection}' did not map to a \ + replicated write" + ), + })? + // Every replica applies the write as restored: AFTER triggers do not + // fire again for it. + .with_event_source(crate::event::EventSource::Restore) + .with_restore_id(target.restore_id); + let (_, write_version) = + crate::control::wal_replication::propose_replicated_entry(state, proposer, entry).await?; + tracing::debug!( + collection, + vshard_id = vshard.as_u32(), + write_version = write_version.as_u64(), + "restore: re-issued write applied on this node" + ); Ok(()) } @@ -118,7 +96,6 @@ pub(super) fn log_reissue_step( group_id = ?group_id, rows, proposer_node = state.node_id, - replicated = state.async_raft_proposer().is_some(), "restore: re-issuing rows" ); } diff --git a/nodedb/src/control/backup/restore/guard.rs b/nodedb/src/control/backup/restore/guard.rs index d22c2febe..8e4c64467 100644 --- a/nodedb/src/control/backup/restore/guard.rs +++ b/nodedb/src/control/backup/restore/guard.rs @@ -38,7 +38,7 @@ use crate::control::security::auth_fence::cluster::{ confirmed_read_index, hosts_group, routed_groups, wait_applied, }; use crate::control::state::SharedState; -use crate::control::state::tenant_marks::{GroupMark, LOCAL_MARK_GROUP}; +use crate::control::state::tenant_marks::GroupMark; use crate::types::{DatabaseId, TraceId}; /// First wait before the guard asks again for the marks of groups whose @@ -60,14 +60,19 @@ pub(super) struct NewestWrite { } /// One group's mark on the wire: `(group_id, commit_hlc, site_code, -/// collection)`. -type WireMark = (u64, u64, u8, String); +/// collection, restore_id)`. +type WireMark = (u64, u64, u8, String, u64); /// The newest committed write of `tenant_id` across every data group, and /// this node's own mark of writes no data group carries. +/// +/// A write RESTORE `own_restore_id` re-issued is this restore's own: a retry +/// of the same envelope skips it. Every other write counts, including one an +/// earlier attempt of another restore re-issued. pub(super) async fn newest_committed_write( state: &Arc, tenant_id: u64, + own_restore_id: u64, ) -> Result, Error> { let mut newest = state.tenant_write_mark(tenant_id).map(|mark| NewestWrite { hlc: mark.hlc, @@ -75,6 +80,9 @@ pub(super) async fn newest_committed_write( collection: mark.origin.collection, }); let mut consider = |mark: GroupMark| { + if is_own_restore_write(&mark, own_restore_id) { + return; + } if newest.as_ref().is_none_or(|current| mark.hlc > current.hlc) { newest = Some(NewestWrite { hlc: mark.hlc, @@ -83,11 +91,6 @@ pub(super) async fn newest_committed_write( }); } }; - // The durable mark of this node's writes while it ran with no Raft groups. - if let Some(mark) = state.tenant_marks.get(LOCAL_MARK_GROUP, tenant_id) { - consider(mark); - } - // The statement deadline, shared by every group and every attempt. let deadline = tokio::time::Instant::now() + Duration::from_secs(state.tuning.network.default_deadline_secs); @@ -257,7 +260,7 @@ pub(crate) async fn local_tenant_marks( for &group_id in group_ids { let index = confirmed_read_index(state, group_id, remaining(deadline)?).await?; wait_applied(state, group_id, index, remaining(deadline)?).await?; - if let Some(mark) = state.tenant_marks.get(group_id, tenant_id) { + for mark in state.tenant_marks.get_all(group_id, tenant_id) { marks.push((group_id, mark)); } } @@ -274,6 +277,7 @@ pub(crate) fn encode_marks(marks: &[(u64, GroupMark)]) -> Result, Error> mark.hlc, mark.site.code(), mark.collection.clone().unwrap_or_default(), + mark.restore_id, ) }) .collect(); @@ -290,19 +294,25 @@ fn decode_marks(bytes: &[u8]) -> Result, Error> { })?; Ok(wire .into_iter() - .map(|(group_id, hlc, site, collection)| { + .map(|(group_id, hlc, site, collection, restore_id)| { ( group_id, GroupMark { hlc, site: crate::control::state::tenant_marks::MarkSite::from_code(site), collection: (!collection.is_empty()).then_some(collection), + restore_id, }, ) }) .collect()) } +/// Whether RESTORE `own_restore_id` re-issued the write `mark` records. +fn is_own_restore_write(mark: &GroupMark, own_restore_id: u64) -> bool { + mark.restore_id != 0 && mark.restore_id == own_restore_id +} + /// The time left before `deadline`, or the deadline error once it passed. fn remaining(deadline: tokio::time::Instant) -> Result { let left = deadline.saturating_duration_since(tokio::time::Instant::now()); @@ -365,6 +375,8 @@ async fn remote_tenant_marks( trace_id: TraceId::generate().0, descriptor_versions: Vec::new(), txn_id: None, + vshard_id: None, + read_groups: Vec::new(), }); let response = tokio::time::timeout_at(deadline, transport.send_rpc(node_id, request)) .await @@ -420,3 +432,38 @@ async fn remote_tenant_marks( .into()), } } + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::state::tenant_marks::MarkSite; + + fn mark(restore_id: u64) -> GroupMark { + GroupMark { + hlc: 10, + site: MarkSite::Restore, + collection: None, + restore_id, + } + } + + #[test] + fn only_the_same_restore_owns_a_mark() { + assert!(is_own_restore_write(&mark(7), 7)); + assert!( + !is_own_restore_write(&mark(8), 7), + "another restore's write counts" + ); + assert!( + !is_own_restore_write(&mark(0), 0), + "a user write always counts" + ); + } + + #[test] + fn marks_keep_their_restore_id_on_the_wire() { + let marks = vec![(3, mark(7)), (4, mark(0))]; + let bytes = encode_marks(&marks).expect("encode"); + assert_eq!(decode_marks(&bytes).expect("decode"), marks); + } +} diff --git a/nodedb/src/control/backup/restore/kv_reissue.rs b/nodedb/src/control/backup/restore/kv_reissue.rs index 061962b06..e187e1839 100644 --- a/nodedb/src/control/backup/restore/kv_reissue.rs +++ b/nodedb/src/control/backup/restore/kv_reissue.rs @@ -2,11 +2,10 @@ //! Durable re-issue of restored KV rows. //! -//! The per-node snapshot install puts a KV row straight into the target -//! node's memtable, with no WAL record and no Raft entry: only that node holds -//! it, and it is gone after a restart. RESTORE re-issues each row as a -//! `KvOp::Put` instead, so every replica of the collection's group applies it -//! and its WAL makes it durable. +//! RESTORE re-issues each captured KV row as a `KvOp::Put`. Every replica of +//! the collection's group applies it, and its WAL makes it durable. The put +//! binds a destination surrogate. The captured surrogate belongs to the +//! source database, so RESTORE ignores it. use nodedb_physical::physical_plan::KvOp; @@ -17,9 +16,10 @@ use crate::types::TenantId; use super::target::DatabaseTarget; -/// One restored KV table's rows: `(key, value, expire_at_ms)`, the shape the -/// KV snapshot captures. `expire_at_ms` is `0` for a row with no TTL. -type KvRows = Vec<(Vec, Vec, u64)>; +/// One restored KV table's rows, in the shape the KV snapshot captures. The +/// carried surrogate is the source database's identity. RESTORE ignores it +/// and binds each row in the destination catalog. +type KvRows = Vec; /// The TTL a restored row keeps at `now_ms`: `Some(0)` for no TTL, the time /// left for a row that has not expired, `None` for a row already expired. @@ -59,13 +59,23 @@ pub(in crate::control::backup::restore) async fn reissue_kv_tables( .duration_since(std::time::UNIX_EPOCH) .map(|d| d.as_millis() as u64) .unwrap_or(0); - for (key, value, expire_at_ms) in rows { - let Some(ttl_ms) = remaining_ttl_ms(expire_at_ms, now_ms) else { - continue; - }; - let surrogate = state - .surrogate_assigner - .assign(name.key(target.dest), tenant, &key)?; + let live: Vec<(Vec, Vec, u64)> = rows + .into_iter() + .filter_map(|(key, value, expire_at_ms, _source_surrogate)| { + remaining_ttl_ms(expire_at_ms, now_ms).map(|ttl_ms| (key, value, ttl_ms)) + }) + .collect(); + // Every live row's surrogate in one batch at the table's home. + let keys: Vec<&[u8]> = live.iter().map(|(key, _, _)| key.as_slice()).collect(); + let surrogates = crate::control::server::surrogate_exchange::assign_surrogates_routed( + state, + name.key(target.dest), + tenant, + &keys, + crate::types::TraceId::ZERO, + ) + .await?; + for ((key, value, ttl_ms), surrogate) in live.into_iter().zip(surrogates) { let plan = PhysicalPlan::Kv(KvOp::Put { collection: name.stored.clone(), key, @@ -76,8 +86,7 @@ pub(in crate::control::backup::restore) async fn reissue_kv_tables( rls_filters: Vec::new(), provenance: None, }); - super::durable::reissue_plan_durably(state, tenant, target.dest, &name.bare, plan) - .await?; + super::durable::reissue_plan_durably(state, tenant, target, &name.bare, plan).await?; reissued += 1; } } diff --git a/nodedb/src/control/backup/restore/mod.rs b/nodedb/src/control/backup/restore/mod.rs index 0ff463c4a..db1a39f23 100644 --- a/nodedb/src/control/backup/restore/mod.rs +++ b/nodedb/src/control/backup/restore/mod.rs @@ -8,6 +8,8 @@ //! resolved in `databases`; `target` maps a source collection name to its //! destination database. +mod array_reissue; +pub(crate) mod bind_conflicts; pub mod columnar_reissue; pub(crate) mod crdt_reissue; mod databases; @@ -17,9 +19,11 @@ mod kv_reissue; mod orchestrate; mod quorum; mod redo_reissue; -mod sections; +pub(in crate::control::backup) mod sections; +mod surrogate_floor; mod target; pub mod timeseries_reissue; +mod validate; pub mod vector_reissue; -pub use orchestrate::{RestoreStats, restore_tenant}; +pub use orchestrate::{CollectionRows, RestoreStats, reissue_into_database, restore_tenant}; diff --git a/nodedb/src/control/backup/restore/orchestrate/database.rs b/nodedb/src/control/backup/restore/orchestrate/database.rs index 6e6b64bc8..f8d4615a0 100644 --- a/nodedb/src/control/backup/restore/orchestrate/database.rs +++ b/nodedb/src/control/backup/restore/orchestrate/database.rs @@ -5,26 +5,57 @@ //! Every section re-issues as durable writes: Raft-replicated to every //! replica of its group in cluster mode, WAL-appended then installed on a //! single node. None is installed straight into a Data-Plane map, which -//! would hold it on one node only and lose it on restart. Every write names +//! holds it on one node only and loses it on restart. Every write names //! the destination database, so each row lands on the vShard and core its //! destination collection key homes to. -use std::sync::Arc; - use crate::Error; use crate::control::state::SharedState; -use crate::types::TenantDataSnapshot; +use crate::types::{DatabaseId, TenantDataSnapshot}; +use super::super::bind_conflicts::{CarriedBind, check_carried_binds}; use super::super::target::DatabaseTarget; use super::rebind; use super::reissue; use super::stats::RestoreStats; +/// Re-issue `snap`, one tenant's data of database `source` gathered from +/// every vShard, into database `dest`. Each collection of `dest` must already +/// exist. Every Raft group must have a reachable majority, or nothing is +/// written. +pub async fn reissue_into_database( + state: &SharedState, + tenant_id: u64, + source: DatabaseId, + dest: DatabaseId, + snap: TenantDataSnapshot, +) -> Result { + super::super::quorum::require_quorum(state)?; + let mut stats = RestoreStats { + tenant_id, + ..Default::default() + }; + reissue_database( + state, + tenant_id, + // MOVE TENANT is no RESTORE: its writes mark as user writes. + DatabaseTarget { + source, + dest, + restore_id: 0, + }, + snap, + &mut stats, + ) + .await?; + Ok(stats) +} + /// Re-issue every section of `snap`, the merged backup of the source /// database `target.source`, into `target.dest`. Any failure is fatal — no /// warn-and-continue. pub(super) async fn reissue_database( - state: &Arc, + state: &SharedState, tenant_id: u64, target: DatabaseTarget, mut snap: TenantDataSnapshot, @@ -36,16 +67,53 @@ pub(super) async fn reissue_database( let crdt_state = std::mem::take(&mut snap.crdt_state); let kv_tables = std::mem::take(&mut snap.kv_tables); let vector_snapshots = std::mem::take(&mut snap.vectors); + let vector_multi_documents = std::mem::take(&mut snap.vector_multi_documents); // Vector-index config re-issues as `VectorOp::SetParams` before the first // vector `Insert`: the Data Plane creates a (collection, field) HNSW index // on its first `Insert`, from whatever params it holds by then. let vector_params_snapshots = std::mem::take(&mut snap.vector_params); let index_config_snapshots = std::mem::take(&mut snap.index_configs); + let array_cells = std::mem::take(&mut snap.arrays); // The PK→surrogate identity map. It is bound on this node before any // re-issue, so a re-issued row keeps the surrogate the backup stored it // under unless this node already binds its key. let surrogate_binds = std::mem::take(&mut snap.surrogate_pk); + // No node issues a carried surrogate again: the floor rises above the + // highest one before the first bind, on every node. + let highest = surrogate_binds + .iter() + .map(|bind| bind.surrogate) + .max() + .unwrap_or(0) + .max(super::super::redo_reissue::max_row_surrogate( + tenant_id, + &snap.documents, + &snap.documents_versioned, + )?); + super::super::surrogate_floor::raise_surrogate_floor(state, tenant_id, highest).await?; + + // Every surrogate the re-issue binds, checked against this node and each + // one's home leader before any bind: another key holding one fails the + // re-issue with nothing bound. + let documents = super::super::redo_reissue::prepare_documents( + state, + tenant_id, + target, + std::mem::take(&mut snap.documents), + std::mem::take(&mut snap.documents_versioned), + &surrogate_binds, + )?; + let carried: Vec = surrogate_binds + .iter() + .map(|bind| CarriedBind { + collection: bind.collection.clone(), + pk: bind.pk.clone(), + surrogate: bind.surrogate, + }) + .chain(documents.iter().flat_map(|rows| rows.carried())) + .collect(); + check_carried_binds(state, tenant_id, target.dest, carried).await?; rebind::rebind_surrogates(state, target, &surrogate_binds)?; // Document rows, their versions and graph edges re-issue as committed @@ -58,10 +126,8 @@ pub(super) async fn reissue_database( tenant_id, target, super::super::redo_reissue::RestoredRows { - documents: std::mem::take(&mut snap.documents), - documents_versioned: std::mem::take(&mut snap.documents_versioned), + documents, edges: std::mem::take(&mut snap.edges), - binds: &surrogate_binds, }, ) .await?; @@ -107,9 +173,23 @@ pub(super) async fn reissue_database( ) .await?; - // Vector rows, one `VectorOp::Insert` per restored vector. - stats.vectors_reissued += - reissue::reissue_vector_snapshots(state, tenant_id, target, vector_snapshots).await?; + // Vector rows: one `VectorOp::Insert` per single-vector row, and one + // delete-then-insert per multi-vector document. + stats.vectors_reissued += reissue::reissue_vector_snapshots( + state, + tenant_id, + target, + vector_snapshots, + vector_multi_documents, + &surrogate_binds, + ) + .await?; + + // Array cell versions, in system-time order per array and vShard. Each + // array's catalog row was restored before any database re-issues. + stats.array_cells_reissued += + super::super::array_reissue::reissue_array_cells(state, tenant_id, target, array_cells) + .await?; Ok(()) } diff --git a/nodedb/src/control/backup/restore/orchestrate/mod.rs b/nodedb/src/control/backup/restore/orchestrate/mod.rs index 63b9632ff..a1b39206d 100644 --- a/nodedb/src/control/backup/restore/orchestrate/mod.rs +++ b/nodedb/src/control/backup/restore/orchestrate/mod.rs @@ -16,5 +16,6 @@ mod reissue; mod restore; mod stats; +pub use database::reissue_into_database; pub use restore::restore_tenant; -pub use stats::RestoreStats; +pub use stats::{CollectionRows, RestoreStats}; diff --git a/nodedb/src/control/backup/restore/orchestrate/rebind.rs b/nodedb/src/control/backup/restore/orchestrate/rebind.rs index 31a2c8782..dd58aa712 100644 --- a/nodedb/src/control/backup/restore/orchestrate/rebind.rs +++ b/nodedb/src/control/backup/restore/orchestrate/rebind.rs @@ -4,7 +4,6 @@ //! [`super::restore_tenant`]. use std::collections::BTreeSet; -use std::sync::Arc; use nodedb_types::{CollectionKey, Surrogate}; @@ -29,7 +28,7 @@ use super::super::target::DatabaseTarget; /// allocation here reuses it. Every replica binds the identities a re-issued /// write carries as it applies the write. Any bind error is fatal. pub(super) fn rebind_surrogates( - state: &Arc, + state: &SharedState, target: DatabaseTarget, binds: &[SurrogateBindEntry], ) -> Result<(), Error> { @@ -56,7 +55,7 @@ pub(super) fn rebind_surrogates( } pub(super) fn warn_on_tombstoned_restores( - state: &Arc, + state: &SharedState, tenant_id: u64, target: DatabaseTarget, merged: &TenantDataSnapshot, @@ -145,6 +144,7 @@ mod collection_name_tests { const DEFAULT_TARGET: DatabaseTarget = DatabaseTarget { source: DatabaseId::DEFAULT, dest: DatabaseId::DEFAULT, + restore_id: 0, }; #[test] @@ -179,6 +179,7 @@ mod collection_name_tests { let target = DatabaseTarget { source: DatabaseId::new(1025), dest: DatabaseId::new(1030), + restore_id: 0, }; let snap = TenantDataSnapshot { documents: vec![("1025:7:1025/users:0000002a".into(), vec![])], diff --git a/nodedb/src/control/backup/restore/orchestrate/reissue.rs b/nodedb/src/control/backup/restore/orchestrate/reissue.rs index 6bb6f4b5e..b6502334b 100644 --- a/nodedb/src/control/backup/restore/orchestrate/reissue.rs +++ b/nodedb/src/control/backup/restore/orchestrate/reissue.rs @@ -8,16 +8,42 @@ //! the destination-qualified collection, and the write routes by the bare //! name in the destination database. -use std::sync::Arc; - +use nodedb_physical::physical_plan::{ColumnarOp, TimeseriesOp, VectorOp}; use nodedb_types::surrogate::Surrogate; use crate::Error; +use crate::bridge::envelope::PhysicalPlan; use crate::control::state::SharedState; use crate::engine::vector::index_config::IndexConfig; -use crate::types::TenantId; +use crate::types::{SurrogateBindEntry, TenantId}; + +use super::super::target::{DatabaseTarget, RestoredName}; +use super::super::vector_reissue; -use super::super::target::DatabaseTarget; +/// Durably clear a destination collection of an append-only engine before its +/// rows re-issue. +/// +/// A columnar or timeseries ingest appends, so a second re-issue of the same +/// rows, after a failed restore or cutover, holds every row twice. The +/// re-issue replaces the collection's contents instead: it clears the +/// collection, then writes the captured rows. `truncate` is the engine's +/// whole-collection truncate of `name`. +async fn clear_before_append( + state: &SharedState, + tenant_id: u64, + target: DatabaseTarget, + name: &RestoredName, + truncate: PhysicalPlan, +) -> Result<(), Error> { + super::super::durable::reissue_plan_durably( + state, + TenantId::new(tenant_id), + target, + &name.bare, + truncate, + ) + .await +} /// Decode and durably re-issue every restored timeseries collection. /// @@ -28,7 +54,7 @@ use super::super::target::DatabaseTarget; /// of the two key sets is re-issued once per collection (memtable + flushed rows /// merged into a single ingest). pub(super) async fn reissue_timeseries_snapshots( - state: &Arc, + state: &SharedState, tenant_id: u64, target: DatabaseTarget, memtables: Vec<(String, Vec)>, @@ -80,10 +106,21 @@ pub(super) async fn reissue_timeseries_snapshots( name.stored.as_str(), rows, )?; + clear_before_append( + state, + tenant_id, + target, + &name, + PhysicalPlan::Timeseries(TimeseriesOp::Truncate { + collection: name.stored.clone(), + restart_identity: false, + }), + ) + .await?; super::super::durable::reissue_plan_durably( state, TenantId::new(tenant_id), - target.dest, + target, &name.bare, plan, ) @@ -99,7 +136,7 @@ pub(super) async fn reissue_timeseries_snapshots( /// were re-issued. `entries` are `("{db}:{tid}:{collection}", msgpack)` pairs /// (the `ColumnarEngineSnapshot` wire shape). pub(super) async fn reissue_columnar_snapshots( - state: &Arc, + state: &SharedState, tenant_id: u64, target: DatabaseTarget, entries: Vec<(String, Vec)>, @@ -136,10 +173,21 @@ pub(super) async fn reissue_columnar_snapshots( name.stored.as_str(), decoded, )?; + clear_before_append( + state, + tenant_id, + target, + &name, + PhysicalPlan::Columnar(ColumnarOp::Truncate { + collection: name.stored.clone(), + restart_identity: false, + }), + ) + .await?; super::super::durable::reissue_plan_durably( state, TenantId::new(tenant_id), - target.dest, + target, &name.bare, plan, ) @@ -149,13 +197,13 @@ pub(super) async fn reissue_columnar_snapshots( Ok(reissued) } -/// Decode and durably re-issue every restored vector as an individual -/// `VectorOp::Insert`. +/// Decode and durably re-issue every restored vector: a single-vector row as +/// one `VectorOp::Insert`, a multi-vector document as one `MultiVectorDelete` +/// then one `MultiVectorInsert` of its full set. `multi_documents` names each +/// index's multi-vector documents, keyed like `entries`. /// -/// Returns the number of vectors re-issued. Unlike columnar/timeseries, -/// `VectorOp::Insert` is a single-row op (there is no named-field-aware batch -/// variant), so — unlike the collection-level counts above — this counts -/// individual vectors, one re-issue per restored row. +/// Returns the number of vectors re-issued, counted per vector, not per +/// collection. /// /// `entries` are `("{db}:{tid}:{coll_key}", msgpack)` pairs where `coll_key` /// is `collection` or `collection:field_name` (see @@ -163,12 +211,27 @@ pub(super) async fn reissue_columnar_snapshots( /// `Vec<(u32, Vec, Option)>` — the raw HNSW export shape /// (`node_id`, vector data, surrogate) `VectorCollection::export_snapshot` /// produces. +/// +/// `binds` are the backup's surrogate binds. A single-vector row or a +/// multi-vector document whose surrogate the backup binds to a key re-issues +/// bound by that key. pub(super) async fn reissue_vector_snapshots( - state: &Arc, + state: &SharedState, tenant_id: u64, target: DatabaseTarget, entries: Vec<(String, Vec)>, + multi_documents: Vec<(String, Vec)>, + binds: &[SurrogateBindEntry], ) -> Result { + let keys: std::collections::HashMap<(&str, u32), &[u8]> = binds + .iter() + .map(|bind| { + ( + (bind.collection.as_str(), bind.surrogate), + bind.pk.as_slice(), + ) + }) + .collect(); let mut reissued = 0usize; for (key, bytes) in entries { let coll_key = target.scoped_rest(&key, tenant_id)?; @@ -189,23 +252,54 @@ pub(super) async fn reissue_vector_snapshots( continue; } - for (_node_id, vector, surrogate) in vectors { - let surrogate = surrogate.unwrap_or(Surrogate::ZERO); - let plan = super::super::vector_reissue::build_vector_insert_plan( - name.stored.as_str(), + let members: std::collections::HashSet = multi_documents + .iter() + .filter(|(members_key, _)| *members_key == key) + .flat_map(|(_, documents)| documents.iter().copied()) + .collect(); + let grouped = vector_reissue::group_restored_vectors(vectors, &members)?; + let tenant = TenantId::new(tenant_id); + let stored = name.stored.as_str(); + let mut plans = Vec::new(); + for (surrogate, vector) in grouped.single { + let pk_bytes = keys + .get(&(collection, surrogate.as_u32())) + .map(|pk| pk.to_vec()); + plans.push(vector_reissue::build_vector_insert_plan( + stored, &field_name, vector, surrogate, - ); - super::super::durable::reissue_plan_durably( - state, - TenantId::new(tenant_id), - target.dest, - &name.bare, - plan, - ) - .await?; - reissued += 1; + pk_bytes, + )); + } + for (surrogate, group) in grouped.multi { + // A multi-vector document, whatever its vector count: clear it, + // then insert its full set, so a repeated re-issue neither drops + // nor doubles one. + reissued += group.len(); + plans.push(vector_reissue::build_multi_vector_delete_plan( + stored, + &field_name, + surrogate, + )); + let pk_bytes = keys + .get(&(collection, surrogate.as_u32())) + .map(|pk| pk.to_vec()); + plans.push(vector_reissue::build_multi_vector_insert_plan( + stored, + &field_name, + surrogate, + pk_bytes, + group, + )); + } + for plan in plans { + if matches!(plan, PhysicalPlan::Vector(VectorOp::Insert { .. })) { + reissued += 1; + } + super::super::durable::reissue_plan_durably(state, tenant, target, &name.bare, plan) + .await?; } } Ok(reissued) @@ -218,7 +312,7 @@ pub(super) async fn reissue_vector_snapshots( /// (`handlers/vector.rs`) lazily creates the Data Plane HNSW index from /// `self.vector_params` on the FIRST `VectorOp::Insert` it sees for a /// (collection, field) — falling back to `HnswParams::default()` when no -/// `SetParams` has landed yet. Re-issuing params after inserts would be a +/// `SetParams` has landed yet. Re-issuing params after inserts is a /// no-op for the already-created index. /// /// `params` are `("{db}:{tid}:{coll_key}", msgpack)` pairs decoding to @@ -232,7 +326,7 @@ pub(super) async fn reissue_vector_snapshots( /// number of (collection, field) configs re-issued. Any failure is fatal — no /// warn-and-continue. pub(super) async fn reissue_vector_params( - state: &Arc, + state: &SharedState, tenant_id: u64, target: DatabaseTarget, params: Vec<(String, Vec)>, @@ -293,7 +387,7 @@ pub(super) async fn reissue_vector_params( super::super::durable::reissue_plan_durably( state, TenantId::new(tenant_id), - target.dest, + target, &name.bare, plan, ) diff --git a/nodedb/src/control/backup/restore/orchestrate/restore.rs b/nodedb/src/control/backup/restore/orchestrate/restore.rs index 9aa51fda5..9fea1e8ea 100644 --- a/nodedb/src/control/backup/restore/orchestrate/restore.rs +++ b/nodedb/src/control/backup/restore/orchestrate/restore.rs @@ -3,22 +3,29 @@ //! `restore_tenant`: validates a backup envelope, maps every backed-up //! database to its destination, merges the sections of each database into //! one `TenantDataSnapshot`, and re-issues every section as durable, -//! replicated writes into its destination database. +//! replicated writes into its destination database. It then captures the +//! destination and checks every collection's row count and digest against +//! the backup's. A DRY RUN validates the envelope and verifies nothing. +use std::collections::BTreeMap; use std::sync::Arc; use nodedb_types::backup_envelope::{ - DEFAULT_MAX_TOTAL_BYTES, EnvelopeError, parse_encrypted as parse_envelope_encrypted, + DEFAULT_MAX_TOTAL_BYTES, Envelope, EnvelopeError, VerificationPhase, + parse_encrypted as parse_envelope_encrypted, }; use crate::Error; +use crate::control::backup::verify::destination::verify_destination; +use crate::control::backup::verify::expect::Expectation; use crate::control::server::shared::ddl::neutral::collection::dispatch_register_from_stored; use crate::control::state::SharedState; -use super::super::databases::{decode_databases, resolve_databases}; -use super::super::sections::{apply_metadata_sections, merge_sections}; +use super::super::databases::{DatabaseMap, resolve_databases}; +use super::super::sections::apply_metadata_sections; +use super::super::validate::{ValidatedEnvelope, validate_envelope}; use super::rebind; -use super::stats::RestoreStats; +use super::stats::{CollectionRows, RestoreStats}; /// Restore a tenant from a fully-buffered backup envelope. pub async fn restore_tenant( @@ -28,23 +35,7 @@ pub async fn restore_tenant( dry_run: bool, force: bool, ) -> Result { - let env = match &state.backup_kek { - Some(kek) => parse_envelope_encrypted(envelope_bytes, DEFAULT_MAX_TOTAL_BYTES, kek)?, - None => { - return Err(Error::Internal { - detail: "restore: envelope is encrypted but no backup KEK is configured; \ - set [backup_encryption] in the server config" - .into(), - }); - } - }; - if env.meta.tenant_id != tenant_id { - return Err(EnvelopeError::TenantMismatch { - expected: tenant_id, - actual: env.meta.tenant_id, - } - .into()); - } + let env = open_envelope(state, tenant_id, envelope_bytes)?; // Every group the restore reads or writes has a reachable majority, or // the restore fails here, before it proposes anything. @@ -52,41 +43,11 @@ pub async fn restore_tenant( super::super::quorum::require_quorum(state)?; } - let newest = if !dry_run && env.meta.snapshot_watermark != 0 { - super::super::guard::newest_committed_write(state, tenant_id).await? - } else { - None - }; - if let Some(mark) = newest { - let current_high_water = mark.hlc; - if env.meta.snapshot_watermark < current_high_water { - if force { - tracing::warn!( - tenant_id, - envelope_watermark = env.meta.snapshot_watermark, - current_high_water, - newest_write_site = mark.site.as_str(), - newest_write_collection = mark.collection.as_deref().unwrap_or(""), - "restore staleness protection explicitly overridden via FORCE: \ - envelope watermark is older than the destination cluster's last \ - observed write-HLC for this tenant — newer writes will be overwritten" - ); - } else { - return Err(Error::Internal { - detail: format!( - "restore refused: envelope watermark {} is older than the \ - destination cluster's last observed write-HLC {} for tenant \ - {} (newest write: {} on collection '{}') — newer writes would \ - be silently overwritten", - env.meta.snapshot_watermark, - current_high_water, - tenant_id, - mark.site, - mark.collection.as_deref().unwrap_or(""), - ), - }); - } - } + // A retry of the same envelope carries the same id, so the guard below + // skips the writes an earlier, failed attempt of it re-issued. + let restore_id = restore_id_of(envelope_bytes); + if !dry_run { + refuse_stale_envelope(state, tenant_id, &env, restore_id, force).await?; } let mut stats = RestoreStats { @@ -97,59 +58,28 @@ pub async fn restore_tenant( ..Default::default() }; + // Every refusal the envelope's content can raise runs here, before the + // first proposal. A refused envelope changes nothing on this cluster. + let ValidatedEnvelope { + databases: database_blobs, + merged, + expectation, + } = validate_envelope(state, tenant_id, &env)?; + // Map every backed-up database to its destination, creating each one the // destination lacks. Every other section names its database by source id. - let database_blobs = decode_databases(&env)?; - let databases = resolve_databases(state, tenant_id, &database_blobs, dry_run)?; + let databases = + resolve_databases(state, tenant_id, &database_blobs, dry_run, restore_id).await?; stats.databases = database_blobs.len(); stats.databases_created = databases.created(); if !dry_run { - let restored_collections = apply_metadata_sections(state, tenant_id, &env, &databases)?; - // Every restored collection's declaration reaches this node's Data - // Plane before any of its rows do. The catalog row alone leaves - // `doc_configs` empty for the collection, and the re-issue below - // would then ingest a timeseries collection's rows into an inferred - // shape: the declared time key becomes an integer field and the row - // is stamped with the restore-time clock. This is the same - // registration a committed DDL and the boot rehydration dispatch, - // and it replaces any registration already present, so a cluster - // applier's own register hook and a later boot seed are both - // idempotent with it. A registration failure fails the restore. - for coll in &restored_collections { - // A classified error keeps its class. Only a machinery failure - // gains the restore context. - dispatch_register_from_stored(state, coll) - .await - .map_err(|e| { - if crate::error_classify::is_unclassified_failure(&e) { - Error::Internal { - detail: format!( - "restore: Data Plane registration of collection '{}' failed: {e}", - coll.name - ), - } - } else { - e - } - })?; - } + restore_metadata(state, tenant_id, &env, &databases).await?; + stats.arrays = + super::super::array_reissue::restore_array_rows(state, tenant_id, &env, &databases) + .await?; } - let merged = merge_sections(&env.sections)?; - for source in merged.keys() { - if !database_blobs - .iter() - .any(|blob| blob.database_id == *source) - { - return Err(Error::Internal { - detail: format!( - "invalid backup format: a data section names database {source}, which the \ - backup's database section does not list" - ), - }); - } - } for (source, snap) in &merged { stats.count_sections(snap); if dry_run { @@ -169,6 +99,9 @@ pub async fn restore_tenant( } if dry_run { + stats.collection_rows = envelope_collection_rows(&expectation, |source| { + databases.get(source).map(|target| target.dest.as_u64()) + }); return Ok(stats); } @@ -177,5 +110,197 @@ pub async fn restore_tenant( let target = databases.target(source)?; super::database::reissue_database(state, tenant_id, target, snap, &mut stats).await?; } + + verify_restored(state, tenant_id, expectation, &databases, &mut stats).await?; Ok(stats) } + +/// Decrypt and parse the envelope, and check that it holds `tenant_id`. +fn open_envelope( + state: &SharedState, + tenant_id: u64, + envelope_bytes: &[u8], +) -> Result { + let Some(kek) = &state.backup_kek else { + return Err(Error::Internal { + detail: "restore: envelope is encrypted but no backup KEK is configured; \ + set [backup_encryption] in the server config" + .into(), + }); + }; + let env = parse_envelope_encrypted(envelope_bytes, DEFAULT_MAX_TOTAL_BYTES, kek)?; + if env.meta.tenant_id != tenant_id { + return Err(EnvelopeError::TenantMismatch { + expected: tenant_id, + actual: env.meta.tenant_id, + } + .into()); + } + Ok(env) +} + +/// Refuse an envelope older than the tenant's newest committed write, unless +/// `force` overrides it. An envelope with no watermark passes. +async fn refuse_stale_envelope( + state: &Arc, + tenant_id: u64, + env: &Envelope, + restore_id: u64, + force: bool, +) -> Result<(), Error> { + if env.meta.snapshot_watermark == 0 { + return Ok(()); + } + let Some(mark) = + super::super::guard::newest_committed_write(state, tenant_id, restore_id).await? + else { + return Ok(()); + }; + let current_high_water = mark.hlc; + if env.meta.snapshot_watermark >= current_high_water { + return Ok(()); + } + if !force { + return Err(Error::Internal { + detail: format!( + "restore refused: envelope watermark {} is older than the \ + destination cluster's last observed write-HLC {} for tenant \ + {} (newest write: {} on collection '{}') — newer writes would \ + be silently overwritten", + env.meta.snapshot_watermark, + current_high_water, + tenant_id, + mark.site, + mark.collection.as_deref().unwrap_or(""), + ), + }); + } + tracing::warn!( + tenant_id, + envelope_watermark = env.meta.snapshot_watermark, + current_high_water, + newest_write_site = mark.site.as_str(), + newest_write_collection = mark.collection.as_deref().unwrap_or(""), + "restore staleness protection explicitly overridden via FORCE: \ + envelope watermark is older than the destination cluster's last \ + observed write-HLC for this tenant — newer writes will be overwritten" + ); + Ok(()) +} + +/// Apply the metadata sections and register every restored collection with +/// this node's Data Plane. +/// +/// Every restored collection's declaration reaches this node's Data Plane +/// before any of its rows do. The catalog row alone leaves `doc_configs` empty +/// for the collection, and the re-issue then ingests a timeseries +/// collection's rows into an inferred shape: the declared time key becomes an +/// integer field and the row is stamped with the restore-time clock. This is +/// the same registration a committed DDL and the boot rehydration dispatch, +/// and it replaces any registration already present, so a cluster applier's +/// own register hook and a later boot seed are both idempotent with it. A +/// registration failure fails the restore. +async fn restore_metadata( + state: &Arc, + tenant_id: u64, + env: &Envelope, + databases: &DatabaseMap, +) -> Result<(), Error> { + let restored_collections = apply_metadata_sections(state, tenant_id, env, databases).await?; + for coll in &restored_collections { + // A classified error keeps its class. Only a machinery failure + // gains the restore context. + dispatch_register_from_stored(state, coll) + .await + .map_err(|e| { + if crate::error_classify::is_unclassified_failure(&e) { + Error::Internal { + detail: format!( + "restore: Data Plane registration of collection '{}' failed: {e}", + coll.name + ), + } + } else { + e + } + })?; + } + Ok(()) +} + +/// Check that the destination holds every backed-up row, and record the +/// verified counts in `stats`. A mismatch fails the restore and leaves the +/// restored data in place. +async fn verify_restored( + state: &Arc, + tenant_id: u64, + expectation: Expectation, + databases: &DatabaseMap, + stats: &mut RestoreStats, +) -> Result<(), Error> { + let verified = verify_destination( + state, + tenant_id, + expectation, + |source| databases.target(source).map(|target| target.dest), + VerificationPhase::Destination, + ) + .await?; + stats.verified_collections = verified.collections; + stats.verified_rows = verified.rows; + stats.collection_rows = verified + .per_collection + .into_iter() + .map(|((database_id, collection), rows)| CollectionRows { + database_id: Some(database_id), + collection, + rows, + }) + .collect(); + Ok(()) +} + +/// The rows per collection the envelope holds. `dest_of` maps a source +/// database id to its destination, `None` for one this cluster lacks. +fn envelope_collection_rows( + expectation: &Expectation, + dest_of: impl Fn(u64) -> Option, +) -> Vec { + let mut rows: BTreeMap<(Option, &str), u64> = BTreeMap::new(); + for (source, database) in &expectation.databases { + let dest = dest_of(*source); + for ((collection, _), tally) in &database.rows.tallies { + *rows.entry((dest, collection.as_str())).or_default() += tally.count; + } + } + rows.into_iter() + .map(|((database_id, collection), rows)| CollectionRows { + database_id, + collection: collection.to_string(), + rows, + }) + .collect() +} + +/// The id of a RESTORE of `envelope_bytes`: the first eight bytes of the +/// envelope's SHA-256, never `0`. Every retry of one envelope gets the same id. +fn restore_id_of(envelope_bytes: &[u8]) -> u64 { + use sha2::{Digest, Sha256}; + let digest = Sha256::digest(envelope_bytes); + let mut head = [0u8; 8]; + head.copy_from_slice(&digest[..8]); + u64::from_le_bytes(head) | 1 +} + +#[cfg(test)] +mod tests { + use super::restore_id_of; + + #[test] + fn one_envelope_has_one_nonzero_restore_id() { + let a = restore_id_of(b"envelope a"); + assert_eq!(a, restore_id_of(b"envelope a")); + assert_ne!(a, restore_id_of(b"envelope b")); + assert_ne!(restore_id_of(b""), 0); + } +} diff --git a/nodedb/src/control/backup/restore/orchestrate/stats.rs b/nodedb/src/control/backup/restore/orchestrate/stats.rs index fc9559539..a9a995205 100644 --- a/nodedb/src/control/backup/restore/orchestrate/stats.rs +++ b/nodedb/src/control/backup/restore/orchestrate/stats.rs @@ -47,6 +47,28 @@ pub struct RestoreStats { pub edges_reissued: usize, /// Redo records the document and edge re-issue committed. pub redo_records: usize, + /// Array catalog rows created or applied to an existing array. + pub arrays: usize, + /// Array cell versions re-issued. + pub array_cells_reissued: usize, + /// Collection parts whose destination row count and digest matched the + /// backup's. `0` on a dry run, which verifies nothing. + pub verified_collections: usize, + /// Rows those collection parts hold. + pub verified_rows: u64, + /// Rows per restored collection. A restore lists the rows it verified. A + /// dry run lists the rows the envelope holds. + pub collection_rows: Vec, +} + +/// The rows one collection of a restore holds. +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct CollectionRows { + /// The destination database, `None` on a dry run for a database this + /// cluster lacks. + pub database_id: Option, + pub collection: String, + pub rows: u64, } impl RestoreStats { @@ -65,4 +87,67 @@ impl RestoreStats { self.flushed_ts_segments += snap.flushed_ts_segments.len(); self.surrogate_pk += snap.surrogate_pk.len(); } + + /// Add the counts of another tenant's restore. A database restore sums + /// its tenants. `tenant_id`, `dry_run` and `source_vshard_count` stay. + pub fn absorb(&mut self, other: &RestoreStats) { + // Destructure exhaustively so a new count is not dropped here. + let RestoreStats { + tenant_id: _, + dry_run: _, + sections, + databases, + databases_created, + source_vshard_count: _, + documents, + indexes, + edges, + vectors, + kv_tables, + crdt_state, + timeseries, + columnar_engines, + flushed_ts_segments, + timeseries_reissued, + crdt_reissued, + vectors_reissued, + kv_reissued, + vector_params_reissued, + surrogate_pk, + documents_reissued, + edges_reissued, + redo_records, + arrays, + array_cells_reissued, + verified_collections, + verified_rows, + collection_rows, + } = other; + self.sections = self.sections.saturating_add(*sections); + self.databases = self.databases.max(*databases); + self.databases_created += databases_created; + self.documents += documents; + self.indexes += indexes; + self.edges += edges; + self.vectors += vectors; + self.kv_tables += kv_tables; + self.crdt_state += crdt_state; + self.timeseries += timeseries; + self.columnar_engines += columnar_engines; + self.flushed_ts_segments += flushed_ts_segments; + self.timeseries_reissued += timeseries_reissued; + self.crdt_reissued += crdt_reissued; + self.vectors_reissued += vectors_reissued; + self.kv_reissued += kv_reissued; + self.vector_params_reissued += vector_params_reissued; + self.surrogate_pk += surrogate_pk; + self.documents_reissued += documents_reissued; + self.edges_reissued += edges_reissued; + self.redo_records += redo_records; + self.arrays += arrays; + self.array_cells_reissued += array_cells_reissued; + self.verified_collections += verified_collections; + self.verified_rows += verified_rows; + self.collection_rows.extend(collection_rows.iter().cloned()); + } } diff --git a/nodedb/src/control/backup/restore/redo_reissue/commit.rs b/nodedb/src/control/backup/restore/redo_reissue/commit.rs index eae5c35bb..5bf546e52 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/commit.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/commit.rs @@ -1,52 +1,60 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Commit restored units as redo records through their vShard's apply log. +//! Commit restored units as Calvin transactions. //! -//! A restored row installs exactly as a committed transaction's row does. With -//! Raft the record is proposed to the collection's data group, and every -//! replica binds its identities, appends it to its own WAL and installs it. -//! With no Raft this node runs the same apply alone. Either way the call -//! returns once the record is durable and installed here. +//! A RESTORE re-issues its rows and edge versions in the Calvin sequence, the +//! one ordering domain every other write of them takes. Each batch is a +//! `MetaOp::RestoreRedo` plan. Its transaction locks the rows and edges it +//! writes, as their live writers lock them. Its resolve appends them to the +//! transaction's redo record, each edge version applied at the transaction's +//! ordinal, so a TRUNCATE sequenced before the RESTORE leaves them visible +//! and one sequenced after it hides them. Every replica binds the batch's +//! identities, installs the record as a RESTORE, and raises the tenant's +//! restore mark under the restore's id. The call returns once the +//! transaction committed. -use std::collections::HashSet; +use std::collections::{BTreeMap, HashSet}; -use nodedb_physical::physical_plan::RedoOrigin; +use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan, RestoredIdentity, RestoredRedo}; +use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; -use crate::bridge::envelope::{ErrorCode, Status}; +use crate::control::planner::calvin::submit::submit_calvin_routed; +use crate::control::planner::calvin::tx_class::build_single_vshard_tx_class; use crate::control::state::SharedState; use crate::control::surrogate::CarriedIdentity; -use crate::control::wal_replication::encode::transaction_redo_entry; -use crate::control::wal_replication::propose_replicated_entry; -use crate::control::wal_replication::transaction_redo::{ - RedoTarget, TransactionRedoPayload, apply_transaction_redo, -}; use crate::event::EventSource; -use crate::types::TenantId; -use crate::wal::{RedoRecord, RedoSubRecord}; +use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::wal::RedoRecord; -use super::units::{CollectionUnits, RowUnit}; +use super::units::{CollectionEdges, CollectionUnits, EdgeUnit, RowUnit}; -/// Most sub-records one restore record carries. -const MAX_OPS_PER_RECORD: usize = 512; +/// Most sub-records or edge versions one restore batch carries. +const MAX_UNITS_PER_BATCH: usize = 512; -/// Most encoded bytes one restore record carries. A single unit larger than -/// this still commits, alone in its own record. -const MAX_BYTES_PER_RECORD: usize = 4 * 1024 * 1024; +/// Most encoded bytes one restore batch carries. A single unit larger than +/// this still commits, alone in its own batch. +const MAX_BYTES_PER_BATCH: usize = 4 * 1024 * 1024; -/// Split `units` into record-sized batches, in order. A unit never splits. -fn batch_units(units: Vec) -> Vec> { +/// The (collection, key) identities one plan already carries. +type SeenIdentities = HashSet<(String, Vec)>; + +/// Split `units` into batch-sized groups, in order. A unit never splits. +/// `weight` names a unit's count toward [`MAX_UNITS_PER_BATCH`] and its +/// encoded size. +fn batch_units(units: Vec, weight: impl Fn(&T) -> (usize, usize)) -> Vec> { let mut batches = Vec::new(); - let mut current: Vec = Vec::new(); - let (mut ops, mut bytes) = (0usize, 0usize); + let mut current: Vec = Vec::new(); + let (mut count, mut bytes) = (0usize, 0usize); for unit in units { - let (unit_ops, unit_bytes) = (unit.ops.len(), unit.byte_len()); + let (unit_count, unit_bytes) = weight(&unit); if !current.is_empty() - && (ops + unit_ops > MAX_OPS_PER_RECORD || bytes + unit_bytes > MAX_BYTES_PER_RECORD) + && (count + unit_count > MAX_UNITS_PER_BATCH + || bytes + unit_bytes > MAX_BYTES_PER_BATCH) { batches.push(std::mem::take(&mut current)); - (ops, bytes) = (0, 0); + (count, bytes) = (0, 0); } - ops += unit_ops; + count += unit_count; bytes += unit_bytes; current.push(unit); } @@ -56,119 +64,212 @@ fn batch_units(units: Vec) -> Vec> { batches } -/// One batch as the payload every replica applies. `stored` is the name the -/// Data Plane stores the collection under, as a committed transaction's -/// written-collection list names it. -fn batch_payload(stored: &str, batch: Vec) -> TransactionRedoPayload { - let mut ops: Vec = Vec::new(); - let mut identities: Vec = Vec::new(); - let mut seen: HashSet<(String, Vec)> = HashSet::new(); +/// The identities of `carried`, each once, in order. +fn push_identities( + out: &mut Vec, + seen: &mut SeenIdentities, + carried: Vec, +) { + for identity in carried { + if seen.insert((identity.collection.clone(), identity.pk_bytes.clone())) { + out.push(RestoredIdentity { + collection: identity.collection, + pk_bytes: identity.pk_bytes, + surrogate: identity.surrogate.as_u32(), + }); + } + } +} + +/// One batch of rows of the collection stored as `stored`, on `vshard_id`. +fn row_batch( + stored: &str, + vshard_id: VShardId, + batch: Vec, +) -> crate::Result { + let mut ops = Vec::new(); + let mut row_changes = Vec::new(); + let mut rows = Vec::new(); + let mut identities = Vec::new(); + let mut seen = HashSet::new(); for unit in batch { ops.extend(unit.ops); - for identity in unit.identities { - if seen.insert((identity.collection.clone(), identity.pk_bytes.clone())) { - identities.push(identity); - } - } + row_changes.extend(unit.changes); + rows.extend(unit.rows); + push_identities(&mut identities, &mut seen, unit.identities); } - TransactionRedoPayload { - redo: RedoRecord { - version: 1, - ops, - calvin_stamp: None, - }, + let rows_redo = RedoRecord { + version: 1, + ops, + calvin_stamp: None, + cross_shard_applied: None, + row_sources: Vec::new(), + publishes: Vec::new(), + row_changes, + } + .to_bytes()?; + Ok(RestoredRedo { + vshard: vshard_id.as_u32(), + rows_redo, + rows, + edges: Vec::new(), collections: vec![stored.to_string()], - // The backup holds every target row with its total already folded in. - sum_targets: Vec::new(), identities, - // Every replica applies the rows as restored: AFTER triggers fired - // when the rows were first written, and do not fire again. - event_source: EventSource::Restore, - origin: RedoOrigin::Restore, + }) +} + +/// One batch of edge versions of the collection stored as `stored`, as one +/// plan per home: each version on every home it lives on, in order. +fn edge_batch(stored: &str, batch: Vec) -> Vec { + let mut homes: BTreeMap = BTreeMap::new(); + for unit in batch { + for home in unit.homes.iter() { + let (plan, seen) = homes.entry(home).or_insert_with(|| { + ( + RestoredRedo { + vshard: home.as_u32(), + rows_redo: Vec::new(), + rows: Vec::new(), + edges: Vec::new(), + collections: vec![stored.to_string()], + identities: Vec::new(), + }, + HashSet::new(), + ) + }); + plan.edges.push(unit.version.clone()); + push_identities(&mut plan.identities, seen, unit.identities.clone()); + } } + homes.into_values().map(|(plan, _)| plan).collect() } -/// Commit one record and wait until it is durable and installed here. -async fn commit_record( +/// Commit `plans` as one Calvin transaction of RESTORE `restore_id`, and +/// wait until it committed. +async fn commit_batch( state: &SharedState, - target: RedoTarget, - payload: &TransactionRedoPayload, + tenant_id: TenantId, + database_id: DatabaseId, + restore_id: u64, + plans: Vec, ) -> crate::Result<()> { - super::super::durable::log_reissue_step( - state, - "redo", - payload.collections.first().map_or("", String::as_str), - target.vshard_id, - payload.redo.ops.len(), - ); - if let Some(proposer) = state.async_raft_proposer() { - let entry = transaction_redo_entry( - target.tenant_id, - target.database_id, - target.vshard_id, - payload, - ); - propose_replicated_entry(state, proposer, entry).await?; - return Ok(()); - } - let outcome = apply_transaction_redo(state, target, payload, 0, None).await?; - if outcome.response.status == Status::Ok { - return Ok(()); - } - Err(crate::Error::DataPlane( - outcome - .response - .error_code - .as_deref() - .cloned() - .unwrap_or_else(|| ErrorCode::Internal { - detail: "restore redo apply returned an error status with no error code".into(), - }), - )) + let tasks: Vec = plans + .into_iter() + .map(|plan| PhysicalTask { + tenant_id, + vshard_id: VShardId::new(plan.vshard), + database_id, + plan: PhysicalPlan::Meta(MetaOp::RestoreRedo(Box::new(plan))), + post_set_op: PostSetOp::None, + txn_id: None, + }) + .collect(); + let mut tx_class = build_single_vshard_tx_class(&tasks, tenant_id, &[])?; + // Every replica installs the rows as restored: AFTER triggers fired when + // the rows were first written, and do not fire again. + tx_class.set_event_source(EventSource::Restore.wal_code()); + tx_class.set_restore_id(restore_id); + if let Some(response) = submit_calvin_routed(state, tx_class).await? { + crate::control::local_dispatch::reject_data_plane_error(&response)?; + } + Ok(()) +} + +/// A machinery failure with the restore's context. A classified error keeps +/// its class. +fn in_context(e: crate::Error, what: &str) -> crate::Error { + if crate::error_classify::is_unclassified_failure(&e) { + crate::Error::Internal { + detail: format!("restore: re-issuing {what} failed: {e}"), + } + } else { + e + } } -/// Commit every unit of `units` in order. Returns the records committed. +/// Commit every row unit of `units` in order, marked under `restore_id`. +/// Returns the transactions committed. pub(super) async fn commit_collection( state: &SharedState, tenant_id: TenantId, + restore_id: u64, units: CollectionUnits, ) -> crate::Result { let CollectionUnits { database_id, collection, + vshard_id, units, } = units; - let target = RedoTarget { - tenant_id, + let stored = nodedb_types::QualifiedCollection::new(database_id, &collection); + let mut transactions = 0usize; + for batch in batch_units(units, |unit| (unit.ops.len(), unit.byte_len())) { + let plan = row_batch(stored.as_str(), vshard_id, batch)?; + super::super::durable::log_reissue_step( + state, + "redo", + stored.as_str(), + vshard_id, + plan.rows.len(), + ); + commit_batch(state, tenant_id, database_id, restore_id, vec![plan]) + .await + .map_err(|e| { + in_context( + e, + &format!("rows of '{collection}' to vShard {}", vshard_id.as_u32()), + ) + })?; + transactions += 1; + } + Ok(transactions) +} + +/// Commit every edge version of `edges` in order, marked under +/// `restore_id`. Each transaction writes a batch of versions on every home +/// they live on, so an error never leaves a version on one home only. +/// Returns the transactions committed. +pub(super) async fn commit_edges( + state: &SharedState, + tenant_id: TenantId, + restore_id: u64, + edges: CollectionEdges, +) -> crate::Result { + let CollectionEdges { database_id, - vshard_id: nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(), - }; + collection, + units, + } = edges; let stored = nodedb_types::QualifiedCollection::new(database_id, &collection); - let mut records = 0usize; - for batch in batch_units(units) { - let payload = batch_payload(stored.as_str(), batch); - // A classified error keeps its class. Only a machinery failure gains - // the restore context. - commit_record(state, target, &payload).await.map_err(|e| { - if crate::error_classify::is_unclassified_failure(&e) { - crate::Error::Internal { - detail: format!("restore: re-issuing rows of '{collection}' failed: {e}"), - } - } else { - e - } - })?; - records += 1; - } - Ok(records) + let mut transactions = 0usize; + for batch in batch_units(units, |unit| (1, unit.byte_len())) { + let versions = batch.len(); + let plans = edge_batch(stored.as_str(), batch); + if let Some(first) = plans.first() { + super::super::durable::log_reissue_step( + state, + "edges", + stored.as_str(), + VShardId::new(first.vshard), + versions, + ); + } + commit_batch(state, tenant_id, database_id, restore_id, plans) + .await + .map_err(|e| in_context(e, &format!("edges of '{collection}'")))?; + transactions += 1; + } + Ok(transactions) } #[cfg(test)] mod tests { + use nodedb_physical::physical_plan::RestoredEdgeVersion; use nodedb_types::Surrogate; use super::*; - use crate::types::VShardId; + use crate::types::RecordHomes; + use crate::wal::RedoSubRecord; fn unit(ops: usize, payload_len: usize, pk: &str) -> RowUnit { RowUnit { @@ -183,15 +284,52 @@ mod tests { pk_bytes: pk.as_bytes().to_vec(), surrogate: Surrogate::new(1), }], + changes: Vec::new(), + rows: Vec::new(), } } + fn edge(dst: &str, system_from: i64) -> EdgeUnit { + EdgeUnit { + version: RestoredEdgeVersion { + collection: "follows".into(), + src_id: "n0".into(), + label: "L".into(), + dst_id: dst.into(), + src_surrogate: 1, + dst_surrogate: 2, + system_from, + properties: Some(Vec::new()), + }, + identities: vec![ + CarriedIdentity { + collection: "follows".into(), + pk_bytes: b"n0".to_vec(), + surrogate: Surrogate::new(1), + }, + CarriedIdentity { + collection: "follows".into(), + pk_bytes: dst.as_bytes().to_vec(), + surrogate: Surrogate::new(2), + }, + ], + homes: RecordHomes::edge("n0", dst), + } + } + + fn cross_shard_peer() -> String { + (1..4096) + .map(|i| format!("n{i}")) + .find(|peer| !RecordHomes::edge("n0", peer).is_single()) + .expect("a cross-shard edge") + } + #[test] - fn batches_cut_between_units_at_the_op_limit() { + fn batches_cut_between_units_at_the_unit_limit() { let units = (0..3) - .map(|i| unit(MAX_OPS_PER_RECORD / 2, 1, &i.to_string())) + .map(|i| unit(MAX_UNITS_PER_BATCH / 2, 1, &i.to_string())) .collect(); - let batches = batch_units(units); + let batches = batch_units(units, |unit: &RowUnit| (unit.ops.len(), unit.byte_len())); let sizes: Vec = batches.iter().map(Vec::len).collect(); assert_eq!(sizes, vec![2, 1]); } @@ -200,42 +338,90 @@ mod tests { fn an_oversized_unit_commits_alone() { let units = vec![ unit(1, 1, "a"), - unit(1, MAX_BYTES_PER_RECORD + 1, "b"), + unit(1, MAX_BYTES_PER_BATCH + 1, "b"), unit(1, 1, "c"), ]; - let sizes: Vec = batch_units(units).iter().map(Vec::len).collect(); + let sizes: Vec = + batch_units(units, |unit: &RowUnit| (unit.ops.len(), unit.byte_len())) + .iter() + .map(Vec::len) + .collect(); assert_eq!(sizes, vec![1, 1, 1]); } #[test] - fn a_payload_carries_each_identity_once_and_restores_without_folds() { - let payload = batch_payload("c", vec![unit(1, 1, "a"), unit(1, 1, "a")]); - assert_eq!(payload.redo.ops.len(), 2); - assert_eq!(payload.identities.len(), 1); - assert!(payload.sum_targets.is_empty()); - assert_eq!(payload.origin, RedoOrigin::Restore); + fn a_row_batch_carries_each_identity_once() { + let plan = row_batch( + "c", + VShardId::new(3), + vec![unit(1, 1, "a"), unit(1, 1, "a")], + ) + .expect("row batch"); + assert_eq!(plan.vshard, 3); + assert_eq!(plan.identities.len(), 1); + let redo = RedoRecord::from_bytes(&plan.rows_redo).expect("decode rows"); + assert_eq!(redo.ops.len(), 2); + assert!(plan.edges.is_empty()); } + /// A cross-shard edge's versions go to both endpoint homes, each in + /// order, within one transaction's plans. #[test] - fn a_restored_record_carries_the_restore_source_to_every_replica() { - let payload = batch_payload("c", vec![unit(1, 1, "a")]); - assert_eq!(payload.event_source, EventSource::Restore); - let entry = transaction_redo_entry( - TenantId::new(1), - crate::types::DatabaseId::DEFAULT, - VShardId::new(0), - &payload, - ); + fn a_cross_shard_edge_batch_writes_both_homes() { + let peer = cross_shard_peer(); + let homes = RecordHomes::edge("n0", &peer); + let plans = edge_batch("follows", vec![edge(&peer, 1), edge(&peer, 2)]); + let mut want: Vec = homes.iter().map(VShardId::as_u32).collect(); + want.sort(); + let targets: Vec = plans.iter().map(|plan| plan.vshard).collect(); + assert_eq!(targets, want); + for plan in &plans { + let order: Vec = plan.edges.iter().map(|v| v.system_from).collect(); + assert_eq!(order, vec![1, 2], "versions keep their order"); + assert_eq!(plan.identities.len(), 2, "each endpoint binds once"); + } + } + + #[test] + fn a_same_shard_edge_batch_writes_one_home() { + let plans = edge_batch("follows", vec![edge("n0", 1)]); + assert_eq!(plans.len(), 1); assert_eq!( - crate::event::EventSource::from(entry.event_source), - EventSource::Restore + plans[0].vshard, + RecordHomes::edge("n0", "n0").owner().as_u32() ); - match entry.write { - crate::control::wal_replication::ReplicatedWrite::TransactionRedo { - event_source, - .. - } => assert_eq!(EventSource::from(event_source), EventSource::Restore), - other => panic!("expected a transaction redo, got {other:?}"), - } + } + + /// Every plan of a restore transaction names its batch's vShard, and + /// the transaction's write set locks the batch's edges on both homes. + #[test] + fn a_restore_transaction_locks_its_edges_on_both_homes() { + let peer = cross_shard_peer(); + let plans = edge_batch("follows", vec![edge(&peer, 1)]); + let tasks: Vec = plans + .into_iter() + .map(|plan| PhysicalTask { + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(plan.vshard), + database_id: DatabaseId::DEFAULT, + plan: PhysicalPlan::Meta(MetaOp::RestoreRedo(Box::new(plan))), + post_set_op: PostSetOp::None, + txn_id: None, + }) + .collect(); + let tx_class = + build_single_vshard_tx_class(&tasks, TenantId::new(1), &[]).expect("tx class"); + let mut participants: Vec = tx_class + .participating_vshards() + .iter() + .map(|v| v.as_u32()) + .collect(); + participants.sort(); + let mut homes: Vec = RecordHomes::edge("n0", &peer) + .iter() + .map(VShardId::as_u32) + .collect(); + homes.sort(); + assert_eq!(participants, homes); } } diff --git a/nodedb/src/control/backup/restore/redo_reissue/documents.rs b/nodedb/src/control/backup/restore/redo_reissue/documents.rs index 0d13cd069..c1252002e 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/documents.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/documents.rs @@ -26,14 +26,13 @@ use nodedb_types::columnar::StrictSchema; use nodedb_types::{CollectionType, DocumentMode, RowIdentity, StorageKey}; use crate::control::state::SharedState; -use crate::control::surrogate::CarriedIdentity; use crate::data::executor::strict_format::{binary_tuple_to_msgpack, undecodable_strict_row}; use crate::engine::sparse::btree_versioned::{TAG_LIVE, TAG_TOMBSTONE, decode_value}; use crate::types::{DatabaseId, SurrogateBindEntry, TenantId}; use super::super::target::DatabaseTarget; -use super::sub_record::{VersionStamp, document_put, document_tombstone}; -use super::units::{CollectionUnits, RowUnit}; +use super::prepared::{PendingBody, PendingRow, PreparedRows}; +use super::sub_record::VersionStamp; /// A row as the backup stored it. enum StoredRow { @@ -47,6 +46,9 @@ enum StoredRow { struct CollectionShape { strict: Option, declared_primary_key: Option, + /// The collection declares `HASH_CHAIN`: its rows re-issue in their + /// install order, so the destination relinks the source chain. + hash_chain: bool, } fn malformed(key: &str) -> crate::Error { @@ -72,6 +74,28 @@ fn split_key(key: &str, tenant_id: u64) -> crate::Result<(u64, &str, &str)> { Ok((db, collection, rest)) } +/// The highest surrogate any restored row's storage key carries, `0` for +/// none. A row the backup carries no bind for keeps its key's surrogate. +pub(in crate::control::backup::restore) fn max_row_surrogate( + tenant_id: u64, + documents: &[(String, Vec)], + documents_versioned: &[(String, Vec)], +) -> crate::Result { + let mut highest = 0u32; + for (key, _) in documents { + let (_, _, rest) = split_key(key, tenant_id)?; + let storage_key = StorageKey::parse(rest).ok_or_else(|| malformed(key))?; + highest = highest.max(storage_key.surrogate().as_u32()); + } + for (key, _) in documents_versioned { + let (_, _, rest) = split_key(key, tenant_id)?; + let (hex, _) = rest.split_once('\x00').ok_or_else(|| malformed(key))?; + let storage_key = StorageKey::parse(hex).ok_or_else(|| malformed(key))?; + highest = highest.max(storage_key.surrogate().as_u32()); + } + Ok(highest) +} + /// Group every restored row by `(database, collection)`, then by storage key. fn group_rows( tenant_id: u64, @@ -145,9 +169,45 @@ fn collection_shape( Ok(CollectionShape { strict, declared_primary_key: stored.declared_primary_key, + hash_chain: stored.hash_chain, }) } +/// A `HASH_CHAIN` row's install-order position, read from its stored body: +/// the current body, or the first live version. +fn chain_seq( + shape: &CollectionShape, + collection: &str, + key: StorageKey, + row: &StoredRow, +) -> crate::Result { + let body = match row { + StoredRow::Current(body) => Some(body_msgpack(shape, collection, key, body)?), + StoredRow::Versions(versions) => { + let mut live = None; + for (_, raw) in versions { + let version = decode_value(raw)?; + if version.tag == TAG_LIVE { + live = Some(body_msgpack(shape, collection, key, version.body)?); + break; + } + } + live + } + }; + body.and_then(|body| nodedb_types::json_from_msgpack(&body).ok()) + .and_then(|doc| { + doc.get(crate::types::hash_chain::CHAIN_SEQ_FIELD) + .and_then(|seq| seq.as_u64()) + }) + .ok_or_else(|| crate::Error::Serialization { + format: "backup".into(), + detail: format!( + "restore: row {key} of hash-chained '{collection}' carries no chain position" + ), + }) +} + /// A stored body as the MessagePack a put carries. fn body_msgpack( shape: &CollectionShape, @@ -162,17 +222,10 @@ fn body_msgpack( } } -/// Builds each row's unit for one collection. +/// Decodes each row of one collection and names its identity. struct RowBuilder<'a> { - state: &'a SharedState, - /// The destination database. - database_id: DatabaseId, - tenant: TenantId, - /// Bare catalog name: it keys the identity binds. + /// Bare catalog name. collection: &'a str, - /// The name the destination Data Plane stores the collection under: the - /// sub-records carry it. - stored: &'a str, shape: CollectionShape, /// `storage surrogate → primary key` the backup bound for this collection. binds: HashMap, @@ -200,41 +253,17 @@ impl RowBuilder<'_> { }) } - /// Bind the row's identity on this node. The backup's surrogate wins - /// unless this node already binds the identity: the row then installs - /// under that surrogate, over the row it names. - fn bind(&self, identity: &RowIdentity, key: StorageKey) -> crate::Result { - let surrogate = self.state.surrogate_assigner.bind( - nodedb_types::CollectionKey::from_bare(self.database_id, self.collection), - self.tenant, - identity.as_str().as_bytes(), - key.surrogate(), - )?; - Ok(CarriedIdentity { - collection: self.collection.to_string(), - pk_bytes: identity.as_str().as_bytes().to_vec(), - surrogate, - }) - } - - fn current(&self, key: StorageKey, body: &[u8]) -> crate::Result { + fn current(&self, key: StorageKey, body: &[u8]) -> crate::Result { let value = body_msgpack(&self.shape, self.collection, key, body)?; let identity = self.identity(key, Some(&value))?; - let carried = self.bind(&identity, key)?; - let op = document_put( - self.stored, - identity.as_str(), - value, - carried.surrogate.as_u32(), - None, - )?; - Ok(RowUnit { - ops: vec![op], - identities: vec![carried], + Ok(PendingRow { + key, + identity, + body: PendingBody::Current(value), }) } - fn versions(&self, key: StorageKey, versions: &[(i64, Vec)]) -> crate::Result { + fn versions(&self, key: StorageKey, versions: &[(i64, Vec)]) -> crate::Result { // Decode every version first: the identity comes from a live body. let mut decoded = Vec::with_capacity(versions.len()); for (sys_from_ms, raw) in versions { @@ -267,44 +296,26 @@ impl RowBuilder<'_> { } let first_live = decoded.iter().find_map(|(_, body)| body.as_deref()); let identity = self.identity(key, first_live)?; - let carried = self.bind(&identity, key)?; - let surrogate = carried.surrogate.as_u32(); - let mut ops = Vec::with_capacity(decoded.len()); - for (stamp, body) in decoded { - ops.push(match body { - Some(value) => document_put( - self.stored, - identity.as_str(), - value, - surrogate, - Some(stamp), - )?, - None => document_tombstone( - self.stored, - identity.as_str(), - surrogate, - stamp.sys_from_ms, - )?, - }); - } - Ok(RowUnit { - ops, - identities: vec![carried], + Ok(PendingRow { + key, + identity, + body: PendingBody::Versions(decoded), }) } } -/// Every restored row of `tenant_id` in one database, one unit per row, -/// grouped by collection. Every row key must name `target.source`. `binds` -/// is the backup's primary-key section of that database. -pub(super) fn document_units( +/// Every restored row of `tenant_id` in one database, decoded and identified, +/// grouped by collection in install order. Nothing is bound yet. Every row key +/// must name `target.source`. `binds` is the backup's primary-key section of +/// that database. +pub(in crate::control::backup::restore) fn prepare_documents( state: &SharedState, tenant_id: u64, target: DatabaseTarget, documents: Vec<(String, Vec)>, documents_versioned: Vec<(String, Vec)>, binds: &[SurrogateBindEntry], -) -> crate::Result> { +) -> crate::Result> { let grouped = group_rows(tenant_id, documents, documents_versioned)?; let mut out = Vec::with_capacity(grouped.len()); for ((db, stored), rows) in grouped { @@ -320,11 +331,7 @@ pub(super) fn document_units( let name = target.resolve(&stored)?; let database_id = target.dest; let builder = RowBuilder { - state, - database_id, - tenant: TenantId::new(tenant_id), collection: &name.bare, - stored: name.stored.as_str(), shape: collection_shape(state, database_id, tenant_id, &name.bare)?, binds: binds .iter() @@ -332,17 +339,31 @@ pub(super) fn document_units( .map(|b| (b.surrogate, b.pk.as_slice())) .collect(), }; - let mut units = Vec::with_capacity(rows.len()); - for (key, row) in &rows { - units.push(match row { + let mut ordered: Vec<(&StorageKey, &StoredRow)> = rows.iter().collect(); + if builder.shape.hash_chain { + let mut positioned = Vec::with_capacity(ordered.len()); + for (key, row) in ordered { + positioned.push((chain_seq(&builder.shape, &name.bare, *key, row)?, key, row)); + } + positioned.sort_by_key(|(seq, _, _)| *seq); + ordered = positioned + .into_iter() + .map(|(_, key, row)| (key, row)) + .collect(); + } + let mut pending = Vec::with_capacity(rows.len()); + for (key, row) in ordered { + pending.push(match row { StoredRow::Current(body) => builder.current(*key, body)?, StoredRow::Versions(versions) => builder.versions(*key, versions)?, }); } - out.push(CollectionUnits { + out.push(PreparedRows { database_id, - collection: name.bare.clone(), - units, + tenant: TenantId::new(tenant_id), + bare: name.bare.clone(), + stored: name.stored.clone(), + rows: pending, }); } Ok(out) @@ -360,6 +381,22 @@ mod tests { assert!(split_key("0:7:users", 7).is_err()); } + #[test] + fn the_highest_row_surrogate_spans_current_and_versioned_rows() { + let highest = max_row_surrogate( + 7, + &[("0:7:plain:0000002a".into(), vec![])], + &[("0:7:ledger:000000ff\x0000000000000000000100".into(), vec![])], + ) + .unwrap(); + assert_eq!( + highest, + StorageKey::parse("000000ff").unwrap().surrogate().as_u32() + ); + assert_eq!(max_row_surrogate(7, &[], &[]).unwrap(), 0); + assert!(max_row_surrogate(7, &[("0:7:plain:zz".into(), vec![])], &[]).is_err()); + } + #[test] fn versions_group_under_their_row_in_system_time_order() { let grouped = group_rows( diff --git a/nodedb/src/control/backup/restore/redo_reissue/edges.rs b/nodedb/src/control/backup/restore/redo_reissue/edges.rs index b3dcbd958..8a5e0eb82 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/edges.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/edges.rs @@ -1,31 +1,37 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Restored graph edges as redo units. +//! Restored graph edges as edge units. //! //! A backup carries every edge version under its versioned key, //! `"{collection}\x00{src}\x00{label}\x00{dst}\x00{system_from:020}"`. Each //! version re-issues at its original `system_from`, so the restored edge keeps //! its history and its valid-from time. A tombstone version re-issues as a -//! delete at its `system_from`. Every replica updates its CSR index and its +//! tombstone at its `system_from`. Every replica updates its CSR index and its //! node identities as it installs each version. //! +//! Each version re-issues to both endpoint homes, `from_key(src)` and +//! `from_key(dst)`, as every live edge write reaches them. Endpoint identities +//! bind through the routed surrogate exchange, as a live edge write binds +//! them. +//! //! The key's collection is the name the source Data Plane stored it under. //! Each version re-issues under the destination-qualified name, and binds its //! node identities under the bare name in the destination database. use std::collections::BTreeMap; +use nodedb_physical::physical_plan::RestoredEdgeVersion; + +use crate::control::server::surrogate_exchange::assign_surrogate_routed; use crate::control::state::SharedState; use crate::control::surrogate::CarriedIdentity; use crate::engine::graph::edge_store::{ EdgeValuePayload, is_gdpr_erasure, is_tombstone, parse_versioned_edge_key, }; -use crate::types::{DatabaseId, TenantId}; -use crate::wal::{EdgeDeleteRedo, EdgePutRedo}; +use crate::types::{DatabaseId, RecordHomes, TenantId, TraceId}; use super::super::target::{DatabaseTarget, RestoredName}; -use super::sub_record::{edge_delete, edge_put}; -use super::units::{CollectionUnits, RowUnit}; +use super::units::{CollectionEdges, EdgeUnit}; fn malformed(key: &str) -> crate::Error { let prefix: String = key.chars().take(64).collect(); @@ -35,20 +41,31 @@ fn malformed(key: &str) -> crate::Error { } } -/// The identity of node `node_id` in edge collection `collection`, bound -/// through the surrogate assigner as a live edge write binds it. -fn node_identity( +/// The identity of node `node_id` in edge collection `collection`. +/// +/// The backup's bind wins: the restore rebinds it on this node before any +/// re-issue. A node the backup carries no bind for is bound on the leader of +/// the collection's home, as a live edge write binds it. Either way the batch +/// carries the identity, and every replica of each home binds it first-wins. +async fn node_identity( state: &SharedState, database_id: DatabaseId, tenant: TenantId, collection: &str, node_id: &str, ) -> crate::Result { - let surrogate = state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(database_id, collection), - tenant, - node_id.as_bytes(), - )?; + let key = nodedb_types::CollectionKey::from_bare(database_id, collection); + // The backup's bind is in this node's catalog. Any other key goes to the + // home through the async exchange, which answers an existing bind too. + let surrogate = match state + .surrogate_assigner + .lookup_bound(key, tenant, node_id.as_bytes())? + { + Some(surrogate) => surrogate, + None => { + assign_surrogate_routed(state, key, tenant, node_id.as_bytes(), TraceId::ZERO).await? + } + }; Ok(CarriedIdentity { collection: collection.to_string(), pk_bytes: node_id.as_bytes().to_vec(), @@ -57,30 +74,17 @@ fn node_identity( } /// One edge version of the collection `name` as a unit. -fn edge_unit( +async fn edge_unit( state: &SharedState, database_id: DatabaseId, tenant: TenantId, name: &RestoredName, key: &str, value: &[u8], -) -> crate::Result { +) -> crate::Result { let (_, src_id, label, dst_id, system_from) = parse_versioned_edge_key(key).ok_or_else(|| malformed(key))?; let collection = name.bare.as_str(); - if is_tombstone(value) { - let op = edge_delete(&EdgeDeleteRedo { - collection: name.stored.to_string(), - src_id: src_id.to_string(), - label: label.to_string(), - dst_id: dst_id.to_string(), - system_from: Some(system_from), - })?; - return Ok(RowUnit { - ops: vec![op], - identities: Vec::new(), - }); - } if is_gdpr_erasure(value) { return Err(crate::Error::Serialization { format: "backup".into(), @@ -90,21 +94,25 @@ fn edge_unit( ), }); } - let payload = EdgeValuePayload::decode(value)?; - let src = node_identity(state, database_id, tenant, collection, src_id)?; - let dst = node_identity(state, database_id, tenant, collection, dst_id)?; - let op = edge_put(&EdgePutRedo { - collection: name.stored.to_string(), - src_id: src_id.to_string(), - label: label.to_string(), - dst_id: dst_id.to_string(), - properties: payload.properties, - src_surrogate: src.surrogate.as_u32(), - dst_surrogate: dst.surrogate.as_u32(), - system_from: Some(system_from), - })?; - Ok(RowUnit { - ops: vec![op], + let properties = if is_tombstone(value) { + None + } else { + Some(EdgeValuePayload::decode(value)?.properties) + }; + let src = node_identity(state, database_id, tenant, collection, src_id).await?; + let dst = node_identity(state, database_id, tenant, collection, dst_id).await?; + Ok(EdgeUnit { + version: RestoredEdgeVersion { + collection: name.stored.to_string(), + src_id: src_id.to_string(), + label: label.to_string(), + dst_id: dst_id.to_string(), + src_surrogate: src.surrogate.as_u32(), + dst_surrogate: dst.surrogate.as_u32(), + system_from, + properties, + }, + homes: RecordHomes::edge(src_id, dst_id), identities: vec![src, dst], }) } @@ -113,31 +121,35 @@ fn edge_unit( /// version, grouped by edge collection in key order: each edge's versions in /// system-time order. The edge section of a database's data section holds /// that database's edges only. -pub(super) fn edge_units( +/// +/// Returns the number of distinct edge versions and the grouped units. +pub(super) async fn edge_units( state: &SharedState, tenant_id: u64, target: DatabaseTarget, edges: Vec<(String, Vec)>, -) -> crate::Result> { +) -> crate::Result<(usize, Vec)> { let database_id = target.dest; let tenant = TenantId::new(tenant_id); let mut by_key: BTreeMap> = BTreeMap::new(); for (key, value) in edges { by_key.insert(key, value); } - let mut grouped: BTreeMap> = BTreeMap::new(); + let mut grouped: BTreeMap> = BTreeMap::new(); for (key, value) in &by_key { let (stored, ..) = parse_versioned_edge_key(key).ok_or_else(|| malformed(key))?; let name = target.resolve(stored)?; - let unit = edge_unit(state, database_id, tenant, &name, key, value)?; + let unit = edge_unit(state, database_id, tenant, &name, key, value).await?; grouped.entry(name.bare).or_default().push(unit); } - Ok(grouped + let versions = by_key.len(); + let collections = grouped .into_iter() - .map(|(collection, units)| CollectionUnits { + .map(|(collection, units)| CollectionEdges { database_id, collection, units, }) - .collect()) + .collect(); + Ok((versions, collections)) } diff --git a/nodedb/src/control/backup/restore/redo_reissue/mod.rs b/nodedb/src/control/backup/restore/redo_reissue/mod.rs index 1a4a1478a..4516ddf58 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/mod.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/mod.rs @@ -1,13 +1,15 @@ // SPDX-License-Identifier: BUSL-1.1 //! Durable, replicated RESTORE of document rows, their index entries, and -//! graph edges, as committed redo records. +//! graph edges, as Calvin transactions. mod commit; mod documents; mod edges; +mod prepared; mod reissue; mod sub_record; mod units; +pub(in crate::control::backup::restore) use documents::{max_row_surrogate, prepare_documents}; pub(in crate::control::backup::restore) use reissue::{RestoredRows, reissue_rows_and_edges}; diff --git a/nodedb/src/control/backup/restore/redo_reissue/prepared.rs b/nodedb/src/control/backup/restore/redo_reissue/prepared.rs new file mode 100644 index 000000000..cff07501e --- /dev/null +++ b/nodedb/src/control/backup/restore/redo_reissue/prepared.rs @@ -0,0 +1,130 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Restored document rows, decoded and identified but not yet bound. +//! +//! Preparing a row reads its identity and the surrogate its storage key +//! carries, and binds nothing. The restore checks every carried surrogate +//! against its home before any bind, then [`document_units`] binds each row +//! and builds its unit. + +use nodedb_types::{QualifiedCollection, RowIdentity, StorageKey}; + +use crate::control::state::SharedState; +use crate::control::surrogate::CarriedIdentity; +use crate::types::{DatabaseId, TenantId}; +use crate::wal::{RedoRowChange, RedoRowKind}; + +use super::super::bind_conflicts::CarriedBind; +use super::sub_record::{VersionStamp, document_put, document_tombstone}; +use super::units::{CollectionUnits, RowUnit}; + +/// One collection's restored rows in the destination database. +pub(in crate::control::backup::restore) struct PreparedRows { + pub(super) database_id: DatabaseId, + pub(super) tenant: TenantId, + /// Bare catalog name: it keys the identity binds. + pub(super) bare: String, + /// The name the destination Data Plane stores the collection under. + pub(super) stored: QualifiedCollection, + pub(super) rows: Vec, +} + +/// One decoded row and the identity it binds. +pub(super) struct PendingRow { + pub(super) key: StorageKey, + pub(super) identity: RowIdentity, + pub(super) body: PendingBody, +} + +/// A row's MessagePack body, or every version of a bitemporal row with the +/// body of each live version. +pub(super) enum PendingBody { + Current(Vec), + Versions(Vec<(VersionStamp, Option>)>), +} + +impl PreparedRows { + /// Every surrogate the rows carry, with the key each one names. + pub(in crate::control::backup::restore) fn carried( + &self, + ) -> impl Iterator + '_ { + self.rows.iter().map(|row| CarriedBind { + collection: self.bare.clone(), + pk: row.identity.as_str().as_bytes().to_vec(), + surrogate: row.key.surrogate().as_u32(), + }) + } +} + +/// Bind each prepared row's identity on this node and build its unit. The +/// backup's surrogate wins unless this node already binds the identity: the +/// row then installs under that surrogate, over the row it names. +pub(super) fn document_units( + state: &SharedState, + prepared: Vec, +) -> crate::Result> { + let mut out = Vec::with_capacity(prepared.len()); + for collection in prepared { + let key = nodedb_types::CollectionKey::from_bare(collection.database_id, &collection.bare); + let mut units = Vec::with_capacity(collection.rows.len()); + for row in collection.rows { + let pk = row.identity.as_str(); + let surrogate = state.surrogate_assigner.bind( + key, + collection.tenant, + pk.as_bytes(), + row.key.surrogate(), + )?; + let carried = CarriedIdentity { + collection: collection.bare.clone(), + pk_bytes: pk.as_bytes().to_vec(), + surrogate, + }; + let raw = surrogate.as_u32(); + let stored = collection.stored.as_str(); + let (ops, ends_live) = match row.body { + PendingBody::Current(value) => { + (vec![document_put(stored, pk, value, raw, None)?], true) + } + PendingBody::Versions(versions) => { + let ends_live = versions.last().is_some_and(|(_, body)| body.is_some()); + let mut ops = Vec::with_capacity(versions.len()); + for (stamp, body) in versions { + ops.push(match body { + Some(value) => document_put(stored, pk, value, raw, Some(stamp))?, + None => document_tombstone(stored, pk, raw, stamp.sys_from_ms)?, + }); + } + (ops, ends_live) + } + }; + // A restored row that ends live installs as a new row. One whose + // last version is a tombstone ends absent and publishes nothing. + let changes = if ends_live { + vec![RedoRowChange { + collection: stored.to_owned(), + row: pk.to_owned(), + kind: RedoRowKind::Insert, + }] + } else { + Vec::new() + }; + units.push(RowUnit { + ops, + identities: vec![carried], + changes, + rows: vec![nodedb_physical::physical_plan::RestoredRow { + collection: stored.to_owned(), + document_id: pk.to_owned(), + surrogate: raw, + }], + }); + } + out.push(CollectionUnits::rows( + collection.database_id, + collection.bare, + units, + )); + } + Ok(out) +} diff --git a/nodedb/src/control/backup/restore/redo_reissue/reissue.rs b/nodedb/src/control/backup/restore/redo_reissue/reissue.rs index b7a32bf4f..a896b27fe 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/reissue.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/reissue.rs @@ -1,27 +1,28 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Re-issue every restored document row and graph edge as committed redo. +//! Re-issue every restored document row and graph edge as Calvin +//! transactions. //! //! Rows go first, then edges, so an edge's endpoints are in place when it -//! installs. A backup's index entries are not re-issued: every replica -//! derives a row's secondary index entries as it installs the row, exactly as -//! for a committed transaction's row. +//! installs. Each row goes to its collection's home, each edge to both of its +//! endpoint homes. Every one of them runs in the Calvin sequence, so one +//! RESTORE has one ordering domain. A backup's index entries are not +//! re-issued: every replica derives a row's secondary index entries as it +//! installs the row, exactly as for a committed transaction's row. use crate::control::state::SharedState; -use crate::types::{SurrogateBindEntry, TenantId}; +use crate::types::TenantId; use super::super::target::DatabaseTarget; -use super::commit::commit_collection; -use super::documents::document_units; +use super::commit::{commit_collection, commit_edges}; use super::edges::edge_units; +use super::prepared::{PreparedRows, document_units}; /// The backup sections this re-issue consumes. -pub(in crate::control::backup::restore) struct RestoredRows<'a> { - pub documents: Vec<(String, Vec)>, - pub documents_versioned: Vec<(String, Vec)>, +pub(in crate::control::backup::restore) struct RestoredRows { + /// The document rows, prepared and checked against their homes. + pub documents: Vec, pub edges: Vec<(String, Vec)>, - /// The backup's primary-key section: each row's client identity. - pub binds: &'a [SurrogateBindEntry], } /// What the re-issue committed. @@ -29,9 +30,10 @@ pub(in crate::control::backup::restore) struct RestoredRows<'a> { pub(in crate::control::backup::restore) struct RedoReissueStats { /// Document sub-records: one per current row, one per version. pub documents: usize, - /// Edge sub-records: one per edge version. + /// Edge versions, each counted once whatever the number of homes it + /// re-issues to. pub edges: usize, - /// Redo records committed. + /// Calvin transactions committed. pub records: usize, } @@ -41,25 +43,28 @@ pub(in crate::control::backup::restore) async fn reissue_rows_and_edges( state: &SharedState, tenant_id: u64, target: DatabaseTarget, - rows: RestoredRows<'_>, + rows: RestoredRows, ) -> crate::Result { let tenant = TenantId::new(tenant_id); let mut stats = RedoReissueStats::default(); - let documents = document_units( - state, - tenant_id, - target, - rows.documents, - rows.documents_versioned, - rows.binds, - )?; + let documents = document_units(state, rows.documents)?; for collection in documents { stats.documents += collection.units.iter().map(|u| u.ops.len()).sum::(); - stats.records += commit_collection(state, tenant, collection).await?; + stats.records += commit_collection(state, tenant, target.restore_id, collection).await?; } - for collection in edge_units(state, tenant_id, target, rows.edges)? { - stats.edges += collection.units.iter().map(|u| u.ops.len()).sum::(); - stats.records += commit_collection(state, tenant, collection).await?; + // Fails the re-issue after the rows committed and before any edge. + crate::fail_point_err!("restore::reissue::before_edges", |detail: String| { + crate::Error::Internal { + detail: format!("fail point: {detail}"), + } + }); + let (versions, edges) = edge_units(state, tenant_id, target, rows.edges).await?; + stats.edges = versions; + // Both homes of an edge commit in one transaction. Each version is + // absolute per home (a versioned edge key, a first-wins bind), so a + // retry rewrites both homes to the same state. + for collection in edges { + stats.records += commit_edges(state, tenant, target.restore_id, collection).await?; } Ok(stats) } diff --git a/nodedb/src/control/backup/restore/redo_reissue/sub_record.rs b/nodedb/src/control/backup/restore/redo_reissue/sub_record.rs index 84bb6f763..15df26f95 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/sub_record.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/sub_record.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Redo sub-record encoders for restored document rows and graph edges. +//! Redo sub-record encoders for restored document rows. //! //! Each encoder writes the exact payload shape the transaction resolver emits //! and the replay arms decode, so a restored row installs through the same @@ -10,13 +10,15 @@ //! `(sys_from_ms, valid_from_ms, valid_until_ms)` for a version of a //! `bitemporal=true` collection; //! * document delete: `(collection, document_id, prov, surrogate, sys_from_ms)`, -//! a tombstone version of a `bitemporal=true` collection; -//! * edge put and delete: [`EdgePutRedo`] and [`EdgeDeleteRedo`]. +//! a tombstone version of a `bitemporal=true` collection. +//! +//! A restored edge version travels typed, and its transaction's resolve +//! encodes it with the transaction's ordinal as its applied ordinal. use nodedb_types::sync::wire::SyncProvenance; use nodedb_wal::record::RecordType; -use crate::wal::{EdgeDeleteRedo, EdgePutRedo, RedoSubRecord}; +use crate::wal::RedoSubRecord; /// The system and valid time a document version was stored at. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -80,22 +82,6 @@ pub(super) fn document_tombstone( }) } -/// One edge version put at its original `system_from`. -pub(super) fn edge_put(put: &EdgePutRedo) -> crate::Result { - Ok(RedoSubRecord { - record_type: RecordType::Put as u32, - payload: zerompk::to_msgpack_vec(put).map_err(|e| encode_error("edge put", e))?, - }) -} - -/// One edge tombstone at its original `system_from`. -pub(super) fn edge_delete(delete: &EdgeDeleteRedo) -> crate::Result { - Ok(RedoSubRecord { - record_type: RecordType::Delete as u32, - payload: zerompk::to_msgpack_vec(delete).map_err(|e| encode_error("edge delete", e))?, - }) -} - #[cfg(test)] mod tests { use super::*; diff --git a/nodedb/src/control/backup/restore/redo_reissue/units.rs b/nodedb/src/control/backup/restore/redo_reissue/units.rs index 27161bd26..f4fbf4a24 100644 --- a/nodedb/src/control/backup/restore/redo_reissue/units.rs +++ b/nodedb/src/control/backup/restore/redo_reissue/units.rs @@ -1,36 +1,98 @@ // SPDX-License-Identifier: BUSL-1.1 -//! The unit a restored row or edge re-issues as. +//! The units a restored row or edge re-issues as. + +use nodedb_physical::physical_plan::{RestoredEdgeVersion, RestoredRow}; use crate::control::surrogate::CarriedIdentity; -use crate::types::DatabaseId; -use crate::wal::RedoSubRecord; +use crate::types::{DatabaseId, HomedRecord, RecordHomes, VShardId}; +use crate::wal::{RedoRowChange, RedoSubRecord}; -/// One restored row or edge: its sub-records in apply order, and the -/// identities every replica binds before it installs them. A unit never -/// splits across two redo records. +/// One restored row: its sub-records in apply order, the identities every +/// replica binds before it installs them, the row change its install +/// publishes, and the row its writers lock. A unit never splits across two +/// transactions. +#[derive(Clone)] pub(super) struct RowUnit { pub ops: Vec, pub identities: Vec, + /// The change events the unit's install publishes: a restored row that + /// ends live is an insert. + pub changes: Vec, + /// The row the unit writes, as its writers lock it. + pub rows: Vec, } impl RowUnit { - /// The unit's encoded size, for sizing the records it goes into. + /// The unit's encoded size, for sizing the batches it goes into. pub(super) fn byte_len(&self) -> usize { - self.ops.iter().map(|op| op.payload.len()).sum::() - + self - .identities - .iter() - .map(|identity| identity.collection.len() + identity.pk_bytes.len()) - .sum::() + self.ops.iter().map(|op| op.payload.len()).sum::() + identities_len(&self.identities) } } -/// Every unit of one collection. All of them write that collection's vShard. +/// Units of one collection that all write one vShard. pub(super) struct CollectionUnits { /// The destination database. pub database_id: DatabaseId, /// Bare catalog name of the collection. pub collection: String, + /// The vShard every unit writes. + pub vshard_id: VShardId, pub units: Vec, } + +impl CollectionUnits { + /// Rows of `collection`, written to its home vShard. + pub(super) fn rows(database_id: DatabaseId, collection: String, units: Vec) -> Self { + let vshard_id = RecordHomes::of(HomedRecord::Row(nodedb_types::CollectionKey::from_bare( + database_id, + &collection, + ))) + .owner(); + Self { + database_id, + collection, + vshard_id, + units, + } + } +} + +/// One restored edge version, the identities of its endpoints, and the +/// homes it re-issues to. +#[derive(Clone)] +pub(super) struct EdgeUnit { + pub version: RestoredEdgeVersion, + pub identities: Vec, + pub homes: RecordHomes, +} + +impl EdgeUnit { + /// The unit's encoded size, for sizing the batches it goes into. + pub(super) fn byte_len(&self) -> usize { + let version = &self.version; + version.collection.len() + + version.src_id.len() + + version.label.len() + + version.dst_id.len() + + version.properties.as_ref().map_or(0, Vec::len) + + identities_len(&self.identities) + } +} + +/// The restored edge versions of one collection, in key order: each edge's +/// versions in system-time order. +pub(super) struct CollectionEdges { + /// The destination database. + pub database_id: DatabaseId, + /// Bare catalog name of the collection. + pub collection: String, + pub units: Vec, +} + +fn identities_len(identities: &[CarriedIdentity]) -> usize { + identities + .iter() + .map(|identity| identity.collection.len() + identity.pk_bytes.len()) + .sum() +} diff --git a/nodedb/src/control/backup/restore/sections.rs b/nodedb/src/control/backup/restore/sections.rs index 464dceeae..fc83afab6 100644 --- a/nodedb/src/control/backup/restore/sections.rs +++ b/nodedb/src/control/backup/restore/sections.rs @@ -6,14 +6,15 @@ use std::collections::BTreeMap; use std::sync::Arc; use nodedb_types::backup_envelope::{ - DatabaseDataSection, Envelope, SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_DATABASES, - SECTION_ORIGIN_SOURCE_TOMBSTONES, SECTION_ORIGIN_SURROGATE_PK, Section, SourceTombstoneEntry, - StoredCollectionBlob, SurrogateBindBlob, + DatabaseDataSection, Envelope, SECTION_ORIGIN_ARRAY_CATALOG, SECTION_ORIGIN_CATALOG_ROWS, + SECTION_ORIGIN_DATABASES, SECTION_ORIGIN_SOURCE_TOMBSTONES, SECTION_ORIGIN_SURROGATE_PK, + SECTION_ORIGIN_VERIFICATION, Section, SourceTombstoneEntry, StoredCollectionBlob, + SurrogateBindBlob, }; use crate::Error; use crate::control::catalog_entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::catalog::StoredCollection; use crate::control::state::SharedState; use crate::types::{SurrogateBindEntry, TenantDataSnapshot}; @@ -70,7 +71,10 @@ pub(super) fn merge_sections( /// Concatenate every section of `snap` onto `into`. Each source node holds /// disjoint vShards, so the sections never overlap. -fn append_snapshot(into: &mut TenantDataSnapshot, snap: TenantDataSnapshot) { +pub(in crate::control::backup) fn append_snapshot( + into: &mut TenantDataSnapshot, + snap: TenantDataSnapshot, +) { // Destructure exhaustively so a new field fails to compile here rather // than being dropped from the restore. let TenantDataSnapshot { @@ -89,10 +93,34 @@ fn append_snapshot(into: &mut TenantDataSnapshot, snap: TenantDataSnapshot) { surrogate_pk, tenant_edges, group_write_marks, + group_proposal_keys, + group_proposal_keys_complete_from, documents_versioned, indexes_versioned, + vector_multi_documents, + arrays, + metadata_floor, + // A backup carries no cut: it serves Raft installs only. + group_cut_index: _, + // A backup carries no lane state: the destination fires and + // delivers its own writes. + group_event_lane: _, + // A backup carries no Calvin cut: the restore re-applies its rows + // as new writes on the destination. + group_calvin: _, + // A backup carries visible edge versions only. The restore re-applies + // them at its own ordinal, so the source's cuts, hidden versions and + // applied ordinals never reach the destination. + edge_hidden: _, + edge_cuts: _, + edge_applied: _, + tenant_edge_cuts: _, + tenant_edge_applied: _, } = snap; into.documents.extend(documents); + // Several nodes can list one index's members; the re-issue reads the + // union. + into.vector_multi_documents.extend(vector_multi_documents); into.indexes.extend(indexes); into.documents_versioned.extend(documents_versioned); into.indexes_versioned.extend(indexes_versioned); @@ -109,18 +137,28 @@ fn append_snapshot(into: &mut TenantDataSnapshot, snap: TenantDataSnapshot) { into.flushed_ts_segments.extend(flushed_ts_segments); into.columnar_engines.extend(columnar_engines); into.surrogate_pk.extend(surrogate_pk); + into.arrays.extend(arrays); into.tenant_edges.extend(tenant_edges); // A backup carries no marks: the guard reads the destination's marks. into.group_write_marks.extend(group_write_marks); + // A backup carries no proposal keys: they serve Raft installs only. + into.group_proposal_keys.extend(group_proposal_keys); + into.group_proposal_keys_complete_from = into + .group_proposal_keys_complete_from + .max(group_proposal_keys_complete_from); + // A backup carries no floor: it serves Raft installs only. + into.metadata_floor = into.metadata_floor.max(metadata_floor); } pub(super) fn is_metadata_section(section: &Section) -> bool { matches!( section.origin_node_id, SECTION_ORIGIN_CATALOG_ROWS + | SECTION_ORIGIN_ARRAY_CATALOG | SECTION_ORIGIN_SOURCE_TOMBSTONES | SECTION_ORIGIN_SURROGATE_PK | SECTION_ORIGIN_DATABASES + | SECTION_ORIGIN_VERIFICATION ) } @@ -138,15 +176,14 @@ pub(super) fn is_metadata_section(section: &Section) -> bool { /// Returns every collection written to the catalog, in section order. The /// caller registers each one with this node's Data Plane before any restored /// row is installed or reissued: a catalog row alone leaves `doc_configs` -/// without the collection's declaration, and a reissued timeseries row would -/// then be ingested into an inferred shape. -pub(super) fn apply_metadata_sections( +/// without the collection's declaration, and a reissued timeseries row +/// is then ingested into an inferred shape. +pub(super) async fn apply_metadata_sections( state: &Arc, tenant_id: u64, env: &Envelope, databases: &DatabaseMap, ) -> Result, Error> { - let catalog = state.credentials.catalog(); let mut restored: Vec = Vec::new(); for section in &env.sections { @@ -168,22 +205,15 @@ pub(super) fn apply_metadata_sections( } })?; coll.database_id = databases.target(blob.database_id)?.dest; - // Propose the collection through the metadata Raft - // group so every node's applier (`catalog_entry:: - // apply::collection::put`) writes the row — mirroring - // CREATE COLLECTION and DROP COLLECTION. The proposer - // blocks on its local applied-index watcher, so on the - // cluster path it has already applied the put via the - // same applier — we must NOT also put locally (double-put). + // The source cluster's incarnation names nothing here: an + // existing collection keeps its own, a new one gets one. + coll.incarnation = nodedb_types::Hlc::ZERO; + // Propose the collection so every node's applier + // (`catalog_entry::apply::collection::put`) writes the + // row — mirroring CREATE COLLECTION and DROP COLLECTION. + // The proposer returns once this node applied it. let entry = CatalogEntry::PutCollection(Box::new(coll.clone())); - if propose_catalog_entry(state, &entry)?.needs_local_apply() { - // Single-node / no-cluster fallback: apply the - // catalog mutation directly, matching what the - // applier would have done on a clustered deployment. - // A failure here is FATAL — the collection would be - // unqueryable otherwise. - catalog.put_collection(coll.database_id, &coll)?; - } + propose_catalog_entry_async(state, &entry).await?; restored.push(coll); } } @@ -205,18 +235,7 @@ pub(super) fn apply_metadata_sections( collection: t.collection.clone(), purge_lsn: t.purge_lsn, }; - if propose_catalog_entry(state, &entry)?.needs_local_apply() { - // Single-node / no-cluster fallback: apply directly, - // matching the applier. A failure here is FATAL — a - // silently-skipped tombstone means purged writes resurrect - // on restart. - catalog.record_wal_tombstone( - database_id, - tenant_id, - &t.collection, - t.purge_lsn, - )?; - } + propose_catalog_entry_async(state, &entry).await?; } } _ => {} diff --git a/nodedb/src/control/backup/restore/surrogate_floor.rs b/nodedb/src/control/backup/restore/surrogate_floor.rs new file mode 100644 index 000000000..f8a6b3f8f --- /dev/null +++ b/nodedb/src/control/backup/restore/surrogate_floor.rs @@ -0,0 +1,76 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Raise the surrogate high-water mark above every surrogate a re-issue binds. +//! +//! A RESTORE binds the surrogates the source cluster issued, and a MOVE TENANT +//! binds the ones its capture carries. The allocator here must never issue one +//! of them again. Before any re-issue binds a carried surrogate, the re-issue +//! proposes `MetadataEntry::SurrogateAlloc` at the highest carried surrogate. +//! Every node applies it in log order: it raises the global watermark, so every +//! later reservation carves above it, and it retires the unissued part of the +//! node's reserved batch at or below it. The re-issue then waits until every +//! node applied the entry. The metadata log replays in full on a restart, so +//! the raise survives it. + +use futures::future::join_all; +use nodedb_physical::physical_plan::ClusterEventOp; + +use crate::Error; +use crate::bridge::envelope::PhysicalPlan; +use crate::control::state::SharedState; +use crate::types::DatabaseId; + +use super::super::node_snapshot::snapshot_remote; + +/// Make sure no node issues a surrogate at or below `highest` from now on. +/// `tenant_id` frames the requests to the other nodes. +pub(super) async fn raise_surrogate_floor( + state: &SharedState, + tenant_id: u64, + highest: u32, +) -> crate::Result<()> { + if highest == 0 { + return Ok(()); + } + let index = crate::control::metadata_proposer::propose_surrogate_hwm(state, highest).await?; + await_applied_on_every_node(state, tenant_id, index).await +} + +/// Wait until every active node other than this one applied the metadata log +/// through `index`. This node applied it before the propose returned. +async fn await_applied_on_every_node( + state: &SharedState, + tenant_id: u64, + index: u64, +) -> crate::Result<()> { + let peers: Vec = match state.cluster_topology.as_ref() { + Some(topology) => topology + .read() + .unwrap_or_else(|p| p.into_inner()) + .active_nodes() + .iter() + .map(|node| node.node_id) + .filter(|node_id| *node_id != state.node_id) + .collect(), + None => Vec::new(), + }; + let plan = PhysicalPlan::ClusterEvent(ClusterEventOp::MetadataApplied { index }); + let answers = join_all(peers.iter().map(|node_id| { + let plan = &plan; + async move { + snapshot_remote(state, *node_id, tenant_id, DatabaseId::DEFAULT, plan) + .await + .map_err(|e| Error::Internal { + detail: format!( + "restore: node {node_id} did not confirm the surrogate floor at \ + metadata index {index}: {e}" + ), + }) + } + })) + .await; + for answer in answers { + answer?; + } + Ok(()) +} diff --git a/nodedb/src/control/backup/restore/target.rs b/nodedb/src/control/backup/restore/target.rs index d1ea9e20a..d2f32b0c4 100644 --- a/nodedb/src/control/backup/restore/target.rs +++ b/nodedb/src/control/backup/restore/target.rs @@ -18,6 +18,11 @@ pub(crate) struct DatabaseTarget { pub source: DatabaseId, /// The database id the rows restore into. pub dest: DatabaseId, + /// The id of the RESTORE that re-issues the rows. Every re-issued write + /// stamps it on its write mark, so a retry of the same envelope knows its + /// own writes. `0` for a re-issue that is no RESTORE: its writes mark as + /// user writes. + pub restore_id: u64, } /// A restored collection's names in the destination database. @@ -104,6 +109,7 @@ mod tests { const TARGET: DatabaseTarget = DatabaseTarget { source: DatabaseId::new(1025), dest: DatabaseId::new(2048), + restore_id: 0, }; #[test] @@ -130,6 +136,7 @@ mod tests { let target = DatabaseTarget { source: DatabaseId::DEFAULT, dest: DatabaseId::DEFAULT, + restore_id: 0, }; let name = target .resolve("orders") diff --git a/nodedb/src/control/backup/restore/validate.rs b/nodedb/src/control/backup/restore/validate.rs new file mode 100644 index 000000000..33c35c78b --- /dev/null +++ b/nodedb/src/control/backup/restore/validate.rs @@ -0,0 +1,269 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Whole-envelope validation for RESTORE TENANT. +//! +//! Every refusal the restore can raise from the envelope's own content runs +//! here, before the restore proposes anything. A refused envelope leaves the +//! destination catalog unchanged. + +use std::collections::{BTreeMap, BTreeSet}; + +use nodedb_types::backup_envelope::{ + DatabaseBlob, Envelope, SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_SOURCE_TOMBSTONES, + SourceTombstoneEntry, StoredCollectionBlob, +}; + +use crate::Error; +use crate::control::backup::metadata::refuse_unmaterialized_clone; +use crate::control::backup::verify::expect::{Expectation, expect_envelope}; +use crate::control::security::catalog::{DatabaseDescriptor, StoredCollection, SystemCatalog}; +use crate::control::state::SharedState; +use crate::types::{TenantDataSnapshot, TenantId}; +use nodedb_types::QuotaRecord; + +use super::databases::{decode_databases, require_writable}; +use super::sections::merge_sections; + +/// An envelope every restore step accepts. +pub(super) struct ValidatedEnvelope { + pub databases: Vec, + /// Every data section, merged per source database. + pub merged: BTreeMap, + /// What the destination must hold once every row is re-issued. + pub expectation: Expectation, +} + +/// Check the whole envelope against the destination. Proposes nothing. +pub(super) fn validate_envelope( + state: &SharedState, + tenant_id: u64, + env: &Envelope, +) -> Result { + let catalog = state.credentials.catalog(); + let databases = decode_databases(env)?; + validate_databases(state, catalog, TenantId::new(tenant_id), &databases)?; + + let listed: BTreeSet = databases.iter().map(|b| b.database_id).collect(); + validate_metadata_sections(env, &listed)?; + + let merged = merge_sections(&env.sections)?; + if let Some(source) = merged.keys().find(|source| !listed.contains(source)) { + return Err(unlisted("a data section", *source)); + } + super::array_reissue::validate_array_rows( + catalog, + TenantId::new(tenant_id), + env, + &databases, + &merged, + )?; + // The rows must match the counts and digests the backup recorded. + let expectation = expect_envelope(state, tenant_id, env, &merged)?; + Ok(ValidatedEnvelope { + databases, + merged, + expectation, + }) +} + +/// Every database the restore maps or creates, and every quota it installs. +fn validate_databases( + state: &SharedState, + catalog: &SystemCatalog, + tenant: TenantId, + blobs: &[DatabaseBlob], +) -> Result<(), Error> { + let mut new_quotas = Vec::new(); + for blob in blobs { + match catalog.get_database_id_by_name(&blob.name)? { + Some(id) => { + require_writable(state, id, &blob.name)?; + if let Some(record) = &blob.tenant_quota + && catalog.get_tenant_quota(id, tenant)?.is_none() + { + catalog.check_tenant_quota(id, tenant, record)?; + } + } + None => { + zerompk::from_msgpack::(&blob.descriptor).map_err(|_| { + Error::Internal { + detail: format!( + "invalid backup format: descriptor of database '{}' is not decodable", + blob.name + ), + } + })?; + if let Some(record) = &blob.tenant_quota { + // Blobs of the same name land in the same new database. + let others: Vec<&QuotaRecord> = blobs + .iter() + .filter(|other| { + other.name == blob.name && other.database_id != blob.database_id + }) + .filter_map(|other| other.tenant_quota.as_ref()) + .collect(); + SystemCatalog::check_tenant_quota_in_new_database( + blob.database_quota.as_ref(), + &others, + record, + )?; + } + if let Some(record) = &blob.database_quota { + new_quotas.push(record); + } + } + } + } + // The created databases' quotas count against the ceiling together. + catalog.check_new_database_quotas(&new_quotas, &state.quota_ceiling_snapshot())?; + Ok(()) +} + +/// Every catalog row and source tombstone decodes, names a listed database, +/// and is restorable. +fn validate_metadata_sections(env: &Envelope, listed: &BTreeSet) -> Result<(), Error> { + for section in &env.sections { + match section.origin_node_id { + SECTION_ORIGIN_CATALOG_ROWS => { + let blobs = zerompk::from_msgpack::>(§ion.body) + .map_err(|_| Error::Internal { + detail: "invalid backup format: catalog-rows section is not decodable" + .into(), + })?; + for blob in blobs { + if !listed.contains(&blob.database_id) { + return Err(unlisted("a catalog row", blob.database_id)); + } + let coll = + zerompk::from_msgpack::(&blob.bytes).map_err(|_| { + Error::Internal { + detail: format!( + "invalid backup format: catalog row of '{}' is not decodable", + blob.name + ), + } + })?; + refuse_unmaterialized_clone(&coll)?; + } + } + SECTION_ORIGIN_SOURCE_TOMBSTONES => { + let tombs = zerompk::from_msgpack::>(§ion.body) + .map_err(|_| Error::Internal { + detail: "invalid backup format: source-tombstones section is not \ + decodable" + .into(), + })?; + if let Some(t) = tombs.iter().find(|t| !listed.contains(&t.database_id)) { + return Err(unlisted("a source tombstone", t.database_id)); + } + } + _ => {} + } + } + Ok(()) +} + +fn unlisted(what: &str, source: u64) -> Error { + Error::Internal { + detail: format!( + "invalid backup format: {what} names database {source}, which the backup's \ + database section does not list" + ), + } +} + +#[cfg(test)] +mod tests { + use nodedb_types::backup_envelope::{ + EnvelopeMeta, SECTION_ORIGIN_DATABASES, Section, StoredCollectionBlob, + }; + use nodedb_types::{CloneOrigin, DatabaseId, Lsn}; + + use super::*; + + fn envelope(sections: Vec
) -> Envelope { + Envelope { + meta: EnvelopeMeta { + tenant_id: 1, + source_vshard_count: 1, + hash_seed: 0, + snapshot_watermark: 0, + }, + sections, + } + } + + fn databases_section(names: &[(u64, &str)]) -> Section { + let blobs: Vec = names + .iter() + .map(|(id, name)| DatabaseBlob { + database_id: *id, + name: (*name).to_string(), + descriptor: Vec::new(), + database_quota: None, + tenant_quota: None, + }) + .collect(); + Section { + origin_node_id: SECTION_ORIGIN_DATABASES, + body: zerompk::to_msgpack_vec(&blobs).unwrap(), + } + } + + fn rows_section(rows: &[StoredCollection]) -> Section { + let blobs: Vec = rows + .iter() + .map(|coll| StoredCollectionBlob { + database_id: DatabaseId::DEFAULT.as_u64(), + name: coll.name.clone(), + bytes: zerompk::to_msgpack_vec(coll).unwrap(), + }) + .collect(); + Section { + origin_node_id: SECTION_ORIGIN_CATALOG_ROWS, + body: zerompk::to_msgpack_vec(&blobs).unwrap(), + } + } + + /// A clone row anywhere in the envelope refuses it as a whole, before any + /// earlier row is proposed: validation reads every section first. + #[test] + fn a_late_clone_row_refuses_the_whole_envelope() { + let plain = StoredCollection::new(1, "plain", "admin"); + let mut clone = StoredCollection::new(1, "cloned", "admin"); + clone.cloned_from = Some(CloneOrigin { + source_database: DatabaseId::new(1030), + source_collection: "cloned".into(), + as_of_lsn: Lsn::new(1), + clone_created_at: Lsn::new(2), + kv_surrogate_ceiling: None, + }); + let env = envelope(vec![ + databases_section(&[(0, "default")]), + rows_section(std::slice::from_ref(&plain)), + rows_section(&[clone]), + ]); + let listed = BTreeSet::from([0]); + assert!(matches!( + validate_metadata_sections(&env, &listed), + Err(Error::BadRequest { .. }) + )); + + let env = envelope(vec![ + databases_section(&[(0, "default")]), + rows_section(&[plain]), + ]); + validate_metadata_sections(&env, &listed).expect("plain rows validate"); + } + + /// A row naming a database the envelope does not list refuses it. + #[test] + fn a_row_in_an_unlisted_database_refuses_the_envelope() { + let env = envelope(vec![rows_section(&[StoredCollection::new( + 1, "orphan", "admin", + )])]); + let err = validate_metadata_sections(&env, &BTreeSet::new()) + .expect_err("an unlisted database refuses"); + assert!(err.to_string().contains("does not list"), "{err}"); + } +} diff --git a/nodedb/src/control/backup/restore/vector_reissue.rs b/nodedb/src/control/backup/restore/vector_reissue.rs index 391449082..24835cbb9 100644 --- a/nodedb/src/control/backup/restore/vector_reissue.rs +++ b/nodedb/src/control/backup/restore/vector_reissue.rs @@ -4,8 +4,9 @@ //! //! Snapshot-install lands vector state in in-memory-only Data Plane maps with //! no WAL record or Raft entry — lost on restart, never replicated. RESTORE -//! re-issues each vector as a durable `VectorOp::Insert`: Raft-proposed on -//! cluster, WAL-appended + dispatched on single-node. +//! re-issues each vector as a durable `VectorOp::Insert`, proposed through +//! Raft. Each plan names its collection by the stored, database-qualified +//! name. use nodedb_types::surrogate::Surrogate; @@ -26,14 +27,22 @@ pub fn split_vector_coll_key(coll_key: &str) -> (&str, &str) { /// Build the durable `VectorOp::Insert` plan for one restored vector row. /// -/// The snapshot format carries no PK and no sync provenance for vector rows -/// (see `TenantDataSnapshot::vectors`), so both are `None` — matching what -/// the raw snapshot-install path (`restore_vector_collection`) does today. +/// The snapshot carries no sync provenance for vector rows. `surrogate` is the +/// one the source bound, never `Surrogate::ZERO`: an insert under the same +/// surrogate replaces the node it bound, so a repeated re-issue lands each +/// vector once. +/// +/// `pk_bytes` is the key the backup binds `surrogate` to, when it binds one. +/// The insert binds by it, as the source's insert did. `None` names a +/// headless row, which self-keys. Self-keying a row the backup binds to a +/// key binds its surrogate to a second key, and the next restore of the +/// same backup refuses it as a surrogate conflict. pub fn build_vector_insert_plan( collection: &str, field_name: &str, vector: Vec, surrogate: Surrogate, + pk_bytes: Option>, ) -> PhysicalPlan { let dim = vector.len(); PhysicalPlan::Vector(VectorOp::Insert { @@ -42,11 +51,98 @@ pub fn build_vector_insert_plan( dim, field_name: field_name.to_string(), surrogate, - pk_bytes: None, + pk_bytes, provenance: None, }) } +/// Delete every vector of one multi-vector document. Idempotent on the Data +/// Plane, so it clears a partial earlier re-issue and is a no-op otherwise. +pub fn build_multi_vector_delete_plan( + collection: &str, + field_name: &str, + document_surrogate: Surrogate, +) -> PhysicalPlan { + PhysicalPlan::Vector(VectorOp::MultiVectorDelete { + collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), + field_name: field_name.to_string(), + document_surrogate, + }) +} + +/// Insert the full vector set of one multi-vector document as one op. +/// Every vector has the width of the first. +/// +/// `pk_bytes` is the key the backup binds `document_surrogate` to, when it +/// binds one, so the insert binds by it as [`build_vector_insert_plan`] does. +pub fn build_multi_vector_insert_plan( + collection: &str, + field_name: &str, + document_surrogate: Surrogate, + pk_bytes: Option>, + vectors: Vec>, +) -> PhysicalPlan { + let dim = vectors.first().map_or(0, Vec::len); + let count = vectors.len(); + PhysicalPlan::Vector(VectorOp::MultiVectorInsert { + collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), + field_name: field_name.to_string(), + document_surrogate, + pk_bytes, + vectors: vectors.into_iter().flatten().collect(), + count, + dim, + }) +} + +/// A restored vector index's rows, grouped for re-issue. +#[derive(Debug, Default, PartialEq)] +pub struct RestoredVectors { + /// Single-vector rows with their surrogate. + pub single: Vec<(Surrogate, Vec)>, + /// Multi-vector documents by surrogate, in first-seen order, each with + /// its full vector set. Membership comes from the snapshot's + /// `vector_multi_documents`, never from a vector count: a one-vector + /// document is still a multi-vector document. + pub multi: Vec<(Surrogate, Vec>)>, +} + +/// Group the `(node_id, vector, surrogate)` export rows of one index. +/// `multi_documents` are the index's multi-vector document surrogates. +/// +/// Every stored vector is bound, so a row with no surrogate (or +/// `Surrogate::ZERO`) is a corrupt capture and fails the group. +pub fn group_restored_vectors( + rows: Vec<(u32, Vec, Option)>, + multi_documents: &std::collections::HashSet, +) -> crate::Result { + let mut grouped = RestoredVectors::default(); + let mut slot_of: std::collections::HashMap = std::collections::HashMap::new(); + for (node_id, vector, surrogate) in rows { + match surrogate.filter(|s| *s != Surrogate::ZERO) { + Some(surrogate) if multi_documents.contains(&surrogate) => { + let slot = *slot_of.entry(surrogate).or_insert_with(|| { + grouped.multi.push((surrogate, Vec::new())); + grouped.multi.len() - 1 + }); + if let Some((_, group)) = grouped.multi.get_mut(slot) { + group.push(vector); + } + } + Some(surrogate) => grouped.single.push((surrogate, vector)), + None => { + return Err(crate::Error::Internal { + detail: format!( + "restored vector node {node_id} carries no surrogate; every stored \ + vector is bound" + ), + }); + } + } + } + Ok(grouped) +} + /// Map a decoded `DistanceMetric` back to the string form `VectorOp::SetParams` /// and `execute_set_vector_params` expect (mirrors the inverse mapping in /// `execute_set_vector_params`, `handlers/vector_params.rs`). @@ -100,3 +196,148 @@ pub fn build_vector_set_params_plan( ivf_nprobe: config.ivf_nprobe, }) } + +#[cfg(test)] +mod tests { + use super::*; + + /// A multi-vector document re-issued with its backup key binds its + /// surrogate to that key, so a repeated restore of the same backup + /// verifies. Re-issued headless, the surrogate self-keys, and the repeated + /// restore's bind check refuses it. + #[tokio::test(flavor = "current_thread")] + async fn a_repeated_restore_of_a_multi_vector_document_verifies() { + use std::sync::Arc; + + use crate::bridge::dispatch::Dispatcher; + use crate::control::backup::restore::bind_conflicts::{CarriedBind, check_carried_binds}; + use crate::control::state::SharedState; + use crate::types::{DatabaseId, TenantId}; + use crate::wal::WalManager; + + let dir = tempfile::tempdir().expect("tempdir"); + let wal = + Arc::new(WalManager::open_for_testing(&dir.path().join("mv.wal")).expect("open wal")); + let (dispatcher, _sides) = Dispatcher::new(1, 64); + let state = SharedState::new(dispatcher, wal).expect("shared state"); + let db = DatabaseId::new(1024); + let tenant = TenantId::new(1); + + // One restore: bind the backup's key, then re-issue the document + // through the production binder. The plan carries the stored name, + // as `reissue_vector_snapshots` builds it. + let restore = |collection: &str, pk: &[u8], surrogate: u32, keyed: bool| { + state + .surrogate_assigner + .bind( + nodedb_types::CollectionKey::from_bare(db, collection), + tenant, + pk, + Surrogate::new(surrogate), + ) + .expect("rebind the backup key"); + let stored = nodedb_types::QualifiedCollection::new(db, collection); + let mut plan = build_multi_vector_insert_plan( + stored.as_str(), + "emb", + Surrogate::new(surrogate), + keyed.then(|| pk.to_vec()), + vec![vec![1.0, 0.0], vec![0.0, 1.0]], + ); + crate::control::surrogate::bind_plan_identities( + &state.surrogate_assigner, + db, + tenant, + &mut plan, + ) + .expect("bind the re-issued document"); + }; + let carried = |collection: &str, pk: &[u8], surrogate: u32| { + vec![CarriedBind { + collection: collection.to_string(), + pk: pk.to_vec(), + surrogate, + }] + }; + + restore("mv", b"d1", 12, true); + check_carried_binds(&state, 1, db, carried("mv", b"d1", 12)) + .await + .expect("a repeated restore verifies"); + + restore("mv_headless", b"d2", 13, false); + assert!( + check_carried_binds(&state, 1, db, carried("mv_headless", b"d2", 13)) + .await + .is_err(), + "a self-keyed re-issue binds the surrogate to a second key" + ); + } + + fn members(surrogates: &[u32]) -> std::collections::HashSet { + surrogates.iter().map(|s| Surrogate::new(*s)).collect() + } + + #[test] + fn a_multi_vector_document_keeps_every_vector_in_one_group() { + let rows = vec![ + (0, vec![1.0, 0.0], Some(Surrogate::new(7))), + (1, vec![0.0, 1.0], Some(Surrogate::new(9))), + (2, vec![0.5, 0.5], Some(Surrogate::new(7))), + ]; + let grouped = group_restored_vectors(rows, &members(&[7])).unwrap(); + assert_eq!( + grouped.multi, + vec![(Surrogate::new(7), vec![vec![1.0, 0.0], vec![0.5, 0.5]])] + ); + assert_eq!(grouped.single, vec![(Surrogate::new(9), vec![0.0, 1.0])]); + } + + #[test] + fn a_one_vector_member_is_still_a_multi_vector_document() { + let rows = vec![ + (0, vec![1.0, 0.0], Some(Surrogate::new(4))), + (1, vec![0.0, 1.0], Some(Surrogate::new(5))), + ]; + let grouped = group_restored_vectors(rows, &members(&[4])).unwrap(); + assert_eq!( + grouped.multi, + vec![(Surrogate::new(4), vec![vec![1.0, 0.0]])] + ); + assert_eq!(grouped.single, vec![(Surrogate::new(5), vec![0.0, 1.0])]); + } + + #[test] + fn a_row_without_a_bound_surrogate_fails_the_group() { + for surrogate in [None, Some(Surrogate::ZERO)] { + assert!( + group_restored_vectors(vec![(0, vec![1.0], surrogate)], &members(&[])).is_err(), + "{surrogate:?} names no bound vector" + ); + } + } + + #[test] + fn the_multi_vector_insert_carries_the_full_set() { + let plan = build_multi_vector_insert_plan( + "docs", + "emb", + Surrogate::new(7), + Some(b"doc-7".to_vec()), + vec![vec![1.0, 0.0], vec![0.5, 0.5], vec![0.0, 1.0]], + ); + let PhysicalPlan::Vector(VectorOp::MultiVectorInsert { + vectors, + count, + dim, + pk_bytes, + .. + }) = plan + else { + panic!("expected a MultiVectorInsert"); + }; + assert_eq!(pk_bytes.as_deref(), Some(b"doc-7".as_slice())); + assert_eq!((count, dim), (3, 2)); + assert_eq!(vectors, vec![1.0, 0.0, 0.5, 0.5, 0.0, 1.0]); + } +} diff --git a/nodedb/src/control/backup/schedule/blocking.rs b/nodedb/src/control/backup/schedule/blocking.rs new file mode 100644 index 000000000..da1ea26c7 --- /dev/null +++ b/nodedb/src/control/backup/schedule/blocking.rs @@ -0,0 +1,22 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Run a blocking step of a scheduled backup off the async runtime. +//! +//! Catalog reads and writes, job history writes, and durable audit appends +//! all block their thread. A scheduled backup runs as an async task, so each +//! of these runs on the blocking pool. A metadata proposal is awaited on the +//! task through the async proposer. + +/// Run `work` on the blocking pool and return its result. `what` names the +/// step in the error when the pool task does not finish. +pub async fn off_runtime(what: &'static str, work: F) -> crate::Result +where + T: Send + 'static, + F: FnOnce() -> crate::Result + Send + 'static, +{ + tokio::task::spawn_blocking(work) + .await + .map_err(|e| crate::Error::Internal { + detail: format!("{what} did not finish: {e}"), + })? +} diff --git a/nodedb/src/control/backup/schedule/envelopes.rs b/nodedb/src/control/backup/schedule/envelopes.rs new file mode 100644 index 000000000..269ebc30a --- /dev/null +++ b/nodedb/src/control/backup/schedule/envelopes.rs @@ -0,0 +1,227 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The envelopes a scheduled backup writes under its target, and `keep` +//! retention over them. +//! +//! An envelope is named `-.ndbb`, the time zero-padded to +//! 20 digits. Retention reads only names of that shape, for the schedule's +//! own database, directly under the target. It never deletes anything else. +//! On a local root it lists and deletes through directory fds and follows no +//! symlink, so a symlink under the target is never listed and never deleted +//! through. + +use object_store::path::Path as ObjectPath; + +use crate::control::backup::store::BackupStore; + +/// File extension of a database backup envelope. +const ENVELOPE_EXTENSION: &str = ".ndbb"; + +/// Digits of the zero-padded time in an envelope name. +const TIME_DIGITS: usize = 20; + +/// The name of the envelope of `database` taken at `at_unix_ms`. +pub fn envelope_name(database: &str, at_unix_ms: u64) -> String { + format!("{database}-{at_unix_ms:020}{ENVELOPE_EXTENSION}") +} + +/// The time an envelope name of `database` carries, or `None` when `name` is +/// no envelope of `database`. +fn envelope_time(database: &str, name: &str) -> Option { + let time = name + .strip_prefix(database)? + .strip_prefix('-')? + .strip_suffix(ENVELOPE_EXTENSION)?; + if time.len() != TIME_DIGITS || !time.bytes().all(|b| b.is_ascii_digit()) { + return None; + } + time.parse().ok() +} + +/// One envelope under a target. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Envelope { + pub path: ObjectPath, + pub at_unix_ms: u64, +} + +/// Every envelope of `database` directly under `prefix`, oldest first. +pub async fn list_envelopes( + store: &BackupStore, + prefix: &ObjectPath, + database: &str, +) -> crate::Result> { + let listed = store.list_files(prefix).await?; + let mut envelopes: Vec = listed + .into_iter() + .filter_map(|path| { + let at_unix_ms = envelope_time(database, path.filename()?)?; + Some(Envelope { path, at_unix_ms }) + }) + .collect(); + envelopes.sort_by(|a, b| (a.at_unix_ms, &a.path).cmp(&(b.at_unix_ms, &b.path))); + Ok(envelopes) +} + +/// The envelopes beyond the newest `keep`, oldest first. +fn expired(mut envelopes: Vec, keep: u64) -> Vec { + let keep = usize::try_from(keep).unwrap_or(usize::MAX); + let excess = envelopes.len().saturating_sub(keep); + envelopes.truncate(excess); + envelopes +} + +/// Delete the envelopes of `database` under `prefix` beyond the newest +/// `keep`, oldest first. Returns the number deleted. +pub async fn apply_keep( + store: &BackupStore, + prefix: &ObjectPath, + database: &str, + keep: u64, +) -> crate::Result { + let mut deleted = 0; + for envelope in expired(list_envelopes(store, prefix, database).await?, keep) { + store.delete(&envelope.path).await?; + deleted += 1; + } + Ok(deleted) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use object_store::memory::InMemory; + use object_store::{ObjectStore, ObjectStoreExt, PutPayload}; + + use super::*; + use crate::config::server::BackupStorageSettings; + use crate::control::backup::store::BackupObject; + + #[test] + fn only_names_of_the_schedule_database_parse() { + let name = envelope_name("sales", 1_700_000_000_123); + assert_eq!(name, "sales-00000001700000000123.ndbb"); + assert_eq!(envelope_time("sales", &name), Some(1_700_000_000_123)); + assert_eq!(envelope_time("sal", &name), None); + let other = envelope_name("sales-eu", 5); + assert_eq!(envelope_time("sales", &other), None); + assert_eq!(envelope_time("sales", "sales-12.ndbb"), None); + assert_eq!( + envelope_time("sales", "sales-00000001700000000123.bak"), + None + ); + } + + #[tokio::test] + async fn keep_deletes_the_oldest_envelopes_of_its_database_only() { + let store: Arc = Arc::new(InMemory::new()); + let prefix = ObjectPath::from("nightly"); + let put = |name: String| { + let store = Arc::clone(&store); + async move { + let path = ObjectPath::from(format!("nightly/{name}")); + store + .put(&path, PutPayload::from_static(b"env")) + .await + .unwrap(); + } + }; + for at in [30, 10, 20] { + put(envelope_name("sales", at)).await; + } + put(envelope_name("hr", 1)).await; + put("notes.txt".into()).await; + put(format!("deeper/{}", envelope_name("sales", 5))).await; + + let backup_store = BackupStore::Remote(Arc::clone(&store)); + assert_eq!( + apply_keep(&backup_store, &prefix, "sales", 2) + .await + .unwrap(), + 1 + ); + let left: Vec = list_envelopes(&backup_store, &prefix, "sales") + .await + .unwrap() + .into_iter() + .map(|envelope| envelope.at_unix_ms) + .collect(); + assert_eq!(left, [20, 30]); + assert_eq!( + list_envelopes(&backup_store, &prefix, "hr") + .await + .unwrap() + .len(), + 1 + ); + assert!( + store + .head(&ObjectPath::from("nightly/notes.txt")) + .await + .is_ok() + ); + let deeper = format!("nightly/deeper/{}", envelope_name("sales", 5)); + assert!(store.head(&ObjectPath::from(deeper)).await.is_ok()); + } + + /// A symlink under a local target that points out of the root is never + /// listed, and retention never deletes through it. + #[cfg(unix)] + #[tokio::test] + async fn keep_never_lists_or_deletes_through_a_symlink_out_of_the_root() { + let outside = tempfile::tempdir().unwrap(); + let dir = tempfile::tempdir().unwrap(); + let root = dir.path().canonicalize().unwrap(); + let nightly = root.join("nightly"); + std::fs::create_dir(&nightly).unwrap(); + for at in [20, 30] { + std::fs::write(nightly.join(envelope_name("sales", at)), b"env").unwrap(); + } + let secret = outside.path().join(envelope_name("sales", 10)); + std::fs::write(&secret, b"outside").unwrap(); + std::os::unix::fs::symlink(&secret, nightly.join(envelope_name("sales", 10))).unwrap(); + std::os::unix::fs::symlink(outside.path(), nightly.join("escape")).unwrap(); + + let storage = BackupStorageSettings { + local_root: Some(root.clone()), + ..Default::default() + }; + let target = BackupObject::resolve( + &format!("file://{}/nightly/", root.display()), + Some(&storage), + ) + .unwrap(); + let listed: Vec = list_envelopes(target.store(), target.path(), "sales") + .await + .unwrap() + .into_iter() + .map(|envelope| envelope.at_unix_ms) + .collect(); + assert_eq!(listed, [20, 30], "the symlink is never listed"); + + assert_eq!( + apply_keep(target.store(), target.path(), "sales", 1) + .await + .unwrap(), + 1 + ); + assert!(!nightly.join(envelope_name("sales", 20)).exists()); + assert!(nightly.join(envelope_name("sales", 30)).exists()); + assert_eq!(std::fs::read(&secret).unwrap(), b"outside"); + assert!( + std::fs::symlink_metadata(nightly.join(envelope_name("sales", 10))).is_ok(), + "retention leaves the symlink alone" + ); + + let through = target + .path() + .clone() + .join(envelope_name("sales", 10).as_str()); + assert!( + target.store().delete(&through).await.is_err(), + "a deletion refuses a symlink" + ); + assert_eq!(std::fs::read(&secret).unwrap(), b"outside"); + } +} diff --git a/nodedb/src/control/backup/schedule/marks.rs b/nodedb/src/control/backup/schedule/marks.rs new file mode 100644 index 000000000..bd98f7b9a --- /dev/null +++ b/nodedb/src/control/backup/schedule/marks.rs @@ -0,0 +1,83 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Durable, replicated progress of each scheduled backup. +//! +//! A schedule's mark is the scheduled minute through which every run is +//! settled. The node running scheduled backups raises it through the +//! metadata group after each completed run, so a node that takes that role +//! later reads the same mark. +//! +//! The scheduler reads a mark only after this node applied the metadata +//! group through a read index its leader confirmed. A node whose catalog +//! lags therefore never mistakes a finished minute for a due one. + +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use super::blocking::off_runtime; +use crate::config::server::BackupScheduleSettings; +use crate::control::catalog_entry::CatalogEntry; +use crate::control::cluster::linearizable_read::confirm_linearizable_read; +use crate::control::metadata_proposer::propose_catalog_entry_async; +use crate::control::security::catalog::backup_schedule_marks::StoredBackupScheduleMark; +use crate::control::state::SharedState; +use crate::event::scheduler::coordinator::ensure_system_coordinator; + +/// How long a mark read waits for its read index and for this node to apply +/// through it. A read that does not finish in time skips its tick. +const MARK_READ_TIMEOUT: Duration = Duration::from_secs(5); + +/// The minute through which `schedule` is settled in this node's catalog, or +/// `None` when it was never armed at its current config incarnation. The +/// value can lag the metadata group: the scheduler uses +/// [`settled_through_linearizable`]. +pub fn settled_through( + state: &SharedState, + schedule: &BackupScheduleSettings, +) -> crate::Result> { + Ok(state + .credentials + .catalog() + .backup_schedule_mark(&schedule.job_name(), schedule.incarnation())? + .map(|mark| mark.through_minute)) +} + +/// The mark of `schedule` as of now: every mark committed before this call is +/// applied here first. Fails when the metadata group confirms no read index, +/// or this node does not apply through it within [`MARK_READ_TIMEOUT`]. +pub async fn settled_through_linearizable( + state: &Arc, + schedule: &BackupScheduleSettings, +) -> crate::Result> { + confirm_linearizable_read( + state, + &[nodedb_cluster::METADATA_GROUP_ID], + Instant::now() + MARK_READ_TIMEOUT, + ) + .await?; + let (state, schedule) = (Arc::clone(state), schedule.clone()); + off_runtime("backup schedule mark read", move || { + settled_through(&state, &schedule) + }) + .await +} + +/// Raise the mark of `schedule` to `through_minute` on every node. +/// +/// Fails without proposing when this node is no longer the `_system` +/// coordinator. Returns once this node applied the mark. +pub async fn raise( + state: &Arc, + schedule: &BackupScheduleSettings, + through_minute: u64, +) -> crate::Result<()> { + ensure_system_coordinator(state)?; + let entry = CatalogEntry::PutBackupScheduleMark(Box::new(StoredBackupScheduleMark { + job: schedule.job_name(), + incarnation: schedule.incarnation(), + through_minute, + })); + // The proposer awaits every wait of the write on this task. + propose_catalog_entry_async(state, &entry).await?; + Ok(()) +} diff --git a/nodedb/src/control/backup/schedule/mod.rs b/nodedb/src/control/backup/schedule/mod.rs new file mode 100644 index 000000000..04ad68283 --- /dev/null +++ b/nodedb/src/control/backup/schedule/mod.rs @@ -0,0 +1,9 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod blocking; +pub mod envelopes; +pub mod marks; +pub mod run; + +pub use envelopes::{Envelope, apply_keep, envelope_name, list_envelopes}; +pub use run::{BackupRun, run_scheduled_backup, write_and_retain}; diff --git a/nodedb/src/control/backup/schedule/run.rs b/nodedb/src/control/backup/schedule/run.rs new file mode 100644 index 000000000..3302aa3f1 --- /dev/null +++ b/nodedb/src/control/backup/schedule/run.rs @@ -0,0 +1,224 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One scheduled backup run: `BACKUP DATABASE` to a new envelope under the +//! target, then `keep` retention under the target. + +use std::sync::Arc; + +use super::blocking::off_runtime; +use super::envelopes::{apply_keep, envelope_name}; +use crate::config::server::{BackupScheduleSettings, BackupStorageSettings}; +use crate::control::backup::database::{backup_database, database_tenants}; +use crate::control::backup::store::BackupObject; +use crate::control::state::SharedState; + +/// What one run did. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct BackupRun { + pub envelope_uri: String, + pub bytes: u64, + /// Envelopes `keep` retention deleted. + pub deleted: u64, +} + +/// Back up `schedule.database` to the envelope named for +/// `scheduled_unix_ms`, then apply `keep`. +/// +/// `scheduled_unix_ms` is the scheduled minute, not the clock at run time. +/// Two runs of one minute, such as a run repeated after a leader change, +/// write the same object. +pub async fn run_scheduled_backup( + state: &Arc, + schedule: &BackupScheduleSettings, + scheduled_unix_ms: u64, +) -> crate::Result { + let (database_id, tenants) = { + let (state, database) = (Arc::clone(state), schedule.database.clone()); + off_runtime("scheduled backup database lookup", move || { + let database_id = state + .credentials + .catalog() + .get_database_id_by_name(&database)? + .ok_or_else(|| crate::Error::BadRequest { + detail: format!( + "scheduled backup: database '{database}' does not exist; fix or \ + remove its backup.schedule entry" + ), + })?; + let tenants = database_tenants(&state, database_id)?; + Ok((database_id, tenants)) + }) + .await? + }; + let bytes = backup_database(state, database_id, &schedule.database, &tenants).await?; + // A former coordinator whose lease lapsed during the capture writes no + // envelope. + crate::event::scheduler::coordinator::ensure_system_coordinator(state)?; + let run = write_and_retain( + schedule, + state.backup_storage.as_deref(), + scheduled_unix_ms, + bytes, + ) + .await?; + // The audit append is durable, so it runs on the blocking pool. + let detail = format!( + "scheduled BACKUP DATABASE {} TO '{}' wrote {} bytes, deleted {} old envelopes", + schedule.database, run.envelope_uri, run.bytes, run.deleted + ); + let audit_state = Arc::clone(state); + off_runtime("scheduled backup audit record", move || { + audit_state.audit_record( + crate::control::security::audit::AuditEvent::AdminAction, + None, + "_system_scheduler", + &detail, + ); + Ok(()) + }) + .await?; + Ok(run) +} + +/// Write `bytes` as the envelope of `scheduled_unix_ms` under the target, +/// replacing an envelope of the same minute, then delete the oldest +/// envelopes of the database beyond `keep`. +/// +/// A retention error comes after the write: the new envelope is kept, and +/// the next run retries retention. +pub async fn write_and_retain( + schedule: &BackupScheduleSettings, + storage: Option<&BackupStorageSettings>, + scheduled_unix_ms: u64, + bytes: Vec, +) -> crate::Result { + let prefix = schedule.target_prefix(); + let envelope_uri = format!( + "{prefix}/{}", + envelope_name(&schedule.database, scheduled_unix_ms) + ); + let envelope = resolve(envelope_uri.clone(), storage).await?; + let size = bytes.len() as u64; + envelope.put(bytes).await?; + + let target = resolve(prefix.to_string(), storage).await?; + let deleted = apply_keep( + target.store(), + target.path(), + &schedule.database, + schedule.keep, + ) + .await + .map_err(|e| crate::Error::Storage { + engine: "backup".into(), + detail: format!( + "scheduled backup wrote '{envelope_uri}', but keep retention under \ + '{prefix}' did not finish: {e}" + ), + })?; + Ok(BackupRun { + envelope_uri, + bytes: size, + deleted, + }) +} + +/// Resolve `uri` on the blocking pool: a `file://` URI opens its local root. +async fn resolve( + uri: String, + storage: Option<&BackupStorageSettings>, +) -> crate::Result { + let storage = storage.cloned(); + off_runtime("backup URI resolve", move || { + BackupObject::resolve(&uri, storage.as_ref()).map_err(crate::Error::from) + }) + .await +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::backup::schedule::envelopes::list_envelopes; + + fn schedule(root: &std::path::Path, keep: u64) -> BackupScheduleSettings { + BackupScheduleSettings { + database: "sales".into(), + target: format!("file://{}/nightly/", root.display()), + cron: "0 3 * * *".into(), + keep, + } + } + + fn storage(root: &std::path::Path) -> BackupStorageSettings { + BackupStorageSettings { + local_root: Some(root.to_path_buf()), + ..Default::default() + } + } + + #[tokio::test] + async fn each_run_writes_an_envelope_and_keep_removes_the_oldest() { + let dir = tempfile::tempdir().unwrap(); + let root = dir.path().canonicalize().unwrap(); + let (schedule, storage) = (schedule(&root, 2), storage(&root)); + + let mut deleted = Vec::new(); + for (at, body) in [(1_000, b"one"), (2_000, b"two"), (3_000, b"thr")] { + let run = write_and_retain(&schedule, Some(&storage), at, body.to_vec()) + .await + .unwrap(); + assert!(run.envelope_uri.ends_with(&envelope_name("sales", at))); + assert_eq!(run.bytes, 3); + deleted.push(run.deleted); + } + assert_eq!(deleted, [0, 0, 1]); + + let nightly = root.join("nightly"); + assert!(!nightly.join(envelope_name("sales", 1_000)).exists()); + assert_eq!( + std::fs::read(nightly.join(envelope_name("sales", 3_000))).unwrap(), + b"thr" + ); + let target = BackupObject::resolve(schedule.target_prefix(), Some(&storage)).unwrap(); + let left: Vec = list_envelopes(target.store(), target.path(), "sales") + .await + .unwrap() + .into_iter() + .map(|envelope| envelope.at_unix_ms) + .collect(); + assert_eq!(left, [2_000, 3_000]); + } + + #[tokio::test] + async fn a_repeated_minute_rewrites_its_own_envelope() { + let dir = tempfile::tempdir().unwrap(); + let root = dir.path().canonicalize().unwrap(); + let (schedule, storage) = (schedule(&root, 5), storage(&root)); + for body in [b"first", b"again"] { + let run = write_and_retain(&schedule, Some(&storage), 60_000, body.to_vec()) + .await + .unwrap(); + assert_eq!(run.deleted, 0); + } + let nightly = root.join("nightly"); + assert_eq!(std::fs::read_dir(&nightly).unwrap().count(), 1); + assert_eq!( + std::fs::read(nightly.join(envelope_name("sales", 60_000))).unwrap(), + b"again" + ); + } + + #[tokio::test] + async fn a_target_outside_the_local_root_writes_nothing() { + let dir = tempfile::tempdir().unwrap(); + let root = dir.path().canonicalize().unwrap(); + let elsewhere = tempfile::tempdir().unwrap(); + let schedule = schedule(&elsewhere.path().canonicalize().unwrap(), 1); + assert!( + write_and_retain(&schedule, Some(&storage(&root)), 1, vec![1]) + .await + .is_err() + ); + assert!(std::fs::read_dir(&root).unwrap().next().is_none()); + } +} diff --git a/nodedb/src/control/backup/snapshot_keys.rs b/nodedb/src/control/backup/snapshot_keys.rs index d3b13685b..96904b714 100644 --- a/nodedb/src/control/backup/snapshot_keys.rs +++ b/nodedb/src/control/backup/snapshot_keys.rs @@ -11,42 +11,93 @@ //! no suffix. The collection never contains `':'` or `'\0'`. Use //! [`extract_db_tenant_scoped_collection`]. //! - **db-scoped (collection-last)** — `"{db}:{tid}:{collection}"` where the -//! collection is the remainder and may itself contain `':'` (flushed-ts +//! collection is the remainder and can itself contain `':'` (flushed-ts //! segments, columnar engines, kv tables). Use [`extract_db_scoped_collection`]. //! //! Every extracted collection is the name the Data Plane stores it under: //! database-qualified (`"{db}/{name}"`) outside the default database. -//! [`vshard_of_stored`] maps such a name to its vShard. +//! [`homes_of_stored`] maps a stored record to its homes by the shared +//! [`RecordHomes`] rule: a row by its collection, an edge by its endpoints. //! -//! The Raft snapshot SEND builder filters sections by which vshard each -//! entry's collection routes to, so the parsing lives here once. +//! The Raft snapshot SEND builder filters sections by the homes of each +//! entry, so the parsing lives here once. //! -//! The backup orchestrator additionally needs to filter a fully-gathered, -//! single-tenant [`TenantDataSnapshot`] *in place* to a set of source vshards -//! (the vshards a given node is the assigned source for), preserving the -//! per-tenant section shapes the RESTORE merge path consumes. That per-section -//! classification is the SAME vshard-of-collection logic the Raft snapshot SEND -//! builder applies, so it lives here once as [`retain_tenant_data_for_vshards`] -//! and is shared rather than duplicated. +//! The backup orchestrator and MOVE TENANT capture filter a fully-gathered, +//! single-tenant [`TenantDataSnapshot`] *in place* to the vShards one node is +//! the assigned source for, preserving the section shapes the RESTORE merge +//! path consumes. [`retain_tenant_data_for_vshards`] keeps each record on the +//! source of its owner home, so every record is captured exactly once. use std::collections::HashSet; use nodedb_types::{CollectionKey, DatabaseId}; use crate::engine::graph::edge_store::parse_versioned_edge_key; -use crate::types::TenantDataSnapshot; +use crate::types::{HomedRecord, RecordHomes, TenantDataSnapshot}; -/// The vShard of a collection named as the Data Plane stores it in -/// `database_id`. +/// One snapshot entry, named as the Data Plane stores it. +#[derive(Debug, Clone, Copy)] +pub enum StoredRecord<'a> { + /// A row, or a per-collection section, of `collection`. + Row { collection: &'a str }, + /// A graph edge of `collection` from `src` to `dst`. + Edge { + collection: &'a str, + src: &'a str, + dst: &'a str, + }, + /// Cells of the array `array` that route to `vshard`. + ArrayCells { array: &'a str, vshard: u32 }, +} + +impl<'a> StoredRecord<'a> { + /// The collection the entry belongs to, as the Data Plane stores it. + pub fn collection(self) -> &'a str { + match self { + Self::Row { collection } | Self::Edge { collection, .. } => collection, + Self::ArrayCells { array, .. } => array, + } + } + + /// The edge a versioned edge key names, or `None` for a malformed key. + pub fn from_edge_key(key: &'a str) -> Option { + parse_versioned_edge_key(key).map(|(collection, src, _, dst, _)| Self::Edge { + collection, + src, + dst, + }) + } +} + +/// The catalog key of a collection named as the Data Plane stores it in +/// `database_id`. A name without the qualifier resolves as a bare name. +pub fn stored_collection_key(database_id: DatabaseId, stored: &str) -> CollectionKey<'_> { + CollectionKey::from_qualified_str(database_id, stored) + .unwrap_or_else(|_| CollectionKey::from_bare(database_id, stored)) +} + +/// The homes of a record stored in `database_id`. /// -/// A database-qualified name routes by its bare name, exactly as the write -/// that stored it did. A name without the qualifier routes as a bare name. -/// Both are deterministic, so every source node that filters the same entry -/// assigns it the same vShard. +/// Deterministic, so every node that filters the same entry assigns it the +/// same homes. +pub fn homes_of_stored(database_id: DatabaseId, record: StoredRecord<'_>) -> RecordHomes { + match record { + StoredRecord::Row { collection } => RecordHomes::of(HomedRecord::Row( + stored_collection_key(database_id, collection), + )), + StoredRecord::Edge { src, dst, .. } => RecordHomes::of(HomedRecord::Edge { src, dst }), + StoredRecord::ArrayCells { vshard, .. } => RecordHomes::of(HomedRecord::ArrayCells { + vshard: crate::types::VShardId::new(vshard), + }), + } +} + +/// The vShard of a collection named as the Data Plane stores it in +/// `database_id`: the home of its rows. pub fn vshard_of_stored(database_id: DatabaseId, stored: &str) -> u32 { - let key = CollectionKey::from_qualified_str(database_id, stored) - .unwrap_or_else(|_| CollectionKey::from_bare(database_id, stored)); - nodedb_cluster::routing::vshard_for_collection(key) + homes_of_stored(database_id, StoredRecord::Row { collection: stored }) + .owner() + .as_u32() } /// Extract the collection from a `"{db}:{tid}:{collection}[:suffix...]"` key. @@ -72,7 +123,7 @@ pub fn extract_db_tenant_scoped_collection(key: &str, tenant_id: u64) -> Option< /// Extract the collection from a db-scoped `"{db}:{tid}:{collection}"` key, /// verifying the embedded tenant matches `tenant_id`. /// -/// The first two ':' are structural (db, tid); the collection may itself +/// The first two ':' are structural (db, tid); the collection can itself /// contain ':'. Returns `None` when the key has fewer than three parts or the /// tenant does not match. pub fn extract_db_scoped_collection(key: &str, tenant_id: u64) -> Option<&str> { @@ -86,49 +137,50 @@ pub fn extract_db_scoped_collection(key: &str, tenant_id: u64) -> Option<&str> { Some(coll) } -/// Filter a single-tenant [`TenantDataSnapshot`] in place to only those -/// sections whose collection routes to a vshard in `source_vshards`. +/// Filter a single-tenant [`TenantDataSnapshot`] in place to the records whose +/// owner home is in `source_vshards`. /// /// The backup orchestrator gathers a full per-node snapshot (under RF>1 every /// replica holds the full vshard data), then calls this so each node /// contributes EXACTLY the vshards it is the assigned source for — the union -/// over nodes covers each vshard once (no duplication, no loss). The retained -/// section shapes are unchanged, so the RESTORE merge path (`merge_sections`) -/// consumes the output exactly as before. +/// over nodes covers each record once (no duplication, no loss). A graph edge +/// is kept by the source of its `from_key(src)` home, never by the source of +/// its collection: that node holds the edge whatever the node count and RF. +/// The retained section shapes are unchanged, so the RESTORE merge path +/// (`merge_sections`) consumes the output exactly as before. /// -/// `vshard_of` maps a collection name to its vshard (the caller passes -/// [`vshard_of_stored`] for the snapshot's database), matching the Raft -/// snapshot SEND builder. Every section kind the snapshot carries is -/// classified here so adding a section without updating this filter is -/// impossible to miss: +/// `homes_of` maps a record to its homes, or to `None` to drop it (the caller +/// passes [`homes_of_stored`] for the snapshot's database). Every section kind +/// the snapshot carries is classified here so adding a section without +/// updating this filter is impossible to miss: /// /// - db-tenant-scoped keys (`documents`, `indexes`, `documents_versioned`, /// `indexes_versioned`, `vectors`, `timeseries`) via /// [`extract_db_tenant_scoped_collection`]. /// - db-scoped keys (`flushed_ts_segments`, `columnar_engines`, `kv_tables`) /// via [`extract_db_scoped_collection`]. -/// - graph `edges` via [`parse_versioned_edge_key`] (key embeds the collection). +/// - graph `edges` via [`StoredRecord::from_edge_key`], homed on endpoints. +/// The hidden edge versions, cuts and applied ordinals are dropped. /// - `surrogate_pk` by its explicit `collection` field (the bare name). /// - CRDT (`crdt_state`): per-collection, tenant-explicit. Each entry carries /// its single collection, so it is kept iff that collection's vshard is in /// `source_vshards` — the node owning the collection keeps it, every other /// node drops it (captured exactly once, never duplicated). +/// - `arrays` by the vShard each blob's cells route to. pub fn retain_tenant_data_for_vshards( snap: &mut TenantDataSnapshot, tenant_id: u64, source_vshards: &HashSet, - vshard_of: impl Fn(&str) -> u32, + homes_of: impl Fn(StoredRecord<'_>) -> Option, ) { - let in_group_db_tenant_scoped = |key: &str| { - extract_db_tenant_scoped_collection(key, tenant_id) - .map(|c| source_vshards.contains(&vshard_of(c))) - .unwrap_or(false) - }; - let in_group_db_scoped = |key: &str| { - extract_db_scoped_collection(key, tenant_id) - .map(|c| source_vshards.contains(&vshard_of(c))) - .unwrap_or(false) + let owned = |record: StoredRecord<'_>| { + homes_of(record).is_some_and(|homes| homes.owned_by(source_vshards)) }; + let owned_row = |collection: &str| owned(StoredRecord::Row { collection }); + let in_group_db_tenant_scoped = + |key: &str| extract_db_tenant_scoped_collection(key, tenant_id).is_some_and(owned_row); + let in_group_db_scoped = + |key: &str| extract_db_scoped_collection(key, tenant_id).is_some_and(owned_row); snap.documents.retain(|(k, _)| in_group_db_tenant_scoped(k)); snap.indexes.retain(|(k, _)| in_group_db_tenant_scoped(k)); @@ -144,27 +196,39 @@ pub fn retain_tenant_data_for_vshards( snap.columnar_engines.retain(|(k, _)| in_group_db_scoped(k)); snap.kv_tables.retain(|(k, _)| in_group_db_scoped(k)); // surrogate_pk: the field IS the collection name. - snap.surrogate_pk - .retain(|e| source_vshards.contains(&vshard_of(&e.collection))); - // Graph edges: collection is the first '\0'-delimited key component. An - // unparseable key has no determinable vshard; drop it from EVERY node's - // retained set rather than duplicate it across all sources. - snap.edges.retain(|(k, _)| { - parse_versioned_edge_key(k) - .map(|(collection, ..)| source_vshards.contains(&vshard_of(collection))) - .unwrap_or(false) - }); + snap.surrogate_pk.retain(|e| owned_row(&e.collection)); + // An unparseable edge key has no determinable home; drop it from EVERY + // node's retained set rather than duplicate it across all sources. + snap.edges + .retain(|(k, _)| StoredRecord::from_edge_key(k).is_some_and(owned)); + // A backup carries the edge versions a current read reaches. Its RESTORE + // re-issues them, each applied at the restore's own ordinal, so the + // versions a TRUNCATE hides, the cuts and the applied ordinals of the + // source stay behind. + snap.edge_hidden.clear(); + snap.edge_cuts.clear(); + snap.edge_applied.clear(); // CRDT (`crdt_state`): each entry carries its single collection. Keep it iff // that collection's vshard is in this source set — exactly one source node // (the collection's owner) retains each entry. snap.crdt_state - .retain(|(_, _, collection, _)| source_vshards.contains(&vshard_of(collection))); + .retain(|(_, _, collection, _)| owned_row(collection)); + snap.arrays.retain(|blob| { + owned(StoredRecord::ArrayCells { + array: &blob.array, + vshard: blob.vshard, + }) + }); } #[cfg(test)] mod tests { - use super::{extract_db_scoped_collection, extract_db_tenant_scoped_collection}; + use super::{ + StoredRecord, extract_db_scoped_collection, extract_db_tenant_scoped_collection, + homes_of_stored, vshard_of_stored, + }; + use nodedb_types::DatabaseId; /// A qualified Data-Plane name in a named database routes to the vShard /// of its bare catalog key, never to the vShard of the qualified string. @@ -218,7 +282,7 @@ mod tests { extract_db_scoped_collection("0:7:metrics", 7), Some("metrics") ); - // Collection may itself contain ':'. + // Collection can itself contain ':'. assert_eq!( extract_db_scoped_collection("0:7:a:b", 7), Some("a:b"), @@ -243,11 +307,10 @@ mod tests { use std::collections::HashSet; const TID: u64 = 1; - // Two collections deterministically mapped to two distinct vshards via - // the test `vshard_of` closure (first char's code). - let vshard_of = |c: &str| c.bytes().next().map(u32::from).unwrap_or(0); - let va = vshard_of("alpha"); // 97 - let vb = vshard_of("beta"); // 98 + let homes_of = |r: StoredRecord<'_>| Some(homes_of_stored(DatabaseId::DEFAULT, r)); + let va = vshard_of_stored(DatabaseId::DEFAULT, "alpha"); + let vb = vshard_of_stored(DatabaseId::DEFAULT, "beta"); + assert_ne!(va, vb); let template = || TenantDataSnapshot { timeseries: vec![ @@ -275,7 +338,7 @@ mod tests { // Node owning only vshard(alpha) keeps alpha sections, drops beta. let mut node_a = template(); let only_a: HashSet = [va].into_iter().collect(); - retain_tenant_data_for_vshards(&mut node_a, TID, &only_a, vshard_of); + retain_tenant_data_for_vshards(&mut node_a, TID, &only_a, homes_of); assert_eq!(node_a.timeseries.len(), 1); assert_eq!(node_a.timeseries[0].0, format!("0:{TID}:alpha")); assert_eq!(node_a.columnar_engines.len(), 1); @@ -285,7 +348,7 @@ mod tests { // Node owning only vshard(beta) keeps beta sections, drops alpha. let mut node_b = template(); let only_b: HashSet = [vb].into_iter().collect(); - retain_tenant_data_for_vshards(&mut node_b, TID, &only_b, vshard_of); + retain_tenant_data_for_vshards(&mut node_b, TID, &only_b, homes_of); assert_eq!(node_b.timeseries.len(), 1); assert_eq!(node_b.timeseries[0].0, format!("0:{TID}:beta")); assert_eq!(node_b.kv_tables.len(), 0, "alpha kv not owned by beta node"); @@ -293,12 +356,12 @@ mod tests { // Third replica owns neither → contributes nothing for these vshards. let mut node_c = template(); let none: HashSet = HashSet::new(); - retain_tenant_data_for_vshards(&mut node_c, TID, &none, vshard_of); + retain_tenant_data_for_vshards(&mut node_c, TID, &none, homes_of); assert!(node_c.timeseries.is_empty()); assert!(node_c.columnar_engines.is_empty()); // Union over the three replicas = each collection's append section - // exactly once (the fix: no ~3× multiplication). + // exactly once, with no ~3× multiplication. let union_ts = node_a.timeseries.len() + node_b.timeseries.len() + node_c.timeseries.len(); assert_eq!( union_ts, 2, @@ -314,7 +377,7 @@ mod tests { use crate::types::TenantDataSnapshot; use std::collections::HashSet; - let vshard_of = |c: &str| c.bytes().next().map(u32::from).unwrap_or(0); + let homes_of = |r: StoredRecord<'_>| Some(homes_of_stored(DatabaseId::DEFAULT, r)); let mut snap = TenantDataSnapshot { timeseries: vec![("0:1:alpha".into(), b"a".to_vec())], columnar_engines: vec![("0:1:beta".into(), b"b".to_vec())], @@ -323,11 +386,79 @@ mod tests { }; let all: HashSet = ["alpha", "beta", "gamma"] .iter() - .map(|c| vshard_of(c)) + .map(|c| vshard_of_stored(DatabaseId::DEFAULT, c)) .collect(); - retain_tenant_data_for_vshards(&mut snap, 1, &all, vshard_of); + retain_tenant_data_for_vshards(&mut snap, 1, &all, homes_of); assert_eq!(snap.timeseries.len(), 1); assert_eq!(snap.columnar_engines.len(), 1); assert_eq!(snap.kv_tables.len(), 1); } + + /// An edge is kept by the source of its `from_key(src)` home, never by the + /// source of its collection's home. + #[test] + fn an_edge_is_kept_by_the_source_of_its_src_home() { + use super::retain_tenant_data_for_vshards; + use crate::types::{TenantDataSnapshot, VShardId}; + use std::collections::HashSet; + + let collection_home = vshard_of_stored(DatabaseId::DEFAULT, "follows"); + let src = (0..4096) + .map(|i| format!("n{i}")) + .find(|k| VShardId::from_key(k.as_bytes()).as_u32() != collection_home) + .expect("a node key off the collection home"); + let src_home = VShardId::from_key(src.as_bytes()).as_u32(); + let key = format!("follows\x00{src}\x00L\x00{src}\x00{:020}", 1); + let homes_of = |r: StoredRecord<'_>| Some(homes_of_stored(DatabaseId::DEFAULT, r)); + let template = || TenantDataSnapshot { + edges: vec![(key.clone(), vec![])], + ..Default::default() + }; + + let mut by_collection = template(); + let collection_only: HashSet = [collection_home].into_iter().collect(); + retain_tenant_data_for_vshards(&mut by_collection, 1, &collection_only, homes_of); + assert!(by_collection.edges.is_empty()); + + let mut by_src = template(); + let src_only: HashSet = [src_home].into_iter().collect(); + retain_tenant_data_for_vshards(&mut by_src, 1, &src_only, homes_of); + assert_eq!(by_src.edges.len(), 1); + } + + /// Array cells are kept by the source of the vShard they route to. + #[test] + fn array_cells_are_kept_by_the_source_of_their_vshard() { + use super::retain_tenant_data_for_vshards; + use crate::types::{ArrayCellsBlob, TenantDataSnapshot}; + use std::collections::HashSet; + + let blob = |vshard: u32| ArrayCellsBlob { + database_id: 0, + tenant_id: 1, + array: "grid".into(), + vshard, + cells: vec![vshard as u8], + }; + let mut snap = TenantDataSnapshot { + arrays: vec![blob(3), blob(9)], + ..Default::default() + }; + let homes_of = |r: StoredRecord<'_>| Some(homes_of_stored(DatabaseId::DEFAULT, r)); + let source: HashSet = [9].into_iter().collect(); + retain_tenant_data_for_vshards(&mut snap, 1, &source, homes_of); + assert_eq!(snap.arrays, vec![blob(9)]); + } + + #[test] + fn a_stored_edge_key_names_its_endpoints() { + let key = format!("1025/follows\x00a\x00L\x00b\x00{:020}", 7); + let record = StoredRecord::from_edge_key(&key).expect("versioned edge key"); + assert_eq!(record.collection(), "1025/follows"); + assert_eq!( + homes_of_stored(DatabaseId::new(1025), record), + crate::types::RecordHomes::edge("a", "b") + ); + assert!(StoredRecord::from_edge_key("not-an-edge-key").is_none()); + } } diff --git a/nodedb/src/control/backup/store.rs b/nodedb/src/control/backup/store.rs new file mode 100644 index 000000000..bf4cfbbe8 --- /dev/null +++ b/nodedb/src/control/backup/store.rs @@ -0,0 +1,568 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The object a `BACKUP DATABASE` writes and a `RESTORE DATABASE` reads. +//! +//! A URI names only the object. Credentials and the local root come from the +//! `[backup_storage]` config section, never from the SQL text: +//! +//! * `file:///` must lie inside `local_root`, with every symlink on it +//! followed. Without `local_root` every `file://` URI is refused, so SQL +//! cannot name an arbitrary server path. Reads, writes, listings and +//! deletions open the path below a directory fd of the canonical root and +//! follow no symlink, so a component swapped for a symlink after the URI +//! resolved refuses them (see [`super::store_local`]). +//! * `s3:///` uses the section's endpoint, region and keys. + +use std::path::{Component, Path, PathBuf}; +use std::sync::Arc; + +use object_store::aws::AmazonS3Builder; +use object_store::path::Path as ObjectPath; +use object_store::{ObjectStore, ObjectStoreExt, PutPayload}; + +use crate::Error; +use crate::config::server::BackupStorageSettings; + +use super::store_local::{self, LocalIoError}; + +/// A backup URI the server refuses before it touches any store. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub enum BackupUriError { + #[error("backup URI '{uri}': unsupported scheme; use file:/// or s3:///")] + UnsupportedScheme { uri: String }, + #[error("backup URI '{uri}': {detail}")] + Malformed { uri: String, detail: String }, + #[error( + "backup URI '{uri}': file:// URIs need [backup_storage] local_root in the server config" + )] + NoLocalRoot { uri: String }, + #[error("backup URI '{uri}': the path must lie inside [backup_storage] local_root '{root}'")] + OutsideLocalRoot { uri: String, root: String }, + #[error("backup URI '{uri}': cannot open the store: {detail}")] + Store { uri: String, detail: String }, +} + +impl From for Error { + fn from(e: BackupUriError) -> Self { + Error::BadRequest { + detail: e.to_string(), + } + } +} + +/// Why a read or write of a backup object did not complete. +#[derive(Debug, thiserror::Error)] +pub enum BackupIoError { + /// The path left the local root at the open, through a symlink placed on + /// it after the URI resolved. + #[error(transparent)] + Refused(BackupUriError), + #[error(transparent)] + Failed(Error), +} + +impl From for Error { + fn from(e: BackupIoError) -> Self { + match e { + BackupIoError::Refused(refusal) => refusal.into(), + BackupIoError::Failed(error) => error, + } + } +} + +/// One backup object in one store. +pub struct BackupObject { + uri: String, + store: BackupStore, + path: ObjectPath, +} + +impl std::fmt::Debug for BackupObject { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("BackupObject") + .field("uri", &self.uri) + .field("path", &self.path) + .finish() + } +} + +/// The store a backup object lives in. +#[derive(Debug, Clone)] +pub enum BackupStore { + /// Below the canonical `[backup_storage] local_root`. Every read, write, + /// listing and deletion opens its path below a directory fd of the root + /// and follows no symlink (see [`super::store_local`]). + Local(PathBuf), + /// An object store reached over the network. + Remote(Arc), +} + +impl BackupStore { + /// Every object directly under `dir`. A local root lists only regular + /// files, never a symlink. + pub async fn list_files(&self, dir: &ObjectPath) -> Result, BackupIoError> { + let uri = self.uri_of(dir); + match self { + Self::Local(root) => { + let components = components(dir); + let names = local_io(root, &uri, "list", move |fd| { + store_local::list_files(fd, &components) + }) + .await?; + Ok(names + .iter() + .map(|name| dir.clone().join(name.as_str())) + .collect()) + } + Self::Remote(store) => { + let listed = store + .list_with_delimiter(Some(dir)) + .await + .map_err(|e| failed("list", &uri, e))?; + Ok(listed + .objects + .into_iter() + .map(|meta| meta.location) + .collect()) + } + } + } + + /// Delete the object at `path`. A missing object counts as deleted. A + /// local root refuses a symlink, so no deletion reaches its target. + pub async fn delete(&self, path: &ObjectPath) -> Result<(), BackupIoError> { + let uri = self.uri_of(path); + match self { + Self::Local(root) => { + let components = components(path); + local_io(root, &uri, "delete", move |fd| { + store_local::delete_beneath(fd, &components) + }) + .await + } + Self::Remote(store) => match store.delete(path).await { + Ok(()) | Err(object_store::Error::NotFound { .. }) => Ok(()), + Err(e) => Err(failed("delete", &uri, e)), + }, + } + } + + async fn put(&self, uri: &str, path: &ObjectPath, bytes: Vec) -> Result<(), BackupIoError> { + match self { + Self::Local(root) => { + let components = components(path); + local_io(root, uri, "write", move |fd| { + store_local::write_beneath(fd, &components, &bytes) + }) + .await + } + Self::Remote(store) => store + .put(path, PutPayload::from(bytes)) + .await + .map(drop) + .map_err(|e| failed("write", uri, e)), + } + } + + async fn get(&self, uri: &str, path: &ObjectPath) -> Result, BackupIoError> { + match self { + Self::Local(root) => { + let components = components(path); + local_io(root, uri, "read", move |fd| { + store_local::read_beneath(fd, &components) + }) + .await + } + Self::Remote(store) => { + let result = store.get(path).await.map_err(|e| failed("read", uri, e))?; + let bytes = result.bytes().await.map_err(|e| failed("read", uri, e))?; + Ok(bytes.to_vec()) + } + } + } + + /// The URI error messages name `path` by. + fn uri_of(&self, path: &ObjectPath) -> String { + match self { + Self::Local(root) => format!("file://{}/{path}", root.display()), + Self::Remote(store) => format!("{store}/{path}"), + } + } +} + +impl BackupObject { + /// Resolve `uri` against `settings`. + pub fn resolve( + uri: &str, + settings: Option<&BackupStorageSettings>, + ) -> Result { + let malformed = |detail: &str| BackupUriError::Malformed { + uri: uri.to_string(), + detail: detail.to_string(), + }; + let store_error = |detail: String| BackupUriError::Store { + uri: uri.to_string(), + detail, + }; + if let Some(rest) = uri.strip_prefix("file://") { + let root = settings + .and_then(|s| s.local_root.as_deref()) + .ok_or_else(|| BackupUriError::NoLocalRoot { + uri: uri.to_string(), + })?; + let (canonical_root, target) = + local_target(root, Path::new(rest)).map_err(|refusal| match refusal { + LocalRefusal::Outside => BackupUriError::OutsideLocalRoot { + uri: uri.to_string(), + root: root.display().to_string(), + }, + LocalRefusal::Unreadable(detail) => store_error(detail), + })?; + let key = target + .strip_prefix(&canonical_root) + .ok() + .and_then(object_key) + .ok_or_else(|| malformed("the path names no file"))?; + return Ok(Self { + uri: uri.to_string(), + store: BackupStore::Local(canonical_root), + path: key, + }); + } + if let Some(rest) = uri.strip_prefix("s3://") { + let (bucket, key) = rest + .split_once('/') + .ok_or_else(|| malformed("expected s3:///"))?; + if bucket.is_empty() || key.trim_matches('/').is_empty() { + return Err(malformed("expected s3:///")); + } + let path = ObjectPath::parse(key.trim_matches('/')) + .map_err(|e| malformed(&format!("invalid object key: {e}")))?; + let defaults = BackupStorageSettings::default(); + let settings = settings.unwrap_or(&defaults); + let mut builder = AmazonS3Builder::new() + .with_bucket_name(bucket) + .with_region(&settings.region); + if !settings.endpoint.is_empty() { + builder = builder + .with_endpoint(&settings.endpoint) + .with_allow_http(settings.endpoint.starts_with("http://")); + } + if !settings.access_key.is_empty() { + builder = builder + .with_access_key_id(&settings.access_key) + .with_secret_access_key(&settings.secret_key); + } + let store = builder + .build() + .map_err(|e| store_error(format!("S3 client: {e}")))?; + return Ok(Self { + uri: uri.to_string(), + store: BackupStore::Remote(Arc::new(store)), + path, + }); + } + Err(BackupUriError::UnsupportedScheme { + uri: uri.to_string(), + }) + } + + pub fn uri(&self) -> &str { + &self.uri + } + + /// The store the object lives in. + pub fn store(&self) -> &BackupStore { + &self.store + } + + /// The object key inside [`Self::store`]. + pub fn path(&self) -> &ObjectPath { + &self.path + } + + /// Write `bytes` as the whole object, replacing any object there. + pub async fn put(&self, bytes: Vec) -> Result<(), BackupIoError> { + self.store.put(&self.uri, &self.path, bytes).await + } + + /// Read the whole object. + pub async fn get(&self) -> Result, BackupIoError> { + self.store.get(&self.uri, &self.path).await + } +} + +/// The components of an object key below a local root. +fn components(path: &ObjectPath) -> Vec { + path.parts().map(|part| part.as_ref().to_string()).collect() +} + +/// Run the blocking local `io` off the async runtime, on a directory fd of +/// `root`. +async fn local_io( + root: &Path, + uri: &str, + action: &str, + io: impl FnOnce(&std::os::fd::OwnedFd) -> Result + Send + 'static, +) -> Result { + let opened = root.to_path_buf(); + let outcome = tokio::task::spawn_blocking(move || { + let fd = store_local::open_root(&opened)?; + io(&fd) + }) + .await; + let failed = |detail: String| { + BackupIoError::Failed(Error::Storage { + engine: "backup".into(), + detail: format!("{action} backup object '{uri}': {detail}"), + }) + }; + match outcome { + Ok(Ok(value)) => Ok(value), + Ok(Err(LocalIoError::Escapes)) => { + Err(BackupIoError::Refused(BackupUriError::OutsideLocalRoot { + uri: uri.to_string(), + root: root.display().to_string(), + })) + } + Ok(Err(LocalIoError::Io(e))) => Err(failed(e.to_string())), + Err(join) => Err(failed(join.to_string())), + } +} + +fn failed(action: &str, uri: &str, e: object_store::Error) -> BackupIoError { + BackupIoError::Failed(Error::Storage { + engine: "backup".into(), + detail: format!("{action} backup object '{uri}': {e}"), + }) +} + +/// Why a `file://` path does not resolve inside the local root. +#[derive(Debug)] +enum LocalRefusal { + /// The path, or a symlink on it, leaves the root. + Outside, + /// The root or an existing component cannot be read. + Unreadable(String), +} + +/// The canonical local root, and `requested` resolved under it with every +/// symlink followed. +/// +/// Refuses a `..` component. Canonicalizes the root, then walks `requested` +/// below it one component at a time: each component that exists is +/// canonicalized and must stay under the canonical root, so a symlink that +/// leaves the root refuses the path. Components past the first missing one +/// do not exist, so no symlink can sit on them. +fn local_target(root: &Path, requested: &Path) -> Result<(PathBuf, PathBuf), LocalRefusal> { + if requested + .components() + .any(|c| matches!(c, Component::ParentDir | Component::Prefix(_))) + { + return Err(LocalRefusal::Outside); + } + let unreadable = |path: &Path, e: std::io::Error| { + LocalRefusal::Unreadable(format!("'{}': {e}", path.display())) + }; + let canonical_root = std::fs::canonicalize(root).map_err(|e| unreadable(root, e))?; + let relative = requested + .strip_prefix(root) + .or_else(|_| requested.strip_prefix(&canonical_root)) + .map_err(|_| LocalRefusal::Outside)?; + let mut target = canonical_root.clone(); + let mut exists = true; + for component in relative.components() { + let Component::Normal(part) = component else { + continue; + }; + target.push(part); + if !exists { + continue; + } + match std::fs::symlink_metadata(&target) { + Ok(_) => { + target = std::fs::canonicalize(&target).map_err(|e| unreadable(&target, e))?; + if !target.starts_with(&canonical_root) { + return Err(LocalRefusal::Outside); + } + } + Err(e) if e.kind() == std::io::ErrorKind::NotFound => exists = false, + Err(e) => return Err(unreadable(&target, e)), + } + } + Ok((canonical_root, target)) +} + +/// The object key of a root-relative path, `None` for an empty one. +fn object_key(relative: &Path) -> Option { + let parts: Vec<&str> = relative + .components() + .filter_map(|c| match c { + Component::Normal(part) => part.to_str(), + _ => None, + }) + .collect(); + if parts.is_empty() { + return None; + } + ObjectPath::parse(parts.join("/")).ok() +} + +#[cfg(test)] +mod tests { + use super::*; + + fn settings(root: &Path) -> BackupStorageSettings { + BackupStorageSettings { + local_root: Some(root.to_path_buf()), + ..Default::default() + } + } + + #[test] + fn a_file_uri_inside_the_root_resolves() { + let dir = tempfile::tempdir().expect("tempdir"); + let uri = format!("file://{}/nightly/db.ndbb", dir.path().display()); + let object = BackupObject::resolve(&uri, Some(&settings(dir.path()))).expect("resolve"); + assert_eq!(object.path.as_ref(), "nightly/db.ndbb"); + } + + #[test] + fn a_file_uri_outside_the_root_is_refused() { + let dir = tempfile::tempdir().expect("tempdir"); + let s = settings(dir.path()); + for uri in [ + "file:///etc/passwd".to_string(), + format!("file://{}/../escape", dir.path().display()), + format!("file://{}", dir.path().display()), + ] { + assert!(BackupObject::resolve(&uri, Some(&s)).is_err(), "{uri}"); + } + } + + #[cfg(unix)] + #[test] + fn a_symlink_out_of_the_root_is_refused() { + let outside = tempfile::tempdir().expect("outside dir"); + let dir = tempfile::tempdir().expect("tempdir"); + std::os::unix::fs::symlink(outside.path(), dir.path().join("escape")) + .expect("symlink a directory out"); + std::fs::write(outside.path().join("file.ndbb"), b"x").expect("outside file"); + std::os::unix::fs::symlink( + outside.path().join("file.ndbb"), + dir.path().join("link.ndbb"), + ) + .expect("symlink a file out"); + let s = settings(dir.path()); + for name in ["escape/new.ndbb", "escape/deeper/new.ndbb", "link.ndbb"] { + let uri = format!("file://{}/{name}", dir.path().display()); + assert!( + matches!( + BackupObject::resolve(&uri, Some(&s)), + Err(BackupUriError::OutsideLocalRoot { .. }) + ), + "{uri} leaves the root through a symlink" + ); + } + } + + #[cfg(unix)] + #[test] + fn a_symlink_that_stays_in_the_root_resolves_to_its_target() { + let dir = tempfile::tempdir().expect("tempdir"); + std::fs::create_dir(dir.path().join("real")).expect("real dir"); + std::os::unix::fs::symlink(dir.path().join("real"), dir.path().join("alias")) + .expect("symlink inside the root"); + let uri = format!("file://{}/alias/db.ndbb", dir.path().display()); + let object = BackupObject::resolve(&uri, Some(&settings(dir.path()))).expect("resolve"); + assert_eq!(object.path.as_ref(), "real/db.ndbb"); + } + + #[test] + fn a_file_uri_without_a_root_is_refused() { + assert!(matches!( + BackupObject::resolve("file:///tmp/x", None), + Err(BackupUriError::NoLocalRoot { .. }) + )); + } + + #[test] + fn an_unknown_scheme_and_a_bare_bucket_are_refused() { + assert!(matches!( + BackupObject::resolve("ftp://host/x", None), + Err(BackupUriError::UnsupportedScheme { .. }) + )); + assert!(matches!( + BackupObject::resolve("s3://bucket", None), + Err(BackupUriError::Malformed { .. }) + )); + } + + /// A component swapped for an out-of-root symlink between the resolve and + /// the write refuses the write, and nothing lands outside the root. + #[cfg(unix)] + #[tokio::test] + async fn a_symlink_swapped_in_after_resolve_refuses_the_write_and_the_read() { + let outside = tempfile::tempdir().expect("outside dir"); + let dir = tempfile::tempdir().expect("tempdir"); + std::fs::create_dir(dir.path().join("nightly")).expect("in-root dir"); + let uri = format!("file://{}/nightly/db.ndbb", dir.path().display()); + let object = BackupObject::resolve(&uri, Some(&settings(dir.path()))).expect("resolve"); + + std::fs::remove_dir(dir.path().join("nightly")).expect("remove the dir"); + std::os::unix::fs::symlink(outside.path(), dir.path().join("nightly")) + .expect("swap in a symlink out of the root"); + + assert!( + matches!( + object.put(vec![1, 2, 3]).await, + Err(BackupIoError::Refused( + BackupUriError::OutsideLocalRoot { .. } + )) + ), + "the write follows the swapped-in symlink" + ); + assert!( + !outside.path().join("db.ndbb").exists(), + "the write landed outside the root" + ); + std::fs::write(outside.path().join("db.ndbb"), b"outside").expect("outside file"); + assert!( + matches!( + object.get().await, + Err(BackupIoError::Refused( + BackupUriError::OutsideLocalRoot { .. } + )) + ), + "the read follows the swapped-in symlink" + ); + } + + /// The final component swapped for a symlink refuses the read. + #[cfg(unix)] + #[tokio::test] + async fn a_file_swapped_for_a_symlink_after_resolve_refuses_the_read() { + let outside = tempfile::tempdir().expect("outside dir"); + std::fs::write(outside.path().join("secret"), b"outside").expect("outside file"); + let dir = tempfile::tempdir().expect("tempdir"); + let uri = format!("file://{}/db.ndbb", dir.path().display()); + let object = BackupObject::resolve(&uri, Some(&settings(dir.path()))).expect("resolve"); + std::os::unix::fs::symlink(outside.path().join("secret"), dir.path().join("db.ndbb")) + .expect("place a symlink at the object"); + assert!(matches!( + object.get().await, + Err(BackupIoError::Refused( + BackupUriError::OutsideLocalRoot { .. } + )) + )); + } + + #[tokio::test] + async fn an_object_round_trips_through_the_local_root() { + let dir = tempfile::tempdir().expect("tempdir"); + let uri = format!("file://{}/a/b.ndbb", dir.path().display()); + let object = BackupObject::resolve(&uri, Some(&settings(dir.path()))).expect("resolve"); + object.put(vec![1, 2, 3]).await.expect("put"); + assert_eq!(object.get().await.expect("get"), vec![1, 2, 3]); + } +} diff --git a/nodedb/src/control/backup/store_local.rs b/nodedb/src/control/backup/store_local.rs new file mode 100644 index 000000000..5c1f3b88a --- /dev/null +++ b/nodedb/src/control/backup/store_local.rs @@ -0,0 +1,379 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Reads and writes of a `file://` backup object through a directory fd of +//! the canonical local root. +//! +//! Every component opens relative to its parent's fd, one at a time, and no +//! open follows a symlink: on Linux through `openat2` with +//! `RESOLVE_BENEATH | RESOLVE_NO_SYMLINKS`, elsewhere through `openat` with +//! `O_NOFOLLOW`. A missing directory is created with `mkdirat` relative to +//! its parent's fd. A component swapped for a symlink after the URI resolved +//! therefore refuses the read or write: there is no gap between a check and +//! the open. +//! +//! A listing names only the regular files of one directory, never a symlink. +//! A deletion unlinks a regular file relative to its parent's fd and refuses +//! a symlink, so no deletion reaches a file outside the root. + +use std::ffi::{CStr, CString}; +use std::fs::File; +use std::io::{self, Read, Write}; +use std::os::fd::{AsFd, AsRawFd, BorrowedFd, FromRawFd, IntoRawFd, OwnedFd}; +use std::path::Path; + +/// Why a local read or write did not complete. +#[derive(Debug)] +pub(super) enum LocalIoError { + /// A component is a symlink, or the path leaves the root. + Escapes, + Io(io::Error), +} + +impl From for LocalIoError { + fn from(e: io::Error) -> Self { + Self::Io(e) + } +} + +/// Open the canonical root `path` as a directory fd. +pub(super) fn open_root(path: &Path) -> io::Result { + use std::os::unix::fs::OpenOptionsExt; + let file = std::fs::OpenOptions::new() + .read(true) + .custom_flags(libc::O_DIRECTORY | libc::O_CLOEXEC) + .open(path)?; + Ok(OwnedFd::from(file)) +} + +/// Write `bytes` as `components` below `root`: every component but the last +/// is a directory, created when missing. The file is written under a +/// temporary name, synced, then renamed over the final name. +pub(super) fn write_beneath( + root: &OwnedFd, + components: &[String], + bytes: &[u8], +) -> Result<(), LocalIoError> { + let (name, dirs) = components.split_last().ok_or(LocalIoError::Escapes)?; + let mut parent = root.try_clone()?; + for dir in dirs { + let dir = c_name(dir)?; + // SAFETY: `parent` is a live directory fd and `dir` a NUL-terminated + // single component. + let made = unsafe { libc::mkdirat(parent.as_raw_fd(), dir.as_ptr(), 0o750) }; + if made != 0 { + let error = io::Error::last_os_error(); + if error.raw_os_error() != Some(libc::EEXIST) { + return Err(error.into()); + } + } + parent = open_dir_at(parent.as_fd(), &dir)?; + } + let final_name = c_name(name)?; + let partial = c_name(&format!(".{name}.partial"))?; + let fd = open_at( + parent.as_fd(), + &partial, + libc::O_WRONLY | libc::O_CREAT | libc::O_TRUNC | libc::O_CLOEXEC, + 0o640, + )?; + let mut file = File::from(fd); + file.write_all(bytes)?; + file.sync_all()?; + // SAFETY: both names are NUL-terminated single components under the same + // live directory fd. `renameat` replaces the final name itself, never a + // symlink's target. + let renamed = unsafe { + libc::renameat( + parent.as_raw_fd(), + partial.as_ptr(), + parent.as_raw_fd(), + final_name.as_ptr(), + ) + }; + if renamed != 0 { + return Err(io::Error::last_os_error().into()); + } + File::from(parent).sync_all()?; + Ok(()) +} + +/// Read the whole file `components` below `root`. +pub(super) fn read_beneath(root: &OwnedFd, components: &[String]) -> Result, LocalIoError> { + let (name, dirs) = components.split_last().ok_or(LocalIoError::Escapes)?; + let mut parent = root.try_clone()?; + for dir in dirs { + parent = open_dir_at(parent.as_fd(), &c_name(dir)?)?; + } + let fd = open_at( + parent.as_fd(), + &c_name(name)?, + libc::O_RDONLY | libc::O_CLOEXEC, + 0, + )?; + let mut bytes = Vec::new(); + File::from(fd).read_to_end(&mut bytes)?; + Ok(bytes) +} + +/// The names of the regular files directly in the directory `components` +/// below `root`, sorted. A missing directory holds none. A symlink, a +/// directory and a name that is not UTF-8 are never listed. +pub(super) fn list_files( + root: &OwnedFd, + components: &[String], +) -> Result, LocalIoError> { + let Some(dir) = open_dirs(root, components)? else { + return Ok(Vec::new()); + }; + let mut files = Vec::new(); + for name in read_dir_names(dir.try_clone()?)? { + let Ok(c) = CString::new(name.as_str()) else { + continue; + }; + if file_type_at(dir.as_fd(), &c)? == Some(libc::S_IFREG) { + files.push(name); + } + } + files.sort(); + Ok(files) +} + +/// Delete the regular file `components` below `root`. A missing file counts +/// as deleted. A symlink is refused as [`LocalIoError::Escapes`]: the +/// deletion never reaches its target. +pub(super) fn delete_beneath(root: &OwnedFd, components: &[String]) -> Result<(), LocalIoError> { + let (name, dirs) = components.split_last().ok_or(LocalIoError::Escapes)?; + let Some(parent) = open_dirs(root, dirs)? else { + return Ok(()); + }; + let name = c_name(name)?; + match file_type_at(parent.as_fd(), &name)? { + None => return Ok(()), + Some(kind) if kind == libc::S_IFLNK => return Err(LocalIoError::Escapes), + Some(_) => {} + } + // SAFETY: `parent` is a live directory fd and `name` a NUL-terminated + // single component. `unlinkat` removes the entry itself and follows no + // symlink. + let unlinked = unsafe { libc::unlinkat(parent.as_raw_fd(), name.as_ptr(), 0) }; + if unlinked != 0 { + let error = io::Error::last_os_error(); + if error.raw_os_error() != Some(libc::ENOENT) { + return Err(error.into()); + } + } + Ok(()) +} + +/// Open the directories `components` below `root`, one at a time. `None` +/// when one is missing. +fn open_dirs(root: &OwnedFd, components: &[String]) -> Result, LocalIoError> { + let mut dir = root.try_clone()?; + for component in components { + dir = match open_dir_at(dir.as_fd(), &c_name(component)?) { + Ok(fd) => fd, + Err(LocalIoError::Io(error)) if error.kind() == io::ErrorKind::NotFound => { + return Ok(None); + } + Err(error) => return Err(error), + }; + } + Ok(Some(dir)) +} + +/// Every entry name of the directory `dir`, except `.` and `..`. +fn read_dir_names(dir: OwnedFd) -> io::Result> { + let raw = dir.into_raw_fd(); + // SAFETY: `raw` is an open directory fd this call owns. `fdopendir` takes + // it over, and `closedir` below closes it. + let stream = unsafe { libc::fdopendir(raw) }; + if stream.is_null() { + let error = io::Error::last_os_error(); + // SAFETY: `fdopendir` failed, so `raw` is still owned here. + unsafe { libc::close(raw) }; + return Err(error); + } + let mut names = Vec::new(); + let outcome = loop { + clear_errno(); + // SAFETY: `stream` is a live directory stream. + let entry = unsafe { libc::readdir(stream) }; + if entry.is_null() { + let error = io::Error::last_os_error(); + break match error.raw_os_error() { + Some(0) | None => Ok(()), + Some(_) => Err(error), + }; + } + // SAFETY: `readdir` returned a live entry whose `d_name` is + // NUL-terminated, valid until the next `readdir`. + let name = unsafe { CStr::from_ptr((*entry).d_name.as_ptr()) }; + if let Ok(name) = name.to_str() + && name != "." + && name != ".." + { + names.push(name.to_owned()); + } + }; + // SAFETY: `stream` is live and closed once. + unsafe { libc::closedir(stream) }; + outcome.map(|()| names) +} + +#[cfg(any(target_os = "linux", target_os = "android"))] +fn clear_errno() { + // SAFETY: the thread's errno slot is always valid to write. + unsafe { *libc::__errno_location() = 0 }; +} + +#[cfg(any(target_os = "macos", target_os = "ios", target_os = "freebsd"))] +fn clear_errno() { + // SAFETY: the thread's errno slot is always valid to write. + unsafe { *libc::__error() = 0 }; +} + +#[cfg(any(target_os = "netbsd", target_os = "openbsd"))] +fn clear_errno() { + // SAFETY: the thread's errno slot is always valid to write. + unsafe { *libc::__errno() = 0 }; +} + +fn c_name(name: &str) -> Result { + if name.is_empty() || name == "." || name == ".." || name.contains('/') { + return Err(LocalIoError::Escapes); + } + CString::new(name).map_err(|_| LocalIoError::Escapes) +} + +fn open_dir_at(parent: BorrowedFd<'_>, name: &CString) -> Result { + open_at( + parent, + name, + libc::O_RDONLY | libc::O_DIRECTORY | libc::O_CLOEXEC, + 0, + ) +} + +/// Open the single component `name` below `parent` without following a +/// symlink. A symlink, or a resolution that leaves `parent`, is +/// [`LocalIoError::Escapes`]. +fn open_at( + parent: BorrowedFd<'_>, + name: &CString, + flags: libc::c_int, + mode: libc::mode_t, +) -> Result { + match open_no_symlinks(parent, name, flags, mode) { + Ok(fd) => Ok(fd), + Err(error) => Err(classify(parent, name, error)), + } +} + +#[cfg(target_os = "linux")] +fn open_no_symlinks( + parent: BorrowedFd<'_>, + name: &CString, + flags: libc::c_int, + mode: libc::mode_t, +) -> io::Result { + // SAFETY: `open_how` is plain data; zero is a valid value of every field. + let mut how: libc::open_how = unsafe { std::mem::zeroed() }; + how.flags = flags as u64; + how.mode = u64::from(mode); + how.resolve = libc::RESOLVE_BENEATH | libc::RESOLVE_NO_SYMLINKS; + // SAFETY: `parent` is a live directory fd, `name` is NUL-terminated, and + // `how` lives for the call with its exact size passed. + let fd = unsafe { + libc::syscall( + libc::SYS_openat2, + parent.as_raw_fd(), + name.as_ptr(), + &how as *const libc::open_how, + std::mem::size_of::(), + ) + }; + if fd >= 0 { + // SAFETY: the kernel returned a new fd this call owns. + return Ok(unsafe { OwnedFd::from_raw_fd(fd as libc::c_int) }); + } + let error = io::Error::last_os_error(); + // A kernel without `openat2` still refuses a symlink through `O_NOFOLLOW` + // on this single component. + if error.raw_os_error() == Some(libc::ENOSYS) { + return openat_nofollow(parent, name, flags, mode); + } + Err(error) +} + +#[cfg(not(target_os = "linux"))] +fn open_no_symlinks( + parent: BorrowedFd<'_>, + name: &CString, + flags: libc::c_int, + mode: libc::mode_t, +) -> io::Result { + openat_nofollow(parent, name, flags, mode) +} + +fn openat_nofollow( + parent: BorrowedFd<'_>, + name: &CString, + flags: libc::c_int, + mode: libc::mode_t, +) -> io::Result { + // SAFETY: `parent` is a live directory fd and `name` a NUL-terminated + // single component. + let fd = unsafe { + libc::openat( + parent.as_raw_fd(), + name.as_ptr(), + flags | libc::O_NOFOLLOW, + libc::c_uint::from(mode), + ) + }; + if fd < 0 { + return Err(io::Error::last_os_error()); + } + // SAFETY: `openat` returned a new fd this call owns. + Ok(unsafe { OwnedFd::from_raw_fd(fd) }) +} + +/// A symlink or an escape is [`LocalIoError::Escapes`]. `ENOTDIR` is one only +/// when the component is a symlink: a regular file in a directory's place is +/// an I/O error. +fn classify(parent: BorrowedFd<'_>, name: &CString, error: io::Error) -> LocalIoError { + match error.raw_os_error() { + Some(libc::ELOOP) | Some(libc::EXDEV) => LocalIoError::Escapes, + Some(libc::ENOTDIR) if is_symlink_at(parent, name) => LocalIoError::Escapes, + _ => LocalIoError::Io(error), + } +} + +fn is_symlink_at(parent: BorrowedFd<'_>, name: &CString) -> bool { + matches!(file_type_at(parent, name), Ok(Some(kind)) if kind == libc::S_IFLNK) +} + +/// The file type bits of `name` below `parent`, without following a symlink. +/// `None` when it is missing. +fn file_type_at(parent: BorrowedFd<'_>, name: &CString) -> io::Result> { + // SAFETY: `stat` is plain data, filled by the call. + let mut stat: libc::stat = unsafe { std::mem::zeroed() }; + // SAFETY: `parent` is a live directory fd, `name` is NUL-terminated, and + // `stat` outlives the call. + let found = unsafe { + libc::fstatat( + parent.as_raw_fd(), + name.as_ptr(), + &mut stat, + libc::AT_SYMLINK_NOFOLLOW, + ) + }; + if found != 0 { + let error = io::Error::last_os_error(); + return match error.raw_os_error() { + Some(libc::ENOENT) => Ok(None), + _ => Err(error), + }; + } + Ok(Some(stat.st_mode & libc::S_IFMT)) +} diff --git a/nodedb/src/control/backup/verify/canonical.rs b/nodedb/src/control/backup/verify/canonical.rs new file mode 100644 index 000000000..c9f78bf6c --- /dev/null +++ b/nodedb/src/control/backup/verify/canonical.rs @@ -0,0 +1,194 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Canonical row hashing for backup verification. +//! +//! A row hashes as its engine part, its canonical key and its canonical value. +//! Every item is written with a type tag and a length prefix, so two different +//! rows never feed the hasher the same bytes. Map entries hash in key order, so +//! a map's iteration order never reaches the digest. + +use loro::LoroValue; +use nodedb_types::Value; +use nodedb_types::backup_envelope::VerifiedPart; +use sha2::{Digest, Sha256}; + +use crate::Error; + +/// The hash of a row's canonical key. Every version of one row shares it, so +/// the restore can match a destination row to a backed-up one. +pub(crate) type KeyHash = u128; + +/// One canonical row of one collection part. +pub(crate) struct Row { + /// Bare catalog name of the collection. + pub collection: String, + pub part: VerifiedPart, + /// `None` for a part whose rows have no key: timeseries and columnar. + pub key: Option, + pub hash: [u8; 32], + /// The row's TTL deadline in Unix milliseconds, `0` for none. + pub expire_at_ms: u64, +} + +/// The hash of `key`, the canonical key parts of a row of `part`. +pub(crate) fn key_hash(part: VerifiedPart, key: &[&[u8]]) -> KeyHash { + let mut hasher = RowHasher::start(b"nodedb-verify-key", part); + for item in key { + hasher.bytes(b'k', item); + } + let digest = hasher.finish(); + let mut head = [0u8; 16]; + head.copy_from_slice(&digest[..16]); + u128::from_le_bytes(head) +} + +/// Streams one row's canonical key and value into SHA-256. +pub(crate) struct RowHasher(Sha256); + +impl RowHasher { + /// A row of `part` keyed by `key`. + pub fn new(part: VerifiedPart, key: &[&[u8]]) -> Self { + let mut hasher = Self::start(b"nodedb-verify-row", part); + hasher.count(b'K', key.len()); + for item in key { + hasher.bytes(b'k', item); + } + hasher + } + + fn start(domain: &[u8], part: VerifiedPart) -> Self { + let mut sha = Sha256::new(); + sha.update(domain); + sha.update([part.tag()]); + Self(sha) + } + + pub fn bytes(&mut self, tag: u8, bytes: &[u8]) { + self.0.update([tag]); + self.0.update((bytes.len() as u64).to_le_bytes()); + self.0.update(bytes); + } + + pub fn int(&mut self, tag: u8, value: i64) { + self.bytes(tag, &value.to_le_bytes()); + } + + fn count(&mut self, tag: u8, len: usize) { + self.0.update([tag]); + self.0.update((len as u64).to_le_bytes()); + } + + /// A native value. Objects hash their fields in name order. + pub fn value(&mut self, value: &Value) -> Result<(), Error> { + match value { + Value::Object(map) => { + let mut fields: Vec<(&String, &Value)> = map.iter().collect(); + fields.sort_by(|a, b| a.0.cmp(b.0)); + self.count(b'O', fields.len()); + for (name, field) in fields { + self.bytes(b's', name.as_bytes()); + self.value(field)?; + } + } + Value::Array(items) => { + self.count(b'A', items.len()); + for item in items { + self.value(item)?; + } + } + Value::Set(items) => { + self.count(b'S', items.len()); + for item in items { + self.value(item)?; + } + } + leaf => { + let bytes = + nodedb_types::value_to_msgpack(leaf).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("backup verification: encode a row value: {e}"), + })?; + self.bytes(b'L', &bytes); + } + } + Ok(()) + } + + /// A CRDT value. Maps hash their entries in key order. + pub fn loro(&mut self, value: &LoroValue) { + match value { + LoroValue::Null => self.bytes(b'n', &[]), + LoroValue::Bool(b) => self.bytes(b'b', &[u8::from(*b)]), + LoroValue::Double(d) => self.bytes(b'd', &d.to_bits().to_le_bytes()), + LoroValue::I64(i) => self.int(b'i', *i), + LoroValue::Binary(bytes) => self.bytes(b'y', bytes.as_slice()), + LoroValue::String(s) => self.bytes(b's', s.as_bytes()), + LoroValue::List(items) => { + self.count(b'A', items.len()); + for item in items.iter() { + self.loro(item); + } + } + LoroValue::Map(map) => { + let mut entries: Vec<(&String, &LoroValue)> = map.iter().collect(); + entries.sort_by(|a, b| a.0.cmp(b.0)); + self.count(b'O', entries.len()); + for (name, entry) in entries { + self.bytes(b's', name.as_bytes()); + self.loro(entry); + } + } + LoroValue::Container(id) => self.bytes(b'c', id.to_string().as_bytes()), + } + } + + pub fn finish(self) -> [u8; 32] { + self.0.finalize().into() + } +} + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + + use super::*; + + fn object(fields: &[(&str, Value)]) -> Value { + Value::Object( + fields + .iter() + .map(|(k, v)| ((*k).to_string(), v.clone())) + .collect::>(), + ) + } + + fn hash(value: &Value) -> [u8; 32] { + let mut hasher = RowHasher::new(VerifiedPart::Documents, &[b"k1"]); + hasher.value(value).expect("hash"); + hasher.finish() + } + + #[test] + fn field_order_does_not_reach_the_hash() { + let a = object(&[("x", Value::Integer(1)), ("y", Value::String("s".into()))]); + let b = object(&[("y", Value::String("s".into())), ("x", Value::Integer(1))]); + assert_eq!(hash(&a), hash(&b)); + let c = object(&[("x", Value::Integer(2)), ("y", Value::String("s".into()))]); + assert_ne!(hash(&a), hash(&c)); + } + + #[test] + fn the_key_and_the_part_reach_the_hash() { + let value = Value::Integer(1); + let mut other_key = RowHasher::new(VerifiedPart::Documents, &[b"k2"]); + other_key.value(&value).expect("hash"); + assert_ne!(hash(&value), other_key.finish()); + let mut other_part = RowHasher::new(VerifiedPart::KeyValue, &[b"k1"]); + other_part.value(&value).expect("hash"); + assert_ne!(hash(&value), other_part.finish()); + assert_ne!( + key_hash(VerifiedPart::Documents, &[b"ab", b"c"]), + key_hash(VerifiedPart::Documents, &[b"a", b"bc"]) + ); + } +} diff --git a/nodedb/src/control/backup/verify/destination.rs b/nodedb/src/control/backup/verify/destination.rs new file mode 100644 index 000000000..10166e9e1 --- /dev/null +++ b/nodedb/src/control/backup/verify/destination.rs @@ -0,0 +1,283 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The destination side of restore verification. +//! +//! After the last re-issue, the restore captures every restored collection +//! from the whole cluster, with the capture a backup takes, and recomputes each +//! collection part's tally. A keyed part counts only the destination rows whose +//! key the envelope holds: a row the destination held before the restore, under +//! another key, is not the restore's. A keyless part, timeseries or columnar, +//! counts every row: its re-issue replaces the collection's contents. +//! +//! A KV row whose TTL has passed by the time of the capture counts on neither +//! side: the restore skips it, and the destination can drop it. + +use std::collections::{BTreeMap, HashSet}; + +use nodedb_types::backup_envelope::{VerificationMismatch, VerificationPhase}; + +use crate::Error; +use crate::control::backup::capture::capture_collections; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantDataSnapshot}; +use nodedb_wal::crypto::WalEncryptionKey; + +use super::expect::{DatabaseExpectation, Expectation, compare}; +use super::walk::{BindIndex, Tallies, Walk}; + +/// What the destination check verified. +#[derive(Debug, Default, Clone, PartialEq, Eq)] +pub(crate) struct Verified { + /// Collection parts whose tally matched. + pub collections: usize, + /// Rows those parts hold. + pub rows: u64, + /// Rows per collection, keyed by (destination database id, collection). + pub per_collection: BTreeMap<(u64, String), u64>, +} + +/// Check that the destination holds every backed-up row. `dest_of` maps a +/// source database id to its destination database. +/// +/// A mismatch fails with [`Error::RestoreVerificationFailed`] in `phase`, +/// naming every mismatched collection. Nothing is rolled back. +pub(crate) async fn verify_destination( + state: &SharedState, + tenant_id: u64, + expectation: Expectation, + dest_of: impl Fn(u64) -> Result, + phase: VerificationPhase, +) -> Result { + let mut verified = Verified::default(); + let mut mismatches = Vec::new(); + for (source, database) in expectation.databases { + let collections = database.collections(); + if collections.is_empty() { + continue; + } + let dest = dest_of(source)?; + let arrays = database + .rows + .tallies + .keys() + .any(|(_, part)| *part == nodedb_types::backup_envelope::VerifiedPart::Array); + let snap = capture_collections(state, tenant_id, dest, &collections, arrays).await?; + // Taken after the capture: a row whose TTL passed during it counts on + // neither side. + let cutoff_ms = now_ms(); + let scope = Scope { + tenant_id, + database_id: dest, + kek: state.wal.encryption_key(), + }; + let found = compare_destination(scope, database, &snap, cutoff_ms)?; + verified.collections += found.verified.collections; + verified.rows += found.verified.rows; + for (key, rows) in found.verified.per_collection { + *verified.per_collection.entry(key).or_default() += rows; + } + mismatches.extend(found.mismatches); + } + if mismatches.is_empty() { + Ok(verified) + } else { + Err(Error::RestoreVerificationFailed { phase, mismatches }) + } +} + +struct Compared { + verified: Verified, + mismatches: Vec, +} + +/// The destination database a capture came from. +#[derive(Clone, Copy)] +struct Scope<'a> { + tenant_id: u64, + database_id: DatabaseId, + kek: Option<&'a WalEncryptionKey>, +} + +/// Compare `snap`, captured from the destination database of `scope`, with +/// `expected`. The capture carries the destination's own binds. +fn compare_destination( + scope: Scope<'_>, + mut expected: DatabaseExpectation, + snap: &TenantDataSnapshot, + cutoff_ms: u64, +) -> Result { + let rows = &mut expected.rows; + for row in &rows.expiring { + if row.expire_at_ms > cutoff_ms { + continue; + } + if let Some(tally) = rows.tallies.get_mut(&row.part) { + tally.remove(&row.hash); + } + if let Some(keys) = rows.keys.get_mut(&row.part) { + keys.remove(&row.key); + } + } + rows.tallies.retain(|_, tally| tally.count != 0); + let rows = &expected.rows; + + let binds = BindIndex::new( + snap.surrogate_pk + .iter() + .filter(|b| b.tenant_id == scope.tenant_id) + .map(|b| (b.collection.as_str(), b.surrogate, b.pk.as_slice())), + ); + let walk = Walk { + tenant_id: scope.tenant_id, + database_id: scope.database_id, + shapes: &expected.shapes, + binds: &binds, + kek: scope.kek, + }; + let no_keys = HashSet::new(); + let mut found = Tallies::new(); + walk.snapshot(snap, &mut |row| { + let part = (row.collection, row.part); + if !rows.tallies.contains_key(&part) { + return; + } + if let Some(key) = row.key + && !rows.keys.get(&part).unwrap_or(&no_keys).contains(&key) + { + return; + } + found.entry(part).or_default().add(&row.hash); + })?; + + let mismatches = compare(scope.database_id.as_u64(), &rows.tallies, &found); + let mut per_collection = BTreeMap::new(); + for ((collection, _), tally) in &rows.tallies { + *per_collection + .entry((scope.database_id.as_u64(), collection.clone())) + .or_default() += tally.count; + } + let verified = Verified { + collections: rows.tallies.len().saturating_sub(mismatches.len()), + rows: rows.tallies.values().map(|t| t.count).sum(), + per_collection, + }; + Ok(Compared { + verified, + mismatches, + }) +} + +fn now_ms() -> u64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_millis() as u64) + .unwrap_or(0) +} + +#[cfg(test)] +mod tests { + use nodedb_types::backup_envelope::VerifiedPart; + + use super::*; + use crate::control::backup::verify::expect::expect_rows; + use crate::control::backup::verify::walk::Shapes; + + const TENANT: u64 = 7; + + fn kv_snapshot(db: u64, rows: &[(&str, &str, u64)]) -> TenantDataSnapshot { + let rows: Vec = rows + .iter() + .zip(1u32..) + .map(|((k, v, at), surrogate)| { + (k.as_bytes().to_vec(), v.as_bytes().to_vec(), *at, surrogate) + }) + .collect(); + TenantDataSnapshot { + kv_tables: vec![( + format!("{db}:{TENANT}:sessions"), + zerompk::to_msgpack_vec(&rows).expect("encode"), + )], + ..Default::default() + } + } + + fn scope() -> Scope<'static> { + Scope { + tenant_id: TENANT, + database_id: DatabaseId::DEFAULT, + kek: None, + } + } + + /// The envelope-side expectation of `snap`. + fn expect(snap: &TenantDataSnapshot) -> DatabaseExpectation { + let (shapes, binds) = (Shapes::new(), BindIndex::default()); + let walk = Walk { + tenant_id: TENANT, + database_id: DatabaseId::DEFAULT, + shapes: &shapes, + binds: &binds, + kek: None, + }; + DatabaseExpectation { + shapes: Shapes::new(), + rows: expect_rows(&walk, snap).expect("walk"), + } + } + + fn check(expected: DatabaseExpectation, found: &TenantDataSnapshot, cutoff: u64) -> Compared { + compare_destination(scope(), expected, found, cutoff).expect("compare") + } + + #[test] + fn a_faithful_destination_verifies() { + let backup = kv_snapshot(0, &[("a", "1", 0), ("b", "2", 0)]); + let dest = kv_snapshot(0, &[("b", "2", 0), ("a", "1", 0)]); + let compared = check(expect(&backup), &dest, 1_000); + assert!(compared.mismatches.is_empty()); + assert_eq!( + compared.verified, + Verified { + collections: 1, + rows: 2, + per_collection: BTreeMap::from([((0, "sessions".to_string()), 2)]), + } + ); + } + + #[test] + fn a_deleted_row_fails_and_names_the_collection() { + let backup = kv_snapshot(0, &[("a", "1", 0), ("b", "2", 0)]); + let dest = kv_snapshot(0, &[("a", "1", 0)]); + let compared = check(expect(&backup), &dest, 1_000); + assert_eq!(compared.mismatches.len(), 1); + let mismatch = &compared.mismatches[0]; + assert_eq!(mismatch.collection, "sessions"); + assert_eq!(mismatch.part, VerifiedPart::KeyValue); + assert_eq!((mismatch.expected.count, mismatch.found.count), (2, 1)); + } + + #[test] + fn a_changed_value_fails() { + let backup = kv_snapshot(0, &[("a", "1", 0)]); + let dest = kv_snapshot(0, &[("a", "2", 0)]); + assert_eq!(check(expect(&backup), &dest, 1_000).mismatches.len(), 1); + } + + #[test] + fn a_row_under_another_key_is_not_the_restores() { + let backup = kv_snapshot(0, &[("a", "1", 0)]); + let dest = kv_snapshot(0, &[("a", "1", 0), ("older", "x", 0)]); + assert!(check(expect(&backup), &dest, 1_000).mismatches.is_empty()); + } + + #[test] + fn an_expired_row_counts_on_neither_side() { + let backup = kv_snapshot(0, &[("a", "1", 0), ("gone", "x", 500)]); + let dest = kv_snapshot(0, &[("a", "1", 0)]); + assert!(check(expect(&backup), &dest, 1_000).mismatches.is_empty()); + // Before the deadline the row must be there. + let compared = check(expect(&backup), &dest, 100); + assert_eq!(compared.mismatches.len(), 1); + } +} diff --git a/nodedb/src/control/backup/verify/expect.rs b/nodedb/src/control/backup/verify/expect.rs new file mode 100644 index 000000000..570e8ebbb --- /dev/null +++ b/nodedb/src/control/backup/verify/expect.rs @@ -0,0 +1,275 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The envelope side of restore verification. +//! +//! Before its first write, a restore recomputes every collection part's tally +//! from the envelope's own rows and compares it with the tally the backup +//! recorded. A mismatch refuses the envelope: nothing is written. The +//! recomputation also keeps what the destination check needs: the expected +//! tallies, the key of every keyed row, and every row with a TTL. + +use std::collections::{BTreeMap, BTreeSet, HashMap, HashSet}; + +use nodedb_types::backup_envelope::{ + CollectionVerification, Envelope, SECTION_ORIGIN_CATALOG_ROWS, SECTION_ORIGIN_VERIFICATION, + StoredCollectionBlob, VerificationMismatch, VerificationPhase, +}; + +use crate::Error; +use crate::control::security::catalog::StoredCollection; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantDataSnapshot}; + +use super::canonical::KeyHash; +use super::walk::{BindIndex, DocShape, PartKey, Shapes, Tallies, Walk}; + +/// A backed-up row with a TTL. +pub(crate) struct ExpiringRow { + pub part: PartKey, + pub key: KeyHash, + pub hash: [u8; 32], + pub expire_at_ms: u64, +} + +/// The rows the destination must hold for one backed-up database. +#[derive(Default)] +pub(crate) struct ExpectedRows { + pub tallies: Tallies, + /// The key of every row of each keyed part. + pub keys: HashMap>, + pub expiring: Vec, +} + +/// Walk `snap` into the tallies, keys and TTL rows the destination must hold. +pub(crate) fn expect_rows( + walk: &Walk<'_>, + snap: &TenantDataSnapshot, +) -> Result { + let mut rows = ExpectedRows::default(); + walk.snapshot(snap, &mut |row| { + let part: PartKey = (row.collection, row.part); + rows.tallies.entry(part.clone()).or_default().add(&row.hash); + if let Some(key) = row.key { + rows.keys.entry(part.clone()).or_default().insert(key); + if row.expire_at_ms != 0 { + rows.expiring.push(ExpiringRow { + part, + key, + hash: row.hash, + expire_at_ms: row.expire_at_ms, + }); + } + } + })?; + Ok(rows) +} + +/// What the destination must hold for one backed-up database. +#[derive(Default)] +pub(crate) struct DatabaseExpectation { + pub shapes: Shapes, + pub rows: ExpectedRows, +} + +impl DatabaseExpectation { + /// Every collection the database's rows belong to. + pub fn collections(&self) -> BTreeSet { + self.rows + .tallies + .keys() + .map(|(name, _)| name.clone()) + .collect() + } +} + +/// What the destination must hold, by source database id. +#[derive(Default)] +pub(crate) struct Expectation { + pub databases: BTreeMap, +} + +/// Recompute every tally of `env` from `merged`, its data sections merged per +/// source database, and compare each with the tally the envelope records. +pub(crate) fn expect_envelope( + state: &SharedState, + tenant_id: u64, + env: &Envelope, + merged: &BTreeMap, +) -> Result { + let recorded = recorded_tallies(env)?; + let mut shapes = envelope_shapes(env)?; + let kek = state.wal.encryption_key(); + + let mut expectation = Expectation::default(); + for (source, snap) in merged { + let binds = BindIndex::new( + snap.surrogate_pk + .iter() + .filter(|b| b.tenant_id == tenant_id) + .map(|b| (b.collection.as_str(), b.surrogate, b.pk.as_slice())), + ); + let mut database = DatabaseExpectation { + shapes: shapes.remove(source).unwrap_or_default(), + ..Default::default() + }; + let walk = Walk { + tenant_id, + database_id: DatabaseId::new(*source), + shapes: &database.shapes, + binds: &binds, + kek, + }; + database.rows = expect_rows(&walk, snap)?; + expectation.databases.insert(*source, database); + } + + let mut mismatches = Vec::new(); + let sources: BTreeSet = recorded + .keys() + .copied() + .chain(expectation.databases.keys().copied()) + .collect(); + let empty = Tallies::new(); + for source in sources { + let found = expectation + .databases + .get(&source) + .map_or(&empty, |d| &d.rows.tallies); + mismatches.extend(compare( + source, + recorded.get(&source).unwrap_or(&empty), + found, + )); + } + if mismatches.is_empty() { + Ok(expectation) + } else { + Err(Error::RestoreVerificationFailed { + phase: VerificationPhase::Envelope, + mismatches, + }) + } +} + +/// What the destination must hold once `snap` is re-issued: `snap` is the +/// capture of `tenant_id`'s `collections` in `database_id`. MOVE TENANT +/// checks its target with it. +pub(crate) fn expect_capture( + state: &SharedState, + tenant_id: u64, + database_id: DatabaseId, + collections: &[StoredCollection], + snap: &TenantDataSnapshot, +) -> Result { + let binds = BindIndex::new( + snap.surrogate_pk + .iter() + .filter(|b| b.tenant_id == tenant_id) + .map(|b| (b.collection.as_str(), b.surrogate, b.pk.as_slice())), + ); + let mut database = DatabaseExpectation { + shapes: collections + .iter() + .filter(|coll| coll.tenant_id == tenant_id) + .map(|coll| (coll.name.clone(), DocShape::of(coll))) + .collect(), + ..Default::default() + }; + let walk = Walk { + tenant_id, + database_id, + shapes: &database.shapes, + binds: &binds, + kek: state.wal.encryption_key(), + }; + database.rows = expect_rows(&walk, snap)?; + let mut expectation = Expectation::default(); + expectation.databases.insert(database_id.as_u64(), database); + Ok(expectation) +} + +/// Every collection part whose tally in `found` differs from `expected`. +pub(crate) fn compare( + database_id: u64, + expected: &Tallies, + found: &Tallies, +) -> Vec { + let parts: BTreeSet<&PartKey> = expected.keys().chain(found.keys()).collect(); + parts + .into_iter() + .filter_map(|part| { + let expected = expected.get(part).copied().unwrap_or_default(); + let found = found.get(part).copied().unwrap_or_default(); + (expected != found).then(|| VerificationMismatch { + database_id, + collection: part.0.clone(), + part: part.1, + expected, + found, + }) + }) + .collect() +} + +/// The tallies the envelope's one verification section records, by source +/// database id. +fn recorded_tallies(env: &Envelope) -> Result, Error> { + let mut sections = env + .sections + .iter() + .filter(|s| s.origin_node_id == SECTION_ORIGIN_VERIFICATION); + let (Some(section), None) = (sections.next(), sections.next()) else { + return Err(Error::Internal { + detail: "invalid backup format: the backup must carry exactly one verification \ + section" + .into(), + }); + }; + let records: Vec = + zerompk::from_msgpack(§ion.body).map_err(|_| Error::Internal { + detail: "invalid backup format: verification section is not decodable".into(), + })?; + let mut recorded: BTreeMap = BTreeMap::new(); + for record in records { + let database = recorded.entry(record.database_id).or_default(); + if database + .insert((record.collection, record.part), record.tally) + .is_some() + { + return Err(Error::Internal { + detail: "invalid backup format: verification section repeats a collection".into(), + }); + } + } + Ok(recorded) +} + +/// The document shape of every collection in the envelope's catalog rows, by +/// source database id. +fn envelope_shapes(env: &Envelope) -> Result, Error> { + let mut shapes: HashMap = HashMap::new(); + for section in env + .sections + .iter() + .filter(|s| s.origin_node_id == SECTION_ORIGIN_CATALOG_ROWS) + { + let blobs: Vec = + zerompk::from_msgpack(§ion.body).map_err(|_| Error::Internal { + detail: "invalid backup format: catalog-rows section is not decodable".into(), + })?; + for blob in blobs { + let coll: StoredCollection = + zerompk::from_msgpack(&blob.bytes).map_err(|_| Error::Internal { + detail: format!( + "invalid backup format: catalog row of '{}' is not decodable", + blob.name + ), + })?; + shapes + .entry(blob.database_id) + .or_default() + .insert(coll.name.clone(), DocShape::of(&coll)); + } + } + Ok(shapes) +} diff --git a/nodedb/src/control/backup/verify/mod.rs b/nodedb/src/control/backup/verify/mod.rs new file mode 100644 index 000000000..a252c0039 --- /dev/null +++ b/nodedb/src/control/backup/verify/mod.rs @@ -0,0 +1,10 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod canonical; +pub mod destination; +pub mod expect; +pub mod record; +pub mod walk; +mod walk_arrays; +mod walk_engines; +mod walk_rows; diff --git a/nodedb/src/control/backup/verify/record.rs b/nodedb/src/control/backup/verify/record.rs new file mode 100644 index 000000000..0f1d1617e --- /dev/null +++ b/nodedb/src/control/backup/verify/record.rs @@ -0,0 +1,99 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The verification section a backup records. +//! +//! Each data section is one node's disjoint share of one database, so each +//! collection part's tally is the sum of its sections' tallies. + +use std::collections::{BTreeMap, HashMap}; + +use nodedb_types::backup_envelope::{ + CollectionVerification, DatabaseDataSection, SurrogateBindBlob, +}; + +use crate::Error; +use crate::control::backup::metadata::TenantDatabase; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantDataSnapshot}; + +use super::walk::{BindIndex, DocShape, Shapes, Tallies, Walk}; + +/// Encode the verification section of a backup of `tenant_id`. +/// +/// `data_sections` are the `(origin_node_id, body)` data sections, each body +/// an encoded [`DatabaseDataSection`]. `binds` are the backup's primary-key +/// binds. +pub(in crate::control::backup) fn verification_section( + state: &SharedState, + tenant_id: u64, + databases: &[TenantDatabase], + data_sections: &[(u64, Vec)], + binds: &[SurrogateBindBlob], +) -> Result, Error> { + let kek = state.wal.encryption_key(); + let shapes: HashMap = databases + .iter() + .map(|database| { + let shapes = database + .collections + .iter() + .map(|coll| (coll.name.clone(), DocShape::of(coll))) + .collect(); + (database.id().as_u64(), shapes) + }) + .collect(); + let mut bind_index: HashMap = HashMap::new(); + for database in databases { + let id = database.id().as_u64(); + let own = binds + .iter() + .filter(|b| b.database_id == id && b.tenant_id == tenant_id) + .map(|b| (b.collection.as_str(), b.surrogate, b.pk.as_slice())); + bind_index.insert(id, BindIndex::new(own)); + } + + let no_shapes = Shapes::new(); + let no_binds = BindIndex::default(); + let mut tallies: BTreeMap = BTreeMap::new(); + for (_node_id, body) in data_sections { + let section: DatabaseDataSection = zerompk::from_msgpack(body).map_err(decode)?; + let snap: TenantDataSnapshot = zerompk::from_msgpack(§ion.snapshot).map_err(decode)?; + let database_id = section.database_id; + let walk = Walk { + tenant_id, + database_id: DatabaseId::new(database_id), + shapes: shapes.get(&database_id).unwrap_or(&no_shapes), + binds: bind_index.get(&database_id).unwrap_or(&no_binds), + kek, + }; + let database = tallies.entry(database_id).or_default(); + walk.snapshot(&snap, &mut |row| { + database + .entry((row.collection, row.part)) + .or_default() + .add(&row.hash); + })?; + } + + let records: Vec = tallies + .into_iter() + .flat_map(|(database_id, parts)| { + parts + .into_iter() + .map(move |((collection, part), tally)| CollectionVerification { + database_id, + collection, + part, + tally, + }) + }) + .collect(); + super::super::metadata::encode_section_part("verification", &records) +} + +fn decode(e: impl std::fmt::Display) -> Error { + Error::Serialization { + format: "msgpack".into(), + detail: format!("backup verification: decode a data section: {e}"), + } +} diff --git a/nodedb/src/control/backup/verify/walk.rs b/nodedb/src/control/backup/verify/walk.rs new file mode 100644 index 000000000..e7fb117ed --- /dev/null +++ b/nodedb/src/control/backup/verify/walk.rs @@ -0,0 +1,263 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Walk a `TenantDataSnapshot` of one database as canonical rows. +//! +//! The backup, the envelope check and the destination check all walk their +//! snapshot with this one walker, so the three digests agree whenever the rows +//! do. A canonical row drops everything a restore re-derives: +//! +//! * a document is keyed by its client identity, never by its surrogate; +//! * a document version keeps its valid time but drops its system time; +//! * a hash-chained document drops its chain fields; +//! * an edge is keyed by its endpoints and label, without its system time or +//! valid time; +//! * a vector is keyed by the primary key its surrogate binds; +//! * a timeseries row drops its server-stamped system column, and a columnar +//! row drops its surrogate; +//! * a KV row drops its TTL deadline from the hash; +//! * an array cell version drops its surrogate. + +use std::collections::{BTreeMap, HashMap}; + +use nodedb_types::columnar::StrictSchema; +use nodedb_types::{CollectionType, DocumentMode}; +use nodedb_wal::crypto::WalEncryptionKey; + +use crate::Error; +use crate::control::backup::snapshot_keys::stored_collection_key; +use crate::control::security::catalog::StoredCollection; +use crate::types::{DatabaseId, TenantDataSnapshot}; + +use super::canonical::Row; + +/// What canonicalising a collection's documents reads from its catalog entry. +#[derive(Debug, Clone, Default)] +pub(crate) struct DocShape { + pub strict: Option, + pub declared_primary_key: Option, + pub hash_chain: bool, +} + +impl DocShape { + /// The storage mode the Data Plane registers for `coll`, so a row decodes + /// with the schema it was encoded with. + pub fn of(coll: &StoredCollection) -> Self { + let strict = match &coll.collection_type { + CollectionType::Document(DocumentMode::Strict(schema)) => Some(schema.clone()), + CollectionType::KeyValue(config) => Some(config.schema.clone()), + CollectionType::Document(DocumentMode::Schemaless) | CollectionType::Columnar(_) => { + None + } + }; + Self { + strict, + declared_primary_key: coll.declared_primary_key.clone(), + hash_chain: coll.hash_chain, + } + } +} + +/// Document shapes by bare collection name. +pub(crate) type Shapes = HashMap; + +/// `surrogate → primary key` per bare collection name. +#[derive(Debug, Default)] +pub(crate) struct BindIndex(HashMap>>); + +impl BindIndex { + pub fn new<'a>(binds: impl IntoIterator) -> Self { + let mut index: HashMap>> = HashMap::new(); + for (collection, surrogate, pk) in binds { + index + .entry(collection.to_string()) + .or_default() + .insert(surrogate, pk.to_vec()); + } + Self(index) + } + + pub fn pk(&self, collection: &str, surrogate: u32) -> Option<&[u8]> { + self.0 + .get(collection) + .and_then(|binds| binds.get(&surrogate)) + .map(Vec::as_slice) + } +} + +/// One database's snapshot, walked in the names of `database_id`. +pub(crate) struct Walk<'a> { + pub tenant_id: u64, + /// The database whose stored collection names the snapshot keys carry. + pub database_id: DatabaseId, + pub shapes: &'a Shapes, + pub binds: &'a BindIndex, + /// The segment encryption key: the WAL encryption key, `None` when at-rest + /// encryption is off. + pub kek: Option<&'a WalEncryptionKey>, +} + +impl Walk<'_> { + /// Emit every canonical row of `snap` into `sink`. + pub fn snapshot( + &self, + snap: &TenantDataSnapshot, + sink: &mut dyn FnMut(Row), + ) -> Result<(), Error> { + self.documents(&snap.documents, &snap.documents_versioned, sink)?; + self.edges(&snap.edges, sink)?; + self.kv_tables(&snap.kv_tables, sink)?; + self.vectors(&snap.vectors, sink)?; + self.timeseries(&snap.timeseries, &snap.flushed_ts_segments, sink)?; + self.columnar(&snap.columnar_engines, sink)?; + self.crdt(&snap.crdt_state, sink)?; + self.arrays(&snap.arrays, sink) + } + + /// The bare catalog name of a collection the snapshot names as `stored`. + pub(super) fn bare(&self, stored: &str) -> String { + stored_collection_key(self.database_id, stored) + .name() + .to_string() + } + + /// The part after `"{db}:{tid}:"` of a scoped section key. + pub(super) fn scoped_rest<'k>(&self, key: &'k str) -> Result<&'k str, Error> { + let mut parts = key.splitn(3, ':'); + match (parts.next(), parts.next(), parts.next()) { + (Some(_), Some(tid), Some(rest)) + if tid.parse::().ok() == Some(self.tenant_id) && !rest.is_empty() => + { + Ok(rest) + } + _ => Err(malformed(key)), + } + } +} + +pub(super) fn malformed(key: &str) -> Error { + let prefix: String = key.chars().take(64).collect(); + Error::Serialization { + format: "backup".into(), + detail: format!("backup verification: section key {prefix:?} is malformed"), + } +} + +/// The per-part tallies of rows, keyed by `(bare collection, part)`. +pub(crate) type PartKey = (String, nodedb_types::backup_envelope::VerifiedPart); + +/// Tallies of canonical rows by collection part. +pub(crate) type Tallies = BTreeMap; + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + + use nodedb_types::Value; + use nodedb_types::surrogate::Surrogate; + + use super::*; + use crate::engine::graph::edge_store::{EdgeValuePayload, versioned_edge_key}; + + const TENANT: u64 = 3; + + fn tallies(snap: &TenantDataSnapshot, binds: &[(&str, u32, &[u8])]) -> Tallies { + let shapes = Shapes::new(); + let binds = BindIndex::new(binds.iter().copied()); + let walk = Walk { + tenant_id: TENANT, + database_id: DatabaseId::DEFAULT, + shapes: &shapes, + binds: &binds, + kek: None, + }; + let mut tallies = Tallies::new(); + walk.snapshot(snap, &mut |row| { + tallies + .entry((row.collection, row.part)) + .or_default() + .add(&row.hash); + }) + .expect("walk"); + tallies + } + + fn doc(surrogate: u32, city: &str) -> (String, Vec) { + let body = Value::Object(HashMap::from([( + "city".to_string(), + Value::String(city.into()), + )])); + ( + format!("0:{TENANT}:people:{surrogate:08x}"), + nodedb_types::value_to_msgpack(&body).expect("encode"), + ) + } + + fn edge(src: &str, dst: &str, system_from: i64) -> (String, Vec) { + let key = versioned_edge_key("people", src, "knows", dst, system_from).expect("key"); + let value = EdgeValuePayload::new(system_from, i64::MAX, b"props".to_vec()) + .encode() + .expect("payload"); + (key, value) + } + + fn vectors(rows: Vec<(u32, Vec, Option)>) -> (String, Vec) { + ( + format!("0:{TENANT}:people:emb"), + zerompk::to_msgpack_vec(&rows).expect("encode"), + ) + } + + /// The source and the restored destination hold the same rows under other + /// surrogates, other system times and in another order. Every part's tally + /// agrees. + #[test] + fn the_digest_ignores_order_surrogates_and_system_time() { + let source = TenantDataSnapshot { + documents: vec![doc(0x2a, "paris"), doc(0x2b, "rome")], + edges: vec![edge("alice", "bob", 100)], + vectors: vec![vectors(vec![ + (0, vec![1.0, 0.0], Some(Surrogate::new(0x2a))), + (1, vec![0.0, 1.0], Some(Surrogate::new(0x2b))), + ])], + ..Default::default() + }; + let source_binds: [(&str, u32, &[u8]); 2] = + [("people", 0x2a, b"alice"), ("people", 0x2b, b"bob")]; + let restored = TenantDataSnapshot { + documents: vec![doc(0x91, "rome"), doc(0x90, "paris")], + edges: vec![edge("alice", "bob", 9_999)], + vectors: vec![vectors(vec![ + (7, vec![0.0, 1.0], Some(Surrogate::new(0x91))), + (3, vec![1.0, 0.0], Some(Surrogate::new(0x90))), + ])], + ..Default::default() + }; + let restored_binds: [(&str, u32, &[u8]); 2] = + [("people", 0x90, b"alice"), ("people", 0x91, b"bob")]; + + let expected = tallies(&source, &source_binds); + assert_eq!(expected.len(), 3, "documents, edges and vectors"); + assert_eq!(expected, tallies(&restored, &restored_binds)); + } + + /// A row whose value or identity changes changes its part's digest. + #[test] + fn the_digest_sees_a_changed_row() { + let binds: [(&str, u32, &[u8]); 2] = [("people", 0x2a, b"alice"), ("people", 0x2b, b"bob")]; + let base = TenantDataSnapshot { + documents: vec![doc(0x2a, "paris"), doc(0x2b, "rome")], + ..Default::default() + }; + let moved = TenantDataSnapshot { + documents: vec![doc(0x2a, "paris"), doc(0x2b, "oslo")], + ..Default::default() + }; + let swapped = TenantDataSnapshot { + documents: vec![doc(0x2a, "rome"), doc(0x2b, "paris")], + ..Default::default() + }; + let base_tallies = tallies(&base, &binds); + assert_ne!(base_tallies, tallies(&moved, &binds)); + assert_ne!(base_tallies, tallies(&swapped, &binds)); + } +} diff --git a/nodedb/src/control/backup/verify/walk_arrays.rs b/nodedb/src/control/backup/verify/walk_arrays.rs new file mode 100644 index 000000000..10bbfc36b --- /dev/null +++ b/nodedb/src/control/backup/verify/walk_arrays.rs @@ -0,0 +1,123 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Canonical rows of the array section. +//! +//! One row per cell version, keyed by its coordinate and system time. The +//! restore keeps both, and assigns the cell's surrogate anew, so the +//! surrogate stays out of the hash. + +use nodedb_types::backup_envelope::VerifiedPart; + +use crate::Error; +use crate::engine::array::export::ArrayCellVersion; +use crate::types::ArrayCellsBlob; + +use super::canonical::{Row, RowHasher, key_hash}; +use super::walk::Walk; + +const LIVE: u8 = b'l'; +const TOMBSTONE: u8 = b't'; +const ERASED: u8 = b'e'; + +fn encode_error(array: &str, e: impl std::fmt::Display) -> Error { + Error::Serialization { + format: "msgpack".into(), + detail: format!("backup verification: array '{array}': {e}"), + } +} + +impl Walk<'_> { + pub(super) fn arrays( + &self, + blobs: &[ArrayCellsBlob], + sink: &mut dyn FnMut(Row), + ) -> Result<(), Error> { + for blob in blobs { + let versions: Vec = + zerompk::from_msgpack(&blob.cells).map_err(|e| encode_error(&blob.array, e))?; + for version in versions { + sink(array_row(&blob.array, &version)?); + } + } + Ok(()) + } +} + +fn array_row(array: &str, version: &ArrayCellVersion) -> Result { + let coord = zerompk::to_msgpack_vec(&version.coord).map_err(|e| encode_error(array, e))?; + let system = version.system_from_ms.to_le_bytes(); + let key: [&[u8]; 2] = [&coord, &system]; + let mut hasher = RowHasher::new(VerifiedPart::Array, &key); + match &version.payload { + Some(payload) => { + hasher.bytes(b'r', &[LIVE]); + hasher.int(b'f', payload.valid_from_ms); + hasher.int(b'u', payload.valid_until_ms); + let attrs = + zerompk::to_msgpack_vec(&payload.attrs).map_err(|e| encode_error(array, e))?; + hasher.bytes(b'a', &attrs); + } + None if version.erased => hasher.bytes(b'r', &[ERASED]), + None => hasher.bytes(b'r', &[TOMBSTONE]), + } + Ok(Row { + collection: array.to_string(), + part: VerifiedPart::Array, + key: Some(key_hash(VerifiedPart::Array, &key)), + hash: hasher.finish(), + expire_at_ms: 0, + }) +} + +#[cfg(test)] +mod tests { + use nodedb_array::tile::cell_payload::CellPayload; + use nodedb_array::types::cell_value::value::CellValue; + use nodedb_array::types::coord::value::CoordValue; + use nodedb_types::{OPEN_UPPER, Surrogate}; + + use super::*; + + fn live(x: i64, v: i64, surrogate: u32) -> ArrayCellVersion { + ArrayCellVersion { + coord: vec![CoordValue::Int64(x)], + system_from_ms: 10, + payload: Some(CellPayload { + valid_from_ms: 0, + valid_until_ms: OPEN_UPPER, + attrs: vec![CellValue::Int64(v)], + surrogate: Surrogate::new(surrogate), + }), + erased: false, + } + } + + fn row(version: &ArrayCellVersion) -> Row { + array_row("grid", version).expect("row") + } + + /// A restored cell under another surrogate hashes the same. A changed + /// value, system time or row kind does not. + #[test] + fn the_digest_ignores_surrogates_only() { + let source = row(&live(1, 7, 3)); + assert_eq!(source.hash, row(&live(1, 7, 99)).hash); + assert_eq!(source.key, row(&live(1, 7, 99)).key); + assert_ne!(source.hash, row(&live(1, 8, 3)).hash); + let later = ArrayCellVersion { + system_from_ms: 11, + ..live(1, 7, 3) + }; + assert_ne!(source.key, row(&later).key); + let tombstone = ArrayCellVersion { + payload: None, + ..live(1, 7, 3) + }; + let erased = ArrayCellVersion { + erased: true, + ..tombstone.clone() + }; + assert_ne!(row(&tombstone).hash, row(&erased).hash); + assert_ne!(row(&tombstone).hash, source.hash); + } +} diff --git a/nodedb/src/control/backup/verify/walk_engines.rs b/nodedb/src/control/backup/verify/walk_engines.rs new file mode 100644 index 000000000..f01030c43 --- /dev/null +++ b/nodedb/src/control/backup/verify/walk_engines.rs @@ -0,0 +1,206 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Canonical rows of the KV, vector, timeseries, columnar and CRDT sections. +//! +//! Timeseries and columnar rows decode with the same decoders RESTORE re-issues +//! them from, so a row restore writes hashes the same on both sides. + +use std::collections::BTreeMap; + +use loro::{LoroDoc, LoroValue}; +use nodedb_types::backup_envelope::VerifiedPart; +use nodedb_types::surrogate::Surrogate; + +use crate::Error; +use crate::control::backup::restore::columnar_reissue::decode_snapshot_live_rows; +use crate::control::backup::restore::timeseries_reissue::decode_timeseries_live_rows; +use crate::control::backup::restore::vector_reissue::split_vector_coll_key; +use crate::types::TsFlushedCollectionBlob; + +use super::canonical::{Row, RowHasher, key_hash}; +use super::walk::Walk; + +/// One KV table's rows, in the KV snapshot shape. +type KvRows = Vec; + +/// One vector index's export rows: `(node_id, vector, surrogate)`. +type VectorRows = Vec<(u32, Vec, Option)>; + +/// How a vector row is keyed. +const VECTOR_BOUND: u8 = b'p'; +/// A vector with no surrogate. Every stored vector is bound, so a capture that +/// holds one is corrupt, and RESTORE refuses it (`group_restored_vectors`). +const VECTOR_UNBOUND: u8 = b'u'; +/// A vector whose surrogate binds no primary key. +const VECTOR_UNKEYED: u8 = b's'; + +fn decode_error(what: &str, collection: &str, e: impl std::fmt::Display) -> Error { + Error::Serialization { + format: "msgpack".into(), + detail: format!("backup verification: decode {what} of '{collection}': {e}"), + } +} + +impl Walk<'_> { + pub(super) fn kv_tables( + &self, + tables: &[(String, Vec)], + sink: &mut dyn FnMut(Row), + ) -> Result<(), Error> { + for (table_key, bytes) in tables { + let bare = self.bare(self.scoped_rest(table_key)?); + let rows: KvRows = + zerompk::from_msgpack(bytes).map_err(|e| decode_error("KV table", &bare, e))?; + // The surrogate is not hashed: RESTORE binds each row in the + // destination catalog, so source and destination identities differ. + for (key, value, expire_at_ms, _surrogate) in rows { + let row_key: [&[u8]; 1] = [key.as_slice()]; + let mut hasher = RowHasher::new(VerifiedPart::KeyValue, &row_key); + hasher.bytes(b'v', &value); + sink(Row { + collection: bare.clone(), + part: VerifiedPart::KeyValue, + key: Some(key_hash(VerifiedPart::KeyValue, &row_key)), + hash: hasher.finish(), + expire_at_ms, + }); + } + } + Ok(()) + } + + pub(super) fn vectors( + &self, + indexes: &[(String, Vec)], + sink: &mut dyn FnMut(Row), + ) -> Result<(), Error> { + for (index_key, bytes) in indexes { + let (stored, field) = split_vector_coll_key(self.scoped_rest(index_key)?); + let bare = self.bare(stored); + let rows: VectorRows = + zerompk::from_msgpack(bytes).map_err(|e| decode_error("vector index", &bare, e))?; + for (_node_id, vector, surrogate) in rows { + let bound = surrogate + .filter(|s| *s != Surrogate::ZERO) + .map(|s| self.binds.pk(&bare, s.as_u32())); + let (kind, pk): (u8, &[u8]) = match bound { + Some(Some(pk)) => (VECTOR_BOUND, pk), + None => (VECTOR_UNBOUND, &[]), + Some(None) => (VECTOR_UNKEYED, &[]), + }; + let kind = [kind]; + let key: [&[u8]; 3] = [field.as_bytes(), &kind, pk]; + let mut hasher = RowHasher::new(VerifiedPart::Vectors, &key); + let data: Vec = vector.iter().flat_map(|v| v.to_le_bytes()).collect(); + hasher.bytes(b'f', &data); + sink(Row { + collection: bare.clone(), + part: VerifiedPart::Vectors, + key: Some(key_hash(VerifiedPart::Vectors, &key)), + hash: hasher.finish(), + expire_at_ms: 0, + }); + } + } + Ok(()) + } + + /// Each collection's memtable rows plus every flushed partition's rows. + pub(super) fn timeseries( + &self, + memtables: &[(String, Vec)], + flushed: &[TsFlushedCollectionBlob], + sink: &mut dyn FnMut(Row), + ) -> Result<(), Error> { + let mut by_key: BTreeMap<&str, (Option<&[u8]>, Option<&TsFlushedCollectionBlob>)> = + BTreeMap::new(); + for (key, bytes) in memtables { + by_key.entry(key.as_str()).or_default().0 = Some(bytes.as_slice()); + } + for blob in flushed { + by_key.entry(blob.collection_key.as_str()).or_default().1 = Some(blob); + } + let empty = TsFlushedCollectionBlob::default(); + for (key, (memtable, flushed)) in by_key { + let bare = self.bare(self.scoped_rest(key)?); + let rows = + decode_timeseries_live_rows(&bare, memtable, flushed.unwrap_or(&empty), self.kek)?; + for row in rows { + let mut hasher = RowHasher::new(VerifiedPart::Timeseries, &[]); + hasher.value(&row)?; + sink(keyless(bare.clone(), VerifiedPart::Timeseries, hasher)); + } + } + Ok(()) + } + + pub(super) fn columnar( + &self, + engines: &[(String, Vec)], + sink: &mut dyn FnMut(Row), + ) -> Result<(), Error> { + for (key, bytes) in engines { + let bare = self.bare(self.scoped_rest(key)?); + let snap: nodedb_columnar::ColumnarEngineSnapshot = zerompk::from_msgpack(bytes) + .map_err(|e| decode_error("columnar snapshot", &bare, e))?; + let decoded = decode_snapshot_live_rows(&bare, snap, self.kek)?; + for row in decoded.rows { + let mut hasher = RowHasher::new(VerifiedPart::Columnar, &[]); + hasher.value(&row)?; + sink(keyless(bare.clone(), VerifiedPart::Columnar, hasher)); + } + } + Ok(()) + } + + /// Each collection's Loro document, one row per entry of each root map. + pub(super) fn crdt( + &self, + states: &[(u64, u64, String, Vec)], + sink: &mut dyn FnMut(Row), + ) -> Result<(), Error> { + for (_database_id, _tenant_id, stored, bytes) in states { + let bare = self.bare(stored); + let doc = LoroDoc::new(); + doc.import(bytes) + .map_err(|e| decode_error("CRDT state", &bare, e))?; + let mut emit = |key: &[&[u8]], value: &LoroValue| { + let mut hasher = RowHasher::new(VerifiedPart::Crdt, key); + hasher.loro(value); + sink(Row { + collection: bare.clone(), + part: VerifiedPart::Crdt, + key: Some(key_hash(VerifiedPart::Crdt, key)), + hash: hasher.finish(), + expire_at_ms: 0, + }); + }; + match doc.get_deep_value() { + LoroValue::Map(roots) => { + for (root, value) in roots.iter() { + match value { + LoroValue::Map(rows) => { + for (id, row) in rows.iter() { + emit(&[root.as_bytes(), id.as_bytes()][..], row); + } + } + other => emit(&[root.as_bytes()][..], other), + } + } + } + other => emit(&[][..], &other), + } + } + Ok(()) + } +} + +fn keyless(collection: String, part: VerifiedPart, hasher: RowHasher) -> Row { + Row { + collection, + part, + key: None, + hash: hasher.finish(), + expire_at_ms: 0, + } +} diff --git a/nodedb/src/control/backup/verify/walk_rows.rs b/nodedb/src/control/backup/verify/walk_rows.rs new file mode 100644 index 000000000..233e3a68d --- /dev/null +++ b/nodedb/src/control/backup/verify/walk_rows.rs @@ -0,0 +1,227 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Canonical rows of the document and graph-edge sections. +//! +//! A document row is keyed by the identity RESTORE binds it under: the backup's +//! primary-key bind of its surrogate, else the identity INSERT derives from its +//! body. A bitemporal row emits one canonical row per version. + +use std::collections::BTreeMap; + +use nodedb_types::backup_envelope::VerifiedPart; +use nodedb_types::{RowIdentity, StorageKey, Value}; + +use crate::Error; +use crate::data::executor::strict_format::{binary_tuple_to_msgpack, undecodable_strict_row}; +use crate::engine::graph::edge_store::{ + EdgeValuePayload, is_gdpr_erasure, is_tombstone, parse_versioned_edge_key, +}; +use crate::engine::sparse::btree_versioned::{TAG_LIVE, decode_value}; +use crate::types::hash_chain::is_chain_field; + +use super::canonical::{Row, RowHasher, key_hash}; +use super::walk::{DocShape, Walk, malformed}; + +/// A document's storage key as the snapshot carries it. +#[derive(Clone, Copy, PartialEq, Eq, PartialOrd, Ord)] +enum RowKey<'a> { + Surrogate(StorageKey), + /// A key that is no storage key. It is its own identity. + Other(&'a str), +} + +impl<'a> RowKey<'a> { + fn parse(text: &'a str) -> Self { + StorageKey::parse(text).map_or(Self::Other(text), Self::Surrogate) + } +} + +/// A stored body decoded to MessagePack and to a native value. +struct Body { + msgpack: Vec, + value: Option, +} + +impl Walk<'_> { + pub(super) fn documents( + &self, + documents: &[(String, Vec)], + documents_versioned: &[(String, Vec)], + sink: &mut dyn FnMut(Row), + ) -> Result<(), Error> { + // A collection with no catalog row decodes as schemaless, keyed by `id`. + let unknown = DocShape::default(); + for (key, body) in documents { + let (stored, rest) = self.document_key(key)?; + let bare = self.bare(stored); + let shape = self.shapes.get(&bare).unwrap_or(&unknown); + let row_key = RowKey::parse(rest); + let body = decode_body(shape, &bare, row_key, body)?; + let identity = self.identity(&bare, shape, row_key, Some(body.msgpack.as_slice())); + let mut hasher = RowHasher::new(VerifiedPart::Documents, &[identity.as_slice()]); + hash_body(&mut hasher, shape, body)?; + sink(row(bare, &identity, hasher)); + } + + // Every version of a row, in system-time order: the identity comes from + // the first live version, as RESTORE derives it. + type Versions<'v> = Vec<(i64, &'v [u8])>; + let mut versioned: BTreeMap<(&str, RowKey<'_>), Versions<'_>> = BTreeMap::new(); + for (key, value) in documents_versioned { + let (stored, rest) = self.document_key(key)?; + let (text, sys) = rest.split_once('\x00').ok_or_else(|| malformed(key))?; + let sys = sys.parse::().map_err(|_| malformed(key))?; + versioned + .entry((stored, RowKey::parse(text))) + .or_default() + .push((sys, value.as_slice())); + } + for ((stored, row_key), mut versions) in versioned { + versions.sort_by_key(|(sys, _)| *sys); + let bare = self.bare(stored); + let shape = self.shapes.get(&bare).unwrap_or(&unknown); + let mut decoded = Vec::with_capacity(versions.len()); + for (_, raw) in versions { + let version = decode_value(raw)?; + let body = if version.tag == TAG_LIVE { + Some(decode_body(shape, &bare, row_key, version.body)?) + } else { + None + }; + decoded.push((version, body)); + } + let first_live = decoded + .iter() + .find_map(|(_, body)| body.as_ref().map(|b| b.msgpack.as_slice())); + let identity = self.identity(&bare, shape, row_key, first_live); + for (version, body) in decoded { + let mut hasher = RowHasher::new(VerifiedPart::Documents, &[identity.as_slice()]); + hasher.bytes(b't', &[version.tag]); + hasher.int(b'f', version.valid_from_ms); + hasher.int(b'u', version.valid_until_ms); + match body { + Some(body) => hash_body(&mut hasher, shape, body)?, + None => hasher.bytes(b'R', version.body), + } + sink(row(bare.clone(), &identity, hasher)); + } + } + Ok(()) + } + + pub(super) fn edges( + &self, + edges: &[(String, Vec)], + sink: &mut dyn FnMut(Row), + ) -> Result<(), Error> { + for (key, value) in edges { + let (stored, src, label, dst, _system_from) = + parse_versioned_edge_key(key).ok_or_else(|| malformed(key))?; + let endpoints: [&[u8]; 3] = [src.as_bytes(), label.as_bytes(), dst.as_bytes()]; + let mut hasher = RowHasher::new(VerifiedPart::Edges, &endpoints); + if is_tombstone(value) { + hasher.bytes(b't', &[]); + } else if is_gdpr_erasure(value) { + hasher.bytes(b'g', value); + } else { + hasher.bytes(b'p', &EdgeValuePayload::decode(value)?.properties); + } + sink(Row { + collection: self.bare(stored), + part: VerifiedPart::Edges, + key: Some(key_hash(VerifiedPart::Edges, &endpoints)), + hash: hasher.finish(), + expire_at_ms: 0, + }); + } + Ok(()) + } + + /// Split `"{db}:{tid}:{collection}:{rest}"` into the stored collection and + /// the rest, checking the tenant. + fn document_key<'k>(&self, key: &'k str) -> Result<(&'k str, &'k str), Error> { + let (stored, rest) = self + .scoped_rest(key)? + .split_once(':') + .ok_or_else(|| malformed(key))?; + if stored.is_empty() { + return Err(malformed(key)); + } + Ok((stored, rest)) + } + + /// The identity RESTORE binds the row under. + fn identity( + &self, + bare: &str, + shape: &DocShape, + key: RowKey<'_>, + body: Option<&[u8]>, + ) -> Vec { + let key = match key { + RowKey::Surrogate(key) => key, + RowKey::Other(text) => return text.as_bytes().to_vec(), + }; + if let Some(pk) = self.binds.pk(bare, key.surrogate().as_u32()) { + return pk.to_vec(); + } + let identity = match body { + Some(body) => { + RowIdentity::of_stored_row(body, shape.declared_primary_key.as_deref(), key) + } + None => key.to_identity(), + }; + identity.as_str().as_bytes().to_vec() + } +} + +fn row(collection: String, identity: &[u8], hasher: RowHasher) -> Row { + Row { + collection, + part: VerifiedPart::Documents, + key: Some(key_hash(VerifiedPart::Documents, &[identity])), + hash: hasher.finish(), + expire_at_ms: 0, + } +} + +/// Decode a stored body: a strict row's Binary Tuple with the collection's +/// schema, a schemaless row as it is. +fn decode_body( + shape: &DocShape, + collection: &str, + key: RowKey<'_>, + body: &[u8], +) -> Result { + let msgpack = match &shape.strict { + Some(schema) => binary_tuple_to_msgpack(body, schema).ok_or_else(|| { + let identity = match key { + RowKey::Surrogate(key) => key.to_identity().as_str().to_string(), + RowKey::Other(text) => text.to_string(), + }; + undecodable_strict_row(collection, &identity) + })?, + None => body.to_vec(), + }; + let value = nodedb_types::value_from_msgpack(&msgpack).ok(); + Ok(Body { msgpack, value }) +} + +/// Hash a decoded body. A hash-chained row drops its chain fields: the restore +/// relinks the chain under the destination's storage keys. A body that is no +/// MessagePack value hashes as its bytes. +fn hash_body(hasher: &mut RowHasher, shape: &DocShape, body: Body) -> Result<(), Error> { + match body.value { + Some(Value::Object(mut fields)) => { + if shape.hash_chain { + fields.retain(|name, _| !is_chain_field(name)); + } + hasher.value(&Value::Object(fields)) + } + Some(value) => hasher.value(&value), + None => { + hasher.bytes(b'R', &body.msgpack); + Ok(()) + } + } +} diff --git a/nodedb/src/control/cascade/orchestrator.rs b/nodedb/src/control/cascade/orchestrator.rs index 46b3dffb6..4d0086f35 100644 --- a/nodedb/src/control/cascade/orchestrator.rs +++ b/nodedb/src/control/cascade/orchestrator.rs @@ -237,6 +237,7 @@ mod tests { owner: "admin".into(), created_at: 0, subscriber_roles: Vec::new(), + modification_hlc: nodedb_types::Hlc::ZERO, }) .unwrap(); catalog @@ -254,7 +255,7 @@ mod tests { .unwrap(); } - // Periodic and streaming MVs live in separate catalogs and may share a + // Periodic and streaming MVs live in separate catalogs and can share a // name. The cascade must retain both instead of collapsing them in the // visited set. catalog diff --git a/nodedb/src/control/catalog_entry/apply/array.rs b/nodedb/src/control/catalog_entry/apply/array.rs new file mode 100644 index 000000000..96f614751 --- /dev/null +++ b/nodedb/src/control/catalog_entry/apply/array.rs @@ -0,0 +1,223 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Apply array catalog entries to `_system.arrays`. + +use nodedb_array::types::ArrayId; + +use crate::control::array_catalog::{ArrayCatalogEntry, persist}; +use crate::control::security::catalog::{SystemCatalog, catalog_err}; +use crate::types::{DatabaseId, TenantId}; + +/// Apply `PutArray`: write the full stored definition. +pub fn put(entry: &ArrayCatalogEntry, catalog: &SystemCatalog) -> crate::Result<()> { + persist::persist(catalog, entry).map_err(|e| { + catalog_err( + &format!( + "put_array '{}' (database {}, tenant {})", + entry.name, + entry.array_id.database_id.as_u64(), + entry.array_id.tenant_id.as_u64() + ), + e, + ) + }) +} + +/// Apply `DeleteArray`: remove the row and its surrogate bindings in one +/// transaction. A move carries the bindings to `moved_to` instead. +pub fn delete( + database_id: u64, + tenant_id: u64, + name: &str, + moved_to: Option, + catalog: &SystemCatalog, +) -> crate::Result<()> { + let array_id = + ArrayId::in_database(TenantId::new(tenant_id), DatabaseId::new(database_id), name); + let removed = match moved_to { + Some(to) => persist::move_with_surrogates(catalog, &array_id, DatabaseId::new(to)), + None => persist::remove_with_surrogates(catalog, &array_id), + }; + removed.map_err(|e| { + catalog_err( + &format!("delete_array '{name}' (database {database_id}, tenant {tenant_id})"), + e, + ) + }) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_types::{Hlc, HlcClock}; + + use super::*; + use crate::control::catalog_entry::CatalogEntry; + use crate::control::catalog_entry::descriptor_stamp::{stamp, stamp_batch}; + use crate::control::catalog_entry::descriptor_validate::{ValidationOutcome, validate}; + use crate::control::security::credential::CredentialStore; + + fn make_catalog() -> (Arc, tempfile::TempDir) { + let tmp = tempfile::tempdir().expect("tmpdir"); + let store = Arc::new(CredentialStore::open(&tmp.path().join("system.redb")).expect("open")); + (store, tmp) + } + + fn array(name: &str, hlc: Hlc) -> ArrayCatalogEntry { + ArrayCatalogEntry { + array_id: ArrayId::in_database(TenantId::new(1), DatabaseId::DEFAULT, name), + name: name.to_string(), + schema_msgpack: vec![0x90], + schema_hash: 7, + created_at_ms: 0, + prefix_bits: 8, + audit_retain_ms: None, + minimum_audit_retain_ms: None, + modification_hlc: hlc, + incarnation: nodedb_types::Hlc::ZERO, + } + } + + fn delete_entry(name: &str, target_hlc: Hlc) -> CatalogEntry { + CatalogEntry::DeleteArray { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: name.to_string(), + target_hlc, + moved_to: None, + } + } + + /// A `DeleteArray` replayed after a same-name recreate names the prior + /// incarnation, so it must not remove the recreated array. + #[test] + fn replayed_delete_of_a_prior_incarnation_is_already_applied() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + put(&array("grid", Hlc::new(30, 0)), catalog).expect("seed recreated array"); + + let replayed = delete_entry("grid", Hlc::new(10, 0)); + assert_eq!( + validate(&replayed, catalog).expect("validate"), + ValidationOutcome::AlreadyApplied + ); + + let current = delete_entry("grid", Hlc::new(30, 0)); + assert_eq!( + validate(¤t, catalog).expect("validate"), + ValidationOutcome::Apply + ); + } + + /// A `PutArray` replayed from before a drop and recreate must not + /// overwrite the recreated definition. + #[test] + fn replayed_put_of_a_prior_incarnation_is_already_applied() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + put(&array("grid", Hlc::new(30, 0)), catalog).expect("seed recreated array"); + + let replayed = CatalogEntry::PutArray(Box::new(array("grid", Hlc::new(10, 0)))); + assert_eq!( + validate(&replayed, catalog).expect("validate"), + ValidationOutcome::AlreadyApplied + ); + } + + /// The proposer freezes a delete's target from the committed row, and a + /// put orders after the row it replaces. + #[test] + fn stamps_target_the_committed_incarnation() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + let clock = HlcClock::new(); + let committed = Hlc::new(u64::MAX / 2, 0); + put(&array("grid", committed), catalog).expect("seed array"); + + let CatalogEntry::DeleteArray { target_hlc, .. } = + stamp(delete_entry("grid", Hlc::ZERO), &clock, catalog).expect("stamp delete") + else { + unreachable!("stamp keeps the variant"); + }; + assert_eq!(target_hlc, committed); + + let CatalogEntry::PutArray(altered) = stamp( + CatalogEntry::PutArray(Box::new(array("grid", Hlc::ZERO))), + &clock, + catalog, + ) + .expect("stamp put") else { + unreachable!("stamp keeps the variant"); + }; + assert!(altered.modification_hlc > committed); + } + + /// `CREATE ARRAY g; DROP ARRAY g` in one batch: the drop targets the + /// create stamped before it. + #[test] + fn batched_delete_targets_the_preceding_put() { + let (store, _tmp) = make_catalog(); + let stamped = stamp_batch( + vec![ + CatalogEntry::PutArray(Box::new(array("grid", Hlc::ZERO))), + delete_entry("grid", Hlc::ZERO), + ], + &HlcClock::new(), + store.catalog(), + ) + .expect("stamp batch"); + let (CatalogEntry::PutArray(created), CatalogEntry::DeleteArray { target_hlc, .. }) = + (&stamped[0], &stamped[1]) + else { + unreachable!("stamp keeps the variants"); + }; + assert_eq!(*target_hlc, created.modification_hlc); + } + + #[test] + fn delete_removes_the_row() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + put(&array("grid", Hlc::new(5, 0)), catalog).expect("seed array"); + delete(DatabaseId::DEFAULT.as_u64(), 1, "grid", None, catalog).expect("delete"); + assert!( + catalog + .get_array_in_database(TenantId::new(1), DatabaseId::DEFAULT, "grid") + .expect("read") + .is_none() + ); + } + + /// A moving delete survives the metadata log encoding with its target. + #[test] + fn moving_delete_roundtrips_through_the_log_codec() { + let entry = CatalogEntry::DeleteArray { + database_id: 3, + tenant_id: 1, + name: "grid".to_string(), + target_hlc: Hlc::new(7, 1), + moved_to: Some(crate::control::array_catalog::ArrayMove { + target_db_id: 4, + mover_tenant_id: 9, + }), + }; + let bytes = crate::control::catalog_entry::encode(&entry).expect("encode"); + let CatalogEntry::DeleteArray { + target_hlc, + moved_to, + .. + } = crate::control::catalog_entry::decode(&bytes).expect("decode") + else { + unreachable!("the codec keeps the variant"); + }; + assert_eq!(target_hlc, Hlc::new(7, 1)); + assert_eq!( + moved_to, + Some(crate::control::array_catalog::ArrayMove { + target_db_id: 4, + mover_tenant_id: 9, + }) + ); + } +} diff --git a/nodedb/src/control/catalog_entry/apply/backup_schedule.rs b/nodedb/src/control/catalog_entry/apply/backup_schedule.rs new file mode 100644 index 000000000..a928eeac9 --- /dev/null +++ b/nodedb/src/control/catalog_entry/apply/backup_schedule.rs @@ -0,0 +1,16 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Apply backup schedule marks to `SystemCatalog` redb. + +use crate::control::security::catalog::backup_schedule_marks::StoredBackupScheduleMark; +use crate::control::security::catalog::{SystemCatalog, catalog_err}; + +/// Apply a `PutBackupScheduleMark` entry. A mark below the stored one of the +/// same incarnation writes nothing, so a replay or a late proposal from a +/// former leader never moves the mark back. +pub fn raise_mark(mark: &StoredBackupScheduleMark, catalog: &SystemCatalog) -> crate::Result<()> { + catalog + .raise_backup_schedule_mark(mark) + .map(drop) + .map_err(|e| catalog_err(&format!("put_backup_schedule_mark '{}'", mark.job), e)) +} diff --git a/nodedb/src/control/catalog_entry/apply/checkpoint.rs b/nodedb/src/control/catalog_entry/apply/checkpoint.rs index 3b3b79cf2..4cb5948a0 100644 --- a/nodedb/src/control/catalog_entry/apply/checkpoint.rs +++ b/nodedb/src/control/catalog_entry/apply/checkpoint.rs @@ -4,10 +4,12 @@ //! //! Writes only. The leader reports the duplicate and the missing checkpoint //! before proposing, so apply runs the unvalidated catalog path: a rejection -//! here would diverge a follower from a statement the leader already accepted. +//! here diverges a follower from a statement the leader already accepted. use crate::control::security::catalog::types::{CheckpointDoc, CheckpointRecord}; -use crate::control::security::catalog::{SystemCatalog, catalog_err}; +use crate::control::security::catalog::{ + StoredCompactionPoint, StoredPendingHistoryCompaction, SystemCatalog, catalog_err, +}; /// Apply a `PutCheckpoint` entry. A re-delivery rewrites the same row. pub fn put(record: &CheckpointRecord, catalog: &SystemCatalog) -> crate::Result<()> { @@ -53,6 +55,38 @@ pub fn delete_before( .map(|_| ()) } +/// Apply a `CompactHistory` entry: record the collection's compaction point +/// and the owed oplog compaction, then delete the checkpoint rows below the +/// boundary. +/// +/// The compaction point is replicated, so a node that installs a metadata +/// image learns every compaction the image covers. The owed row is +/// node-local. It is written before the checkpoint delete and removed by +/// post-apply once every local core compacted durably. A crash between +/// apply and that removal leaves the row for the boot drain. +pub fn compact_history( + doc: CheckpointDoc<'_>, + before_timestamp: u64, + target_version_json: &str, + catalog: &SystemCatalog, +) -> crate::Result<()> { + catalog.put_compaction_point(&StoredCompactionPoint { + database_id: doc.database_id, + tenant_id: doc.tenant_id, + collection: doc.collection.to_string(), + target_version_json: target_version_json.to_string(), + })?; + catalog.enqueue_pending_history_compaction(&StoredPendingHistoryCompaction { + database_id: doc.database_id, + tenant_id: doc.tenant_id, + collection: doc.collection.to_string(), + target_version_json: target_version_json.to_string(), + last_error: String::new(), + attempts: 0, + })?; + delete_before(doc, before_timestamp, catalog) +} + #[cfg(test)] mod tests { use super::*; @@ -269,6 +303,32 @@ mod tests { ); } + /// Applying `CompactHistory` records the owed compaction before post-apply + /// runs, so a crash before the fan-out leaves it for the boot drain. + #[test] + fn apply_records_the_owed_compaction() { + let (_dir, catalog) = open_catalog(); + apply::apply_to( + &CatalogEntry::CompactHistory { + tenant_id: TENANT, + database_id: DATABASE, + collection: COLLECTION.to_string(), + doc_id: DOC.to_string(), + before_timestamp: 100, + target_version_json: "{\"n1\":4}".to_string(), + }, + &catalog, + ) + .unwrap(); + + let owed = catalog.load_pending_history_compactions().unwrap(); + assert_eq!(owed.len(), 1); + assert_eq!(owed[0].database_id, DATABASE); + assert_eq!(owed[0].tenant_id, TENANT); + assert_eq!(owed[0].collection, COLLECTION); + assert_eq!(owed[0].target_version_json, "{\"n1\":4}"); + } + #[test] fn range_delete_leaves_other_documents_alone() { let (_dir, catalog) = open_catalog(); diff --git a/nodedb/src/control/catalog_entry/apply/clone_cow.rs b/nodedb/src/control/catalog_entry/apply/clone_cow.rs new file mode 100644 index 000000000..d69d1f668 --- /dev/null +++ b/nodedb/src/control/catalog_entry/apply/clone_cow.rs @@ -0,0 +1,134 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Apply clone copy-on-write catalog entries. +//! +//! Each row is keyed by the clone collection's database-qualified name. A +//! row is written only while that collection is still a clone: once it is +//! materialized, purged, or dropped, the rows it had are gone and a replay of +//! an older entry writes nothing. Every write is an upsert, so a re-delivery +//! leaves the same state. + +use nodedb_types::DatabaseId; + +use crate::control::planner::sql_plan_convert::convert::db_qualified; +use crate::control::security::catalog::SystemCatalog; + +/// The clone collection a copy-on-write entry names. +#[derive(Debug, Clone, Copy)] +pub struct CloneTarget<'a> { + pub database_id: u64, + pub tenant_id: u64, + pub collection: &'a str, +} + +impl CloneTarget<'_> { + /// The database-qualified key the copy-on-write tables use, or `None` + /// when the collection is no longer a clone. + fn live_key(&self, catalog: &SystemCatalog) -> crate::Result> { + let database_id = DatabaseId::new(self.database_id); + let is_clone = catalog + .get_collection(database_id, self.tenant_id, self.collection)? + .is_some_and(|coll| coll.cloned_from.is_some()); + Ok(is_clone.then(|| db_qualified(database_id, self.collection))) + } +} + +/// Apply `PutCloneCopyup`. +pub fn put_copyup( + target: CloneTarget<'_>, + source_surrogate: u32, + target_surrogate: u32, + catalog: &SystemCatalog, +) -> crate::Result<()> { + if let Some(key) = target.live_key(catalog)? { + catalog.put_clone_copyup(&key, source_surrogate, target_surrogate)?; + } + Ok(()) +} + +/// Apply `PutCloneTombstone`. +pub fn put_tombstone( + target: CloneTarget<'_>, + source_surrogate: u32, + catalog: &SystemCatalog, +) -> crate::Result<()> { + if let Some(key) = target.live_key(catalog)? { + catalog.put_clone_tombstone(&key, source_surrogate)?; + } + Ok(()) +} + +/// Apply `PutKvCloneTombstone`. +pub fn put_kv_tombstone( + target: CloneTarget<'_>, + kv_key: &str, + catalog: &SystemCatalog, +) -> crate::Result<()> { + if let Some(key) = target.live_key(catalog)? { + catalog.put_kv_clone_tombstone(&key, kv_key)?; + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use nodedb_types::{CloneOrigin, CloneStatus, Lsn}; + + use super::*; + use crate::control::security::catalog::StoredCollection; + + fn open_catalog() -> (tempfile::TempDir, SystemCatalog) { + let dir = tempfile::tempdir().unwrap(); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).unwrap(); + (dir, catalog) + } + + fn clone_collection(db: DatabaseId) -> StoredCollection { + let mut coll = StoredCollection::stamped_for_test(1, "orders", "admin"); + coll.database_id = db; + coll.cloned_from = Some(CloneOrigin { + source_database: DatabaseId::new(1024), + source_collection: "orders".into(), + as_of_lsn: Lsn::new(1), + clone_created_at: Lsn::new(2), + kv_surrogate_ceiling: None, + }); + coll + } + + /// Rows land while the collection is a clone, twice without change, and + /// never once it is materialized. + #[test] + fn writes_only_while_the_collection_is_a_clone() { + let (_dir, catalog) = open_catalog(); + let db = DatabaseId::new(1030); + let mut coll = clone_collection(db); + catalog.put_collection(db, &coll).unwrap(); + let target = CloneTarget { + database_id: db.as_u64(), + tenant_id: 1, + collection: "orders", + }; + let key = db_qualified(db, "orders"); + + for _ in 0..2 { + put_tombstone(target, 7, &catalog).unwrap(); + put_kv_tombstone(target, "k1", &catalog).unwrap(); + put_copyup(target, 9, 90, &catalog).unwrap(); + } + assert!(catalog.is_clone_tombstoned(&key, 7).unwrap()); + assert!( + catalog + .list_kv_clone_tombstones(&key) + .unwrap() + .contains("k1") + ); + assert!(catalog.get_clone_copyup(&key, 9).unwrap().is_some()); + + coll.cloned_from = None; + coll.clone_status = CloneStatus::Materialized; + catalog.put_collection(db, &coll).unwrap(); + put_tombstone(target, 8, &catalog).unwrap(); + assert!(!catalog.is_clone_tombstoned(&key, 8).unwrap()); + } +} diff --git a/nodedb/src/control/catalog_entry/apply/collection.rs b/nodedb/src/control/catalog_entry/apply/collection.rs index eb5631a6b..ade6cdada 100644 --- a/nodedb/src/control/catalog_entry/apply/collection.rs +++ b/nodedb/src/control/catalog_entry/apply/collection.rs @@ -9,6 +9,7 @@ use crate::control::security::catalog::auth_types::object_type; use crate::control::security::catalog::{StoredCollection, SystemCatalog, catalog_err}; pub fn put(stored: &StoredCollection, catalog: &SystemCatalog) -> crate::Result<()> { + reap_clone_aux_on_materialize(stored, catalog)?; catalog .put_collection(stored.database_id, stored) .map_err(|e| { @@ -36,6 +37,32 @@ pub fn put(stored: &StoredCollection, catalog: &SystemCatalog) -> crate::Result< sync_index_visibility(stored, catalog) } +/// Drop this node's copy-on-write rows of a clone the materializer finished. +/// +/// The flip to `Materialized` with no `cloned_from` is the materializer's last +/// write. Every node applies it, so every node drops its own copy-up and +/// tombstone rows here. The rows go before the collection row: a crash in +/// between re-applies the whole entry on replay, because the persisted row +/// still differs from it. +fn reap_clone_aux_on_materialize( + stored: &StoredCollection, + catalog: &SystemCatalog, +) -> crate::Result<()> { + if stored.cloned_from.is_some() + || stored.clone_status != nodedb_types::CloneStatus::Materialized + { + return Ok(()); + } + let qualified = crate::control::planner::sql_plan_convert::convert::db_qualified( + stored.database_id, + &stored.name, + ); + catalog.delete_all_clone_copyups_for_collection(&qualified)?; + catalog.delete_all_clone_tombstones_for_collection(&qualified)?; + catalog.delete_all_kv_clone_tombstones_for_collection(&qualified)?; + Ok(()) +} + /// Align the collection's index records with its own `is_active` state, so a /// soft-dropped collection hides its indexes and an undropped one brings them /// back. Indexes are never deleted here — that happens only at purge. @@ -114,7 +141,7 @@ pub fn put_if_absent(stored: &StoredCollection, catalog: &SystemCatalog) -> crat /// from crossing an incomplete reclaim. /// /// Returns whether a row was found and deactivated. `false` is legitimate -/// only for the replicated applier (may never have held the row) — a caller +/// only for the replicated applier (it can lack the row) — a caller /// that already read the row must use [`prepare_purge_checked`] instead. pub fn prepare_purge( database_id: u64, @@ -140,8 +167,8 @@ pub fn prepare_purge( } /// Fail-closed [`prepare_purge`] for callers that resolved the collection -/// before asking for the purge. A miss is never benign: the reclaim would run -/// while the row is active and a same-name CREATE could register over keys +/// before asking for the purge. A miss is never benign: the reclaim runs +/// while the row is active and a same-name CREATE can register over keys /// the old incarnation still owns. Raises rather than reporting success. pub fn prepare_purge_checked( database_id: u64, @@ -186,6 +213,20 @@ pub fn finalize_purge( // collection; the Data Plane storage itself is reclaimed by the // `UnregisterCollection` half of the purge. purge_index_records(database_id.as_u64(), tenant_id, name, catalog)?; + // Statistics, checkpoints, and a compaction point left behind describe + // a later collection of the same name. + catalog.delete_column_stats_for_collection(database_id.as_u64(), tenant_id, name)?; + catalog.delete_checkpoints_for_collection(database_id.as_u64(), tenant_id, name)?; + catalog.delete_compaction_point(database_id.as_u64(), tenant_id, name)?; + // A grant row left behind reloads on the next boot and opens a later + // collection of the same name. + catalog.delete_permissions_for_target( + &crate::control::security::permission::collection_target( + database_id, + nodedb_types::TenantId::new(tenant_id), + name, + ), + )?; let removed = catalog.delete_collection(database_id, tenant_id, name)?; debug!( collection = %name, @@ -329,7 +370,7 @@ pub fn deactivate( } // Intentionally preserve the `StoredOwner` row on soft-delete: the // primary record's `owner` field stays populated, and stripping the - // owner row would break `UNDROP COLLECTION`'s ownership restore. + // owner row breaks `UNDROP COLLECTION`'s ownership restore. Ok(()) } @@ -402,7 +443,7 @@ mod tests { let (credentials, _tmp) = open_catalog(); let catalog = credentials.catalog(); - let stored = StoredCollection::new(1, "widgets", "carol"); + let stored = StoredCollection::stamped_for_test(1, "widgets", "carol"); apply_to(&CatalogEntry::PutCollection(Box::new(stored)), catalog) .expect("apply put_collection"); @@ -422,7 +463,7 @@ mod tests { // Set up through `apply_to` so the owner row is written alongside the // primary row, avoiding an orphan-row integrity trip on deactivate. - let stored = StoredCollection::new(1, "archived", "carol"); + let stored = StoredCollection::stamped_for_test(1, "archived", "carol"); apply_to(&CatalogEntry::PutCollection(Box::new(stored)), catalog) .expect("apply put_collection"); @@ -446,9 +487,9 @@ mod tests { } /// DROP COLLECTION is a soft delete, so it must advance the same ordering - /// metadata a CREATE or ALTER would — a replayed CREATE cannot be ordered + /// metadata a CREATE or ALTER does — a replayed CREATE cannot be ordered /// against the current row otherwise, and retention (which reads - /// `modification_hlc` as the drop time) would measure from the original + /// `modification_hlc` as the drop time) measures from the original /// CREATE instead. Drives the entry through the exact production path: /// `descriptor_stamp::stamp` then `apply_to`. #[test] @@ -462,7 +503,8 @@ mod tests { CatalogEntry::PutCollection(Box::new(stored)), &clock, catalog, - ); + ) + .expect("stamp"); let CatalogEntry::PutCollection(created) = &create else { panic!("expected PutCollection"); }; @@ -480,7 +522,8 @@ mod tests { }, &clock, catalog, - ); + ) + .expect("stamp"); apply_to(&deactivate, catalog).expect("apply deactivate_collection"); let loaded = catalog @@ -500,9 +543,9 @@ mod tests { } /// `deactivate()` must stamp `deactivated_at_ns` from the same - /// `modification_hlc.wall_ns` it stamps on the row — this is the field - /// `resolve_retention` reads instead of `modification_hlc`, so a mismatch - /// here reproduces the pre-fix bug one field over. + /// `modification_hlc.wall_ns` it stamps on the row. `resolve_retention` + /// reads this field instead of `modification_hlc`, so a mismatch here + /// skews retention. #[test] fn apply_deactivate_collection_stamps_deactivated_at_ns_from_modification_hlc() { let (credentials, _tmp) = open_catalog(); @@ -514,7 +557,8 @@ mod tests { CatalogEntry::PutCollection(Box::new(stored)), &clock, catalog, - ); + ) + .expect("stamp"); apply_to(&create, catalog).expect("apply put_collection"); let deactivate = stamp( @@ -527,7 +571,8 @@ mod tests { }, &clock, catalog, - ); + ) + .expect("stamp"); apply_to(&deactivate, catalog).expect("apply deactivate_collection"); let loaded = catalog @@ -548,8 +593,8 @@ mod tests { fn purge_collection_is_scoped_to_database() { let (credentials, _tmp) = open_catalog(); let catalog = credentials.catalog(); - let default = StoredCollection::new(1, "shared", "default_owner"); - let mut other = StoredCollection::new(1, "shared", "other_owner"); + let default = StoredCollection::stamped_for_test(1, "shared", "default_owner"); + let mut other = StoredCollection::stamped_for_test(1, "shared", "other_owner"); other.database_id = DatabaseId::new(9); apply_to(&CatalogEntry::PutCollection(Box::new(default)), catalog) .expect("apply put_collection"); @@ -563,6 +608,8 @@ mod tests { database_id: 9, tenant_id: 1, name: "shared".into(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }, catalog, ) @@ -594,7 +641,7 @@ mod tests { let catalog = credentials.catalog(); catalog.fail_next_collection_write_for_test(); - let stored = StoredCollection::new(1, "wedged", "carol"); + let stored = StoredCollection::stamped_for_test(1, "wedged", "carol"); let error = apply_to(&CatalogEntry::PutCollection(Box::new(stored)), catalog) .expect_err("a failed catalog write must raise, not be swallowed"); @@ -634,7 +681,7 @@ mod tests { fn prepare_purge_reports_the_row_it_deactivated() { let (credentials, _tmp) = open_catalog(); let catalog = credentials.catalog(); - let stored = StoredCollection::new(1, "orders", "tester"); + let stored = StoredCollection::stamped_for_test(1, "orders", "tester"); apply_to(&CatalogEntry::PutCollection(Box::new(stored)), catalog).expect("apply"); let found = prepare_purge(0, 1, "orders", catalog).expect("prepare purge"); @@ -648,7 +695,7 @@ mod tests { ); } - /// The replicated applier runs on nodes that may never have held the row, so + /// The replicated applier runs on nodes that can lack the row, so /// `prepare_purge` reports the miss instead of raising. #[test] fn prepare_purge_reports_a_missing_row_without_raising() { @@ -665,7 +712,7 @@ mod tests { fn prepare_purge_checked_rejects_a_database_id_that_holds_no_row() { let (credentials, _tmp) = open_catalog(); let catalog = credentials.catalog(); - let stored = StoredCollection::new(1, "orders", "tester"); + let stored = StoredCollection::stamped_for_test(1, "orders", "tester"); apply_to(&CatalogEntry::PutCollection(Box::new(stored)), catalog).expect("apply"); let err = prepare_purge_checked(1024, 1, "orders", catalog) @@ -692,7 +739,7 @@ mod tests { fn prepare_purge_checked_accepts_the_database_that_holds_the_row() { let (credentials, _tmp) = open_catalog(); let catalog = credentials.catalog(); - let mut stored = StoredCollection::new(1, "orders", "tester"); + let mut stored = StoredCollection::stamped_for_test(1, "orders", "tester"); stored.database_id = DatabaseId::new(1024); apply_to(&CatalogEntry::PutCollection(Box::new(stored)), catalog).expect("apply"); diff --git a/nodedb/src/control/catalog_entry/apply/consumer_group.rs b/nodedb/src/control/catalog_entry/apply/consumer_group.rs index ef00367e3..06ec0fc4f 100644 --- a/nodedb/src/control/catalog_entry/apply/consumer_group.rs +++ b/nodedb/src/control/catalog_entry/apply/consumer_group.rs @@ -91,6 +91,7 @@ mod tests { stream_name: STREAM.to_string(), owner: "admin".to_string(), created_at: 1_000, + modification_hlc: nodedb_types::Hlc::ZERO, } } @@ -100,6 +101,7 @@ mod tests { tenant_id: TENANT, stream_name: STREAM.to_string(), name: GROUP.to_string(), + target_hlc: nodedb_types::Hlc::new(7, 0), } } @@ -129,11 +131,13 @@ mod tests { tenant_id, stream_name, name, + target_hlc, } => { assert_eq!(database_id, DB); assert_eq!(tenant_id, TENANT); assert_eq!(stream_name, STREAM); assert_eq!(name, GROUP); + assert_eq!(target_hlc, nodedb_types::Hlc::new(7, 0)); } other => panic!("unexpected variant: {}", other.kind()), } diff --git a/nodedb/src/control/catalog_entry/apply/continuous_aggregate.rs b/nodedb/src/control/catalog_entry/apply/continuous_aggregate.rs index 78bb7abd5..aa5110261 100644 --- a/nodedb/src/control/catalog_entry/apply/continuous_aggregate.rs +++ b/nodedb/src/control/catalog_entry/apply/continuous_aggregate.rs @@ -104,6 +104,8 @@ mod tests { database_id: 9, tenant_id: 1, name: "shared".into(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }, catalog, ) diff --git a/nodedb/src/control/catalog_entry/apply/database.rs b/nodedb/src/control/catalog_entry/apply/database.rs index 9e1d7e4ac..5b23d6317 100644 --- a/nodedb/src/control/catalog_entry/apply/database.rs +++ b/nodedb/src/control/catalog_entry/apply/database.rs @@ -3,9 +3,9 @@ //! Apply database catalog entries to `SystemCatalog` redb. use crate::control::security::catalog::auth_types::object_type; -use crate::control::security::catalog::database_types::DatabaseDescriptor; +use crate::control::security::catalog::database_types::{DatabaseDescriptor, ParentCloneRef}; use crate::control::security::catalog::{SystemCatalog, catalog_err}; -use nodedb_types::DatabaseId; +use nodedb_types::{DatabaseId, Hlc}; /// Apply a `PutDatabase` entry — upsert the descriptor into /// `_system.databases` and `_system.databases_by_name`. @@ -23,11 +23,22 @@ pub fn put(descriptor: &DatabaseDescriptor, catalog: &SystemCatalog) -> crate::R } /// Apply a `DeleteDatabase` entry — remove the descriptor, its -/// reverse-lookup row, and the quota rows of the dropped scope. +/// reverse-lookup row, the quota rows of the dropped scope, and its mirror +/// collection map and lag rows. pub fn delete(db_id: u64, catalog: &SystemCatalog) -> crate::Result<()> { + let id = DatabaseId::new(db_id); catalog - .delete_database(DatabaseId::new(db_id)) + .delete_database(id) .map_err(|e| catalog_err(&format!("delete_database (database {db_id})"), e))?; + catalog.delete_mirror_collection_map(id).map_err(|e| { + catalog_err( + &format!("delete_mirror_collection_map (database {db_id})"), + e, + ) + })?; + catalog + .delete_mirror_lag(id) + .map_err(|e| catalog_err(&format!("delete_mirror_lag (database {db_id})"), e))?; // A stale quota row keeps consuming the sum-of-quotas ceiling. super::quota::purge_database_scope(db_id, catalog) } @@ -56,12 +67,30 @@ pub fn put_grant( /// /// Every step raises on failure: a half-stamped clone answers queries this /// node's peers answer differently. +/// +/// The lineage edge is written last. `descriptor_validate` reads it as the +/// mark of a completed clone, so a replay after an interrupted apply re-runs +/// the whole clone. +/// +/// Every shadow takes `incarnation`, the fresh one the proposer stamped. pub fn clone_apply( target_descriptor: &DatabaseDescriptor, source_db_id: u64, + incarnation: Hlc, catalog: &SystemCatalog, ) -> crate::Result<()> { + if incarnation == Hlc::ZERO { + return Err(crate::Error::Internal { + detail: format!( + "clone_database '{}' (database {}) carries no shadow incarnation; \ + the proposer stamps every clone entry", + target_descriptor.name, + target_descriptor.id.as_u64() + ), + }); + } let child = target_descriptor.id; + let source = DatabaseId::new(source_db_id); catalog.put_database(target_descriptor).map_err(|e| { catalog_err( &format!( @@ -72,7 +101,15 @@ pub fn clone_apply( e, ) })?; - let source = DatabaseId::new(source_db_id); + if let Some(parent_clone) = &target_descriptor.parent_clone { + stamp_shadows( + target_descriptor, + parent_clone, + source, + incarnation, + catalog, + )?; + } catalog.add_clone_child(source, child).map_err(|e| { catalog_err( &format!( @@ -81,15 +118,20 @@ pub fn clone_apply( ), e, ) - })?; + }) +} - // Determine the as_of and clone_created_at LSN values from the target - // descriptor's parent_clone reference. - let Some(parent_clone) = &target_descriptor.parent_clone else { - // No parent clone ref — nothing to stamp. Descriptor was written - // above; non-clone databases are complete. - return Ok(()); - }; +/// Write a shadow descriptor for every active source collection into the +/// child, then copy the source's other database-scoped catalog rows. +fn stamp_shadows( + target_descriptor: &DatabaseDescriptor, + parent_clone: &ParentCloneRef, + source: DatabaseId, + incarnation: Hlc, + catalog: &SystemCatalog, +) -> crate::Result<()> { + let child = target_descriptor.id; + let source_db_id = source.as_u64(); let as_of_lsn = nodedb_types::Lsn::new(parent_clone.as_of_lsn); let clone_created_at = nodedb_types::Lsn::new(target_descriptor.created_at_lsn); let kv_surrogate_ceiling = parent_clone.kv_surrogate_ceiling; @@ -122,8 +164,11 @@ pub fn clone_apply( kv_surrogate_ceiling, }); coll.clone_status = nodedb_types::CloneStatus::Shadowed; - // Reset versioning so the new clone descriptor starts fresh. - coll.descriptor_version = 0; + // A shadow is a new collection under a new key: its first version, and + // an incarnation of its own, identical on every replica. + coll.descriptor_version = 1; + coll.incarnation = incarnation; + coll.modification_hlc = incarnation; catalog.put_collection(child, &coll).map_err(|e| { catalog_err( &format!( @@ -177,6 +222,9 @@ mod tests { use crate::control::security::catalog::database_types::{DatabaseStatus, ParentCloneRef}; use crate::control::security::credential::store::CredentialStore; + /// The incarnation the proposer stamps on the test clone. + const SHADOW: Hlc = Hlc::new(1_000_000, 0); + fn open_catalog() -> (Arc, tempfile::TempDir) { let tmp = tempfile::tempdir().expect("tmpdir"); let store = Arc::new( @@ -205,7 +253,7 @@ mod tests { } /// A shadow-stamp failure aborts the whole clone. Finishing the remaining - /// collections would leave this node answering queries its peers cannot. + /// collections leaves this node answering queries its peers cannot. #[test] fn clone_apply_raises_instead_of_stamping_the_rest() { let (credentials, _tmp) = open_catalog(); @@ -213,15 +261,20 @@ mod tests { let source = DatabaseId::new(1); let child = DatabaseId::new(2); for name in ["orders", "invoices"] { - let mut coll = StoredCollection::new(5, name, "cloner"); + let mut coll = StoredCollection::stamped_for_test(5, name, "cloner"); coll.database_id = source; apply_to(&CatalogEntry::PutCollection(Box::new(coll)), catalog) .expect("seed source collection"); } catalog.fail_next_collection_write_for_test(); - let error = clone_apply(&clone_descriptor(source, child), source.as_u64(), catalog) - .expect_err("a failed shadow stamp must raise"); + let error = clone_apply( + &clone_descriptor(source, child), + source.as_u64(), + SHADOW, + catalog, + ) + .expect_err("a failed shadow stamp must raise"); assert!(error.to_string().contains("clone_database"), "{error}"); let stamped = catalog.load_all_collections(child).expect("load target"); @@ -229,5 +282,107 @@ mod tests { stamped.is_empty(), "a raised clone leaves no partially stamped target: {stamped:?}" ); + // No lineage edge, so a replay re-runs the clone instead of skipping it. + assert!( + catalog + .get_clone_children(source) + .expect("read lineage") + .is_empty() + ); + } + + /// A completed clone leaves the lineage edge that marks it applied. + #[test] + fn clone_apply_writes_the_lineage_edge() { + let (credentials, _tmp) = open_catalog(); + let catalog = credentials.catalog(); + let source = DatabaseId::new(1); + let child = DatabaseId::new(2); + clone_apply( + &clone_descriptor(source, child), + source.as_u64(), + SHADOW, + catalog, + ) + .expect("clone applies"); + assert_eq!( + catalog.get_clone_children(source).expect("read lineage"), + vec![child] + ); + } + + /// A shadow is a new collection: it takes the clone's own incarnation and + /// its first descriptor version, never the source's. + #[test] + fn a_shadow_takes_a_fresh_incarnation() { + let (credentials, _tmp) = open_catalog(); + let catalog = credentials.catalog(); + let source = DatabaseId::new(1); + let child = DatabaseId::new(2); + let mut coll = StoredCollection::stamped_for_test(5, "orders", "cloner"); + coll.database_id = source; + catalog.put_collection(source, &coll).expect("seed source"); + let source_row = catalog + .get_committed_collection(source, 5, "orders") + .expect("read source") + .expect("source row"); + + clone_apply( + &clone_descriptor(source, child), + source.as_u64(), + SHADOW, + catalog, + ) + .expect("clone applies"); + + let shadow = catalog + .get_committed_collection(child, 5, "orders") + .expect("read shadow") + .expect("shadow row"); + assert_eq!(shadow.incarnation, SHADOW); + assert_ne!(shadow.incarnation, source_row.incarnation); + assert_eq!(shadow.descriptor_version, 1); + } + + /// An unstamped clone entry is refused before it writes anything. + #[test] + fn an_unstamped_clone_is_refused() { + let (credentials, _tmp) = open_catalog(); + let catalog = credentials.catalog(); + let source = DatabaseId::new(1); + let child = DatabaseId::new(2); + assert!( + clone_apply( + &clone_descriptor(source, child), + source.as_u64(), + Hlc::ZERO, + catalog, + ) + .is_err() + ); + assert!(catalog.get_database(child).expect("read").is_none()); + } + + /// `DeleteDatabase` removes the mirror rows on every node that applies + /// it, and a replay of it is a no-op. + #[test] + fn delete_removes_mirror_rows_and_replays_cleanly() { + let (credentials, _tmp) = open_catalog(); + let catalog = credentials.catalog(); + let db = DatabaseId::new(1030); + catalog + .apply_ddl_entry_atomic(db, nodedb_types::Lsn::new(5), 7, "src", "local") + .expect("seed mirror rows"); + let entry = CatalogEntry::DeleteDatabase { db_id: db.as_u64() }; + for _ in 0..2 { + apply_to(&entry, catalog).expect("delete database applies"); + assert!(catalog.get_mirror_lag(db).expect("read lag").is_none()); + assert!( + catalog + .get_mirror_collection_mapping(db, "src") + .expect("read map") + .is_none() + ); + } } } diff --git a/nodedb/src/control/catalog_entry/apply/dispatch.rs b/nodedb/src/control/catalog_entry/apply/dispatch.rs index b89cbb37e..0f0a41711 100644 --- a/nodedb/src/control/catalog_entry/apply/dispatch.rs +++ b/nodedb/src/control/catalog_entry/apply/dispatch.rs @@ -11,10 +11,10 @@ use crate::control::security::catalog::types::CheckpointDoc; use super::outcome::ApplyOutcome; use super::{ - alert_rule, api_key, auth_user, change_stream, checkpoint, collection, column_stats, - consumer_group, continuous_aggregate, custom_type, database, function, index_registry, - materialized_view, oidc_provider, owner, permission, procedure, quota, redaction, - retention_policy, rls, role, schedule, scope_grant, scope_quota, sequence, + alert_rule, api_key, array, auth_user, backup_schedule, change_stream, checkpoint, clone_cow, + collection, column_stats, consumer_group, continuous_aggregate, custom_type, database, + function, index_registry, materialized_view, oidc_provider, owner, permission, procedure, + quota, redaction, retention_policy, rls, role, schedule, scope_grant, scope_quota, sequence, streaming_materialized_view, synonym_group, tenant, topic, trigger, user, vector, wal_tombstone, }; @@ -27,7 +27,7 @@ use super::{ /// [`ApplyOutcome::Refused`] reports an entry that breaks a role rule at its /// log position; every node refuses it alike. Debug builds verify /// referential integrity after every apply — release-gated because a full -/// rescan would wedge `raft_tick_loop` on a node with a pre-existing orphan. +/// rescan wedges `raft_tick_loop` on a node with a pre-existing orphan. pub fn apply_to( entry: &CatalogEntry, catalog: &SystemCatalog, @@ -62,10 +62,7 @@ pub fn apply_to( .into_iter() .filter(|d| matches!(d.kind, DivergenceKind::OrphanRow { .. })) .collect(); - if let Some(first) = orphans.first() { - let DivergenceKind::OrphanRow { kind, .. } = &first.kind else { - unreachable!("filtered to OrphanRow above"); - }; + if let Some(DivergenceKind::OrphanRow { kind, .. }) = orphans.first().map(|d| &d.kind) { crate::diag::catalog_apply_orphan_row(entry.kind(), kind, orphans.len()); return Err(crate::Error::CatalogIntegrityViolation { entry_kind: entry.kind().to_string(), @@ -103,29 +100,27 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul database_id, tenant_id, name, + .. } => { // Preserve an inactive row until post-apply storage reclaim // succeeds — the restart-durable same-name lifecycle barrier. - match collection::prepare_purge(*database_id, *tenant_id, name, catalog) { - // A node that never held the row has nothing to fence. - // Interactive purge paths use `prepare_purge_checked` instead. - Ok(found) => { - debug!( - collection = %name, - tenant = *tenant_id, - found, - "catalog_entry: purge preparation" - ); - Ok(()) - } - Err(error) => panic!("collection catalog purge preparation failed: {error}"), - } + // A node that never held the row has nothing to fence. + // Interactive purge paths use `prepare_purge_checked` instead. + let found = collection::prepare_purge(*database_id, *tenant_id, name, catalog)?; + debug!( + collection = %name, + tenant = *tenant_id, + found, + "catalog_entry: purge preparation" + ); + Ok(()) } CatalogEntry::PutSequence(stored) => sequence::put(stored, catalog), CatalogEntry::DeleteSequence { database_id, tenant_id, name, + .. } => sequence::delete(*database_id, *tenant_id, name, catalog), CatalogEntry::PutSequenceState(state) => sequence::put_state(state, catalog), CatalogEntry::PutTrigger(stored) => trigger::put(stored, catalog), @@ -133,18 +128,21 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul database_id, tenant_id, name, + .. } => trigger::delete(*database_id, *tenant_id, name, catalog), CatalogEntry::PutFunction(stored) => function::put(stored, catalog), CatalogEntry::DeleteFunction { database_id, tenant_id, name, + .. } => function::delete(*database_id, *tenant_id, name, catalog), CatalogEntry::PutProcedure(stored) => procedure::put(stored, catalog), CatalogEntry::DeleteProcedure { database_id, tenant_id, name, + .. } => procedure::delete(*database_id, *tenant_id, name, catalog), CatalogEntry::PutSchedule(stored) => schedule::put(stored, catalog), CatalogEntry::DeleteSchedule { @@ -157,6 +155,7 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul database_id, tenant_id, name, + .. } => change_stream::delete(*database_id, *tenant_id, name, catalog), // Applied by `apply_to`, which reports a refused entry. CatalogEntry::PutUser(_) => Ok(()), @@ -173,10 +172,8 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul database_id, tenant_id, name, - } => match materialized_view::delete(*database_id, *tenant_id, name, catalog) { - Ok(()) => Ok(()), - Err(error) => panic!("materialized-view catalog deletion failed: {error}"), - }, + .. + } => materialized_view::delete(*database_id, *tenant_id, name, catalog), CatalogEntry::PutStreamingMaterializedView(definition) => { streaming_materialized_view::put(definition, catalog) } @@ -184,15 +181,13 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul database_id, tenant_id, name, - } => match streaming_materialized_view::delete(*database_id, *tenant_id, name, catalog) { - Ok(()) => Ok(()), - Err(error) => panic!("streaming materialized-view catalog deletion failed: {error}"), - }, + } => streaming_materialized_view::delete(*database_id, *tenant_id, name, catalog), CatalogEntry::PutContinuousAggregate(stored) => continuous_aggregate::put(stored, catalog), CatalogEntry::DeleteContinuousAggregate { database_id, tenant_id, name, + .. } => continuous_aggregate::delete(*database_id, *tenant_id, name, catalog), CatalogEntry::PutTenant(stored) => tenant::put(stored, catalog), // Applied by `apply_to` so its commit outcome can suppress post-apply. @@ -241,6 +236,7 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul database_id, tenant_id, name, + .. } => synonym_group::delete(*database_id, *tenant_id, name, catalog), CatalogEntry::PutCustomType(stored) => custom_type::put(stored, catalog), CatalogEntry::DeleteCustomType { @@ -263,7 +259,8 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul CatalogEntry::CloneDatabase { target_descriptor, source_db_id, - } => database::clone_apply(target_descriptor, *source_db_id, catalog), + incarnation, + } => database::clone_apply(target_descriptor, *source_db_id, *incarnation, catalog), CatalogEntry::PutOidcProvider(provider) => oidc_provider::put(provider, catalog), CatalogEntry::DeleteOidcProvider { name } => oidc_provider::delete(name, catalog), CatalogEntry::RecordWalTombstone { @@ -304,6 +301,7 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul database_id, tenant_id, name, + .. } => topic::delete_with_consumer_groups(*database_id, *tenant_id, name, catalog), CatalogEntry::PutConsumerGroupIfAbsent(def) => consumer_group::put_if_absent(def, catalog), CatalogEntry::DeleteConsumerGroup { @@ -311,6 +309,7 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul tenant_id, stream_name, name, + .. } => consumer_group::delete(*database_id, *tenant_id, stream_name, name, catalog), CatalogEntry::MigrateConsumerGroupStream { def, legacy_stream } => { consumer_group::migrate_stream(def, legacy_stream, catalog) @@ -327,18 +326,17 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul checkpoint_name, catalog, ), - // `target_version_json` drives the post-apply compaction, not the row - // delete. CatalogEntry::CompactHistory { tenant_id, database_id, collection, doc_id, before_timestamp, - .. - } => checkpoint::delete_before( + target_version_json, + } => checkpoint::compact_history( CheckpointDoc::new(*database_id, *tenant_id, collection, doc_id), *before_timestamp, + target_version_json, catalog, ), CatalogEntry::PutVectorModel(entry) => vector::put_model(entry, catalog), @@ -354,8 +352,61 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul tenant_id, collection, field_name, + .. } => vector::delete_params(*database_id, *tenant_id, collection, field_name, catalog), CatalogEntry::PutColumnStats(rows) => column_stats::put_rows(rows, catalog), + CatalogEntry::PutCloneCopyup { + database_id, + tenant_id, + collection, + source_surrogate, + target_surrogate, + } => clone_cow::put_copyup( + clone_target(*database_id, *tenant_id, collection), + *source_surrogate, + *target_surrogate, + catalog, + ), + CatalogEntry::PutCloneTombstone { + database_id, + tenant_id, + collection, + source_surrogate, + } => clone_cow::put_tombstone( + clone_target(*database_id, *tenant_id, collection), + *source_surrogate, + catalog, + ), + CatalogEntry::PutKvCloneTombstone { + database_id, + tenant_id, + collection, + kv_key, + } => clone_cow::put_kv_tombstone( + clone_target(*database_id, *tenant_id, collection), + kv_key, + catalog, + ), + CatalogEntry::PutArray(stored) => array::put(stored, catalog), + CatalogEntry::DeleteArray { + database_id, + tenant_id, + name, + moved_to, + .. + } => array::delete( + *database_id, + *tenant_id, + name, + moved_to.map(|m| m.target_db_id), + catalog, + ), + CatalogEntry::PutCloneSourceDrain(row) => catalog.put_clone_source_drain(row), + CatalogEntry::DeleteCloneSourceDrain { + clone_database, + tenant_id, + clone_collection, + } => catalog.delete_clone_source_drain(*clone_database, *tenant_id, clone_collection), CatalogEntry::MoveTenantCutover { tenant_id, source_db_id, @@ -368,5 +419,18 @@ fn apply_to_inner(entry: &CatalogEntry, catalog: &SystemCatalog) -> crate::Resul collections, catalog, ), + // Offsets live in the Event-Plane offset store, not the catalog. + // The post-apply writes them. + CatalogEntry::CommitConsumerOffsets(_) => Ok(()), + CatalogEntry::PutBackupScheduleMark(mark) => backup_schedule::raise_mark(mark, catalog), + } +} + +/// The clone collection a copy-on-write entry names. +fn clone_target(database_id: u64, tenant_id: u64, collection: &str) -> clone_cow::CloneTarget<'_> { + clone_cow::CloneTarget { + database_id, + tenant_id, + collection, } } diff --git a/nodedb/src/control/catalog_entry/apply/local.rs b/nodedb/src/control/catalog_entry/apply/local.rs deleted file mode 100644 index e1658656b..000000000 --- a/nodedb/src/control/catalog_entry/apply/local.rs +++ /dev/null @@ -1,48 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Single-node DDL catch-up shim. -//! -//! When `metadata_proposer::propose_catalog_entry` reports -//! [`ProposeOutcome::LocalOnly`] — single node or rolling-upgrade compat -//! mode — no Raft applier will run on this node, and the originating -//! handler is solely responsible for landing the catalog write. A -//! `Buffered` entry belongs to an open transaction and applies nothing. -//! -//! [`apply_locally_if_needed`] is the only place that short-circuit -//! happens. It routes through [`apply_to`], whose per-family appliers -//! pair every primary write with the matching owner write, so the -//! orphan-row class is unrepresentable by construction. - -use tracing::warn; - -use crate::control::catalog_entry::CatalogEntry; -use crate::control::propose_outcome::ProposeOutcome; -use crate::control::state::SharedState; - -use super::apply_to; - -/// Apply `entry` locally only for [`ProposeOutcome::LocalOnly`], so the -/// originating node's redb catalog reflects the DDL. No-op for -/// `Replicated` (the Raft applier has run, or will) and for `Buffered` -/// (the open transaction owns the entry until COMMIT). -/// -/// Always returns, whether the apply succeeded or not. Family handlers -/// raise on a redb error; this local-only caller is the one that logs and -/// continues, because the startup integrity repair in -/// `recovery_check::verify_and_repair` reconciles the row on the next boot. -/// A debug-mode orphan-row violation from [`apply_to`] is logged here for -/// the same reason, and `apply_to` files its `faultbox` report at the point -/// of detection, so the failure is never silently lost. -pub fn apply_locally_if_needed(state: &SharedState, entry: &CatalogEntry, outcome: ProposeOutcome) { - if !outcome.needs_local_apply() { - return; - } - let catalog = state.credentials.catalog(); - if let Err(e) = apply_to(entry, catalog) { - warn!( - kind = entry.kind(), - error = %e, - "catalog_entry: apply_locally_if_needed: apply_to failed" - ); - } -} diff --git a/nodedb/src/control/catalog_entry/apply/mod.rs b/nodedb/src/control/catalog_entry/apply/mod.rs index 2a2380265..0055e017a 100644 --- a/nodedb/src/control/catalog_entry/apply/mod.rs +++ b/nodedb/src/control/catalog_entry/apply/mod.rs @@ -8,9 +8,12 @@ pub mod alert_rule; pub mod api_key; +pub mod array; pub mod auth_user; +pub mod backup_schedule; pub mod change_stream; pub mod checkpoint; +pub mod clone_cow; pub mod collection; pub mod column_stats; pub mod consumer_group; @@ -20,7 +23,6 @@ pub mod database; mod dispatch; pub mod function; pub mod index_registry; -pub mod local; pub mod materialized_view; pub mod oidc_provider; pub mod outcome; diff --git a/nodedb/src/control/catalog_entry/apply/sequence.rs b/nodedb/src/control/catalog_entry/apply/sequence.rs index c5bfc10bf..d867d230b 100644 --- a/nodedb/src/control/catalog_entry/apply/sequence.rs +++ b/nodedb/src/control/catalog_entry/apply/sequence.rs @@ -99,6 +99,8 @@ mod tests { database_id: 3, tenant_id: 42, name: "gone".into(), + target_descriptor_version: 4, + target_hlc: nodedb_types::Hlc::new(7, 1), }; let bytes = encode(&entry).unwrap(); match decode(&bytes).unwrap() { @@ -106,7 +108,11 @@ mod tests { database_id, tenant_id, name, + target_descriptor_version, + target_hlc, } => { + assert_eq!(target_descriptor_version, 4); + assert_eq!(target_hlc, nodedb_types::Hlc::new(7, 1)); assert_eq!(database_id, 3); assert_eq!(tenant_id, 42); assert_eq!(name, "gone"); @@ -134,6 +140,8 @@ mod tests { database_id: 3, tenant_id: 1, name: "orders_id_seq".into(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }, catalog, ) @@ -183,6 +191,8 @@ mod tests { database_id: 1, tenant_id: 7, name: "shared_seq".into(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }, catalog, ) diff --git a/nodedb/src/control/catalog_entry/apply/tenant.rs b/nodedb/src/control/catalog_entry/apply/tenant.rs index 094f146cc..e64132339 100644 --- a/nodedb/src/control/catalog_entry/apply/tenant.rs +++ b/nodedb/src/control/catalog_entry/apply/tenant.rs @@ -205,9 +205,9 @@ mod tests { let (credentials, _tmp) = open_catalog(); let catalog = credentials.catalog(); let src = DatabaseId::new(1); - let mut first = StoredCollection::new(5, "orders", "mover"); + let mut first = StoredCollection::stamped_for_test(5, "orders", "mover"); first.database_id = src; - let mut second = StoredCollection::new(5, "invoices", "mover"); + let mut second = StoredCollection::stamped_for_test(5, "invoices", "mover"); second.database_id = src; for coll in [&first, &second] { apply_to( diff --git a/nodedb/src/control/catalog_entry/apply/topic.rs b/nodedb/src/control/catalog_entry/apply/topic.rs index 056e01d8c..576778bd2 100644 --- a/nodedb/src/control/catalog_entry/apply/topic.rs +++ b/nodedb/src/control/catalog_entry/apply/topic.rs @@ -4,7 +4,7 @@ //! //! Writes only. The leader checks the topic name and the duplicate before //! proposing, so apply runs the unvalidated catalog path: a rejection here -//! would leave followers without a topic the leader already accepted. +//! leaves followers without a topic the leader already accepted. use crate::control::security::catalog::{SystemCatalog, catalog_err}; use crate::event::topic::TopicDef; @@ -70,6 +70,8 @@ mod tests { created_at: 1_000, last_sequence: 0, last_lsn: 0, + last_epoch: 0, + modification_hlc: nodedb_types::Hlc::ZERO, } } @@ -78,6 +80,7 @@ mod tests { database_id: DB, tenant_id: TENANT, name: NAME.to_string(), + target_hlc: nodedb_types::Hlc::new(7, 1), } } @@ -106,7 +109,9 @@ mod tests { database_id, tenant_id, name, + target_hlc, } => { + assert_eq!(target_hlc, nodedb_types::Hlc::new(7, 1)); assert_eq!(database_id, DB); assert_eq!(tenant_id, TENANT); assert_eq!(name, NAME); @@ -164,6 +169,7 @@ mod tests { stream_name: format!("topic:{NAME}"), owner: "admin".to_string(), created_at: 1_000, + modification_hlc: nodedb_types::Hlc::ZERO, }) .expect("seed group"); diff --git a/nodedb/src/control/catalog_entry/apply/vector.rs b/nodedb/src/control/catalog_entry/apply/vector.rs index 9718aabc2..95872642c 100644 --- a/nodedb/src/control/catalog_entry/apply/vector.rs +++ b/nodedb/src/control/catalog_entry/apply/vector.rs @@ -5,7 +5,7 @@ //! //! Writes only. The leader resolves the duplicate index, the missing //! collection, and every build-parameter rule before proposing, so apply runs -//! the unvalidated catalog path: a rejection here would diverge a follower +//! the unvalidated catalog path: a rejection here diverges a follower //! from a statement the leader already accepted. //! //! Both tables describe the same object — a collection's embedding column and @@ -133,6 +133,7 @@ mod tests { pq_m: 0, ivf_cells: 0, ivf_nprobe: 0, + modification_hlc: nodedb_types::Hlc::ZERO, } } @@ -180,6 +181,7 @@ mod tests { tenant_id: TENANT, collection: COLLECTION.to_string(), field_name: FIELD.to_string(), + target_hlc: nodedb_types::Hlc::new(7, 1), }; match decode(&encode(&entry).unwrap()).unwrap() { CatalogEntry::DeleteVectorIndexParams { @@ -187,7 +189,9 @@ mod tests { tenant_id, collection, field_name, + target_hlc, } => { + assert_eq!(target_hlc, nodedb_types::Hlc::new(7, 1)); assert_eq!(database_id, DATABASE); assert_eq!(tenant_id, TENANT); assert_eq!(collection, COLLECTION); @@ -305,6 +309,7 @@ mod tests { tenant_id: TENANT, collection: COLLECTION.to_string(), field_name: FIELD.to_string(), + target_hlc: nodedb_types::Hlc::ZERO, }, &catalog, ) diff --git a/nodedb/src/control/catalog_entry/authorization.rs b/nodedb/src/control/catalog_entry/authorization.rs index 4b9c613cf..087d08fe4 100644 --- a/nodedb/src/control/catalog_entry/authorization.rs +++ b/nodedb/src/control/catalog_entry/authorization.rs @@ -2,7 +2,7 @@ //! Which catalog entries change authorization state. //! -//! An entry that changes who may do what, or what a statement may see, is +//! An entry that changes who can do what, or what a statement can see, is //! acknowledged only after it binds every node (see the authorization //! lease). The match is exhaustive, so a new entry kind is classified when //! it is added. @@ -96,7 +96,16 @@ impl CatalogEntry { | Self::DeleteVectorModel { .. } | Self::PutVectorIndexParams(_) | Self::PutColumnStats(_) - | Self::DeleteVectorIndexParams { .. } => false, + | Self::PutCloneCopyup { .. } + | Self::PutCloneTombstone { .. } + | Self::PutKvCloneTombstone { .. } + | Self::PutCloneSourceDrain(_) + | Self::DeleteCloneSourceDrain { .. } + | Self::PutArray(_) + | Self::DeleteArray { .. } + | Self::DeleteVectorIndexParams { .. } + | Self::CommitConsumerOffsets(_) + | Self::PutBackupScheduleMark(_) => false, } } } @@ -117,6 +126,8 @@ mod tests { database_id: 0, tenant_id: 1, name: "c".into(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, } .bears_authorization() ); diff --git a/nodedb/src/control/catalog_entry/descriptor_stamp.rs b/nodedb/src/control/catalog_entry/descriptor_stamp.rs index b4490d15c..4d5a367c1 100644 --- a/nodedb/src/control/catalog_entry/descriptor_stamp.rs +++ b/nodedb/src/control/catalog_entry/descriptor_stamp.rs @@ -13,16 +13,22 @@ //! Every prior read is committed-only. Deriving a version from the //! transaction's own uncommitted DDL overlay stamps one descriptor twice. //! -//! Rolling upgrade: stamping is gated by -//! [`crate::control::rolling_upgrade::DESCRIPTOR_VERSIONING_VERSION`], and the -//! applier skips this helper in compat mode. That gate lives at the call site. +//! Every proposer stamps. There is no unstamped compat path. //! //! Variants without descriptor fields pass through unchanged: `PutUser`, //! `PutRole`, `PutPermission`, `PutOwner`, `PutTenant`, `PutApiKey`, -//! `PutAuthUser`, `PutRlsPolicy`, `PutSchedule`, `PutChangeStream`, -//! `PutSequenceState`, and every `Delete*` / `Purge*` variant. +//! `PutAuthUser`, `PutRlsPolicy`, `PutSchedule`, `PutSequenceState`, and +//! every unfenced `Delete*` variant. //! `DeactivateCollection` does carry them: a soft delete rewrites the row, so it //! consumes a version like any other mutation. +//! +//! A fenced delete (`PurgeCollection` and the `Delete*` of every stored type +//! with an incarnation clock) freezes the target row's incarnation instead. +//! `PutSynonymGroup`, `PutVectorIndexParams`, `CreateTopicIfAbsent`, +//! `PutChangeStream`, `PutConsumerGroupIfAbsent`, +//! `MigrateConsumerGroupStream`, and `PutArray` stamp only `modification_hlc`. See +//! [`super::incarnation`]. +//! `CloneDatabase` stamps the fresh incarnation its shadow collections take. //! The match is exhaustive on [`CatalogEntry`], so a new variant is a compile //! error here. Deciding whether it needs a stamp is deliberate. @@ -38,10 +44,7 @@ fn next_collection_stamp( clock: &HlcClock, hlc: Hlc, ) -> (u64, Hlc) { - let hlc = match prior.map(|c| c.modification_hlc) { - Some(prior_hlc) if prior_hlc >= hlc => clock.update(prior_hlc), - _ => hlc, - }; + let hlc = super::incarnation::stamp::after(prior.map(|c| c.modification_hlc), clock, hlc); let prior_descriptor = prior.map(|c| c.descriptor_version).unwrap_or(0); (prior_descriptor.saturating_add(1), hlc) } @@ -49,17 +52,21 @@ fn next_collection_stamp( /// Read the prior persisted descriptor, assign `descriptor_version = prior + 1` /// (or `1` on create), stamp `modification_hlc = clock.now()`, return the entry. /// -/// Infallible by design: a failed redb read stamps as if the record was absent -/// (version `1`). Version `0` is never emitted — it is strictly the -/// pre-stamping compat-mode sentinel. -pub fn stamp(entry: CatalogEntry, clock: &HlcClock, catalog: &SystemCatalog) -> CatalogEntry { +/// A failed catalog read fails the stamp: a guessed prior emits a +/// version-1 put or an unfenced delete. Version `0` is never emitted. +pub fn stamp( + entry: CatalogEntry, + clock: &HlcClock, + catalog: &SystemCatalog, +) -> crate::Result { let mut hlc = clock.now(); - match entry { + Ok(match entry { CatalogEntry::PutCollection(mut stored) => { - let prior = catalog - .get_committed_collection(stored.database_id, stored.tenant_id, &stored.name) - .ok() - .flatten(); + let prior = catalog.get_committed_collection( + stored.database_id, + stored.tenant_id, + &stored.name, + )?; let (descriptor_version, stamped_hlc) = next_collection_stamp(prior.as_ref(), clock, hlc); stored.descriptor_version = descriptor_version; @@ -80,29 +87,39 @@ pub fn stamp(entry: CatalogEntry, clock: &HlcClock, catalog: &SystemCatalog) -> prior_constraint_version }; stored.modification_hlc = hlc; + stored.incarnation = super::incarnation::stamp::collection_incarnation( + stored.incarnation, + prior.as_ref(), + hlc, + ); CatalogEntry::PutCollection(stored) } CatalogEntry::PutCollectionIfAbsent(mut stored) => { - let prior = catalog - .get_committed_collection(stored.database_id, stored.tenant_id, &stored.name) - .ok() - .flatten(); + let prior = catalog.get_committed_collection( + stored.database_id, + stored.tenant_id, + &stored.name, + )?; // Existing is a semantic no-op: freeze the exact persisted record, so // replay stays payload-identical and later batch entries see the prior. if let Some(prior) = prior { - return CatalogEntry::PutCollectionIfAbsent(Box::new(prior)); + return Ok(CatalogEntry::PutCollectionIfAbsent(Box::new(prior))); } stored.descriptor_version = 1; let new_set = crate::control::security::catalog::collection_constraints(&stored); stored.constraint_version = u64::from(!new_set.is_empty()); stored.modification_hlc = hlc; + stored.incarnation = + super::incarnation::stamp::collection_incarnation(stored.incarnation, None, hlc); CatalogEntry::PutCollectionIfAbsent(stored) } CatalogEntry::PutMaterializedView(mut stored) => { let prior = catalog - .get_committed_materialized_view(stored.database_id, stored.tenant_id, &stored.name) - .ok() - .flatten() + .get_committed_materialized_view( + stored.database_id, + stored.tenant_id, + &stored.name, + )? .map(|v| v.descriptor_version) .unwrap_or(0); stored.descriptor_version = prior.saturating_add(1); @@ -115,9 +132,7 @@ pub fn stamp(entry: CatalogEntry, clock: &HlcClock, catalog: &SystemCatalog) -> stored.database_id, stored.tenant_id, &stored.name, - ) - .ok() - .flatten() + )? .map(|f| f.descriptor_version) .unwrap_or(0); stored.descriptor_version = prior.saturating_add(1); @@ -130,9 +145,7 @@ pub fn stamp(entry: CatalogEntry, clock: &HlcClock, catalog: &SystemCatalog) -> stored.database_id, stored.tenant_id, &stored.name, - ) - .ok() - .flatten() + )? .map(|p| p.descriptor_version) .unwrap_or(0); stored.descriptor_version = prior.saturating_add(1); @@ -145,9 +158,7 @@ pub fn stamp(entry: CatalogEntry, clock: &HlcClock, catalog: &SystemCatalog) -> stored.database_id, stored.tenant_id, &stored.name, - ) - .ok() - .flatten() + )? .map(|t| t.descriptor_version) .unwrap_or(0); stored.descriptor_version = prior.saturating_add(1); @@ -156,9 +167,7 @@ pub fn stamp(entry: CatalogEntry, clock: &HlcClock, catalog: &SystemCatalog) -> } CatalogEntry::PutSequence(mut stored) => { let prior = catalog - .get_sequence(stored.database_id, stored.tenant_id, &stored.name) - .ok() - .flatten() + .get_sequence(stored.database_id, stored.tenant_id, &stored.name)? .map(|s| s.descriptor_version) .unwrap_or(0); stored.descriptor_version = prior.saturating_add(1); @@ -167,9 +176,7 @@ pub fn stamp(entry: CatalogEntry, clock: &HlcClock, catalog: &SystemCatalog) -> } CatalogEntry::PutContinuousAggregate(mut stored) => { let prior = catalog - .get_continuous_aggregate(stored.database_id, stored.tenant_id, &stored.name) - .ok() - .flatten() + .get_continuous_aggregate(stored.database_id, stored.tenant_id, &stored.name)? .map(|c| c.descriptor_version) .unwrap_or(0); stored.descriptor_version = prior.saturating_add(1); @@ -183,15 +190,12 @@ pub fn stamp(entry: CatalogEntry, clock: &HlcClock, catalog: &SystemCatalog) -> .. } => { // A soft delete rewrites the row, advancing the same ordering - // metadata a `PutCollection` would. - let prior = catalog - .get_committed_collection( - crate::types::DatabaseId::new(database_id), - tenant_id, - &name, - ) - .ok() - .flatten(); + // metadata a `PutCollection` does. + let prior = catalog.get_committed_collection( + crate::types::DatabaseId::new(database_id), + tenant_id, + &name, + )?; let (descriptor_version, stamped_hlc) = next_collection_stamp(prior.as_ref(), clock, hlc); CatalogEntry::DeactivateCollection { @@ -202,21 +206,39 @@ pub fn stamp(entry: CatalogEntry, clock: &HlcClock, catalog: &SystemCatalog) -> modification_hlc: stamped_hlc, } } - // Variants without descriptor versioning pass through unchanged. + entry @ CatalogEntry::CloneDatabase { .. } => { + super::incarnation::stamp::stamp_clone(entry, hlc) + } entry @ (CatalogEntry::PurgeCollection { .. } + | CatalogEntry::DeleteSequence { .. } + | CatalogEntry::DeleteTrigger { .. } | CatalogEntry::DeleteFunction { .. } | CatalogEntry::DeleteProcedure { .. } - | CatalogEntry::DeleteTrigger { .. } | CatalogEntry::DeleteMaterializedView { .. } - | CatalogEntry::PutStreamingMaterializedView(_) - | CatalogEntry::DeleteStreamingMaterializedView { .. } | CatalogEntry::DeleteContinuousAggregate { .. } - | CatalogEntry::DeleteSequence { .. } + | CatalogEntry::DeleteSynonymGroup { .. } + | CatalogEntry::DeleteTopicWithConsumerGroups { .. } + | CatalogEntry::DeleteVectorIndexParams { .. } + | CatalogEntry::DeleteChangeStream { .. } + | CatalogEntry::DeleteConsumerGroup { .. } + | CatalogEntry::DeleteArray { .. }) => { + super::incarnation::stamp::stamp_delete(entry, catalog)? + } + entry @ (CatalogEntry::PutSynonymGroup(_) + | CatalogEntry::PutVectorIndexParams(_) + | CatalogEntry::CreateTopicIfAbsent(_) + | CatalogEntry::PutChangeStream(_) + | CatalogEntry::PutConsumerGroupIfAbsent(_) + | CatalogEntry::MigrateConsumerGroupStream { .. } + | CatalogEntry::PutArray(_)) => { + super::incarnation::stamp::stamp_put(entry, clock, catalog, hlc)? + } + // Variants without descriptor versioning pass through unchanged. + entry @ (CatalogEntry::PutStreamingMaterializedView(_) + | CatalogEntry::DeleteStreamingMaterializedView { .. } | CatalogEntry::PutSequenceState(_) | CatalogEntry::PutSchedule(_) | CatalogEntry::DeleteSchedule { .. } - | CatalogEntry::PutChangeStream(_) - | CatalogEntry::DeleteChangeStream { .. } | CatalogEntry::PutUser(_) | CatalogEntry::DropUser { .. } | CatalogEntry::PutRole(_) @@ -239,8 +261,6 @@ pub fn stamp(entry: CatalogEntry, clock: &HlcClock, catalog: &SystemCatalog) -> | CatalogEntry::DeleteIndexRecord { .. } | CatalogEntry::PutOwner(_) | CatalogEntry::DeleteOwner { .. } - | CatalogEntry::PutSynonymGroup(_) - | CatalogEntry::DeleteSynonymGroup { .. } | CatalogEntry::PutCustomType(_) | CatalogEntry::DeleteCustomType { .. } | CatalogEntry::PutDatabase(_) @@ -250,7 +270,6 @@ pub fn stamp(entry: CatalogEntry, clock: &HlcClock, catalog: &SystemCatalog) -> | CatalogEntry::PutOidcProvider(_) | CatalogEntry::DeleteOidcProvider { .. } | CatalogEntry::RecordWalTombstone { .. } - | CatalogEntry::CloneDatabase { .. } | CatalogEntry::PutDatabaseQuota { .. } | CatalogEntry::DeleteDatabaseQuota { .. } | CatalogEntry::PutTenantQuota { .. } @@ -261,21 +280,21 @@ pub fn stamp(entry: CatalogEntry, clock: &HlcClock, catalog: &SystemCatalog) -> | CatalogEntry::DeleteRetentionPolicy { .. } | CatalogEntry::PutAlertRule(_) | CatalogEntry::DeleteAlertRule { .. } - | CatalogEntry::CreateTopicIfAbsent(_) - | CatalogEntry::DeleteTopicWithConsumerGroups { .. } - | CatalogEntry::PutConsumerGroupIfAbsent(_) - | CatalogEntry::DeleteConsumerGroup { .. } - | CatalogEntry::MigrateConsumerGroupStream { .. } | CatalogEntry::PutCheckpoint(_) | CatalogEntry::DeleteCheckpoint { .. } | CatalogEntry::CompactHistory { .. } | CatalogEntry::PutVectorModel(_) | CatalogEntry::DeleteVectorModel { .. } - | CatalogEntry::PutVectorIndexParams(_) - | CatalogEntry::DeleteVectorIndexParams { .. } | CatalogEntry::PutColumnStats(_) - | CatalogEntry::MoveTenantCutover { .. }) => entry, - } + | CatalogEntry::PutCloneCopyup { .. } + | CatalogEntry::PutCloneTombstone { .. } + | CatalogEntry::PutKvCloneTombstone { .. } + | CatalogEntry::PutCloneSourceDrain(_) + | CatalogEntry::DeleteCloneSourceDrain { .. } + | CatalogEntry::MoveTenantCutover { .. } + | CatalogEntry::CommitConsumerOffsets(_) + | CatalogEntry::PutBackupScheduleMark(_)) => entry, + }) } /// Stamp a transactional DDL batch in statement order. Persisted catalog state @@ -284,10 +303,10 @@ pub fn stamp_batch( entries: Vec, clock: &HlcClock, catalog: &SystemCatalog, -) -> Vec { +) -> crate::Result> { let mut stamped_entries = Vec::with_capacity(entries.len()); for entry in entries { - let mut stamped = stamp(entry, clock, catalog); + let mut stamped = stamp(entry, clock, catalog)?; if let Some(prior) = stamped_entries .iter() .rev() @@ -295,9 +314,10 @@ pub fn stamp_batch( { stamped = advance_after(prior, stamped); } + let stamped = super::incarnation::stamp::retarget_in_batch(&stamped_entries, stamped); stamped_entries.push(stamped); } - stamped_entries + Ok(stamped_entries) } /// The `(database, tenant, name)` a versioned collection entry mutates. A soft @@ -454,6 +474,11 @@ fn advance_collection( current: &mut crate::control::security::catalog::StoredCollection, ) { current.descriptor_version = prior.descriptor_version.saturating_add(1); + // A later put of a row an earlier entry of the batch leaves active keeps + // that entry's incarnation. + if prior.is_active && prior.incarnation != Hlc::ZERO { + current.incarnation = prior.incarnation; + } let prior_set = crate::control::security::catalog::collection_constraints(prior); let current_set = crate::control::security::catalog::collection_constraints(current); current.constraint_version = if prior_set == current_set { @@ -495,7 +520,7 @@ mod tests { let stored = StoredCollection::new(1, "orders", "tester"); let entry = CatalogEntry::PutCollection(Box::new(stored)); - let stamped = stamp(entry, &clock, catalog); + let stamped = stamp(entry, &clock, catalog).expect("stamp"); let CatalogEntry::PutCollection(boxed) = stamped else { panic!("expected PutCollection"); }; @@ -513,7 +538,7 @@ mod tests { for expected in 1u64..=5 { let stored = StoredCollection::new(1, "orders", "tester"); let entry = CatalogEntry::PutCollection(Box::new(stored)); - let stamped = stamp(entry, &clock, catalog); + let stamped = stamp(entry, &clock, catalog).expect("stamp"); let CatalogEntry::PutCollection(boxed) = stamped else { panic!("expected PutCollection"); }; @@ -535,7 +560,7 @@ mod tests { CatalogEntry::PutCollection(Box::new(StoredCollection::new(1, "orders", "tester"))), CatalogEntry::PutCollection(Box::new(StoredCollection::new(1, "orders", "tester"))), ]; - let stamped = stamp_batch(entries, &clock, store.catalog()); + let stamped = stamp_batch(entries, &clock, store.catalog()).expect("stamp batch"); let CatalogEntry::PutCollection(first) = &stamped[0] else { panic!("expected first collection"); }; @@ -548,7 +573,7 @@ mod tests { /// Persist a collection at `version` so the next stamp reads it as prior. fn seed_prior(catalog: &SystemCatalog, name: &str, version: u64) { - let mut stored = StoredCollection::new(1, name, "tester"); + let mut stored = StoredCollection::stamped_for_test(1, name, "tester"); stored.descriptor_version = version; catalog .put_collection(DatabaseId::DEFAULT, &stored) @@ -666,7 +691,8 @@ mod tests { ], &clock, store.catalog(), - ); + ) + .expect("stamp batch"); let CatalogEntry::PutCollectionIfAbsent(noop) = &stamped[0] else { panic!("expected create-only entry"); }; @@ -690,7 +716,8 @@ mod tests { ], &clock, store.catalog(), - ); + ) + .expect("stamp batch"); let CatalogEntry::PutSequence(first) = &stamped[0] else { panic!("expected first sequence"); }; @@ -701,11 +728,8 @@ mod tests { assert_eq!(second.descriptor_version, 2); } - /// A soft delete (`DeactivateCollection`) must advance the same - /// descriptor version and HLC a `PutCollection` would — it is a mutation - /// of the row, not a pass-through. `stamp_ignores_deletes` previously - /// asserted the opposite (pass-through unchanged), which encoded the - /// bug this test now guards against. + /// A soft delete (`DeactivateCollection`) advances the descriptor version + /// and HLC the same way a `PutCollection` does. It mutates the row. #[test] fn stamp_advances_deactivate_collection_version_and_hlc() { let (store, _tmp) = make_catalog(); @@ -717,7 +741,8 @@ mod tests { CatalogEntry::PutCollection(Box::new(stored)), &clock, catalog, - ); + ) + .expect("stamp"); let CatalogEntry::PutCollection(created) = &create else { panic!("expected PutCollection"); }; @@ -737,7 +762,8 @@ mod tests { }, &clock, catalog, - ); + ) + .expect("stamp"); let CatalogEntry::DeactivateCollection { descriptor_version, modification_hlc, diff --git a/nodedb/src/control/catalog_entry/descriptor_validate.rs b/nodedb/src/control/catalog_entry/descriptor_validate.rs index bb159dd79..bf57e7230 100644 --- a/nodedb/src/control/catalog_entry/descriptor_validate.rs +++ b/nodedb/src/control/catalog_entry/descriptor_validate.rs @@ -1,11 +1,12 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Validate a stamped `Put*` entry against the locally persisted descriptor -//! version before it is applied. +//! Validate a stamped entry against the locally persisted descriptor before it +//! is applied. //! //! The stamping half lives in [`super::descriptor_stamp`]; this module decides //! whether an entry that arrives at the applier is the next version, a -//! historical replay to acknowledge, or a divergence to reject. +//! historical replay to acknowledge, or a divergence to reject. Fenced deletes +//! and unversioned puts decide by incarnation clock in [`super::incarnation`]. use crate::control::catalog_entry::CatalogEntry; use crate::control::security::catalog::SystemCatalog; @@ -22,16 +23,17 @@ pub enum ValidationOutcome { /// remain loud anomalies. A create (carried version 1) carrying a strictly /// newer clock is the one exception: it is a recreate of the same name and /// applies over the incarnation this node still holds. +/// +/// A fenced delete, a soft delete, or an unversioned put that a later +/// incarnation has superseded is acknowledged without applying. pub fn validate( entry: &CatalogEntry, catalog: &SystemCatalog, ) -> Result { match entry { CatalogEntry::PutCollection(stored) => { - let current = catalog - .get_collection(stored.database_id, stored.tenant_id, &stored.name) - .ok() - .flatten(); + let current = + catalog.get_collection(stored.database_id, stored.tenant_id, &stored.name)?; validate_one( &stored.name, stored.descriptor_version, @@ -45,10 +47,8 @@ pub fn validate( ) } CatalogEntry::PutCollectionIfAbsent(stored) => { - let current = catalog - .get_collection(stored.database_id, stored.tenant_id, &stored.name) - .ok() - .flatten(); + let current = + catalog.get_collection(stored.database_id, stored.tenant_id, &stored.name)?; if current.is_some() { Ok(ValidationOutcome::AlreadyApplied) } else { @@ -64,10 +64,11 @@ pub fn validate( } } CatalogEntry::PutMaterializedView(stored) => { - let current = catalog - .get_materialized_view(stored.database_id, stored.tenant_id, &stored.name) - .ok() - .flatten(); + let current = catalog.get_materialized_view( + stored.database_id, + stored.tenant_id, + &stored.name, + )?; validate_one( &stored.name, stored.descriptor_version, @@ -81,10 +82,11 @@ pub fn validate( ) } CatalogEntry::PutFunction(stored) => { - let current = catalog - .get_function(stored.tenant_id, &stored.name) - .ok() - .flatten(); + let current = catalog.get_function_in_database( + stored.database_id, + stored.tenant_id, + &stored.name, + )?; validate_one( &stored.name, stored.descriptor_version, @@ -98,10 +100,11 @@ pub fn validate( ) } CatalogEntry::PutProcedure(stored) => { - let current = catalog - .get_procedure(stored.tenant_id, &stored.name) - .ok() - .flatten(); + let current = catalog.get_procedure_in_database( + stored.database_id, + stored.tenant_id, + &stored.name, + )?; validate_one( &stored.name, stored.descriptor_version, @@ -115,10 +118,11 @@ pub fn validate( ) } CatalogEntry::PutTrigger(stored) => { - let current = catalog - .get_trigger(stored.tenant_id, &stored.name) - .ok() - .flatten(); + let current = catalog.get_trigger_in_database( + stored.database_id, + stored.tenant_id, + &stored.name, + )?; validate_one( &stored.name, stored.descriptor_version, @@ -132,10 +136,8 @@ pub fn validate( ) } CatalogEntry::PutSequence(stored) => { - let current = catalog - .get_sequence(stored.database_id, stored.tenant_id, &stored.name) - .ok() - .flatten(); + let current = + catalog.get_sequence(stored.database_id, stored.tenant_id, &stored.name)?; validate_one( &stored.name, stored.descriptor_version, @@ -149,10 +151,11 @@ pub fn validate( ) } CatalogEntry::PutContinuousAggregate(stored) => { - let current = catalog - .get_continuous_aggregate(stored.database_id, stored.tenant_id, &stored.name) - .ok() - .flatten(); + let current = catalog.get_continuous_aggregate( + stored.database_id, + stored.tenant_id, + &stored.name, + )?; validate_one( &stored.name, stored.descriptor_version, @@ -165,10 +168,55 @@ pub fn validate( .map_or(nodedb_types::Hlc::ZERO, |value| value.modification_hlc), ) } + CatalogEntry::PurgeCollection { .. } + | CatalogEntry::DeleteSequence { .. } + | CatalogEntry::DeleteTrigger { .. } + | CatalogEntry::DeleteFunction { .. } + | CatalogEntry::DeleteProcedure { .. } + | CatalogEntry::DeleteMaterializedView { .. } + | CatalogEntry::DeleteContinuousAggregate { .. } + | CatalogEntry::DeleteSynonymGroup { .. } + | CatalogEntry::DeleteTopicWithConsumerGroups { .. } + | CatalogEntry::DeleteVectorIndexParams { .. } + | CatalogEntry::DeleteChangeStream { .. } + | CatalogEntry::DeleteConsumerGroup { .. } + | CatalogEntry::DeleteArray { .. } => { + super::incarnation::fence::check_delete(entry, catalog) + } + CatalogEntry::DeactivateCollection { .. } + | CatalogEntry::PutSynonymGroup(_) + | CatalogEntry::PutVectorIndexParams(_) + | CatalogEntry::CreateTopicIfAbsent(_) + | CatalogEntry::PutChangeStream(_) + | CatalogEntry::PutConsumerGroupIfAbsent(_) + | CatalogEntry::MigrateConsumerGroupStream { .. } + | CatalogEntry::PutArray(_) => super::incarnation::fence::check_superseded(entry, catalog), + CatalogEntry::CloneDatabase { + target_descriptor, + source_db_id, + .. + } => check_clone_once(target_descriptor.id, *source_db_id, catalog), _ => Ok(ValidationOutcome::Apply), } } +/// A clone applies once. Its lineage edge is the last write of `clone_apply` +/// and outlives a later drop of the child, so the edge marks a completed +/// clone. Re-applying re-reads the source's current collections into the +/// child. +fn check_clone_once( + child: crate::types::DatabaseId, + source_db_id: u64, + catalog: &SystemCatalog, +) -> Result { + let children = catalog.get_clone_children(crate::types::DatabaseId::new(source_db_id))?; + Ok(if children.contains(&child) { + ValidationOutcome::AlreadyApplied + } else { + ValidationOutcome::Apply + }) +} + fn validate_one( name: &str, carried: u64, @@ -225,7 +273,9 @@ fn validate_one( #[cfg(test)] mod tests { use super::*; - use crate::control::security::catalog::{StoredCollection, StoredSequence}; + use crate::control::security::catalog::{ + StoredCollection, StoredMaterializedView, StoredSequence, StoredSynonymGroup, + }; use crate::control::security::credential::CredentialStore; use nodedb_types::DatabaseId; use std::sync::Arc; @@ -236,14 +286,16 @@ mod tests { (store, tmp) } + /// A stamped entry at `version`. Its clock follows every row seeded + /// before the call. fn collection_with_version(name: &str, version: u64) -> CatalogEntry { - let mut stored = StoredCollection::new(1, name, "tester"); + let mut stored = StoredCollection::stamped_for_test(1, name, "tester"); stored.descriptor_version = version; CatalogEntry::PutCollection(Box::new(stored)) } fn seed_prior(catalog: &SystemCatalog, name: &str, version: u64) { - let mut stored = StoredCollection::new(1, name, "tester"); + let mut stored = StoredCollection::stamped_for_test(1, name, "tester"); stored.descriptor_version = version; catalog .put_collection(DatabaseId::DEFAULT, &stored) @@ -324,9 +376,11 @@ mod tests { fn validate_acknowledges_stale_historical_replay() { let (store, _tmp) = make_catalog(); let catalog = store.catalog(); + // Version 2 was stamped before the row reached version 5. + let historical = collection_with_version("orders", 2); seed_prior(catalog, "orders", 5); assert!(matches!( - validate(&collection_with_version("orders", 2), catalog), + validate(&historical, catalog), Ok(ValidationOutcome::AlreadyApplied) )); } @@ -335,7 +389,7 @@ mod tests { fn validate_treats_older_higher_version_as_prior_incarnation() { let (store, _tmp) = make_catalog(); let catalog = store.catalog(); - let mut current = StoredCollection::new(1, "orders", "new_owner"); + let mut current = StoredCollection::stamped_for_test(1, "orders", "new_owner"); current.descriptor_version = 1; current.modification_hlc = nodedb_types::Hlc::new(20, 0); catalog @@ -355,7 +409,7 @@ mod tests { fn validate_rejects_newer_divergent_equal_version() { let (store, _tmp) = make_catalog(); let catalog = store.catalog(); - let mut current = StoredCollection::new(1, "orders", "first"); + let mut current = StoredCollection::stamped_for_test(1, "orders", "first"); current.descriptor_version = 2; current.modification_hlc = nodedb_types::Hlc::new(10, 0); catalog @@ -409,7 +463,7 @@ mod tests { fn validate_rejects_locally_accreted_fields_at_same_version() { let (store, _tmp) = make_catalog(); let catalog = store.catalog(); - let mut persisted = StoredCollection::new(1, "metrics", "tester"); + let mut persisted = StoredCollection::stamped_for_test(1, "metrics", "tester"); persisted.descriptor_version = 1; persisted.fields = vec![ ("host".to_owned(), "VARCHAR".to_owned()), @@ -443,8 +497,7 @@ mod tests { fn validate_applies_recreate_at_version_one_with_newer_clock() { let (store, _tmp) = make_catalog(); let catalog = store.catalog(); - let mut current = StoredCollection::new(1, "orders", "tester"); - current.descriptor_version = 1; + let mut current = StoredCollection::stamped_for_test(1, "orders", "tester"); current.modification_hlc = nodedb_types::Hlc::new(1_787_734_007_496_107_753, 0); catalog .put_collection(DatabaseId::DEFAULT, ¤t) @@ -463,8 +516,7 @@ mod tests { fn validate_applies_recreate_over_deactivated_row() { let (store, _tmp) = make_catalog(); let catalog = store.catalog(); - let mut current = StoredCollection::new(1, "orders", "tester"); - current.descriptor_version = 1; + let mut current = StoredCollection::stamped_for_test(1, "orders", "tester"); current.is_active = false; current.modification_hlc = nodedb_types::Hlc::new(10, 0); catalog @@ -486,8 +538,7 @@ mod tests { fn validate_acknowledges_stamped_redelivery_at_equal_clock() { let (store, _tmp) = make_catalog(); let catalog = store.catalog(); - let mut current = StoredCollection::new(1, "orders", "tester"); - current.descriptor_version = 1; + let mut current = StoredCollection::stamped_for_test(1, "orders", "tester"); current.modification_hlc = nodedb_types::Hlc::new(1_787_734_007_496_107_753, 0); catalog .put_collection(DatabaseId::DEFAULT, ¤t) @@ -500,13 +551,12 @@ mod tests { } /// A version-1 divergence with no clock advance stays a loud anomaly: the - /// recreate carve-out needs a strictly newer clock, not just version 1. + /// recreate carve-out needs a strictly newer clock, not only version 1. #[test] fn validate_rejects_version_one_divergence_at_equal_clock() { let (store, _tmp) = make_catalog(); let catalog = store.catalog(); - let mut current = StoredCollection::new(1, "orders", "first"); - current.descriptor_version = 1; + let mut current = StoredCollection::stamped_for_test(1, "orders", "first"); current.modification_hlc = nodedb_types::Hlc::new(10, 0); catalog .put_collection(DatabaseId::DEFAULT, ¤t) @@ -531,7 +581,7 @@ mod tests { let (store, _tmp) = make_catalog(); let catalog = store.catalog(); seed_prior(catalog, "orders", 3); - let mut divergent = StoredCollection::new(1, "orders", "different-owner"); + let mut divergent = StoredCollection::stamped_for_test(1, "orders", "different-owner"); divergent.descriptor_version = 3; let err = validate(&CatalogEntry::PutCollection(Box::new(divergent)), catalog) .expect_err("same-version divergent payload must be rejected"); @@ -544,4 +594,398 @@ mod tests { } )); } + + /// Seeds a stored collection at descriptor version 3, stamped h3. + fn seed_current_incarnation(catalog: &SystemCatalog, name: &str, h3: nodedb_types::Hlc) { + let mut current = StoredCollection::stamped_for_test(1, name, "tester"); + current.descriptor_version = 3; + current.modification_hlc = h3; + catalog + .put_collection(DatabaseId::DEFAULT, ¤t) + .expect("seed current incarnation"); + } + + /// A `PurgeCollection` tombstone stamped for an older incarnation (h1) + /// must not remove the currently persisted row (stamped h3): the same + /// name has been dropped and recreated since the purge was proposed. + #[test] + fn validate_acknowledges_purge_of_prior_incarnation() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + let h1 = nodedb_types::Hlc::new(10, 0); + let h3 = nodedb_types::Hlc::new(30, 0); + seed_current_incarnation(catalog, "orders", h3); + + let purge = CatalogEntry::PurgeCollection { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: "orders".to_string(), + target_descriptor_version: 1, + target_hlc: h1, + }; + assert!(matches!( + validate(&purge, catalog), + Ok(ValidationOutcome::AlreadyApplied) + )); + } + + /// A `PurgeCollection` tombstone stamped for the incarnation that is + /// actually persisted (h3) must apply. + #[test] + fn validate_applies_purge_of_current_incarnation() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + let h3 = nodedb_types::Hlc::new(30, 0); + seed_current_incarnation(catalog, "orders", h3); + + let purge = CatalogEntry::PurgeCollection { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: "orders".to_string(), + target_descriptor_version: 3, + target_hlc: h3, + }; + assert!(matches!( + validate(&purge, catalog), + Ok(ValidationOutcome::Apply) + )); + } + + /// The same fencing on the soft-delete step: a `DeactivateCollection` + /// stamped for an older incarnation (h1) must not deactivate the + /// recreated row (stamped h3). + #[test] + fn validate_acknowledges_deactivate_of_prior_incarnation() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + let h1 = nodedb_types::Hlc::new(10, 0); + let h3 = nodedb_types::Hlc::new(30, 0); + seed_current_incarnation(catalog, "orders", h3); + + let deactivate = CatalogEntry::DeactivateCollection { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: "orders".to_string(), + descriptor_version: 1, + modification_hlc: h1, + }; + assert!(matches!( + validate(&deactivate, catalog), + Ok(ValidationOutcome::AlreadyApplied) + )); + } + + fn seed_view(catalog: &SystemCatalog, name: &str, hlc: nodedb_types::Hlc) { + catalog + .put_materialized_view(&StoredMaterializedView { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: name.to_string(), + source: "orders".to_string(), + query_sql: "SELECT id FROM orders".to_string(), + refresh_mode: "auto".to_string(), + owner: "tester".to_string(), + created_at: 0, + descriptor_version: 1, + modification_hlc: hlc, + }) + .expect("seed materialized view"); + } + + fn drop_view(name: &str, target_hlc: nodedb_types::Hlc) -> CatalogEntry { + CatalogEntry::DeleteMaterializedView { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: name.to_string(), + target_descriptor_version: 1, + target_hlc, + } + } + + /// A drop stamped for the first incarnation of a view must not remove a + /// recreated view of the same name: the recreate also restarts at + /// version 1, so only the clock tells the two apart. + #[test] + fn validate_acknowledges_view_drop_of_prior_incarnation() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + seed_view(catalog, "mv_orders", nodedb_types::Hlc::new(30, 0)); + assert!(matches!( + validate( + &drop_view("mv_orders", nodedb_types::Hlc::new(10, 0)), + catalog + ), + Ok(ValidationOutcome::AlreadyApplied) + )); + } + + /// A follower that never held the view applies the drop. + #[test] + fn validate_applies_view_drop_when_row_is_absent() { + let (store, _tmp) = make_catalog(); + assert!(matches!( + validate( + &drop_view("mv_orders", nodedb_types::Hlc::new(10, 0)), + store.catalog() + ), + Ok(ValidationOutcome::Apply) + )); + } + + /// A row older than the drop's target means this node missed a mutation + /// the proposer saw. That divergence stays a loud anomaly. + #[test] + fn validate_rejects_view_drop_ahead_of_the_local_row() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + seed_view(catalog, "mv_orders", nodedb_types::Hlc::new(10, 0)); + assert!(matches!( + validate( + &drop_view("mv_orders", nodedb_types::Hlc::new(30, 0)), + catalog + ), + Err(crate::Error::DescriptorVersionAnomaly { .. }) + )); + } + + /// An unstamped drop was proposed against an absent row, so a stamped row + /// is a later incarnation. + #[test] + fn validate_acknowledges_unstamped_view_drop_against_a_stamped_row() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + seed_view(catalog, "mv_orders", nodedb_types::Hlc::new(30, 0)); + assert!(matches!( + validate(&drop_view("mv_orders", nodedb_types::Hlc::ZERO), catalog), + Ok(ValidationOutcome::AlreadyApplied) + )); + } + + /// A row written before HLC stamping carries `Hlc::ZERO` and matches an + /// unstamped drop. + #[test] + fn validate_applies_unstamped_view_drop_against_an_unstamped_row() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + seed_view(catalog, "mv_orders", nodedb_types::Hlc::ZERO); + assert!(matches!( + validate(&drop_view("mv_orders", nodedb_types::Hlc::ZERO), catalog), + Ok(ValidationOutcome::Apply) + )); + } + + /// A purge proposed against an absent row carries no target. Replayed + /// over a collection created later, it must not remove that collection. + #[test] + fn validate_acknowledges_unstamped_purge_against_a_present_row() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + seed_current_incarnation(catalog, "orders", nodedb_types::Hlc::new(30, 0)); + let purge = CatalogEntry::PurgeCollection { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: "orders".to_string(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, + }; + assert!(matches!( + validate(&purge, catalog), + Ok(ValidationOutcome::AlreadyApplied) + )); + } + + /// Unversioned families fence on the clock alone: a replayed synonym put + /// from before a drop and recreate must not overwrite the recreated row. + #[test] + fn validate_acknowledges_stale_synonym_put() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + let group = |terms: &[&str], hlc| StoredSynonymGroup { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: "db_terms".to_string(), + terms: terms.iter().map(|t| (*t).to_string()).collect(), + created_at: 0, + modification_hlc: hlc, + }; + catalog + .put_synonym_group(&group(&["database", "db"], nodedb_types::Hlc::new(30, 0))) + .expect("seed synonym group"); + let replayed = group(&["database"], nodedb_types::Hlc::new(10, 0)); + assert!(matches!( + validate(&CatalogEntry::PutSynonymGroup(Box::new(replayed)), catalog), + Ok(ValidationOutcome::AlreadyApplied) + )); + let drop = CatalogEntry::DeleteSynonymGroup { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: "db_terms".to_string(), + target_hlc: nodedb_types::Hlc::new(10, 0), + }; + assert!(matches!( + validate(&drop, catalog), + Ok(ValidationOutcome::AlreadyApplied) + )); + } + + /// A `DeleteConsumerGroup` stamped for the first incarnation of a group + /// must not remove a recreated group of the same name: recreate offsets + /// stay committed, so only the clock tells the two incarnations apart. + #[test] + fn validate_acknowledges_consumer_group_drop_of_prior_incarnation() { + use crate::event::cdc::ConsumerGroupDef; + + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + let h1 = nodedb_types::Hlc::new(10, 0); + let h3 = nodedb_types::Hlc::new(30, 0); + let group = ConsumerGroupDef { + tenant_id: 1, + name: "analytics".to_string(), + stream_name: "orders_stream".to_string(), + owner: "tester".to_string(), + created_at: 0, + database_id: DatabaseId::DEFAULT, + modification_hlc: h3, + }; + catalog + .put_consumer_group(&group) + .expect("seed current incarnation"); + + let drop = CatalogEntry::DeleteConsumerGroup { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + stream_name: "orders_stream".to_string(), + name: "analytics".to_string(), + target_hlc: h1, + }; + assert!(matches!( + validate(&drop, catalog), + Ok(ValidationOutcome::AlreadyApplied) + )); + } + + /// A `DeleteChangeStream` stamped for the first incarnation of a stream + /// must not remove a recreated stream of the same name: the recreate's + /// buffer and its groups' offsets stay live, so only the clock tells the + /// two incarnations apart. + #[test] + fn validate_acknowledges_change_stream_drop_of_prior_incarnation() { + use crate::event::cdc::ChangeStreamDef; + use crate::event::cdc::stream_def::{ + CompactionConfig, LateDataPolicy, OpFilter, RetentionConfig, StreamFormat, + }; + + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + let h1 = nodedb_types::Hlc::new(10, 0); + let h3 = nodedb_types::Hlc::new(30, 0); + let stream = ChangeStreamDef { + database_id: DatabaseId::DEFAULT, + tenant_id: 1, + name: "orders_feed".to_string(), + collection: "orders".to_string(), + op_filter: OpFilter::all(), + format: StreamFormat::Json, + retention: RetentionConfig::default(), + compaction: CompactionConfig::default(), + webhook: crate::event::webhook::WebhookConfig::default(), + late_data: LateDataPolicy::default(), + kafka: crate::event::kafka::KafkaDeliveryConfig::default(), + owner: "tester".to_string(), + created_at: 0, + subscriber_roles: Vec::new(), + modification_hlc: h3, + }; + catalog + .put_change_stream(&stream) + .expect("seed current incarnation"); + + let drop = CatalogEntry::DeleteChangeStream { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: "orders_feed".to_string(), + target_hlc: h1, + }; + assert!(matches!( + validate(&drop, catalog), + Ok(ValidationOutcome::AlreadyApplied) + )); + } + + fn clone_entry(source: DatabaseId, child: DatabaseId) -> CatalogEntry { + use crate::control::security::catalog::database_types::{ + DatabaseDescriptor, DatabaseStatus, ParentCloneRef, + }; + + CatalogEntry::CloneDatabase { + target_descriptor: Box::new(DatabaseDescriptor { + id: child, + name: "clone_target".into(), + status: DatabaseStatus::Cloning, + created_at_lsn: 20, + quota_ref: 0, + parent_clone: Some(ParentCloneRef { + source_db_id: source, + as_of_lsn: 10, + as_of_ms: 0, + kv_surrogate_ceiling: None, + }), + mirror_origin: None, + audit_dml: nodedb_types::AuditDmlMode::None, + idle_session_timeout_secs: 0, + }), + source_db_id: source.as_u64(), + incarnation: nodedb_types::Hlc::new(1_000_000, 0), + } + } + + /// A replayed clone whose lineage edge exists must not re-read the source + /// into the child. + #[test] + fn validate_acknowledges_replayed_clone() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + let source = DatabaseId::new(1024); + let child = DatabaseId::new(1025); + catalog + .add_clone_child(source, child) + .expect("seed completed clone"); + assert!(matches!( + validate(&clone_entry(source, child), catalog), + Ok(ValidationOutcome::AlreadyApplied) + )); + } + + /// The edge still marks the clone after the child database is dropped. + #[test] + fn validate_acknowledges_replayed_clone_of_a_dropped_child() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + let source = DatabaseId::new(1024); + let child = DatabaseId::new(1025); + let entry = clone_entry(source, child); + crate::control::catalog_entry::apply::apply_to(&entry, catalog).expect("apply clone"); + catalog.delete_database(child).expect("drop child"); + assert!(matches!( + validate(&entry, catalog), + Ok(ValidationOutcome::AlreadyApplied) + )); + } + + /// The first apply finds no edge and applies. + #[test] + fn validate_applies_first_clone() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + let source = DatabaseId::new(1024); + let other_child = DatabaseId::new(1026); + catalog + .add_clone_child(source, other_child) + .expect("seed a sibling clone"); + assert!(matches!( + validate(&clone_entry(source, DatabaseId::new(1025)), catalog), + Ok(ValidationOutcome::Apply) + )); + } } diff --git a/nodedb/src/control/catalog_entry/entry.rs b/nodedb/src/control/catalog_entry/entry.rs index 7c2beade3..18673fb7b 100644 --- a/nodedb/src/control/catalog_entry/entry.rs +++ b/nodedb/src/control/catalog_entry/entry.rs @@ -13,6 +13,7 @@ //! accepted. Variants appended at the end of the enum stay there to keep //! MessagePack discriminants stable across rolling upgrades. +use crate::control::security::catalog::backup_schedule_marks::StoredBackupScheduleMark; use crate::control::security::catalog::column_stats::StoredColumnStats; use crate::control::security::catalog::types::CheckpointRecord; use crate::control::security::catalog::{ @@ -31,6 +32,7 @@ use crate::control::security::catalog::{ use crate::engine::timeseries::retention_policy::RetentionPolicyDef; use crate::event::alert::types::AlertDef; use crate::event::cdc::consumer_group::ConsumerGroupDef; +use crate::event::cdc::consumer_group::OffsetCommit; use crate::event::cdc::stream_def::ChangeStreamDef; use crate::event::scheduler::types::ScheduleDef; use crate::event::topic::TopicDef; @@ -70,6 +72,10 @@ pub enum CatalogEntry { database_id: u64, tenant_id: u64, name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_descriptor_version: u64, + target_hlc: nodedb_types::Hlc, }, // ── Sequence ─────────────────────────────────────────────────── @@ -81,6 +87,10 @@ pub enum CatalogEntry { database_id: u64, tenant_id: u64, name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_descriptor_version: u64, + target_hlc: nodedb_types::Hlc, }, /// Runtime state (current value, is_called, epoch, period_key). Used by /// ALTER SEQUENCE RESTART to propagate the new counter across nodes. @@ -93,6 +103,10 @@ pub enum CatalogEntry { database_id: DatabaseId, tenant_id: u64, name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_descriptor_version: u64, + target_hlc: nodedb_types::Hlc, }, // ── Function ─────────────────────────────────────────────────── @@ -103,6 +117,10 @@ pub enum CatalogEntry { database_id: DatabaseId, tenant_id: u64, name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_descriptor_version: u64, + target_hlc: nodedb_types::Hlc, }, // ── Procedure ────────────────────────────────────────────────── @@ -113,6 +131,10 @@ pub enum CatalogEntry { database_id: DatabaseId, tenant_id: u64, name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_descriptor_version: u64, + target_hlc: nodedb_types::Hlc, }, // ── Schedule ─────────────────────────────────────────────────── @@ -133,6 +155,9 @@ pub enum CatalogEntry { database_id: u64, tenant_id: u64, name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_hlc: nodedb_types::Hlc, }, // ── Custom type ──────────────────────────────────────────────── @@ -155,6 +180,9 @@ pub enum CatalogEntry { database_id: u64, tenant_id: u64, name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_hlc: nodedb_types::Hlc, }, // ── User ─────────────────────────────────────────────────────── @@ -203,6 +231,10 @@ pub enum CatalogEntry { database_id: u64, tenant_id: u64, name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_descriptor_version: u64, + target_hlc: nodedb_types::Hlc, }, // ── Continuous Aggregate ─────────────────────────────────────── /// Writes the catalog row plus the owner row. Post-apply re-dispatches @@ -214,6 +246,10 @@ pub enum CatalogEntry { database_id: u64, tenant_id: u64, name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_descriptor_version: u64, + target_hlc: nodedb_types::Hlc, }, // ── Tenant ───────────────────────────────────────────────────── @@ -269,8 +305,10 @@ pub enum CatalogEntry { /// `CREATE DATABASE`, `ALTER DATABASE RENAME`, `SET QUOTA`, `MATERIALIZE`, /// `PROMOTE`. PutDatabase(Box), - /// `DROP DATABASE`: removes the descriptor and its `_system.databases_by_name` - /// row. Does not touch collection rows — cascade those before proposing. + /// `DROP DATABASE`: removes the descriptor, its `_system.databases_by_name` + /// row, and its quota and mirror rows. The objects inside the database, + /// arrays included, travel as their own deletes in the same commit, ahead + /// of this entry. DeleteDatabase { /// Numeric database id. db_id: u64, @@ -364,6 +402,10 @@ pub enum CatalogEntry { Box, /// Numeric id of the source database (for lineage update). source_db_id: u64, + /// The incarnation every shadow collection of the clone takes. A + /// shadow is a new collection under a new key, never the source's + /// incarnation. The proposer stamps it. + incarnation: nodedb_types::Hlc, }, /// Streaming MV definition plus its database-scoped owner row. @@ -460,6 +502,9 @@ pub enum CatalogEntry { database_id: u64, tenant_id: u64, name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_hlc: nodedb_types::Hlc, }, // ── Consumer group ───────────────────────────────────────────── @@ -472,9 +517,13 @@ pub enum CatalogEntry { tenant_id: u64, stream_name: String, name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_hlc: nodedb_types::Hlc, }, /// Re-keys a bare-topic group row onto its canonical `topic:` stream. - /// `def` carries the legacy record; `legacy_stream` is its bare topic name. + /// `def` carries the row as the canonical key stores it; `legacy_stream` is + /// the bare topic name of the row it replaces. MigrateConsumerGroupStream { def: Box, legacy_stream: String, @@ -537,6 +586,9 @@ pub enum CatalogEntry { tenant_id: u64, collection: String, field_name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_hlc: nodedb_types::Hlc, }, // ── Column statistics ────────────────────────────────────────── @@ -546,4 +598,79 @@ pub enum CatalogEntry { /// plans from the same figures. One entry carries every column so a /// planner never sees a subset and costs against it. PutColumnStats(Box>), + + // ── Clone copy-on-write rows ─────────────────────────────────── + // Each names the clone collection by `(database_id, tenant_id, + // collection)`. Apply writes only while that collection is still a clone, + // so a replay after its materialization or purge is a no-op. + /// Maps a source row a clone copied up to its target surrogate. + PutCloneCopyup { + database_id: u64, + tenant_id: u64, + collection: String, + source_surrogate: u32, + target_surrogate: u32, + }, + /// Hides a source row, by its surrogate, from the clone's reads. + PutCloneTombstone { + database_id: u64, + tenant_id: u64, + collection: String, + source_surrogate: u32, + }, + /// Hides a source KV key from the clone's reads. + PutKvCloneTombstone { + database_id: u64, + tenant_id: u64, + collection: String, + kv_key: String, + }, + + // ── Clone source drain claims ────────────────────────────────── + /// A materialization claims its KV source's drain. Written before the + /// drain starts, so a later singleton worker can end an orphaned drain. + PutCloneSourceDrain( + Box, + ), + /// Releases a claim once its drain ended or its copy is no longer needed. + DeleteCloneSourceDrain { + clone_database: u64, + tenant_id: u64, + clone_collection: String, + }, + + // ── Array ────────────────────────────────────────────────────── + /// CREATE ARRAY and ALTER ARRAY: the full stored definition. Post-apply + /// opens the array on every core of every node. + PutArray(Box), + /// DROP ARRAY, the DROP DATABASE teardown, and the source side of a MOVE + /// TENANT rekey. Removes the row and its surrogate bindings. + DeleteArray { + database_id: u64, + tenant_id: u64, + name: String, + /// Incarnation this delete targets, frozen at propose time from the + /// stored row. `Hlc::ZERO` applies unfenced. + target_hlc: nodedb_types::Hlc, + /// MOVE TENANT only. Apply rekeys the surrogate bindings to the + /// target database, and post-apply renames the cell store there + /// instead of purging it. + moved_to: Option, + }, + + // ── Consumer offsets ─────────────────────────────────────────── + /// Raises one consumer group's committed offsets on every node, so a + /// consumer that moves to another node resumes where it committed. A + /// node whose registered group is another incarnation than the commit's + /// `group_hlc` skips the entry, so a replayed commit of a dropped group + /// never moves a recreated group's cursor. Apply only raises offsets, so + /// a replay is a no-op. + CommitConsumerOffsets(Box), + + // ── Scheduled backups ────────────────────────────────────────── + /// Raises one scheduled backup's mark on every node, so the node that + /// runs scheduled backups after a leader change neither skips a due + /// minute nor repeats a finished one. The mark is keyed by the schedule's + /// config incarnation, and apply only raises it, so a replay is a no-op. + PutBackupScheduleMark(Box), } diff --git a/nodedb/src/control/catalog_entry/incarnation/fence.rs b/nodedb/src/control/catalog_entry/incarnation/fence.rs new file mode 100644 index 000000000..a9bde528c --- /dev/null +++ b/nodedb/src/control/catalog_entry/incarnation/fence.rs @@ -0,0 +1,65 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Apply-time fence decisions for entries that carry an incarnation clock. + +use nodedb_types::Hlc; +use std::cmp::Ordering; + +use super::target::{carried_target, delete_key, written_row}; +use crate::control::catalog_entry::CatalogEntry; +use crate::control::catalog_entry::descriptor_validate::ValidationOutcome; +use crate::control::security::catalog::SystemCatalog; + +/// Decide a fenced delete against the row it names. +/// +/// - row absent: apply. A follower that never held the row has nothing to fence. +/// - row clock above the target: a later incarnation. Acknowledge. An +/// unstamped target (`Hlc::ZERO`) was proposed against an absent row, so any +/// stamped row is later than it. +/// - row clock equal to the target: the targeted incarnation. Apply. This +/// covers an unstamped target against a row that predates HLC stamping. +/// - row clock below the target: this node missed a mutation the proposer saw. +/// +/// A live delete never meets a later incarnation: the proposer stamps while +/// it holds the replicated DDL preparation lease, after its own node applied +/// the lease acquire, and a `DdlPrepared` entry applies only under the token +/// that owns the lease. Acknowledgement is reached by log replay alone. +pub fn check_delete( + entry: &CatalogEntry, + catalog: &SystemCatalog, +) -> crate::Result { + let (Some(key), Some(target)) = (delete_key(entry), carried_target(entry)) else { + return Ok(ValidationOutcome::Apply); + }; + let Some(current) = key.read(catalog)? else { + return Ok(ValidationOutcome::Apply); + }; + match current.hlc.cmp(&target.hlc) { + Ordering::Greater => Ok(ValidationOutcome::AlreadyApplied), + Ordering::Equal => Ok(ValidationOutcome::Apply), + Ordering::Less => Err(crate::Error::DescriptorVersionAnomaly { + descriptor: key.name().to_string(), + carried: target.descriptor_version, + prior: current.descriptor_version, + }), + } +} + +/// Acknowledge a soft delete or an unversioned put whose row a later +/// incarnation already holds. The entry's own clock is newer than the row it +/// was stamped against, so a row clock at or below it applies. +pub fn check_superseded( + entry: &CatalogEntry, + catalog: &SystemCatalog, +) -> crate::Result { + let Some((key, incoming)) = written_row(entry) else { + return Ok(ValidationOutcome::Apply); + }; + if incoming.hlc == Hlc::ZERO { + return Ok(ValidationOutcome::Apply); + } + Ok(match key.read(catalog)? { + Some(current) if current.hlc > incoming.hlc => ValidationOutcome::AlreadyApplied, + _ => ValidationOutcome::Apply, + }) +} diff --git a/nodedb/src/control/catalog_entry/incarnation/mod.rs b/nodedb/src/control/catalog_entry/incarnation/mod.rs new file mode 100644 index 000000000..3ad6f582c --- /dev/null +++ b/nodedb/src/control/catalog_entry/incarnation/mod.rs @@ -0,0 +1,15 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Incarnation fencing for delete-type catalog entries. +//! +//! The metadata log replays from its start on every boot. A delete proposed +//! against one incarnation of a name must not remove a later incarnation of +//! that name. Each fenced delete carries the `modification_hlc` of the row it +//! targeted, frozen at propose time. Apply acknowledges the delete when the +//! row has since moved past that clock. + +pub mod fence; +pub mod stamp; +pub mod target; + +pub use target::{Incarnation, RowKey}; diff --git a/nodedb/src/control/catalog_entry/incarnation/stamp.rs b/nodedb/src/control/catalog_entry/incarnation/stamp.rs new file mode 100644 index 000000000..205323111 --- /dev/null +++ b/nodedb/src/control/catalog_entry/incarnation/stamp.rs @@ -0,0 +1,221 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Propose-time stamps for fenced deletes and for puts whose stored type has +//! no descriptor version. + +use nodedb_types::{Hlc, HlcClock}; + +use super::target::{Incarnation, RowKey, delete_key, with_target, written_row}; +use crate::control::catalog_entry::CatalogEntry; +use crate::control::security::catalog::SystemCatalog; + +/// A clock strictly above `prior`, so a mutation always orders after the row +/// it replaces even when that row came from a peer whose clock ran ahead. +pub fn after(prior: Option, clock: &HlcClock, now: Hlc) -> Hlc { + match prior { + Some(prior) if prior >= now => clock.update(prior), + _ => now, + } +} + +/// The incarnation a collection put leaves on its row. A put carrying one +/// (ALTER, UNDROP, MOVE TENANT) keeps it. Otherwise an active prior row keeps +/// its own, and a create names a new one with the put's clock. +pub fn collection_incarnation( + carried: Hlc, + prior: Option<&crate::control::security::catalog::StoredCollection>, + stamped: Hlc, +) -> Hlc { + if carried != Hlc::ZERO { + return carried; + } + match prior { + Some(prior) if prior.is_active && prior.incarnation != Hlc::ZERO => prior.incarnation, + _ => stamped, + } +} + +/// Freeze a fenced delete's target from the committed row it names. An absent +/// row leaves the target `UNSTAMPED`. A failed read fails the stamp. +pub fn stamp_delete(entry: CatalogEntry, catalog: &SystemCatalog) -> crate::Result { + let target = match delete_key(&entry) { + Some(key) => key.read(catalog)?, + None => return Ok(entry), + }; + Ok(with_target(entry, target.unwrap_or(Incarnation::UNSTAMPED))) +} + +/// Stamp `modification_hlc` on a put whose stored type carries no descriptor +/// version. A create-only put that finds its row keeps the existing clock, +/// since apply keeps the existing definition. +pub fn stamp_put( + mut entry: CatalogEntry, + clock: &HlcClock, + catalog: &SystemCatalog, + now: Hlc, +) -> crate::Result { + let prior = match written_row(&entry) { + Some((key, _)) => key.read(catalog)?.map(|row| row.hlc), + None => return Ok(entry), + }; + match &mut entry { + CatalogEntry::CreateTopicIfAbsent(def) => def.modification_hlc = prior.unwrap_or(now), + CatalogEntry::PutSynonymGroup(row) => row.modification_hlc = after(prior, clock, now), + CatalogEntry::PutVectorIndexParams(row) => { + row.modification_hlc = after(prior, clock, now); + } + CatalogEntry::PutChangeStream(row) => row.modification_hlc = after(prior, clock, now), + CatalogEntry::PutConsumerGroupIfAbsent(def) => def.modification_hlc = prior.unwrap_or(now), + CatalogEntry::MigrateConsumerGroupStream { def, .. } => { + def.modification_hlc = after(prior, clock, now); + } + CatalogEntry::PutArray(row) => { + row.modification_hlc = after(prior, clock, now); + // A create names its incarnation. ALTER and MOVE TENANT carry the + // incarnation of the row they replace. + if row.incarnation == Hlc::ZERO { + row.incarnation = row.modification_hlc; + } + } + _ => {} + } + Ok(entry) +} + +/// Name the incarnation a clone's shadow collections take: the propose clock, +/// fresh for every clone. A re-stamp keeps the named one. +pub fn stamp_clone(mut entry: CatalogEntry, now: Hlc) -> CatalogEntry { + if let CatalogEntry::CloneDatabase { incarnation, .. } = &mut entry + && *incarnation == Hlc::ZERO + { + *incarnation = now; + } + entry +} + +/// Retarget a fenced delete at the row an earlier entry of the same batch +/// leaves behind. Committed state does not yet hold that row. +pub fn retarget_in_batch(prior: &[CatalogEntry], entry: CatalogEntry) -> CatalogEntry { + let Some(key) = delete_key(&entry) else { + return entry; + }; + let target = prior + .iter() + .rev() + .find_map(|earlier| batch_row(earlier, &key)); + match target { + Some(target) => with_target(entry, target), + None => entry, + } +} + +/// The incarnation `earlier` leaves on `key`'s row. A delete of the same row +/// leaves it absent, so a later delete in the batch applies unfenced. +fn batch_row(earlier: &CatalogEntry, key: &RowKey<'_>) -> Option { + if let Some((written, incarnation)) = written_row(earlier) + && written == *key + { + return Some(incarnation); + } + if delete_key(earlier).as_ref() == Some(key) { + return Some(Incarnation::UNSTAMPED); + } + None +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_types::DatabaseId; + + use super::*; + use crate::control::catalog_entry::descriptor_stamp::{stamp, stamp_batch}; + use crate::control::security::catalog::StoredCollection; + use crate::control::security::credential::CredentialStore; + + fn make_catalog() -> (Arc, tempfile::TempDir) { + let tmp = tempfile::tempdir().expect("tmpdir"); + let store = Arc::new(CredentialStore::open(&tmp.path().join("system.redb")).expect("open")); + (store, tmp) + } + + fn purge(name: &str) -> CatalogEntry { + CatalogEntry::PurgeCollection { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: name.to_string(), + target_descriptor_version: 0, + target_hlc: Hlc::ZERO, + } + } + + fn target(entry: &CatalogEntry) -> Incarnation { + super::super::target::carried_target(entry).expect("fenced delete") + } + + #[test] + fn purge_targets_the_committed_row() { + let (store, _tmp) = make_catalog(); + let catalog = store.catalog(); + let mut row = StoredCollection::stamped_for_test(1, "orders", "tester"); + row.descriptor_version = 3; + row.modification_hlc = Hlc::new(30, 0); + catalog + .put_collection(DatabaseId::DEFAULT, &row) + .expect("seed collection"); + + let stamped = stamp(purge("orders"), &HlcClock::new(), catalog).expect("stamp"); + assert_eq!( + target(&stamped), + Incarnation { + descriptor_version: 3, + hlc: Hlc::new(30, 0), + } + ); + } + + /// An absent row leaves the target unstamped. Apply then acknowledges the + /// purge against any stamped row, which can only be a later incarnation. + #[test] + fn purge_of_an_absent_row_stays_unstamped() { + let (store, _tmp) = make_catalog(); + let stamped = stamp(purge("orders"), &HlcClock::new(), store.catalog()).expect("stamp"); + assert_eq!(target(&stamped), Incarnation::UNSTAMPED); + } + + /// `CREATE t; DROP t PURGE` in one transaction: committed state has no + /// row yet, so the purge targets the create stamped before it. + #[test] + fn batched_purge_targets_the_preceding_create() { + let (store, _tmp) = make_catalog(); + let stamped = stamp_batch( + vec![ + CatalogEntry::PutCollection(Box::new(StoredCollection::new(1, "orders", "tester"))), + purge("orders"), + ], + &HlcClock::new(), + store.catalog(), + ) + .expect("stamp batch"); + let CatalogEntry::PutCollection(created) = &stamped[0] else { + panic!("expected PutCollection"); + }; + assert_eq!( + target(&stamped[1]), + Incarnation { + descriptor_version: created.descriptor_version, + hlc: created.modification_hlc, + } + ); + } + + #[test] + fn unversioned_put_orders_after_the_row_it_replaces() { + let clock = HlcClock::new(); + let ahead = Hlc::new(u64::MAX / 2, 0); + assert!(after(Some(ahead), &clock, clock.now()) > ahead); + let now = clock.now(); + assert_eq!(after(None, &clock, now), now); + } +} diff --git a/nodedb/src/control/catalog_entry/incarnation/target.rs b/nodedb/src/control/catalog_entry/incarnation/target.rs new file mode 100644 index 000000000..2611de127 --- /dev/null +++ b/nodedb/src/control/catalog_entry/incarnation/target.rs @@ -0,0 +1,410 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The stored row a fenced entry names, and the incarnation of that row. + +use nodedb_types::Hlc; + +use crate::control::catalog_entry::CatalogEntry; +use crate::control::security::catalog::SystemCatalog; +use crate::types::DatabaseId; + +/// `(descriptor_version, modification_hlc)` of one stored row. The clock is +/// the fence key: a recreate restarts the version at 1, so an older +/// incarnation can hold a higher version. Families without a descriptor +/// version carry `0`. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Incarnation { + pub descriptor_version: u64, + pub hlc: Hlc, +} + +impl Incarnation { + /// Carried by a delete proposed against an absent row. + pub const UNSTAMPED: Self = Self { + descriptor_version: 0, + hlc: Hlc::ZERO, + }; + + fn versioned(descriptor_version: u64, hlc: Hlc) -> Self { + Self { + descriptor_version, + hlc, + } + } + + fn unversioned(hlc: Hlc) -> Self { + Self { + descriptor_version: 0, + hlc, + } + } +} + +/// Identity of one stored row: `(database_id, tenant_id, name)`, plus the +/// field name for vector index parameters and the stream name for consumer +/// groups. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RowKey<'a> { + Collection(u64, u64, &'a str), + Sequence(u64, u64, &'a str), + Trigger(u64, u64, &'a str), + Function(u64, u64, &'a str), + Procedure(u64, u64, &'a str), + MaterializedView(u64, u64, &'a str), + ContinuousAggregate(u64, u64, &'a str), + SynonymGroup(u64, u64, &'a str), + Topic(u64, u64, &'a str), + VectorIndexParams(u64, u64, &'a str, &'a str), + ChangeStream(u64, u64, &'a str), + /// `(database_id, tenant_id, stream_name, group_name)`. + ConsumerGroup(u64, u64, &'a str, &'a str), + Array(u64, u64, &'a str), +} + +impl<'a> RowKey<'a> { + /// The descriptor name reported in a fence anomaly. + pub fn name(&self) -> &'a str { + match *self { + Self::Collection(_, _, name) + | Self::Sequence(_, _, name) + | Self::Trigger(_, _, name) + | Self::Function(_, _, name) + | Self::Procedure(_, _, name) + | Self::MaterializedView(_, _, name) + | Self::ContinuousAggregate(_, _, name) + | Self::SynonymGroup(_, _, name) + | Self::Topic(_, _, name) + | Self::ChangeStream(_, _, name) + | Self::ConsumerGroup(_, _, _, name) + | Self::Array(_, _, name) => name, + Self::VectorIndexParams(_, _, collection, _) => collection, + } + } + + /// Read the committed row this key names. + pub fn read(&self, catalog: &SystemCatalog) -> crate::Result> { + Ok(match *self { + Self::Collection(db, tenant, name) => catalog + .get_committed_collection(DatabaseId::new(db), tenant, name)? + .map(|row| Incarnation::versioned(row.descriptor_version, row.modification_hlc)), + Self::Sequence(db, tenant, name) => catalog + .get_sequence(db, tenant, name)? + .map(|row| Incarnation::versioned(row.descriptor_version, row.modification_hlc)), + Self::Trigger(db, tenant, name) => catalog + .get_committed_trigger_in_database(DatabaseId::new(db), tenant, name)? + .map(|row| Incarnation::versioned(row.descriptor_version, row.modification_hlc)), + Self::Function(db, tenant, name) => catalog + .get_committed_function_in_database(DatabaseId::new(db), tenant, name)? + .map(|row| Incarnation::versioned(row.descriptor_version, row.modification_hlc)), + Self::Procedure(db, tenant, name) => catalog + .get_committed_procedure_in_database(DatabaseId::new(db), tenant, name)? + .map(|row| Incarnation::versioned(row.descriptor_version, row.modification_hlc)), + Self::MaterializedView(db, tenant, name) => catalog + .get_committed_materialized_view(db, tenant, name)? + .map(|row| Incarnation::versioned(row.descriptor_version, row.modification_hlc)), + Self::ContinuousAggregate(db, tenant, name) => catalog + .get_continuous_aggregate(db, tenant, name)? + .map(|row| Incarnation::versioned(row.descriptor_version, row.modification_hlc)), + Self::SynonymGroup(db, tenant, name) => catalog + .get_synonym_group(db, tenant, name)? + .map(|row| Incarnation::unversioned(row.modification_hlc)), + Self::Topic(db, tenant, name) => catalog + .get_committed_ep_topic(DatabaseId::new(db), tenant, name)? + .map(|row| Incarnation::unversioned(row.modification_hlc)), + Self::VectorIndexParams(db, tenant, collection, field) => catalog + .get_committed_vector_index_params(db, tenant, collection, field)? + .map(|row| Incarnation::unversioned(row.modification_hlc)), + Self::ChangeStream(db, tenant, name) => catalog + .get_change_stream(DatabaseId::new(db), tenant, name)? + .map(|row| Incarnation::unversioned(row.modification_hlc)), + Self::ConsumerGroup(db, tenant, stream, group) => catalog + .get_consumer_group(DatabaseId::new(db), tenant, stream, group)? + .map(|row| Incarnation::unversioned(row.modification_hlc)), + Self::Array(db, tenant, name) => catalog + .get_array_in_database( + nodedb_types::TenantId::new(tenant), + DatabaseId::new(db), + name, + )? + .map(|row| Incarnation::unversioned(row.modification_hlc)), + }) + } +} + +/// The row a fenced delete removes. `None` for every other entry. +pub fn delete_key(entry: &CatalogEntry) -> Option> { + Some(match entry { + CatalogEntry::PurgeCollection { + database_id, + tenant_id, + name, + .. + } => RowKey::Collection(*database_id, *tenant_id, name), + CatalogEntry::DeleteSequence { + database_id, + tenant_id, + name, + .. + } => RowKey::Sequence(*database_id, *tenant_id, name), + CatalogEntry::DeleteTrigger { + database_id, + tenant_id, + name, + .. + } => RowKey::Trigger(database_id.as_u64(), *tenant_id, name), + CatalogEntry::DeleteFunction { + database_id, + tenant_id, + name, + .. + } => RowKey::Function(database_id.as_u64(), *tenant_id, name), + CatalogEntry::DeleteProcedure { + database_id, + tenant_id, + name, + .. + } => RowKey::Procedure(database_id.as_u64(), *tenant_id, name), + CatalogEntry::DeleteMaterializedView { + database_id, + tenant_id, + name, + .. + } => RowKey::MaterializedView(*database_id, *tenant_id, name), + CatalogEntry::DeleteContinuousAggregate { + database_id, + tenant_id, + name, + .. + } => RowKey::ContinuousAggregate(*database_id, *tenant_id, name), + CatalogEntry::DeleteSynonymGroup { + database_id, + tenant_id, + name, + .. + } => RowKey::SynonymGroup(*database_id, *tenant_id, name), + CatalogEntry::DeleteTopicWithConsumerGroups { + database_id, + tenant_id, + name, + .. + } => RowKey::Topic(*database_id, *tenant_id, name), + CatalogEntry::DeleteVectorIndexParams { + database_id, + tenant_id, + collection, + field_name, + .. + } => RowKey::VectorIndexParams(*database_id, *tenant_id, collection, field_name), + CatalogEntry::DeleteChangeStream { + database_id, + tenant_id, + name, + .. + } => RowKey::ChangeStream(*database_id, *tenant_id, name), + CatalogEntry::DeleteConsumerGroup { + database_id, + tenant_id, + stream_name, + name, + .. + } => RowKey::ConsumerGroup(*database_id, *tenant_id, stream_name, name), + CatalogEntry::DeleteArray { + database_id, + tenant_id, + name, + .. + } => RowKey::Array(*database_id, *tenant_id, name), + _ => return None, + }) +} + +/// The incarnation a fenced delete targets. `None` for every other entry. +pub fn carried_target(entry: &CatalogEntry) -> Option { + match entry { + CatalogEntry::PurgeCollection { + target_descriptor_version, + target_hlc, + .. + } + | CatalogEntry::DeleteSequence { + target_descriptor_version, + target_hlc, + .. + } + | CatalogEntry::DeleteTrigger { + target_descriptor_version, + target_hlc, + .. + } + | CatalogEntry::DeleteFunction { + target_descriptor_version, + target_hlc, + .. + } + | CatalogEntry::DeleteProcedure { + target_descriptor_version, + target_hlc, + .. + } + | CatalogEntry::DeleteMaterializedView { + target_descriptor_version, + target_hlc, + .. + } + | CatalogEntry::DeleteContinuousAggregate { + target_descriptor_version, + target_hlc, + .. + } => Some(Incarnation::versioned( + *target_descriptor_version, + *target_hlc, + )), + CatalogEntry::DeleteSynonymGroup { target_hlc, .. } + | CatalogEntry::DeleteTopicWithConsumerGroups { target_hlc, .. } + | CatalogEntry::DeleteVectorIndexParams { target_hlc, .. } + | CatalogEntry::DeleteChangeStream { target_hlc, .. } + | CatalogEntry::DeleteConsumerGroup { target_hlc, .. } + | CatalogEntry::DeleteArray { target_hlc, .. } => { + Some(Incarnation::unversioned(*target_hlc)) + } + _ => None, + } +} + +/// Replace the incarnation a fenced delete targets. Other entries pass through. +pub fn with_target(mut entry: CatalogEntry, target: Incarnation) -> CatalogEntry { + match &mut entry { + CatalogEntry::PurgeCollection { + target_descriptor_version, + target_hlc, + .. + } + | CatalogEntry::DeleteSequence { + target_descriptor_version, + target_hlc, + .. + } + | CatalogEntry::DeleteTrigger { + target_descriptor_version, + target_hlc, + .. + } + | CatalogEntry::DeleteFunction { + target_descriptor_version, + target_hlc, + .. + } + | CatalogEntry::DeleteProcedure { + target_descriptor_version, + target_hlc, + .. + } + | CatalogEntry::DeleteMaterializedView { + target_descriptor_version, + target_hlc, + .. + } + | CatalogEntry::DeleteContinuousAggregate { + target_descriptor_version, + target_hlc, + .. + } => { + *target_descriptor_version = target.descriptor_version; + *target_hlc = target.hlc; + } + CatalogEntry::DeleteSynonymGroup { target_hlc, .. } + | CatalogEntry::DeleteTopicWithConsumerGroups { target_hlc, .. } + | CatalogEntry::DeleteVectorIndexParams { target_hlc, .. } + | CatalogEntry::DeleteChangeStream { target_hlc, .. } + | CatalogEntry::DeleteConsumerGroup { target_hlc, .. } + | CatalogEntry::DeleteArray { target_hlc, .. } => { + *target_hlc = target.hlc; + } + _ => {} + } + entry +} + +/// The row a put, create, or soft delete leaves behind, and its incarnation. +pub fn written_row(entry: &CatalogEntry) -> Option<(RowKey<'_>, Incarnation)> { + Some(match entry { + CatalogEntry::PutCollection(row) | CatalogEntry::PutCollectionIfAbsent(row) => ( + RowKey::Collection(row.database_id.as_u64(), row.tenant_id, &row.name), + Incarnation::versioned(row.descriptor_version, row.modification_hlc), + ), + CatalogEntry::DeactivateCollection { + database_id, + tenant_id, + name, + descriptor_version, + modification_hlc, + } => ( + RowKey::Collection(*database_id, *tenant_id, name), + Incarnation::versioned(*descriptor_version, *modification_hlc), + ), + CatalogEntry::PutSequence(row) => ( + RowKey::Sequence(row.database_id, row.tenant_id, &row.name), + Incarnation::versioned(row.descriptor_version, row.modification_hlc), + ), + CatalogEntry::PutTrigger(row) => ( + RowKey::Trigger(row.database_id.as_u64(), row.tenant_id, &row.name), + Incarnation::versioned(row.descriptor_version, row.modification_hlc), + ), + CatalogEntry::PutFunction(row) => ( + RowKey::Function(row.database_id.as_u64(), row.tenant_id, &row.name), + Incarnation::versioned(row.descriptor_version, row.modification_hlc), + ), + CatalogEntry::PutProcedure(row) => ( + RowKey::Procedure(row.database_id.as_u64(), row.tenant_id, &row.name), + Incarnation::versioned(row.descriptor_version, row.modification_hlc), + ), + CatalogEntry::PutMaterializedView(row) => ( + RowKey::MaterializedView(row.database_id, row.tenant_id, &row.name), + Incarnation::versioned(row.descriptor_version, row.modification_hlc), + ), + CatalogEntry::PutContinuousAggregate(row) => ( + RowKey::ContinuousAggregate(row.database_id, row.tenant_id, &row.name), + Incarnation::versioned(row.descriptor_version, row.modification_hlc), + ), + CatalogEntry::PutSynonymGroup(row) => ( + RowKey::SynonymGroup(row.database_id, row.tenant_id, &row.name), + Incarnation::unversioned(row.modification_hlc), + ), + CatalogEntry::CreateTopicIfAbsent(row) => ( + RowKey::Topic(row.database_id.as_u64(), row.tenant_id, &row.name), + Incarnation::unversioned(row.modification_hlc), + ), + CatalogEntry::PutVectorIndexParams(row) => ( + RowKey::VectorIndexParams( + row.database_id, + row.tenant_id, + &row.collection, + &row.field_name, + ), + Incarnation::unversioned(row.modification_hlc), + ), + CatalogEntry::PutChangeStream(row) => ( + RowKey::ChangeStream(row.database_id.as_u64(), row.tenant_id, &row.name), + Incarnation::unversioned(row.modification_hlc), + ), + CatalogEntry::PutConsumerGroupIfAbsent(row) + | CatalogEntry::MigrateConsumerGroupStream { def: row, .. } => ( + RowKey::ConsumerGroup( + row.database_id.as_u64(), + row.tenant_id, + &row.stream_name, + &row.name, + ), + Incarnation::unversioned(row.modification_hlc), + ), + CatalogEntry::PutArray(row) => ( + RowKey::Array( + row.array_id.database_id.as_u64(), + row.array_id.tenant_id.as_u64(), + &row.name, + ), + Incarnation::unversioned(row.modification_hlc), + ), + _ => return None, + }) +} diff --git a/nodedb/src/control/catalog_entry/kind.rs b/nodedb/src/control/catalog_entry/kind.rs index 6cdff2b22..5c36f74a3 100644 --- a/nodedb/src/control/catalog_entry/kind.rs +++ b/nodedb/src/control/catalog_entry/kind.rs @@ -3,8 +3,8 @@ //! Stable `kind()` label for every [`CatalogEntry`] variant. //! //! Kept out of `entry.rs` so the enum definition stays a pure type -//! declaration: the label table grows one line per variant and would -//! otherwise push the definition file past its size budget. +//! declaration: the label table grows one line per variant and +//! pushes the definition file past its size budget when kept inline. use super::entry::CatalogEntry; @@ -93,7 +93,16 @@ impl CatalogEntry { Self::DeleteVectorModel { .. } => "delete_vector_model", Self::PutVectorIndexParams(_) => "put_vector_index_params", Self::PutColumnStats(_) => "put_column_stats", + Self::PutCloneCopyup { .. } => "put_clone_copyup", + Self::PutCloneTombstone { .. } => "put_clone_tombstone", + Self::PutKvCloneTombstone { .. } => "put_kv_clone_tombstone", + Self::PutCloneSourceDrain(_) => "put_clone_source_drain", + Self::DeleteCloneSourceDrain { .. } => "delete_clone_source_drain", + Self::PutArray(_) => "put_array", + Self::DeleteArray { .. } => "delete_array", Self::DeleteVectorIndexParams { .. } => "delete_vector_index_params", + Self::CommitConsumerOffsets(_) => "commit_consumer_offsets", + Self::PutBackupScheduleMark(_) => "put_backup_schedule_mark", } } } @@ -131,7 +140,9 @@ mod tests { CatalogEntry::DeleteSequence { database_id: 0, tenant_id: 1, - name: "c".into() + name: "c".into(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, } .kind(), "delete_sequence" diff --git a/nodedb/src/control/catalog_entry/mod.rs b/nodedb/src/control/catalog_entry/mod.rs index e175ee82a..1581d731f 100644 --- a/nodedb/src/control/catalog_entry/mod.rs +++ b/nodedb/src/control/catalog_entry/mod.rs @@ -6,7 +6,7 @@ //! The `CatalogEntry` enum is the single source of truth for "the set //! of mutations a pgwire DDL handler can make to the replicated //! catalog". Every DDL handler constructs a `CatalogEntry`, hands it -//! to [`crate::control::metadata_proposer::propose_catalog_entry`] +//! to [`crate::control::metadata_proposer::propose_catalog_entry_async`] //! which encodes it into an opaque `Vec` payload inside a //! `nodedb_cluster::MetadataEntry::CatalogDdl { payload }`, the raft //! log commits it, and on every node the production @@ -36,6 +36,7 @@ pub mod codec; pub mod descriptor_stamp; pub mod descriptor_validate; pub mod entry; +pub mod incarnation; pub mod kind; pub mod persist_collection; pub mod post_apply; diff --git a/nodedb/src/control/catalog_entry/persist_collection.rs b/nodedb/src/control/catalog_entry/persist_collection.rs index dc1b2039d..07210790f 100644 --- a/nodedb/src/control/catalog_entry/persist_collection.rs +++ b/nodedb/src/control/catalog_entry/persist_collection.rs @@ -26,28 +26,25 @@ use super::CatalogEntry; /// times out, which presents as a database that starts cleanly and fails /// every query. /// -/// `LocalOnly` means no metadata raft handle (single-node or mixed-version -/// compat mode); the applier is bypassed there, so the caller's record is -/// written through locally — mirroring the DDL handlers. -pub fn persist_collection_replicated( +/// The apply re-registers the collection on every Data Plane core of this +/// node, and this call awaits that register. A buffered entry applies +/// nothing until COMMIT. The returned outcome tells the caller which of the +/// two happened. +pub async fn persist_collection_replicated( state: &SharedState, - database_id: DatabaseId, coll: &StoredCollection, -) -> crate::Result<()> { +) -> crate::Result { let entry = CatalogEntry::PutCollection(Box::new(coll.clone())); - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry)?; - if outcome.needs_local_apply() { - state - .credentials - .catalog() - .put_collection(database_id, coll)?; - } - Ok(()) + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry).await } /// Union ingest-inferred fields into a collection's schema projection and /// persist the result through the replicated metadata path. /// +/// `time_column` is the column a timeseries ingest inferred for the row +/// time. [`crate::control::security::catalog::merge_inferred_fields`] decides +/// whether the collection's projection takes it. +/// /// Returns `true` when the projection changed and a new descriptor version was /// proposed, `false` when the collection is absent or already carries every /// inferred field (the overwhelmingly common case on a steady ingest stream — @@ -55,7 +52,7 @@ pub fn persist_collection_replicated( /// /// The schema projection is rebuildable control-plane state, but the record it /// lives in is not: it is the replicated collection descriptor. Writing the -/// merged fields straight to local redb would satisfy the projection and break +/// merged fields straight to local redb satisfies the projection and breaks /// the descriptor — see the divergence and wedged-apply-loop reasoning on /// [`persist_collection_replicated`]. Going through the proposer instead makes /// the merge a real descriptor version, which is both replicated to every node @@ -68,20 +65,29 @@ pub fn persist_collection_replicated( /// carrying the dropped field merges it again. Trading that for a descriptor /// that is byte-stable at a given version is the right side of the deal — the /// alternative loses the whole node. -pub fn merge_collection_fields_replicated( +pub async fn merge_collection_fields_replicated( state: &SharedState, database_id: DatabaseId, tenant_id: u64, name: &str, + time_column: Option<&(String, String)>, inferred_fields: &[(String, String)], ) -> crate::Result { - let catalog = state.credentials.catalog(); - let Some(mut coll) = catalog.get_collection(database_id, tenant_id, name)? else { + let Some(mut coll) = + state + .credentials + .catalog() + .get_collection(database_id, tenant_id, name)? + else { return Ok(false); }; - if !crate::control::security::catalog::merge_inferred_fields(&mut coll, inferred_fields) { + if !crate::control::security::catalog::merge_inferred_fields( + &mut coll, + time_column, + inferred_fields, + ) { return Ok(false); } - persist_collection_replicated(state, database_id, &coll)?; + persist_collection_replicated(state, &coll).await?; Ok(true) } diff --git a/nodedb/src/control/catalog_entry/post_apply/array.rs b/nodedb/src/control/catalog_entry/post_apply/array.rs new file mode 100644 index 000000000..8dcc954f8 --- /dev/null +++ b/nodedb/src/control/catalog_entry/post_apply/array.rs @@ -0,0 +1,103 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The in-memory array mirror and the bitemporal retention registry, as the +//! array catalog entries change them. +//! +//! The Data Plane opens an array from the mirror, so the mirror changes +//! before the applied-index watcher advances. A `PutArray` installs its row +//! in the synchronous lane. A `DeleteArray` removes or moves its row in the +//! async lane, under the incarnation's gate, together with the per-core drop or +//! rekey, so a replica never routes a cell write to a key its cores left. + +use nodedb_types::config::retention::BitemporalRetention; + +use crate::control::array_catalog::ArrayCatalogEntry; +use crate::control::state::SharedState; +use crate::engine::bitemporal::BitemporalEngineKind; +use crate::types::{DatabaseId, TenantId}; + +/// Install `entry` in the mirror, replacing any row of the same identity, +/// and register its retention window. +pub fn put_sync(entry: &ArrayCatalogEntry, shared: &SharedState) { + let array_id = &entry.array_id; + { + // A poisoned lock still holds a consistent mirror: every writer + // replaces whole entries. Skipping the install hides the array + // from every planner on this node for good. + let mut mirror = shared + .array_catalog + .write() + .unwrap_or_else(|poisoned| poisoned.into_inner()); + mirror.unregister_in_database(array_id.tenant_id, array_id.database_id, &entry.name); + if let Err(error) = mirror.register(entry.clone()) { + tracing::warn!(array = %entry.name, %error, "array mirror install failed"); + } + } + register_retention(entry, shared); +} + +/// Register `entry`'s retention window, or unregister it when it has none. +fn register_retention(entry: &ArrayCatalogEntry, shared: &SharedState) { + let array_id = &entry.array_id; + let registry = &shared.bitemporal_retention_registry; + let Some(audit_retain_ms) = entry.audit_retain_ms else { + registry.unregister(array_id.database_id, array_id.tenant_id, &entry.name); + return; + }; + let retention = BitemporalRetention { + data_retain_ms: 0, + audit_retain_ms: u64::try_from(audit_retain_ms).unwrap_or(0), + minimum_audit_retain_ms: entry.minimum_audit_retain_ms.unwrap_or(0), + }; + if let Err(error) = registry.register( + array_id.database_id, + array_id.tenant_id, + &entry.name, + BitemporalEngineKind::Array, + retention, + ) { + tracing::warn!(array = %entry.name, %error, "array retention register failed"); + } +} + +/// Remove the array from the mirror and the retention registry. With `to`, +/// install the same definition under database `to` in one mirror write, so +/// the incarnation is never absent from the mirror. +pub fn remove_or_move( + database_id: u64, + tenant_id: u64, + name: &str, + to: Option, + shared: &SharedState, +) { + let (database_id, tenant_id) = (DatabaseId::new(database_id), TenantId::new(tenant_id)); + let moved = { + // See `put_sync`: a poisoned mirror is still consistent. + let mut mirror = shared + .array_catalog + .write() + .unwrap_or_else(|poisoned| poisoned.into_inner()); + let removed = mirror.unregister_in_database(tenant_id, database_id, name); + match (removed, to) { + (Some(entry), Some(to)) => { + let to = DatabaseId::new(to); + let moved = ArrayCatalogEntry { + array_id: nodedb_array::types::ArrayId::in_database(tenant_id, to, name), + ..entry + }; + mirror.unregister_in_database(tenant_id, to, name); + if let Err(error) = mirror.register(moved.clone()) { + tracing::warn!(array = %name, %error, "array mirror move failed"); + } + Some(moved) + } + _ => None, + } + }; + shared + .bitemporal_retention_registry + .unregister(database_id, tenant_id, name); + if let Some(moved) = moved { + register_retention(&moved, shared); + } +} diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/array.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/array.rs new file mode 100644 index 000000000..447bfd416 --- /dev/null +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/array.rs @@ -0,0 +1,301 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-core effects of the array catalog entries on this node. +//! +//! - `PutArray` sends `OpenArray` to every core. The open purges a +//! finalized-drop tombstone of the same identity, so a recreate starts +//! empty on every core. +//! - `DeleteArray` stages the drop on every core, then purges the staged +//! tombstones. The catalog row is already gone, so a failed stage returns +//! `Err` and the re-delivered entry stages again. A failed purge counts as +//! done: the tombstone fences a same-name open until the next `PutArray` +//! purges it. +//! - A `DeleteArray` with `moved_to` rekeys the store on every core instead. +//! Cells route by Hilbert prefix alone, so the vShard of every cell stays +//! the same, and only the store directory carries the database. +//! - Every `DeleteArray` changes the mirror last, under the array +//! incarnation's exclusive gate (see `control::write_gate`). + +use nodedb_array::types::ArrayId; +use nodedb_physical::physical_plan::ArrayOp; + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::array_catalog::ArrayCatalogEntry; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId}; + +use super::core_fanout::{CoreFanout, dispatch_to_every_core}; + +/// Open `entry` on every core of this node. +pub(crate) async fn open_on_every_core( + shared: &SharedState, + entry: &ArrayCatalogEntry, +) -> crate::Result<()> { + let plan = PhysicalPlan::Array(ArrayOp::OpenArray { + array_id: entry.array_id.clone(), + schema_msgpack: entry.schema_msgpack.clone(), + schema_hash: entry.schema_hash, + prefix_bits: entry.prefix_bits, + audit_retain_ms: entry.audit_retain_ms, + minimum_audit_retain_ms: entry.minimum_audit_retain_ms, + }); + dispatch_to_every_core(shared, &fanout(&entry.array_id, "array open"), &plan).await +} + +/// Drop `(database_id, tenant_id, name)` on every core of this node, or +/// rekey it there when the delete is the source side of a move. +pub(crate) async fn delete_on_every_core( + shared: &SharedState, + database_id: u64, + tenant_id: u64, + name: &str, + moved_to: Option, +) -> crate::Result<()> { + // Exclusive on this array's incarnation: no replica routes a cell write + // to it while the cores and the mirror change. The mirror changes last, + // once every core left the key. A key the mirror no longer holds takes + // no write, so a replayed delete holds no gate. + let incarnation = { + let mirror = shared + .array_catalog + .read() + .unwrap_or_else(|poisoned| poisoned.into_inner()); + mirror + .lookup_by_name_in_database( + TenantId::new(tenant_id), + DatabaseId::new(database_id), + name, + ) + .map(|entry| entry.incarnation) + }; + let _gate = match incarnation { + Some(incarnation) => Some( + crate::control::write_gate::exclusive(crate::control::write_gate::GateKey::Array( + incarnation, + )) + .await, + ), + None => None, + }; + match moved_to { + Some(to) => rekey_on_every_core(shared, database_id, tenant_id, name, to).await?, + None => drop_on_every_core(shared, database_id, tenant_id, name).await?, + } + super::super::array::remove_or_move(database_id, tenant_id, name, moved_to, shared); + Ok(()) +} + +/// Stage and purge the drop of `(database_id, tenant_id, name)` on every +/// core of this node. +async fn drop_on_every_core( + shared: &SharedState, + database_id: u64, + tenant_id: u64, + name: &str, +) -> crate::Result<()> { + let array_id = identity(database_id, tenant_id, name); + let stage = PhysicalPlan::Array(ArrayOp::DropArray { + array_id: array_id.clone(), + }); + dispatch_to_every_core(shared, &fanout(&array_id, "array drop stage"), &stage).await?; + let purge = PhysicalPlan::Array(ArrayOp::PurgeArrayDrop { + array_id: array_id.clone(), + }); + if let Err(error) = + dispatch_to_every_core(shared, &fanout(&array_id, "array drop purge"), &purge).await + { + tracing::warn!( + database = database_id, + tenant = tenant_id, + array = %name, + %error, + "array drop post-apply: purge failed; the drop tombstone owns it" + ); + } + Ok(()) +} + +/// Move the store of `(database_id, tenant_id, name)` under `moved_to` on +/// every core of this node. +async fn rekey_on_every_core( + shared: &SharedState, + database_id: u64, + tenant_id: u64, + name: &str, + moved_to: u64, +) -> crate::Result<()> { + let array_id = identity(database_id, tenant_id, name); + let plan = PhysicalPlan::Array(ArrayOp::RekeyArray { + array_id: array_id.clone(), + target: identity(moved_to, tenant_id, name), + }); + dispatch_to_every_core(shared, &fanout(&array_id, "array rekey"), &plan).await +} + +fn identity(database_id: u64, tenant_id: u64, name: &str) -> ArrayId { + ArrayId::in_database(TenantId::new(tenant_id), DatabaseId::new(database_id), name) +} + +fn fanout<'a>(array_id: &'a ArrayId, what: &'a str) -> CoreFanout<'a> { + CoreFanout { + database_id: array_id.database_id.as_u64(), + tenant_id: array_id.tenant_id.as_u64(), + collection: &array_id.name, + what, + detail: "", + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + use std::time::Duration; + + use super::*; + use crate::bridge::dispatch::{BridgeResponse, CoreChannelDataSide, Dispatcher}; + use crate::bridge::envelope::{Payload, Response, Status}; + use crate::types::Lsn; + use crate::wal::WalManager; + + const CORES: usize = 2; + + fn fixture() -> ( + Arc, + Vec, + tempfile::TempDir, + ) { + let directory = tempfile::tempdir().expect("temporary directory"); + let wal = Arc::new( + WalManager::open_for_testing(&directory.path().join("array.wal")).expect("test WAL"), + ); + let (dispatcher, sides) = Dispatcher::new(CORES, 64); + let state = SharedState::new(dispatcher, wal).expect("shared state"); + (state, sides, directory) + } + + fn step(plan: &PhysicalPlan) -> &'static str { + match plan { + PhysicalPlan::Array(ArrayOp::DropArray { .. }) => "drop", + PhysicalPlan::Array(ArrayOp::PurgeArrayDrop { .. }) => "purge", + PhysicalPlan::Array(ArrayOp::RekeyArray { .. }) => "rekey", + PhysicalPlan::Array(ArrayOp::OpenArray { .. }) => "open", + _ => "other", + } + } + + /// Run `work` while answering every core request with the status + /// `status_of` gives its step. Returns the work's result and the steps + /// each core saw, in order. + async fn run_answering( + state: &Arc, + sides: &mut [CoreChannelDataSide], + work: impl std::future::Future + Send + 'static, + status_of: impl Fn(&'static str) -> Status, + ) -> (T, Vec>) { + let task = tokio::spawn(work); + let mut seen = vec![Vec::new(); sides.len()]; + let deadline = std::time::Instant::now() + Duration::from_secs(10); + while !task.is_finished() && std::time::Instant::now() < deadline { + for (core_id, side) in sides.iter_mut().enumerate() { + if let Ok(request) = side.request_rx.try_pop() { + let kind = step(&request.inner.plan); + seen[core_id].push(kind); + side.response_tx + .try_push(BridgeResponse { + inner: Response { + request_id: request.inner.request_id, + status: status_of(kind), + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .expect("fake data-plane response queue has capacity"); + } + } + state.poll_and_route_responses(); + tokio::task::yield_now().await; + } + let result = task.await.expect("post-apply task"); + (result, seen) + } + + #[tokio::test(flavor = "multi_thread")] + async fn a_drop_stages_then_purges_on_every_core() { + let (state, mut sides, _directory) = fixture(); + let shared = Arc::clone(&state); + let (result, seen) = run_answering( + &state, + &mut sides, + async move { drop_on_every_core(&shared, 1, 1, "grid").await }, + |_| Status::Ok, + ) + .await; + result.expect("the drop completes"); + assert!(seen.iter().all(|steps| steps == &["drop", "purge"])); + } + + /// The catalog row is gone before the stage, so a refused stage returns + /// `Err`: the re-delivered entry stages again. + #[tokio::test(flavor = "multi_thread")] + async fn a_refused_stage_fails_the_post_apply() { + let (state, mut sides, _directory) = fixture(); + let shared = Arc::clone(&state); + let (result, seen) = run_answering( + &state, + &mut sides, + async move { drop_on_every_core(&shared, 1, 1, "grid").await }, + |kind| { + if kind == "drop" { + Status::Error + } else { + Status::Ok + } + }, + ) + .await; + assert!(result.is_err(), "a refused stage must fail the post-apply"); + assert!(seen.iter().all(|steps| steps == &["drop"])); + } + + /// A failed purge leaves the tombstone to the next open of the identity. + #[tokio::test(flavor = "multi_thread")] + async fn a_failed_purge_counts_as_done() { + let (state, mut sides, _directory) = fixture(); + let shared = Arc::clone(&state); + let (result, _) = run_answering( + &state, + &mut sides, + async move { drop_on_every_core(&shared, 1, 1, "grid").await }, + |kind| { + if kind == "purge" { + Status::Error + } else { + Status::Ok + } + }, + ) + .await; + result.expect("the tombstone owns a failed purge"); + } + + #[tokio::test(flavor = "multi_thread")] + async fn a_refused_rekey_fails_the_post_apply() { + let (state, mut sides, _directory) = fixture(); + let shared = Arc::clone(&state); + let (result, seen) = run_answering( + &state, + &mut sides, + async move { rekey_on_every_core(&shared, 1, 1, "grid", 2).await }, + |_| Status::Error, + ) + .await; + assert!(result.is_err(), "a refused rekey must fail the post-apply"); + assert!(seen.iter().all(|steps| steps == &["rekey"])); + } +} diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/collection.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/collection.rs index c3b9ace3b..6cb9ce923 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/collection.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/collection.rs @@ -2,11 +2,9 @@ //! Collection-specific async post-apply dispatchers. //! -//! Runs on **every node** (via `spawn_post_apply_async_side_effects` -//! in `apply_replicated`). Each node's local Data Plane observes -//! catalog mutations symmetrically. - -use std::sync::Arc; +//! Runs on **every node**: the metadata applier awaits +//! `run_post_apply_async_side_effects`. Each node's local Data Plane +//! observes catalog mutations symmetrically. use tracing::{debug, warn}; @@ -15,8 +13,24 @@ use crate::control::security::catalog::{StoredCollection, StoredL2CleanupEntry}; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId}; -pub async fn put_async(stored: StoredCollection, shared: Arc) { - collection::put_async(stored, shared).await; +/// Register `stored` on every local Data Plane core. `Err` means a core did +/// not acknowledge the Register. +pub async fn put_async(stored: &StoredCollection, shared: &SharedState) -> crate::Result<()> { + collection::put_async(stored, shared).await +} + +/// Register every shadow collection a `CloneDatabase` entry stamped into +/// `target` on every local Data Plane core. +/// +/// The clone writes its shadow descriptors straight into the catalog, so no +/// `PutCollection` entry registers them. An unregistered strict shadow stores +/// a copied-up or materialized row as MessagePack, and a scan that decodes it +/// as a Binary Tuple once the collection registers matches nothing. +pub async fn clone_shadows_async(target: DatabaseId, shared: &SharedState) -> crate::Result<()> { + for stored in collection::clone_shadows(target, shared)? { + collection::put_async(&stored, shared).await?; + } + Ok(()) } /// Failure outcome of [`reclaim_collection_storage`]. @@ -32,7 +46,7 @@ pub async fn put_async(stored: StoredCollection, shared: Arc) { /// writes failed before any record was queued, or queuing the record itself /// failed). The holder must let its guard `Drop` release the in-memory drain /// so a same-name CREATE can re-acquire the lifecycle and self-heal off the -/// durable inactive catalog row. Leaking the drain here would wedge every +/// durable inactive catalog row. Leaking the drain here wedges every /// future same-name CREATE (and the GC sweeper) until the node restarts. #[derive(Debug)] pub(crate) struct ReclaimFailure { @@ -79,16 +93,31 @@ pub(crate) async fn reclaim_collection_storage( purge_lsn: u64, drain_already_held: bool, ) -> Result<(), ReclaimFailure> { + // No replicated write reaches a core under this key while its storage + // goes: the write routes under the key's gate held shared. + let _gate = + crate::control::write_gate::exclusive(crate::control::write_gate::GateKey::Collection { + database_id, + tenant_id, + name: name.to_string(), + }) + .await; // 1. Persist to redb (every node has its own catalog). A failure here // leaves no durable retry owner, so it is a `no_retry` failure: the caller // releases its lifecycle guard rather than leaking the drain. + crate::fail_point_err!("collection_reclaim::before_tombstone", |detail: String| { + ReclaimFailure::no_retry(crate::Error::Storage { + engine: "catalog".into(), + detail, + }) + }); let catalog = shared.credentials.catalog(); catalog .record_wal_tombstone(database_id, tenant_id, name, purge_lsn) .map_err(ReclaimFailure::no_retry)?; // 1b. Drop the collection's column-redaction policies. Their key carries - // no collection generation, so a survivor would re-attach to a same-name + // no collection generation, so a survivor re-attaches to a same-name // collection created later and redact columns nobody protected. crate::control::catalog_entry::post_apply::redaction::purge_for_collection( shared, @@ -151,13 +180,9 @@ pub(crate) async fn reclaim_collection_storage( // files while a scan is touching an mmap page faults the // whole TPC reactor — drain ordering is a correctness, not // performance, requirement. - if !drain_already_held { - shared.quiesce.begin_drain(database_id, tenant_id, name); - } - shared - .quiesce - .wait_until_drained(database_id, tenant_id, name) - .await; + let hold = + (!drain_already_held).then(|| shared.quiesce.begin_drain(database_id, tenant_id, name)); + wait_drained_reporting_progress(shared, database_id, tenant_id, name).await; // 4. Reclaim on local Data Plane. RESULT-CHECKED: the redb + // versioned engine purge is correctness-critical (the catalog @@ -186,13 +211,13 @@ pub(crate) async fn reclaim_collection_storage( match purge_result { Err(e) => { - // Keep the lifecycle drain marker set ONLY when a durable retry - // record is persisted: a worker then owns the retry and releases - // the drain via `forget`. A same-name CREATE waits until that - // retry succeeds, because engine keys are name-scoped. If recording - // the durable entry itself fails there is no owner to release the - // drain, so this is a `no_retry` failure and the caller must let - // its guard release the in-memory hold. + // Keep the drain ONLY when a durable retry record is persisted: + // the hold passes to the pending-reclaim path, which releases it + // when the retry succeeds. A same-name CREATE waits until then, + // because engine keys are name-scoped. If recording the durable + // entry itself fails there is no owner, so this is a `no_retry` + // failure: this hold drops here, and the caller's guard releases + // its own. match record_pending_reclaim( shared, database_id, @@ -201,7 +226,12 @@ pub(crate) async fn reclaim_collection_storage( purge_lsn, &e.to_string(), ) { - Ok(()) => Err(ReclaimFailure::retry_queued(e)), + Ok(()) => { + if let Some(hold) = hold { + hold.hand_to_reclaim(); + } + Err(ReclaimFailure::retry_queued(e)) + } Err(record_error) => Err(ReclaimFailure::no_retry(crate::Error::Storage { engine: "pending-reclaim".into(), detail: format!( @@ -212,7 +242,7 @@ pub(crate) async fn reclaim_collection_storage( } Ok(()) => { // Broadcast only after every core reclaimed the old incarnation. - // Saturated per-session channels may drop the notification; offline + // Saturated per-session channels can drop the notification; offline // replay remains the fallback. shared.crdt_sync_delivery.broadcast_collection_purged( tenant_id, @@ -221,16 +251,22 @@ pub(crate) async fn reclaim_collection_storage( purge_lsn, ); - // A prior failed attempt may have left a durable entry; a - // succeeding purge clears it, then releases CREATE waiters. + // A prior failed attempt can leave a durable entry; a + // succeeding purge clears it and the hold that entry owned. + // This call's own hold drops on return. shared .credentials .catalog() .remove_pending_reclaim(database_id, tenant_id, name) .map_err(ReclaimFailure::no_retry)?; - if !drain_already_held { - shared.quiesce.forget(database_id, tenant_id, name); - } + shared + .quiesce + .release_reclaim_hold(&crate::bridge::quiesce::ReclaimOwner::new( + database_id, + tenant_id, + name, + )); + drop(hold); debug!( collection = %name, tenant = tenant_id, @@ -242,11 +278,126 @@ pub(crate) async fn reclaim_collection_storage( } } +/// How often a quiesce drain checks for closed scans. +const DRAIN_PROGRESS_TICK: std::time::Duration = std::time::Duration::from_millis(100); + +/// Wait until every open scan of the collection closes. Each closed scan +/// counts as apply progress, so a proposer waiting on this purge keeps +/// waiting while scans close and times out only when none does. +/// +/// A draining collection admits no new scan, so the open count only falls. +async fn wait_drained_reporting_progress( + shared: &SharedState, + database_id: u64, + tenant_id: u64, + name: &str, +) { + let drained = shared + .quiesce + .wait_until_drained(database_id, tenant_id, name); + tokio::pin!(drained); + let mut open = shared.quiesce.open_scans(database_id, tenant_id, name); + loop { + tokio::select! { + () = &mut drained => return, + () = tokio::time::sleep(DRAIN_PROGRESS_TICK) => { + let now = shared.quiesce.open_scans(database_id, tenant_id, name); + if now < open { + shared + .metadata_apply_progress + .fetch_add(1, std::sync::atomic::Ordering::Release); + } + open = now; + } + } + } +} + +/// Clear this node's storage under `name` before a new incarnation of the +/// collection registers there. +/// +/// Data Plane storage is keyed by `(database, tenant, name)`, so rows an +/// earlier incarnation left behind read as the new incarnation's. That +/// happens when a reclaim was dropped or never ran on this node. Both +/// tombstone surfaces are written first, so WAL replay skips the earlier +/// incarnation's records too. +/// +/// It runs on every new incarnation. No local state proves the node never +/// held the name: WAL tombstones are collected once the WAL truncates past +/// them, and a cancelled create or a dropped retry leaves none. Clearing an +/// empty prefix costs one round-trip per core. +/// +/// A pending reclaim for the name is covered by this clear, so its row goes +/// and the pending-reclaim path's hold on the name is released. +/// +/// No write to the new incarnation precedes the clear on any node. Every +/// data-group entry and every Calvin transaction carries the proposer's +/// applied metadata index, and a replica holds it until its own metadata +/// watcher reaches that index. A data-group snapshot carries its builder's +/// applied metadata index, and its install waits the same way. The watcher bumps only after the applier, +/// this clear included, returned. +pub(crate) async fn clear_before_recreate( + shared: &SharedState, + database_id: u64, + tenant_id: u64, + name: &str, +) -> crate::Result<()> { + let catalog = shared.credentials.catalog(); + let purge_lsn = shared.wal.next_lsn().as_u64(); + catalog.record_wal_tombstone(database_id, tenant_id, name, purge_lsn)?; + shared + .wal + .appender(crate::wal::manager::NO_APPLY_KEY) + .append_collection_tombstone( + TenantId::new(tenant_id), + DatabaseId::new(database_id), + name, + purge_lsn, + )?; + // Scans of the earlier incarnation must release before its segments go. + let hold = shared.quiesce.begin_drain(database_id, tenant_id, name); + shared + .quiesce + .wait_until_drained(database_id, tenant_id, name) + .await; + let cleared = + crate::control::server::shared::ddl::neutral::collection::purge::dispatch_unregister_collection( + shared, database_id, tenant_id, name, purge_lsn, + ) + .await; + hold.release(); + cleared?; + + let owed = catalog + .load_pending_reclaim_queue()? + .into_iter() + .any(|entry| { + entry.database_id == database_id && entry.tenant_id == tenant_id && entry.name == name + }); + if owed { + catalog.remove_pending_reclaim(database_id, tenant_id, name)?; + } + shared + .quiesce + .release_reclaim_hold(&crate::bridge::quiesce::ReclaimOwner::new( + database_id, + tenant_id, + name, + )); + debug!( + collection = %name, + tenant = tenant_id, + purge_lsn, + owed, + "catalog_entry: storage under the name cleared before the new incarnation registers" + ); + Ok(()) +} + /// Persist a durable `_system.pending_reclaim` entry so the failed /// engine purge is retried at-least-once by the pending-reclaim worker -/// and the boot-time drain, instead of being lost to a warn log. This -/// is the whole point of the fix: NEVER warn-and-forget a failed -/// engine purge. +/// and the boot-time drain, instead of being lost to a warn log. A failed +/// engine purge is never warn-and-forget. fn record_pending_reclaim( shared: &SharedState, database_id: u64, @@ -260,6 +411,11 @@ fn record_pending_reclaim( .duration_since(std::time::UNIX_EPOCH) .map(|d| d.as_nanos() as u64) .unwrap_or(0); + // The row `prepare_purge` left inactive names the incarnation the retry + // owns. A retry never touches a later row under the same name. + let target_hlc = catalog + .get_committed_collection(DatabaseId::new(database_id), tenant_id, name)? + .map(|row| row.modification_hlc); let entry = crate::control::security::catalog::StoredPendingReclaim { database_id, tenant_id, @@ -268,6 +424,8 @@ fn record_pending_reclaim( enqueued_at_ns: now_ns, last_error: last_error.to_string(), attempts: 0, + target_hlc, + cancelled_create: false, }; catalog.enqueue_pending_reclaim(&entry)?; warn!( diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/continuous_aggregate.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/continuous_aggregate.rs index 47044c696..5953918c3 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/continuous_aggregate.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/continuous_aggregate.rs @@ -8,43 +8,96 @@ //! registration consistent across leader and followers after the raft //! commit. `DeleteContinuousAggregate` dispatches the matching //! `MetaOp::UnregisterContinuousAggregate`. +//! +//! Every core must apply the op. The catalog row is already committed and +//! the post-apply lane cannot propagate, so every failure files a report. +//! Boot re-registration installs every stored aggregate again from redb, so a +//! lost post-apply dispatch is repaired at the next boot. use std::sync::Arc; -use tracing::debug; - -use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Status}; +use crate::bridge::envelope::PhysicalPlan; use crate::control::state::SharedState; use crate::engine::timeseries::continuous_agg::ContinuousAggregateDef; -use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; use nodedb_physical::physical_plan::MetaOp; +use super::core_fanout::{CoreFanout, dispatch_to_every_core}; + +/// Name the fan-out reports a continuous-aggregate dispatch under. The +/// fan-out addresses each core directly, so this name only labels the ack +/// line and the unreached-core error. +const CAGG_SENTINEL_COLLECTION: &str = "_continuous_aggregates"; + /// Dispatch `MetaOp::RegisterContinuousAggregate` to every core on /// this node. `def_bytes` is the MessagePack-encoded /// `ContinuousAggregateDef` from the catalog row. pub async fn put_async(tenant_id: u64, name: String, def_bytes: Vec, shared: Arc) { - let def: ContinuousAggregateDef = match zerompk::from_msgpack(&def_bytes) { - Ok(def) => def, - Err(e) => { - debug!( - tenant_id, - cagg = %name, - error = %e, - "continuous aggregate: failed to deserialize def — skipping register" - ); - return; + if let Err(failure) = register_on_every_core(&shared, tenant_id, &name, &def_bytes).await { + crate::diag::continuous_aggregate_not_applied( + &failure.error, + failure.stage.label(), + failure.database_id, + tenant_id, + &name, + ); + } +} + +/// Why a register did not reach every core, and at which stage. +pub struct RegisterFailure { + pub error: crate::Error, + pub stage: RegisterStage, + /// Zero when the definition did not decode, since the definition names it. + pub database_id: u64, +} + +/// The stage of a register that failed. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum RegisterStage { + /// The stored definition did not decode. + Decode, + /// The register did not reach every core. + Dispatch, +} + +impl RegisterStage { + /// The stage name the diagnostic report carries. + pub fn label(self) -> &'static str { + match self { + Self::Decode => "put_decode", + Self::Dispatch => "put_dispatch", } - }; + } +} + +/// Decode one stored definition and register it on every core on this node. +/// +/// Shared by the post-apply lane and boot re-registration, so both install +/// the aggregate on the same set of cores. +pub async fn register_on_every_core( + shared: &SharedState, + tenant_id: u64, + name: &str, + def_bytes: &[u8], +) -> Result<(), RegisterFailure> { + let def: ContinuousAggregateDef = + zerompk::from_msgpack(def_bytes).map_err(|e| RegisterFailure { + error: crate::Error::Codec { + detail: format!("continuous aggregate '{name}': definition decode: {e}"), + }, + stage: RegisterStage::Decode, + database_id: 0, + })?; let database_id = def.database_id; - dispatch_meta( - shared, - database_id, - tenant_id, - &name, - MetaOp::RegisterContinuousAggregate { def }, - "register", - ) - .await; + let plan = PhysicalPlan::Meta(MetaOp::RegisterContinuousAggregate { def }); + let fanout = fanout_for(database_id, tenant_id, name); + dispatch_to_every_core(shared, &fanout, &plan) + .await + .map_err(|error| RegisterFailure { + error, + stage: RegisterStage::Dispatch, + database_id, + }) } /// Dispatch `MetaOp::UnregisterContinuousAggregate` to every core @@ -56,81 +109,41 @@ pub async fn delete_async( name: String, shared: Arc, ) { - dispatch_meta( - shared, + let plan = PhysicalPlan::Meta(MetaOp::UnregisterContinuousAggregate { name: name.clone() }); + let fanout = fanout_for(database_id, tenant_id, &name); + if let Err(error) = dispatch_to_every_core(&shared, &fanout, &plan).await { + crate::diag::continuous_aggregate_not_applied( + &error, + "delete_dispatch", + database_id, + tenant_id, + &name, + ); + } +} + +fn fanout_for(database_id: u64, tenant_id: u64, name: &str) -> CoreFanout<'_> { + CoreFanout { database_id, tenant_id, - &name, - MetaOp::UnregisterContinuousAggregate { name: name.clone() }, - "unregister", - ) - .await; + collection: CAGG_SENTINEL_COLLECTION, + what: "continuous aggregate change", + detail: name, + } } -async fn dispatch_meta( - shared: Arc, - database_id: u64, - tenant_id: u64, - name: &str, - op: MetaOp, - label: &'static str, -) { - let num_cores = { - let d = shared.dispatcher.lock().unwrap_or_else(|p| p.into_inner()); - d.num_cores() - }; - let timeout = std::time::Duration::from_secs(30); - let mut receivers = Vec::with_capacity(num_cores); +#[cfg(test)] +mod tests { + use super::*; - { - let mut d = shared.dispatcher.lock().unwrap_or_else(|p| p.into_inner()); - for core_id in 0..num_cores { - let request_id = shared.next_request_id(); - let request = Request { - request_id, - tenant_id: TenantId::new(tenant_id), - database_id: DatabaseId::new(database_id), - vshard_id: VShardId::new(core_id as u32), - plan: PhysicalPlan::Meta(op.clone()), - deadline: std::time::Instant::now() + timeout, - priority: Priority::Background, - trace_id: TraceId::generate(), - consistency: ReadConsistency::Eventual, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::AlreadyOrdered, - ), - }; - let rx = shared.tracker.register(request_id); - if d.dispatch_to_core(core_id, request).is_err() { - shared.tracker.cancel(&request_id); - continue; - } - receivers.push((core_id, rx)); - } - } - - for (core_id, mut rx) in receivers { - match tokio::time::timeout(timeout, async { rx.recv().await.ok_or(()) }).await { - Ok(Ok(resp)) if resp.status == Status::Ok => { - debug!(tenant_id, cagg = %name, core_id, %label, "continuous aggregate ack"); - } - _ => { - debug!( - tenant_id, - cagg = %name, - core_id, - %label, - "continuous aggregate dispatch: core did not ack" - ); - } - } + /// The fan-out carries the aggregate's own database, so an aggregate in + /// one database never registers in another's per-core manager map. + #[test] + fn the_fanout_carries_the_aggregates_own_database() { + let fanout = fanout_for(7, 3, "hourly"); + assert_eq!(fanout.database_id, 7); + assert_eq!(fanout.tenant_id, 3); + assert_eq!(fanout.detail, "hourly"); + assert_eq!(fanout.collection, CAGG_SENTINEL_COLLECTION); } } diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/core_fanout.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/core_fanout.rs index f5a8a0804..dfe5b6060 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/core_fanout.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/core_fanout.rs @@ -95,6 +95,26 @@ pub(super) async fn fan_out( target: &CoreFanout<'_>, plan: &PhysicalPlan, ) -> FanoutAnswers { + send_to_every_core(shared, target, plan) + .answers(target) + .await +} + +/// The requests of one fan-out, enqueued and not yet answered. +pub(super) struct SentFanout { + deadline: std::time::Instant, + /// Cores that refused the request at the enqueue. + refused: Vec, + receivers: Vec<(usize, ResponseReceiver)>, +} + +/// Enqueue `plan` on every core on this node. Awaits nothing, so a caller +/// can mark what the requests carry as sent in the same synchronous step. +pub(super) fn send_to_every_core( + shared: &SharedState, + target: &CoreFanout<'_>, + plan: &PhysicalPlan, +) -> SentFanout { let num_cores = { let d = shared.dispatcher.lock().unwrap_or_else(|p| p.into_inner()); d.num_cores() @@ -125,6 +145,7 @@ pub(super) async fn fan_out( txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: crate::bridge::envelope::Admission::Exempt( crate::bridge::envelope::ExemptReason::AlreadyOrdered, ), @@ -139,26 +160,41 @@ pub(super) async fn fan_out( } } - let mut pending: Vec<(usize, ResponseReceiver)> = Vec::new(); - let wait_until = tokio::time::Instant::from_std(deadline); - for (core_id, mut rx) in receivers { - match tokio::time::timeout_at(wait_until, final_response(&mut rx)).await { - Ok(Some(resp)) if resp.status == Status::Ok => { - debug!( - tenant = target.tenant_id, - collection = %target.collection, - detail = target.detail, - core_id, - what = target.what, - "post-apply core ack" - ); + SentFanout { + deadline, + refused, + receivers, + } +} + +impl SentFanout { + /// Collect the answers that arrive by the dispatch deadline. + pub(super) async fn answers(self, target: &CoreFanout<'_>) -> FanoutAnswers { + let SentFanout { + deadline, + mut refused, + receivers, + } = self; + let mut pending: Vec<(usize, ResponseReceiver)> = Vec::new(); + let wait_until = tokio::time::Instant::from_std(deadline); + for (core_id, mut rx) in receivers { + match tokio::time::timeout_at(wait_until, final_response(&mut rx)).await { + Ok(Some(resp)) if resp.status == Status::Ok => { + debug!( + tenant = target.tenant_id, + collection = %target.collection, + detail = target.detail, + core_id, + what = target.what, + "post-apply core ack" + ); + } + Ok(_) => refused.push(core_id), + Err(_) => pending.push((core_id, rx)), } - Ok(_) => refused.push(core_id), - Err(_) => pending.push((core_id, rx)), } + FanoutAnswers { refused, pending } } - - FanoutAnswers { refused, pending } } /// The request's final response, skipping partial frames. `None` once the diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/crdt_compact.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/crdt_compact.rs index 496153064..748909ca3 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/crdt_compact.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/crdt_compact.rs @@ -1,6 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Async post-apply for the COMPACT HISTORY catalog entry. +//! Post-apply for the COMPACT HISTORY catalog entry. //! //! `CompactHistory` deletes the checkpoint rows on every node and //! dispatches `CrdtOp::CompactAtVersion` to every core on that node. The @@ -11,13 +11,17 @@ //! checkpoint row that holds the version vector. This module reads the target //! from the entry, never from the catalog. //! -//! Nothing here can propagate a failure: the catalog rows are already -//! committed. Every failed stage files a `Capture` instead, because a node -//! silently keeping the compacted history is the defect this module exists -//! to stop. +//! Apply records the owed compaction in `_system.pending_history_compaction`. +//! A core answers `Ok` only after it compacted and published a CRDT +//! checkpoint, so the row is removed once every core answered `Ok`. A failed +//! fan-out keeps the row, and the retry worker and the boot drain re-drive +//! it. The row is the durable owner of the retry, so a failure never stops +//! the apply batch. //! -//! The COMPACT HISTORY handler calls this directly on a single node, where no -//! applier runs and the post-apply lane never fires. +//! The COMPACT HISTORY handler calls [`compact_async`] directly on a single +//! node, where no applier runs and the post-apply lane never fires. + +use tracing::warn; use crate::bridge::envelope::PhysicalPlan; use crate::control::state::SharedState; @@ -37,14 +41,18 @@ fn compact_plan(database_id: u64, collection: &str, target_version_json: String) }) } -/// Discard this node's oplog entries below the committed compaction target. +/// Compact this node's oplog to the committed target on every core, then +/// remove the owed-compaction row. +/// +/// On a failed fan-out the row stays with the attempt recorded, and the +/// error is returned. Every failure files a `Capture`. pub async fn compact_async( database_id: u64, tenant_id: u64, collection: &str, target_version_json: &str, shared: &SharedState, -) { +) -> crate::Result<()> { let plan = compact_plan(database_id, collection, target_version_json.to_string()); let target = CoreFanout { database_id, @@ -53,6 +61,7 @@ pub async fn compact_async( what: "history compaction", detail: "", }; + let catalog = shared.credentials.catalog(); if let Err(error) = dispatch_to_every_core(shared, &target, &plan).await { crate::diag::history_compaction_not_applied( @@ -62,12 +71,197 @@ pub async fn compact_async( tenant_id, collection, ); + if let Err(record_error) = catalog.record_pending_history_compaction_attempt( + database_id, + tenant_id, + collection, + target_version_json, + &error.to_string(), + ) { + warn!( + collection, + tenant = tenant_id, + error = %record_error, + "history compaction: failed to record the attempt on the owed row" + ); + } + return Err(error); + } + + catalog + .remove_pending_history_compaction(database_id, tenant_id, collection, target_version_json) + .inspect_err(|error| { + crate::diag::history_compaction_not_applied( + error, + "compact_row_remove", + database_id, + tenant_id, + collection, + ); + }) +} + +/// Re-drive every owed compaction on this node, and return how many are +/// still owed after the pass. +/// +/// A row that failed keeps its place for the next pass. `Err` means the +/// owed rows were not readable. +pub async fn drain_pending_compactions(shared: &SharedState) -> crate::Result { + let owed = shared + .credentials + .catalog() + .load_pending_history_compactions()?; + let mut still_owed = 0usize; + for row in owed { + if let Err(error) = compact_async( + row.database_id, + row.tenant_id, + &row.collection, + &row.target_version_json, + shared, + ) + .await + { + warn!( + collection = %row.collection, + tenant = row.tenant_id, + attempts = row.attempts.saturating_add(1), + error = %error, + "history compaction still owed; the next pass retries it" + ); + still_owed += 1; + } } + Ok(still_owed) } #[cfg(test)] mod tests { + use std::sync::Arc; + use std::sync::atomic::{AtomicBool, Ordering}; + use std::time::{Duration, Instant}; + use super::*; + use crate::bridge::dispatch::{BridgeResponse, CoreChannelDataSide, Dispatcher}; + use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; + use crate::control::catalog_entry::CatalogEntry; + use crate::control::catalog_entry::apply::apply_to; + use crate::types::Lsn; + use crate::wal::WalManager; + + const TARGET: &str = "{\"v\":1,\"vv\":{}}"; + + fn fixture() -> (Arc, CoreChannelDataSide, tempfile::TempDir) { + let directory = tempfile::tempdir().expect("temporary WAL directory"); + let wal = Arc::new( + WalManager::open_for_testing(&directory.path().join("compact.wal")).expect("test WAL"), + ); + let (dispatcher, mut sides) = Dispatcher::new(1, 64); + let side = sides.pop().expect("one data side"); + let state = SharedState::new(dispatcher, wal).expect("shared state"); + (state, side, directory) + } + + /// Answer every request with `status` until `stop` is set, then hand the + /// data side back for the next responder. + async fn answer_all( + state: Arc, + mut side: CoreChannelDataSide, + stop: Arc, + status: Status, + ) -> CoreChannelDataSide { + let deadline = Instant::now() + Duration::from_secs(10); + while !stop.load(Ordering::Relaxed) && Instant::now() < deadline { + if let Ok(request) = side.request_rx.try_pop() { + let error_code = (status == Status::Error).then(|| { + Box::new(ErrorCode::Internal { + detail: "injected compaction refusal".into(), + }) + }); + side.response_tx + .try_push(BridgeResponse { + inner: Response { + request_id: request.inner.request_id, + status, + attempt: 1, + partial: false, + payload: Payload::empty(), + watermark_lsn: Lsn::ZERO, + error_code, + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }, + }) + .expect("fake data-plane response queue has capacity"); + } + state.poll_and_route_responses(); + tokio::task::yield_now().await; + } + side + } + + fn owed( + state: &SharedState, + ) -> Vec { + state + .credentials + .catalog() + .load_pending_history_compactions() + .expect("load owed compactions") + } + + /// Apply writes the owed row. A fan-out a core refuses leaves it with the + /// attempt recorded, as a crash before the fan-out leaves it untouched. + /// The boot drain then compacts and removes it. + #[tokio::test(flavor = "multi_thread")] + async fn an_owed_compaction_survives_a_failed_fan_out_and_the_drain_removes_it() { + let (state, side, _directory) = fixture(); + apply_to( + &CatalogEntry::CompactHistory { + tenant_id: 7, + database_id: 3, + collection: "docs".to_string(), + doc_id: "doc-1".to_string(), + before_timestamp: 100, + target_version_json: TARGET.to_string(), + }, + state.credentials.catalog(), + ) + .expect("apply CompactHistory"); + assert_eq!(owed(&state).len(), 1, "apply records the owed compaction"); + + let stop = Arc::new(AtomicBool::new(false)); + let responder = tokio::spawn(answer_all( + Arc::clone(&state), + side, + Arc::clone(&stop), + Status::Error, + )); + let refused = compact_async(3, 7, "docs", TARGET, &state).await; + stop.store(true, Ordering::Relaxed); + let side = responder.await.expect("refusing responder"); + assert!(refused.is_err(), "a refused fan-out reports its error"); + let rows = owed(&state); + assert_eq!(rows.len(), 1, "a refused fan-out keeps the owed row"); + assert_eq!(rows[0].attempts, 1); + + let stop = Arc::new(AtomicBool::new(false)); + let responder = tokio::spawn(answer_all( + Arc::clone(&state), + side, + Arc::clone(&stop), + Status::Ok, + )); + let still_owed = drain_pending_compactions(&state).await; + stop.store(true, Ordering::Relaxed); + responder.await.expect("acknowledging responder"); + assert_eq!(still_owed.expect("drain reads the owed rows"), 0); + assert!( + owed(&state).is_empty(), + "the drain removes the compacted row" + ); + } /// The plan names the qualified collection and carries the committed /// target verbatim. A rewritten target compacts a node to a version its diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/dispatcher.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/dispatcher.rs index c2afb6bf2..4bae70d60 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/dispatcher.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/dispatcher.rs @@ -14,16 +14,17 @@ //! `DocumentOp::Scan` on the same node must find the collection registered in //! `doc_configs` so Binary Tuple (strict) documents decode correctly. //! -//! `tokio::task::block_in_place` is used for the Register dispatch so it runs -//! synchronously on the calling tokio worker thread. The raft tick loop always -//! runs on a tokio worker thread, so `block_in_place` is valid here. +//! [`run_post_apply_async_side_effects`] awaits the Register dispatch, and +//! the metadata applier awaits it before it advances past the entry. //! -//! Collection purge and materialized-view deletion have the same ordering -//! requirement: all local Data Plane cores must reclaim the old incarnation -//! before the applied-index watcher advances, because a same-name re-CREATE may -//! immediately follow. Reclaim failure is fatal to the applying node; the -//! durable pending-reclaim record is drained on restart before stale state can -//! be served. +//! Collection purge, materialized-view deletion, and the MOVE TENANT cutover +//! have the same ordering requirement: all local Data Plane cores must reclaim +//! the old incarnation before the applied-index watcher advances, because a +//! same-name re-CREATE can immediately follow. A reclaim that queued a durable +//! `_system.pending_reclaim` retry counts as done: the pending-reclaim worker +//! and the boot drain own it. A reclaim that queued nothing returns `Err`, the +//! metadata applier stops the batch at the entry, and the re-delivered entry +//! retries the reclaim. //! //! ## Applied-index contract for the vector-index variants //! @@ -32,7 +33,7 @@ //! the index with default build parameters, and `execute_set_vector_params` //! then refuses to reconfigure a materialized index — so a late `SetParams` //! never applies and the node serves the wrong index for good. The same -//! refusal makes a late `DropIndex` block the same-name re-CREATE that may +//! refusal makes a late `DropIndex` block the same-name re-CREATE that can //! follow it. //! //! ## Applied-index contract for the synonym-group variants @@ -45,20 +46,24 @@ //! both answer with the wrong row set and no error, which no later dispatch //! makes the client aware of. //! -//! ## Ordering for `CompactHistory` +//! ## Durability for `CompactHistory` //! -//! `CrdtOp::CompactAtVersion` has no such refusal: a compaction that lands -//! after a later read still discards the same oplog entries. It is spawned -//! fire-and-forget. +//! Apply records the owed compaction in `_system.pending_history_compaction`. +//! The fan-out is awaited, and the row is removed once every core compacted +//! and checkpointed. A failed fan-out keeps the row for the retry worker and +//! the boot drain, so it never stops the batch. //! -//! Variants without a read-after-apply dependency remain fire-and-forget. +//! Variants without a read-after-apply dependency remain fire-and-forget, +//! and only where boot rebuilds their effect from redb. use std::sync::Arc; +use tracing::warn; + use crate::control::catalog_entry::entry::CatalogEntry; use crate::control::state::SharedState; -use super::collection; +use super::collection::{self, ReclaimFailure}; /// Dispatch post-apply side effects of `entry`. Runs on every node (leader /// and followers) so each node's local Data Plane observes catalog mutations @@ -68,27 +73,48 @@ use super::collection; /// compares the boundary against WAL record LSNs, so it must be a WAL LSN, /// never a Raft log index. Every write of the reclaimed collection on this /// node sits below the next LSN this WAL assigns. -pub fn spawn_post_apply_async_side_effects(entry: CatalogEntry, shared: Arc) { +/// +/// `Err` means a reclaim failed with no durable retry queued. The caller must +/// not advance past the entry, so its re-delivery retries the reclaim. +/// +/// A Register that a Data Plane core did not acknowledge is logged, and the +/// batch advances. +/// +/// The future resolves once every awaited effect completed. A variant with +/// no awaited effect resolves on its first poll. +pub async fn run_post_apply_async_side_effects( + entry: CatalogEntry, + shared: Arc, +) -> crate::Result<()> { match entry { CatalogEntry::PutCollection(stored) => { - // SYNCHRONOUS: Register must complete before the applied-index - // watcher bumps so any subsequent scan on this node finds the - // collection in doc_configs. block_in_place is valid because - // the raft tick loop runs on a tokio worker thread. - tokio::task::block_in_place(|| { - tokio::runtime::Handle::current().block_on(async move { - collection::put_async(*stored, shared).await; - }); - }); + // AWAITED: Register completes before the applied-index watcher + // bumps, so any later scan on this node finds the collection in + // doc_configs. + // + // The stamp carries version 1 only on a create, and the validator + // admits a create only as a new incarnation. Its storage is + // cleared first, so it starts empty. + if stored.descriptor_version == 1 { + collection::clear_before_recreate( + &shared, + stored.database_id.as_u64(), + stored.tenant_id, + &stored.name, + ) + .await?; + } + let registered = collection::put_async(&stored, &shared).await; + register_outcome(registered, &stored.name); } CatalogEntry::PutCollectionIfAbsent(stored) => { // Register from the CANONICAL collection read back from the // catalog after apply — never from the carried entry. On the // no-op path (the collection already existed) the carried - // `stored` may hold a divergent incoming config; the catalog + // `stored` can hold a divergent incoming config; the catalog // holds the authoritative pre-existing one. Post-apply the // collection always exists (created or pre-existing), so the - // read-back is always Some; a None here would mean the redb + // read-back is always Some; a None here means the redb // write silently failed, so warn and skip rather than register // a divergent config. let canonical = shared @@ -99,16 +125,23 @@ pub fn spawn_post_apply_async_side_effects(entry: CatalogEntry, shared: Arc { - // SYNCHRONOUS: Register must complete before the - // applied-index watcher bumps so any subsequent scan on - // this node finds the collection in doc_configs. - // block_in_place is valid because the raft tick loop - // runs on a tokio worker thread. - tokio::task::block_in_place(|| { - tokio::runtime::Handle::current().block_on(async move { - collection::put_async(canonical, shared).await; - }); - }); + // AWAITED: Register completes before the applied-index + // watcher bumps, so any later scan on this node finds the + // collection in doc_configs. + // + // The canonical row carries the entry's clock only when + // this entry created it: a new incarnation, cleared first. + if canonical.modification_hlc == stored.modification_hlc { + collection::clear_before_recreate( + &shared, + canonical.database_id.as_u64(), + canonical.tenant_id, + &canonical.name, + ) + .await?; + } + let registered = collection::put_async(&canonical, &shared).await; + register_outcome(registered, &canonical.name); } None => { tracing::warn!( @@ -124,56 +157,42 @@ pub fn spawn_post_apply_async_side_effects(entry: CatalogEntry, shared: Arc { let purge_lsn = shared.wal.next_lsn().as_u64(); - let result = tokio::task::block_in_place(|| { - tokio::runtime::Handle::current().block_on(async move { - collection::reclaim_collection_storage( - &shared, - database_id, - tenant_id, - &name, - purge_lsn, - false, - ) - .await - }) - }); - if let Err(error) = result { - panic!("collection post-apply reclaim failed: {error}"); - } + let result = collection::reclaim_collection_storage( + &shared, + database_id, + tenant_id, + &name, + purge_lsn, + false, + ) + .await; + reclaim_outcome(result, "collection purge")?; } - // SYNCHRONOUS: every node must clear the view target's per-core state + // AWAITED: every node must clear the view target's per-core state // before its applied-index watcher advances. Otherwise a same-name // re-CREATE can observe cached aggregates from the dropped target. - // A failure is fatal: the metadata deletion is already committed, so - // continuing would serve an inconsistent catalog/Data Plane pair; - // restart safely reconstructs the in-memory cache from empty state. CatalogEntry::DeleteMaterializedView { database_id, tenant_id, name, + .. } => { let purge_lsn = shared.wal.next_lsn().as_u64(); - let result = tokio::task::block_in_place(|| { - tokio::runtime::Handle::current().block_on(async move { - super::materialized_view::delete_async( - database_id, - tenant_id, - name, - purge_lsn, - shared, - ) - .await - }) - }); - if let Err(error) = result { - panic!("materialized-view post-apply reclaim failed: {error}"); - } + let result = super::materialized_view::delete_async( + database_id, + tenant_id, + name, + purge_lsn, + shared, + ) + .await; + reclaim_outcome(result, "materialized-view target reclaim")?; } - // `PutContinuousAggregate` dispatches register to every core on - // this node so the local `continuous_agg_mgr` picks up the new - // definition after a raft commit without re-issuing DDL. + // Fire-and-forget: boot re-registers every stored aggregate from redb, + // so a register lost to a crash is rebuilt at the next boot. CatalogEntry::PutContinuousAggregate(stored) => { let tenant_id = stored.tenant_id; let name = stored.name.clone(); @@ -182,40 +201,27 @@ pub fn spawn_post_apply_async_side_effects(entry: CatalogEntry, shared: Arc { - tokio::task::block_in_place(|| { - tokio::runtime::Handle::current().block_on(async move { - super::vector::put_async(*stored, Arc::clone(&shared)).await; - }); - }); + super::vector::put_async(*stored, shared).await; } - // SYNCHRONOUS: a same-name re-CREATE may follow immediately, and + // AWAITED: a same-name re-CREATE can follow immediately, and // `SetParams` is refused while the dropped index is still materialized. CatalogEntry::DeleteVectorIndexParams { database_id, tenant_id, collection, field_name, + .. } => { - tokio::task::block_in_place(|| { - tokio::runtime::Handle::current().block_on(async move { - super::vector::delete_async( - database_id, - tenant_id, - collection, - field_name, - Arc::clone(&shared), - ) - .await; - }); - }); + super::vector::delete_async(database_id, tenant_id, collection, field_name, shared) + .await; } - // `CompactHistory` dispatches the oplog compaction to every core so - // each node discards the same history the leader does. A late - // compaction still succeeds, so this stays fire-and-forget. + // AWAITED: the owed-compaction row apply wrote is removed once every + // core compacted durably. A failed fan-out leaves the row to the + // retry worker and the boot drain. CatalogEntry::CompactHistory { tenant_id, collection, @@ -223,51 +229,99 @@ pub fn spawn_post_apply_async_side_effects(entry: CatalogEntry, shared: Arc { - tokio::spawn(async move { - super::crdt_compact::compact_async( - database_id, - tenant_id, - &collection, - &target_version_json, - &shared, - ) - .await; - }); + let result = super::crdt_compact::compact_async( + database_id, + tenant_id, + &collection, + &target_version_json, + &shared, + ) + .await; + if let Err(error) = result { + warn!( + collection = %collection, + tenant = tenant_id, + error = %error, + "history compaction post-apply: still owed on this node; the retry worker \ + re-drives it" + ); + } } - // `DeleteContinuousAggregate` dispatches unregister to every - // core so per-node runtime state is reclaimed symmetrically. + // Fire-and-forget: boot re-registers only the aggregates redb still + // holds, so an unregister lost to a crash is rebuilt at the next boot. CatalogEntry::DeleteContinuousAggregate { database_id, tenant_id, name, + .. } => { tokio::spawn(async move { super::continuous_aggregate::delete_async(database_id, tenant_id, name, shared) .await; }); } - // SYNCHRONOUS: the group must reach every core's FTS backend before - // the applied-index watcher bumps. A query that runs first expands + // AWAITED: the group must reach every core's FTS backend before the + // applied-index watcher bumps. A query that runs first expands // nothing and returns fewer rows with no error. CatalogEntry::PutSynonymGroup(stored) => { - tokio::task::block_in_place(|| { - tokio::runtime::Handle::current().block_on(async move { - super::synonym_group::put_async(*stored, &shared).await; - }); - }); + super::synonym_group::put_async(*stored, &shared).await; } - // SYNCHRONOUS: a query that runs before the removal lands keeps - // expanding terms the statement already dropped. + // AWAITED: a query that runs before the removal lands keeps expanding + // terms the statement already dropped. CatalogEntry::DeleteSynonymGroup { database_id, tenant_id, name, + .. } => { - tokio::task::block_in_place(|| { - tokio::runtime::Handle::current().block_on(async move { - super::synonym_group::delete_async(database_id, tenant_id, name, &shared).await; - }); - }); + super::synonym_group::delete_async(database_id, tenant_id, name, &shared).await; + } + // AWAITED: the moved rows must leave the source key before the + // applied-index watcher bumps. Otherwise a same-name CREATE in the + // source database observes them. + CatalogEntry::MoveTenantCutover { + source_db_id, + collections, + .. + } => { + let result = + super::move_tenant::reclaim_moved_sources(&shared, source_db_id, &collections) + .await; + reclaim_outcome(result, "MOVE TENANT source reclaim")?; + } + // AWAITED: every shadow collection registers before the applied-index + // watcher bumps, as a `PutCollection` does, so a write or scan on the + // clone finds its config in doc_configs. + CatalogEntry::CloneDatabase { + target_descriptor, .. + } => { + let registered = collection::clone_shadows_async(target_descriptor.id, &shared).await; + register_outcome(registered, &target_descriptor.name); + } + // AWAITED: every core opens the array before the applied-index + // watcher advances, purging the tombstone of a prior incarnation, so + // the next statement on this node reads the new incarnation. + CatalogEntry::PutArray(stored) => { + super::array::open_on_every_core(&shared, &stored).await?; + } + // AWAITED: a same-name CREATE can follow at once, and a moved store + // must sit under its target key before the target's `PutArray` opens + // it. + CatalogEntry::DeleteArray { + database_id, + tenant_id, + name, + moved_to, + .. + } => { + super::array::delete_on_every_core( + &shared, + database_id, + tenant_id, + &name, + moved_to.map(|m| m.target_db_id), + ) + .await?; } // ── Variants with no async side effect today ───────────────────────── // Listed explicitly (no `_ => {}`) so the compiler forces a decision @@ -327,13 +381,14 @@ pub fn spawn_post_apply_async_side_effects(entry: CatalogEntry, shared: Arc { + // Clone copy-on-write rows have no in-memory mirror and no Data + // Plane side effect. + | CatalogEntry::PutCloneCopyup { .. } + | CatalogEntry::PutCloneTombstone { .. } + | CatalogEntry::PutKvCloneTombstone { .. } + // Clone source drain claims are read only from the catalog by the + // singleton worker's recovery. + | CatalogEntry::PutCloneSourceDrain(_) + | CatalogEntry::DeleteCloneSourceDrain { .. } => { let _ = shared; } } + Ok(()) +} + +/// Log a Register that a Data Plane core did not acknowledge. The metadata +/// applier runs the post-apply lane, and no client waits on it. The batch +/// advances: the catalog row is durable, and boot seeds every core's config +/// from it. +fn register_outcome(result: crate::Result<()>, collection: &str) { + if let Err(error) = result { + tracing::error!( + collection, + error = %error, + "catalog_entry: Register barrier failed — one or more Data Plane cores \ + did not acknowledge the schema update; this node may serve stale schema" + ); + } +} + +/// Map a reclaim result onto the post-apply contract. +/// +/// A queued durable retry is owned by the pending-reclaim worker and the boot +/// drain, so it counts as done. Anything else is `Err`, which stops the apply +/// batch at this entry. +fn reclaim_outcome(result: Result<(), ReclaimFailure>, what: &str) -> crate::Result<()> { + match result { + Ok(()) => Ok(()), + Err(failure) if failure.retry_queued => { + warn!( + error = %failure.error, + "{what} post-apply: reclaim failed on this node; the pending-reclaim worker \ + owns the retry" + ); + Ok(()) + } + Err(failure) => Err(failure.error), + } } diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/materialized_view.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/materialized_view.rs index cf0bda0a3..602904759 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/materialized_view.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/materialized_view.rs @@ -16,9 +16,8 @@ use crate::control::state::SharedState; /// /// `reclaim_collection_storage` provides the durable pending-reclaim fallback /// and dispatches `UnregisterCollection` to every local Data Plane core. The -/// caller treats an error as fatal because the catalog deletion is already -/// committed; serving through an incomplete reclaim would violate object -/// incarnation isolation. +/// catalog deletion is already committed, so an error with no retry queued +/// stops the apply batch, and the re-delivered entry retries the reclaim. pub async fn delete_async( database_id: u64, tenant_id: u64, diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/mod.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/mod.rs index 524aa93f0..2207389b5 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/mod.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/mod.rs @@ -4,8 +4,8 @@ //! //! # Per-node contract //! -//! `spawn_post_apply_async_side_effects` is invoked unconditionally on -//! **every node** (leader and followers alike) from the metadata +//! `run_post_apply_async_side_effects` is awaited unconditionally on +//! **every node** (leader and followers alike) by the metadata //! commit applier — there is NO `is_leader()` gate. Any side effect //! that must fire on every replica (WAL tombstone append, Data Plane //! `MetaOp::UnregisterCollection` dispatch, storage-reclaim spawning, @@ -19,13 +19,15 @@ //! //! [sync]: crate::control::catalog_entry::post_apply::sync +pub mod array; pub mod collection; pub mod continuous_aggregate; mod core_fanout; pub mod crdt_compact; mod dispatcher; pub mod materialized_view; +pub mod move_tenant; pub mod synonym_group; pub mod vector; -pub use dispatcher::spawn_post_apply_async_side_effects; +pub use dispatcher::run_post_apply_async_side_effects; diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/move_tenant.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/move_tenant.rs new file mode 100644 index 000000000..0a1d289a5 --- /dev/null +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/move_tenant.rs @@ -0,0 +1,56 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Source-storage reclaim for `CatalogEntry::MoveTenantCutover`. +//! +//! A collection's storage key and home vShard both hash its database. The +//! cutover re-issues every moved row into the target database before it +//! proposes the entry, so the rows under the source key are dead on every +//! node. Each node reclaims them through the collection purge path: WAL and +//! redb tombstones, then `UnregisterCollection` on every local core. +//! +//! A replayed entry finds nothing under the source key, and the purge of an +//! absent collection is a no-op. + +use crate::control::security::catalog::StoredCollection; +use crate::control::state::SharedState; + +use super::collection::{ReclaimFailure, reclaim_collection_storage}; + +/// Reclaim the source-keyed storage of every moved collection on this node. +/// +/// A collection whose reclaim queued a durable retry does not stop the loop: +/// the pending-reclaim worker owns it, and the remaining collections still +/// need their reclaim. The first such failure is returned once every +/// collection was tried. A failure with no retry queued returns at once. +pub(crate) async fn reclaim_moved_sources( + shared: &SharedState, + source_db_id: u64, + collections: &[StoredCollection], +) -> Result<(), ReclaimFailure> { + let mut queued: Option = None; + for coll in collections { + // The purge boundary is a WAL LSN of this node: every source write + // this node holds sits below it. + let purge_lsn = shared.wal.next_lsn().as_u64(); + match reclaim_collection_storage( + shared, + source_db_id, + coll.tenant_id, + &coll.name, + purge_lsn, + false, + ) + .await + { + Ok(()) => {} + Err(failure) if failure.retry_queued => { + queued.get_or_insert(failure); + } + Err(failure) => return Err(failure), + } + } + match queued { + Some(failure) => Err(failure), + None => Ok(()), + } +} diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/synonym_group.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/synonym_group.rs index 2766af27c..2d8a9193a 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/synonym_group.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/synonym_group.rs @@ -110,6 +110,7 @@ mod tests { name: "db_terms".to_string(), terms: vec!["database".to_string(), "db".to_string()], created_at: 42, + modification_hlc: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs index 25bb5fe9c..8b570ddbc 100644 --- a/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs +++ b/nodedb/src/control/catalog_entry/post_apply/async_dispatch/vector.rs @@ -31,13 +31,15 @@ use std::sync::Arc; use tokio::sync::oneshot; use crate::bridge::envelope::PhysicalPlan; -use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; +use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner, SentRecords}; use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, TenantId}; use nodedb_physical::physical_plan::VectorOp; use nodedb_types::StoredVectorIndexParams; -use super::core_fanout::{CoreFanout, DISPATCH_TIMEOUT, FanoutAnswers, fan_out}; +use super::core_fanout::{ + CoreFanout, DISPATCH_TIMEOUT, FanoutAnswers, fan_out, send_to_every_core, +}; /// The longest a parameter install waits for its cores to answer: the /// `SetParams` dispatch deadline, then the reshape's. Its record's window @@ -162,9 +164,11 @@ async fn install_params( report(&error, "set_params_wal_append", &target); } - // The cores hold the record from here. It closes from their answers. - minted.mark_sent(); - let set_params = fan_out(&shared, &fanout(&target), &plan).await; + // The cores hold the record from the enqueue. It closes from their + // answers, and an aborted task holds it. + let sent = send_to_every_core(&shared, &fanout(&target), &plan); + let minted = SentRecords::sent(minted); + let set_params = sent.answers(&fanout(&target)).await; let stage = if set_params.pending.is_empty() { let refused = set_params.refused; if refused.is_empty() { @@ -197,7 +201,7 @@ async fn finish_put( shared: &SharedState, entry: &StoredVectorIndexParams, stage: PutStage, - minted: MintedRecords, + minted: SentRecords, ) { let target = IndexTarget::of(entry); let (refused, rebuild) = match stage { @@ -222,7 +226,7 @@ fn close_put( target: &IndexTarget<'_>, refused: Vec, reshaped: Vec, - minted: MintedRecords, + minted: SentRecords, ) { let missed: Vec = refused .into_iter() @@ -306,9 +310,11 @@ async fn drop_index(name: IndexName, shared: Arc, ready: oneshot::S Err(error) => report(&error, "drop_index_wal_append", &target), } - // The cores hold the record from here. It closes from their answers. - minted.mark_sent(); - let answers = fan_out(&shared, &fanout(&target), &plan).await; + // The cores hold the record from the enqueue. It closes from their + // answers, and an aborted task holds it. + let sent = send_to_every_core(&shared, &fanout(&target), &plan); + let minted = SentRecords::sent(minted); + let answers = sent.answers(&fanout(&target)).await; // Cores still working past the deadline answer later. The caller moves // on while this task waits for their final answers. let _ = ready.send(()); @@ -317,7 +323,7 @@ async fn drop_index(name: IndexName, shared: Arc, ready: oneshot::S } /// Close a drop's window from the cores that did not drop the index. -fn close_drop(target: &IndexTarget<'_>, refused: Vec, minted: MintedRecords) { +fn close_drop(target: &IndexTarget<'_>, refused: Vec, minted: SentRecords) { if refused.is_empty() { minted.settle(); return; @@ -397,6 +403,7 @@ mod tests { pq_m: 8, ivf_cells: 64, ivf_nprobe: 16, + modification_hlc: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/catalog_entry/post_apply/change_stream.rs b/nodedb/src/control/catalog_entry/post_apply/change_stream.rs index 8c7533970..d832115a2 100644 --- a/nodedb/src/control/catalog_entry/post_apply/change_stream.rs +++ b/nodedb/src/control/catalog_entry/post_apply/change_stream.rs @@ -24,6 +24,8 @@ pub fn put(stored: ChangeStreamDef, shared: Arc) { &stored.owner, &shared, ); + // The sink groups exist before any offset commit of theirs applies. + crate::event::cdc::sink_owner::register_sink_groups(&shared.group_registry, &stored); shared.stream_registry.register(stored); } @@ -42,9 +44,9 @@ pub fn delete(database_id: u64, tenant_id: u64, name: String, shared: Arc) { // Replicate the owner record on every node so cluster-wide - // `is_owner` / `check` evaluations succeed. Handlers no longer - // call `set_owner` directly — ownership is entirely a side - // effect of the parent `PutCollection` apply. + // `is_owner` / `check` evaluations succeed. Handlers never call + // `set_owner` directly. Ownership is a side effect of the parent + // `PutCollection` apply. shared.permissions.install_replicated_owner(&StoredOwner { database_id: stored.database_id.as_u64(), object_type: "collection".into(), @@ -44,21 +45,54 @@ pub fn put_owner_sync(stored: &StoredCollection, shared: Arc) { } } +/// Every shadow collection a `CloneDatabase` entry stamped into `target`. +/// +/// The clone writes the shadows straight into the catalog, never as +/// `PutCollection` entries, so the clone's post-apply runs the +/// `PutCollection` effects for each one. The target is a new database, so +/// every collection it holds is a shadow. +pub fn clone_shadows( + target: DatabaseId, + shared: &SharedState, +) -> crate::Result> { + shared.credentials.catalog().load_all_collections(target) +} + +/// Synchronous half of `CloneDatabase` post-apply: install the owner record +/// and queue the tree definition of every shadow collection, as the +/// synchronous half of `PutCollection` does for one collection. +pub fn clone_shadows_sync(target: DatabaseId, shared: &Arc) { + let shadows = match clone_shadows(target, shared) { + Ok(shadows) => shadows, + Err(e) => { + tracing::error!( + database_id = target.as_u64(), + error = %e, + "post_apply: shadow collections of a committed clone could not be read" + ); + return; + } + }; + for stored in &shadows { + put_owner_sync(stored, Arc::clone(shared)); + queue_tree_def_sync(stored, shared); + } +} + /// Queue the tree-definition change a committed collection descriptor makes. /// Every node runs this, so each node's permission cache learns the tree /// defined through any node. Planning and lease coverage move the queue into /// the cache. pub fn queue_tree_def_sync(stored: &StoredCollection, shared: &SharedState) { match TreeDefChange::from_collection(stored) { - Ok(Some(change)) => { + Ok(change) => { change.note_committed(shared.authorization_fence.sources()); shared.authorization_fence.tree_defs().push(change); } - Ok(None) => {} Err(e) => { // The DDL commits the serialization of a parsed definition, so // this JSON always parses. The prior definition stays in place: - // removing it would drop the filter and expose rows. + // removing it drops the filter and exposes rows. tracing::error!( collection = %stored.name, tenant = stored.tenant_id, @@ -77,12 +111,8 @@ pub fn queue_tree_def_removal_sync( name: &str, shared: &SharedState, ) { - if database_id != DatabaseId::DEFAULT.as_u64() { - return; - } let change = TreeDefChange::Unregister { - tenant_id, - collection: name.to_owned(), + key: TreeKey::new(DatabaseId::new(database_id), tenant_id, name), }; change.note_committed(shared.authorization_fence.sources()); shared.authorization_fence.tree_defs().push(change); @@ -92,39 +122,29 @@ pub fn queue_tree_def_removal_sync( /// Data Plane so subsequent `DocumentOp::Scan` calls find the collection /// in `doc_configs` and decode strict (Binary Tuple) documents correctly. /// -/// Called via `block_in_place` inside `spawn_post_apply_async_side_effects` -/// for `PutCollection` — it completes synchronously before the applied-index -/// watcher bumps, making it part of the applied-index contract. +/// Awaited by `run_post_apply_async_side_effects` for `PutCollection` — it +/// completes before the applied-index watcher bumps, making it part of the +/// applied-index contract. /// /// A collection this entry names as a materialized-sum SOURCE is re-registered /// too. The binding travels on the target, but the config that decides whether a /// write folds it is derived for the source, so propagating the target alone /// leaves every source on this node folding nothing. -pub async fn put_async(stored: StoredCollection, shared: Arc) { +/// +/// `Err` means one or more Data Plane cores did not acknowledge a Register. +/// The dispatcher decides where that error goes, by who awaits the lane. +pub async fn put_async(stored: &StoredCollection, shared: &SharedState) -> crate::Result<()> { use crate::control::server::shared::ddl::neutral::collection::{ dispatch_register_for_sum_sources, dispatch_register_from_stored, }; - let registered = match dispatch_register_from_stored(&shared, &stored).await { - Ok(()) => dispatch_register_for_sum_sources(&shared, &stored).await, - Err(e) => Err(e), - }; - match registered { - Ok(()) => { - debug!( - collection = %stored.name, - "catalog_entry: Register dispatched to all Data Plane cores" - ); - } - Err(e) => { - tracing::error!( - collection = %stored.name, - error = %e, - "catalog_entry: Register barrier failed — one or more Data Plane cores \ - did not acknowledge the schema update; this node may serve stale schema" - ); - } - } + dispatch_register_from_stored(shared, stored).await?; + dispatch_register_for_sum_sources(shared, stored).await?; + debug!( + collection = %stored.name, + "catalog_entry: Register dispatched to all Data Plane cores" + ); + Ok(()) } /// Synchronous half of `PurgeCollection` post-apply: remove the @@ -140,13 +160,13 @@ pub fn purge_sync(database_id: u64, tenant_id: u64, name: String, shared: Arc:" (see grant_target_for_collection - // in the pgwire GRANT handler). - let grant_target = format!("collection:{tenant_id}:{name}"); + // Grants on the purged collection leave the in-memory grant set, or they + // outlive the catalog row they reference. + let grant_target = crate::control::security::permission::collection_target( + crate::types::DatabaseId::new(database_id), + crate::types::TenantId::new(tenant_id), + &name, + ); let grants_removed = shared.permissions.remove_grants_for_target(&grant_target); // The indexes go with the collection, so their in-memory ownership entries // go too. Read before `finalize_purge` removes the registry rows (it runs @@ -203,12 +223,12 @@ pub fn deactivate(tenant_id: u64, name: String, _shared: Arc) { // Ownership is intentionally preserved on soft-delete. The // primary `StoredCollection` record is kept for audit / undrop // (see `CatalogEntry::DeactivateCollection`); removing the - // in-memory owner entry would split truth from the preserved + // in-memory owner entry splits truth from the preserved // primary row's `stored.owner` field and force any future // UNDROP to be admin-only. `is_owner` returning true for a // soft-deleted collection is the correct semantics: the former // owner remains the rightful restorer. Hard deletion of the - // collection (not wired today) would clear both halves via + // collection (not wired today) clears both halves via // `delete_parent_owner` in the applier. debug!( collection = %name, diff --git a/nodedb/src/control/catalog_entry/post_apply/consumer_group.rs b/nodedb/src/control/catalog_entry/post_apply/consumer_group.rs index 1fbc07d13..75fcf92aa 100644 --- a/nodedb/src/control/catalog_entry/post_apply/consumer_group.rs +++ b/nodedb/src/control/catalog_entry/post_apply/consumer_group.rs @@ -7,7 +7,7 @@ //! resolves on all of them without a restart. use crate::control::state::SharedState; -use crate::event::cdc::consumer_group::ConsumerGroupDef; +use crate::event::cdc::consumer_group::{ConsumerGroupDef, OffsetCommit}; use crate::types::DatabaseId; /// Install an applied consumer group. An existing registration is kept, so the @@ -67,6 +67,41 @@ pub fn migrate_stream(def: &ConsumerGroupDef, legacy_stream: &str, shared: &Shar } } +/// Raise an applied commit's offsets on this node, when the registered group +/// is the incarnation the commit targets. The committed-message delivery +/// cursors and the trigger firing cursors have no registered group: every +/// node keeps them. +pub fn commit_offsets(commit: &OffsetCommit, shared: &SharedState) { + let current = shared.group_registry.get( + commit.database_id, + commit.tenant_id, + &commit.stream_name, + &commit.group_name, + ); + let system_cursor = crate::event::topic::committed::is_publish_cursor(commit) + || crate::event::trigger::lane::is_action_cursor(commit); + if !system_cursor && current.is_none_or(|def| def.modification_hlc != commit.group_hlc) { + return; + } + if let Err(error) = shared.offset_store.advance_offsets( + commit.database_id, + commit.tenant_id, + &commit.stream_name, + &commit.group_name, + &commit.offsets, + ) { + tracing::error!( + database_id = commit.database_id.as_u64(), + tenant_id = commit.tenant_id, + stream = %commit.stream_name, + group = %commit.group_name, + %error, + "replicated consumer offset commit did not persist on this node; \ + a consumer served here re-reads events after its previous commit" + ); + } +} + /// Clear one group's durable offsets on this node. /// /// The offset store is node-local, so a failure here leaves a cursor no diff --git a/nodedb/src/control/catalog_entry/post_apply/database.rs b/nodedb/src/control/catalog_entry/post_apply/database.rs index 65964ab61..16a8879cf 100644 --- a/nodedb/src/control/catalog_entry/post_apply/database.rs +++ b/nodedb/src/control/catalog_entry/post_apply/database.rs @@ -1,10 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Synchronous post-apply side effects for database catalog entries. +//! Post-apply side effects for database catalog entries. //! //! Database descriptors and grants are read directly from redb on the hot //! path (no separate in-memory registry), so most arms are no-ops. The -//! `DeleteDatabase` arm releases the dropped scope's quota caps on every node. +//! synchronous `DeleteDatabase` arm releases the dropped scope's quota caps. use std::sync::Arc; @@ -16,8 +16,8 @@ use crate::control::state::SharedState; /// Post-apply for `PutDatabase` — no in-memory cache to update. pub fn put(_descriptor: DatabaseDescriptor, _shared: Arc) {} -/// Post-apply for `DeleteDatabase` — release the quota caps of the dropped -/// scope, its tenants' caps included. +/// Synchronous post-apply for `DeleteDatabase` — release the quota caps of +/// the dropped scope, its tenants' caps included. pub fn delete(db_id: u64, shared: Arc) { super::quota::release_database_scope(DatabaseId::new(db_id), &shared); } diff --git a/nodedb/src/control/catalog_entry/post_apply/gateway_invalidation.rs b/nodedb/src/control/catalog_entry/post_apply/gateway_invalidation.rs index 36f08e97d..2c211d479 100644 --- a/nodedb/src/control/catalog_entry/post_apply/gateway_invalidation.rs +++ b/nodedb/src/control/catalog_entry/post_apply/gateway_invalidation.rs @@ -47,6 +47,7 @@ use crate::control::state::SharedState; /// | PutRetentionPolicy / DeleteRetentionPolicy | ✅ yes | `auto_tier` rewrites a timeseries scan onto tier aggregates, so the policy is baked into the plan | /// | PutAlertRule / DeleteAlertRule | ❌ no | alert rules drive their own eval loop and never enter a PhysicalPlan | /// | Topic / consumer-group variants | ❌ no | Event Plane delivery identities that never enter a PhysicalPlan | +/// | PutArray / DeleteArray | ✅ yes | array schema and routing prefix are baked into the plan | pub(crate) fn invalidate_gateway_cache_for_entry(entry: &CatalogEntry, shared: &Arc) { let Some(inv) = shared.gateway_invalidator.get() else { return; @@ -156,7 +157,7 @@ pub(crate) fn invalidate_gateway_cache_for_entry(entry: &CatalogEntry, shared: & // does not directly modify any PhysicalPlan. The `materialized_sum_sources` // field in DocumentOp::Register is set at collection-register time // (driven by PutCollection), not updated independently by - // PutMaterializedView. Any schema change that would affect plans + // PutMaterializedView. Any schema change that affects plans // cascades through PutCollection instead. } CatalogEntry::DeleteMaterializedView { .. } => { @@ -232,7 +233,7 @@ pub(crate) fn invalidate_gateway_cache_for_entry(entry: &CatalogEntry, shared: & } CatalogEntry::DeleteIndexRecord { collection, .. } => { // A cached plan still holding an IndexLookup against the dropped - // index would read an index the engine no longer has. + // index reads an index the engine no longer has. inv.invalidate(collection, 0); } @@ -315,10 +316,14 @@ pub(crate) fn invalidate_gateway_cache_for_entry(entry: &CatalogEntry, shared: & | CatalogEntry::DeleteTopicWithConsumerGroups { .. } | CatalogEntry::PutConsumerGroupIfAbsent(_) | CatalogEntry::DeleteConsumerGroup { .. } - | CatalogEntry::MigrateConsumerGroupStream { .. } => { + | CatalogEntry::MigrateConsumerGroupStream { .. } + | CatalogEntry::CommitConsumerOffsets(_) => { // no-op: topics and consumer groups are Event Plane delivery // identities and never enter a query plan. } + CatalogEntry::PutBackupScheduleMark(_) => { + // no-op: a backup schedule mark never enters a query plan. + } CatalogEntry::PutCheckpoint(_) | CatalogEntry::DeleteCheckpoint { .. } | CatalogEntry::CompactHistory { .. } => { @@ -332,6 +337,23 @@ pub(crate) fn invalidate_gateway_cache_for_entry(entry: &CatalogEntry, shared: & // no-op: vector build parameters are read by the Data Plane index, // never by a cached plan. } + CatalogEntry::PutCloneCopyup { .. } + | CatalogEntry::PutCloneTombstone { .. } + | CatalogEntry::PutKvCloneTombstone { .. } => { + // no-op: a clone read consults these rows at execution, never + // through a cached plan. + } + CatalogEntry::PutCloneSourceDrain(_) | CatalogEntry::DeleteCloneSourceDrain { .. } => { + // no-op: drain claims reach no plan. + } + CatalogEntry::PutArray(stored) => { + // An array plan carries its schema and routing prefix. Version 0 + // evicts every cached plan over the name. + inv.invalidate(&stored.name, 0); + } + CatalogEntry::DeleteArray { name, .. } => { + inv.invalidate(name, 0); + } CatalogEntry::PutColumnStats(rows) => { // The join and aggregate cost models read these rows, so a cached // plan carries the previous figures. Version 0 evicts every plan @@ -492,8 +514,8 @@ mod tests { // We test each Delete* variant directly (simple { tenant_id, name } shape) and // rely on the compiler's exhaustiveness check for the corresponding Put* arm. // The Put* variants for complex nested types (StoredTrigger, StoredFunction, - // etc.) are covered by the same `// no-op` arm; constructing them would - // require pages of boilerplate without adding behavioral coverage. + // etc.) are covered by the same `// no-op` arm; constructing them + // requires pages of boilerplate without adding behavioral coverage. fn assert_noop( shared: &Arc, @@ -530,6 +552,8 @@ mod tests { database_id: 0, tenant_id: 1, name: "seq".into(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }, "DeleteSequence", ); @@ -559,6 +583,8 @@ mod tests { database_id: crate::types::DatabaseId::DEFAULT, tenant_id: 1, name: "trig".into(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }, "DeleteTrigger", ); @@ -571,6 +597,8 @@ mod tests { database_id: crate::types::DatabaseId::DEFAULT, tenant_id: 1, name: "fn_".into(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }, "DeleteFunction", ); @@ -583,6 +611,8 @@ mod tests { database_id: crate::types::DatabaseId::DEFAULT, tenant_id: 1, name: "proc".into(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }, "DeleteProcedure", ); @@ -607,6 +637,7 @@ mod tests { database_id: crate::types::DatabaseId::DEFAULT.as_u64(), tenant_id: 1, name: "stream".into(), + target_hlc: nodedb_types::Hlc::ZERO, }, "DeleteChangeStream", ); @@ -649,6 +680,8 @@ mod tests { database_id: 2, tenant_id: 1, name: "mv_orders".into(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }, "DeleteMaterializedView", ); diff --git a/nodedb/src/control/catalog_entry/post_apply/mod.rs b/nodedb/src/control/catalog_entry/post_apply/mod.rs index 564e1af61..a18fc8035 100644 --- a/nodedb/src/control/catalog_entry/post_apply/mod.rs +++ b/nodedb/src/control/catalog_entry/post_apply/mod.rs @@ -10,14 +10,16 @@ //! synchronous in-memory cache updates **inline** on the raft //! applier thread, BEFORE the metadata applier bumps the //! `AppliedIndexWatcher`. -//! - [`spawn_post_apply_async_side_effects`] (in `async_dispatch`) -//! spawns tokio tasks for the genuinely async work — runs on -//! **every node** (leader and followers) so each node's local -//! Data Plane observes catalog mutations symmetrically. +//! - [`run_post_apply_async_side_effects`] (in `async_dispatch`) +//! runs the async work — on **every node** (leader and followers) +//! so each node's local Data Plane observes catalog mutations +//! symmetrically. The metadata applier awaits it before it +//! advances past the entry. // Per-family modules (existing). pub mod alert_rule; pub mod api_key; +pub mod array; pub mod auth_user; pub mod change_stream; pub mod collection; @@ -51,11 +53,23 @@ mod async_dispatch; pub(crate) mod gateway_invalidation; mod sync; -pub(crate) use async_dispatch::collection::{ReclaimFailure, reclaim_collection_storage}; -pub(crate) use async_dispatch::crdt_compact::compact_async; -pub use async_dispatch::spawn_post_apply_async_side_effects; +pub(crate) use async_dispatch::array::{ + delete_on_every_core as drop_array_on_every_core, + open_on_every_core as open_array_on_every_core, +}; +pub(crate) use async_dispatch::collection::{ + ReclaimFailure, clear_before_recreate, reclaim_collection_storage, +}; +pub(crate) use async_dispatch::continuous_aggregate::{ + RegisterFailure as ContinuousAggregateRegisterFailure, + delete_async as unregister_continuous_aggregate, + register_on_every_core as register_continuous_aggregate_on_every_core, +}; +pub(crate) use async_dispatch::crdt_compact::{compact_async, drain_pending_compactions}; +pub use async_dispatch::run_post_apply_async_side_effects; pub(crate) use async_dispatch::synonym_group::delete_async as remove_synonym_group; pub(crate) use async_dispatch::synonym_group::put_async as install_synonym_group; +pub(crate) use async_dispatch::vector::delete_async as remove_vector_index_params; pub(crate) use async_dispatch::vector::longest_core_wait as vector_install_longest_core_wait; pub(crate) use async_dispatch::vector::put_async as install_vector_index_params; pub use sync::apply_post_apply_side_effects_sync; diff --git a/nodedb/src/control/catalog_entry/post_apply/sync.rs b/nodedb/src/control/catalog_entry/post_apply/sync.rs index 0f9d854f8..d7d65020d 100644 --- a/nodedb/src/control/catalog_entry/post_apply/sync.rs +++ b/nodedb/src/control/catalog_entry/post_apply/sync.rs @@ -7,19 +7,17 @@ //! readers are guaranteed to see every sync side effect of every //! entry up to N — no tokio spawn race. //! -//! Previously `sync` and `async` were combined into a single -//! `tokio::spawn`, so a freshly-applied `PutUser` could bump the -//! watcher while its `install_replicated_user` task was still queued -//! on the scheduler. Tests that waited on `applied_index` and then -//! immediately polled `credentials.get_user` would flake whenever -//! the scheduler ran them in that order. Keeping this function +//! A `tokio::spawn` here lets a freshly-applied `PutUser` bump the +//! watcher while its `install_replicated_user` task still sat queued on +//! the scheduler. A reader that waited on `applied_index` then polled +//! `credentials.get_user` misses the user. Keeping this function //! **sync** and inline avoids that race by construction. use std::sync::Arc; use super::gateway_invalidation::invalidate_gateway_cache_for_entry; use super::{ - alert_rule, api_key, auth_user, change_stream, collection, consumer_group, + alert_rule, api_key, array, auth_user, change_stream, collection, consumer_group, continuous_aggregate, custom_type, database, function, materialized_view, owner, permission, procedure, quota, redaction, retention_policy, rls, role, schedule, scope_grant, scope_quota, sequence, streaming_materialized_view, synonym_group, tenant, topic, trigger, user, @@ -43,7 +41,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { // Owner record install is sync; Data Plane register is - // the async part, handled by `spawn_post_apply_async_side_effects`. + // the async part, handled by `run_post_apply_async_side_effects`. collection::put_owner_sync(stored, Arc::clone(shared)); collection::queue_tree_def_sync(stored, shared); } @@ -54,7 +52,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { collection::purge_sync(*database_id, *tenant_id, name.clone(), Arc::clone(shared)); collection::queue_tree_def_removal_sync(*database_id, *tenant_id, name, shared); @@ -89,6 +88,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { sequence::delete(*database_id, *tenant_id, name.clone(), Arc::clone(shared)); } @@ -102,6 +102,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { trigger::delete(*database_id, *tenant_id, name.clone(), shared); } @@ -112,6 +113,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { function::delete(*database_id, *tenant_id, name.clone(), shared); } @@ -122,6 +124,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { procedure::delete(*database_id, *tenant_id, name.clone(), Arc::clone(shared)); } @@ -142,6 +145,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { change_stream::delete(*database_id, *tenant_id, name.clone(), Arc::clone(shared)); } @@ -177,6 +181,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { materialized_view::delete(*database_id, *tenant_id, name.clone(), Arc::clone(shared)); } @@ -202,6 +207,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { continuous_aggregate::delete( *database_id, @@ -308,6 +314,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { synonym_group::delete(*database_id, *tenant_id, name.clone(), Arc::clone(shared)); } @@ -352,6 +359,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { // WAL replay barrier only; no in-memory cache to refresh. @@ -411,6 +419,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { topic::delete_with_consumer_groups(*database_id, *tenant_id, name, shared); } @@ -422,6 +431,7 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { consumer_group::delete(*database_id, *tenant_id, stream_name, name, shared); } @@ -448,6 +458,18 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { + // no-op: clone reads look copy-on-write rows up in the catalog. + } + CatalogEntry::PutCloneSourceDrain(_) | CatalogEntry::DeleteCloneSourceDrain { .. } => { + // no-op: only the singleton worker's recovery reads the claims, + // from the catalog. + } + CatalogEntry::PutArray(stored) => array::put_sync(stored, shared), + // The mirror changes in the async lane, under the incarnation's gate. + CatalogEntry::DeleteArray { .. } => {} CatalogEntry::MoveTenantCutover { tenant_id, source_db_id, @@ -462,5 +484,10 @@ pub fn apply_post_apply_side_effects_sync(entry: &CatalogEntry, shared: &Arc { + consumer_group::commit_offsets(commit, shared); + } + // The scheduler reads the mark from the catalog row apply wrote. + CatalogEntry::PutBackupScheduleMark(_) => {} } } diff --git a/nodedb/src/control/catalog_overlay/collection.rs b/nodedb/src/control/catalog_overlay/collection.rs index b61ee17eb..b1144d928 100644 --- a/nodedb/src/control/catalog_overlay/collection.rs +++ b/nodedb/src/control/catalog_overlay/collection.rs @@ -40,6 +40,7 @@ fn targets(entry: &CatalogEntry, database_id: DatabaseId, tenant_id: u64, name: database_id: entry_db, tenant_id: entry_tenant, name: entry_name, + .. } => *entry_db == database_id.as_u64() && *entry_tenant == tenant_id && entry_name == name, _ => false, } @@ -180,6 +181,8 @@ mod tests { database_id: DatabaseId::DEFAULT.as_u64(), tenant_id: TENANT, name: name.to_owned(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/catalog_overlay/function.rs b/nodedb/src/control/catalog_overlay/function.rs index 98259fb17..f6f9442fd 100644 --- a/nodedb/src/control/catalog_overlay/function.rs +++ b/nodedb/src/control/catalog_overlay/function.rs @@ -21,6 +21,7 @@ fn targets(entry: &CatalogEntry, database_id: DatabaseId, tenant_id: u64, name: database_id: entry_db, tenant_id: entry_tenant, name: entry_name, + .. } => *entry_db == database_id && *entry_tenant == tenant_id && entry_name == name, _ => false, } @@ -90,6 +91,8 @@ mod tests { database_id: DatabaseId::DEFAULT, tenant_id: 1, name: name.to_owned(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/catalog_overlay/materialized_view.rs b/nodedb/src/control/catalog_overlay/materialized_view.rs index 2f1c4be00..3e5823d43 100644 --- a/nodedb/src/control/catalog_overlay/materialized_view.rs +++ b/nodedb/src/control/catalog_overlay/materialized_view.rs @@ -21,6 +21,7 @@ fn targets(entry: &CatalogEntry, database_id: u64, tenant_id: u64, name: &str) - database_id: entry_database, tenant_id: entry_tenant, name: entry_name, + .. } => *entry_database == database_id && *entry_tenant == tenant_id && entry_name == name, _ => false, } @@ -81,6 +82,8 @@ mod tests { database_id: 2, tenant_id: 1, name: name.to_owned(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/catalog_overlay/mod.rs b/nodedb/src/control/catalog_overlay/mod.rs index f09f42da5..3d2b5aa29 100644 --- a/nodedb/src/control/catalog_overlay/mod.rs +++ b/nodedb/src/control/catalog_overlay/mod.rs @@ -15,9 +15,8 @@ //! `control::sequence::ddl_overlay` — because `NEXTVAL` mutates shared //! runtime state a rolled-back transaction must never let another connection //! observe. Array DDL is not buffered at all: `CREATE`/`ALTER`/`DROP ARRAY` -//! apply and persist synchronously in the write funnel -//! (`array_catalog::ddl::apply_authorized_ddl`), regardless of transaction -//! state, so there is no uncommitted state for an overlay to replay. +//! refuse to run inside a transaction block (`array_catalog::ddl`), so +//! there is no uncommitted state for an overlay to replay. mod collection; mod core; diff --git a/nodedb/src/control/catalog_overlay/procedure.rs b/nodedb/src/control/catalog_overlay/procedure.rs index d9b1c2c40..79dcef3bc 100644 --- a/nodedb/src/control/catalog_overlay/procedure.rs +++ b/nodedb/src/control/catalog_overlay/procedure.rs @@ -21,6 +21,7 @@ fn targets(entry: &CatalogEntry, database_id: DatabaseId, tenant_id: u64, name: database_id: entry_db, tenant_id: entry_tenant, name: entry_name, + .. } => *entry_db == database_id && *entry_tenant == tenant_id && entry_name == name, _ => false, } @@ -87,6 +88,8 @@ mod tests { database_id: DatabaseId::DEFAULT, tenant_id: 1, name: name.to_owned(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/catalog_overlay/trigger.rs b/nodedb/src/control/catalog_overlay/trigger.rs index 68f024850..e1d753b64 100644 --- a/nodedb/src/control/catalog_overlay/trigger.rs +++ b/nodedb/src/control/catalog_overlay/trigger.rs @@ -21,6 +21,7 @@ fn targets(entry: &CatalogEntry, database_id: DatabaseId, tenant_id: u64, name: database_id: entry_db, tenant_id: entry_tenant, name: entry_name, + .. } => *entry_db == database_id && *entry_tenant == tenant_id && entry_name == name, _ => false, } @@ -94,6 +95,8 @@ mod tests { database_id: DatabaseId::DEFAULT, tenant_id: 1, name: name.to_owned(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/catalog_overlay/vector_index_params.rs b/nodedb/src/control/catalog_overlay/vector_index_params.rs index 791cf3323..a6dbc7749 100644 --- a/nodedb/src/control/catalog_overlay/vector_index_params.rs +++ b/nodedb/src/control/catalog_overlay/vector_index_params.rs @@ -33,6 +33,7 @@ fn targets(entry: &CatalogEntry, target: &Target<'_>) -> bool { tenant_id, collection, field_name, + .. } => { *database_id == target.database_id && *tenant_id == target.tenant_id @@ -92,6 +93,7 @@ mod tests { pq_m: 0, ivf_cells: 0, ivf_nprobe: 0, + modification_hlc: nodedb_types::Hlc::ZERO, } } @@ -119,6 +121,7 @@ mod tests { tenant_id: 1, collection: "docs".to_owned(), field_name: "emb".to_owned(), + target_hlc: nodedb_types::Hlc::ZERO, }); assert!(resolve(Some(params(16))).is_none()); }) diff --git a/nodedb/src/control/change_stream/journal/codec.rs b/nodedb/src/control/change_stream/journal/codec.rs new file mode 100644 index 000000000..b66e97ca6 --- /dev/null +++ b/nodedb/src/control/change_stream/journal/codec.rs @@ -0,0 +1,178 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Key and value encodings of the change-feed journal. +//! +//! Keys are big-endian so a partition's rows sort in position order. + +use nodedb_types::RowIdentity; + +use crate::control::change_stream::{ + ChangeEvent, ChangeOperation, ChangePartition, PositionedChange, SequencedChangeEvent, +}; +use crate::event::cdc::CdcOffset; +use crate::types::{DatabaseId, Lsn, TenantId}; + +pub(super) const PARTITION_KEY_LEN: usize = 9; +const POSITION_LEN: usize = 24; +pub(super) const ROW_KEY_LEN: usize = PARTITION_KEY_LEN + POSITION_LEN; +const FEED_LEN: usize = 2 * POSITION_LEN + 8; + +const TAG_GROUP: u8 = 0; +const TAG_CALVIN: u8 = 1; + +fn word(bytes: &[u8], at: usize) -> Option { + Some(u64::from_be_bytes(bytes.get(at..at + 8)?.try_into().ok()?)) +} + +pub(super) fn partition_key(partition: ChangePartition) -> [u8; PARTITION_KEY_LEN] { + let (tag, id) = match partition { + ChangePartition::Group(group) => (TAG_GROUP, group), + ChangePartition::Calvin(vshard) => (TAG_CALVIN, u64::from(vshard)), + }; + let mut out = [0u8; PARTITION_KEY_LEN]; + out[0] = tag; + out[1..].copy_from_slice(&id.to_be_bytes()); + out +} + +pub(super) fn decode_partition(bytes: &[u8]) -> Option { + let id = word(bytes, 1)?; + match *bytes.first()? { + TAG_GROUP => Some(ChangePartition::Group(id)), + TAG_CALVIN => u32::try_from(id).ok().map(ChangePartition::Calvin), + _ => None, + } +} + +fn encode_position(position: CdcOffset, out: &mut [u8]) { + out[..8].copy_from_slice(&position.epoch.to_be_bytes()); + out[8..16].copy_from_slice(&position.index.to_be_bytes()); + out[16..24].copy_from_slice(&position.sequence.to_be_bytes()); +} + +fn decode_position(bytes: &[u8]) -> Option { + Some(CdcOffset::at( + word(bytes, 0)?, + word(bytes, 8)?, + word(bytes, 16)?, + )) +} + +pub(super) fn row_key(partition: ChangePartition, position: CdcOffset) -> [u8; ROW_KEY_LEN] { + let mut out = [0u8; ROW_KEY_LEN]; + out[..PARTITION_KEY_LEN].copy_from_slice(&partition_key(partition)); + encode_position(position, &mut out[PARTITION_KEY_LEN..]); + out +} + +/// The first and last row key `partition` can hold. +pub(super) fn row_bounds(partition: ChangePartition) -> ([u8; ROW_KEY_LEN], [u8; ROW_KEY_LEN]) { + let mut last = row_key(partition, CdcOffset::ZERO); + last[PARTITION_KEY_LEN..].fill(u8::MAX); + (row_key(partition, CdcOffset::ZERO), last) +} + +pub(super) fn row_position(key: &[u8]) -> Option { + decode_position(key.get(PARTITION_KEY_LEN..)?) +} + +/// What the journal holds of one partition. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) struct FeedRecord { + /// The journal holds every event of the partition above this position. + pub after: CdcOffset, + /// The position up to which the journal holds the partition. + pub through: CdcOffset, + /// Rows the partition holds. + pub rows: u64, +} + +impl FeedRecord { + pub fn encode(&self) -> [u8; FEED_LEN] { + let mut out = [0u8; FEED_LEN]; + encode_position(self.after, &mut out[..POSITION_LEN]); + encode_position(self.through, &mut out[POSITION_LEN..2 * POSITION_LEN]); + out[2 * POSITION_LEN..].copy_from_slice(&self.rows.to_be_bytes()); + out + } + + pub fn decode(bytes: &[u8]) -> Option { + Some(Self { + after: decode_position(bytes.get(..POSITION_LEN)?)?, + through: decode_position(bytes.get(POSITION_LEN..2 * POSITION_LEN)?)?, + rows: word(bytes, 2 * POSITION_LEN)?, + }) + } +} + +/// One journaled change. +#[derive(Debug, Clone, zerompk::ToMessagePack, zerompk::FromMessagePack)] +#[msgpack(map)] +pub(super) struct StoredChange { + pub database_id: u64, + pub tenant_id: u64, + pub collection: String, + pub document_id: String, + pub operation: String, + pub timestamp_ms: u64, + pub lsn: u64, +} + +impl StoredChange { + pub fn of(event: &SequencedChangeEvent) -> Self { + Self { + database_id: event.database_id().as_u64(), + tenant_id: event.tenant_id.as_u64(), + collection: event.collection.clone(), + document_id: event.document_id.to_string(), + operation: event.operation.as_str().to_owned(), + timestamp_ms: event.timestamp_ms, + lsn: event.lsn.as_u64(), + } + } + + pub fn at(self, position: CdcOffset) -> PositionedChange { + PositionedChange { + position, + database_id: DatabaseId::new(self.database_id), + event: ChangeEvent { + lsn: Lsn::new(self.lsn), + tenant_id: TenantId::new(self.tenant_id), + collection: self.collection, + // The journal stores the identity as text; wrap it verbatim. + document_id: RowIdentity::from_user_key(self.document_id), + operation: match self.operation.as_str() { + "UPDATE" => ChangeOperation::Update, + "DELETE" => ChangeOperation::Delete, + _ => ChangeOperation::Insert, + }, + timestamp_ms: self.timestamp_ms, + after: None, + }, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn keys_and_feeds_round_trip_and_sort_by_position() { + for partition in [ChangePartition::Group(7), ChangePartition::Calvin(3)] { + assert_eq!(decode_partition(&partition_key(partition)), Some(partition)); + let low = row_key(partition, CdcOffset::data_event(0, 5, 1)); + let high = row_key(partition, CdcOffset::data_event(0, 5, 2)); + assert!(low < high); + assert_eq!(row_position(&high), Some(CdcOffset::data_event(0, 5, 2))); + let (first, last) = row_bounds(partition); + assert!(first <= low && high <= last); + } + let feed = FeedRecord { + after: CdcOffset::whole_index(4), + through: CdcOffset::whole_index(9), + rows: 3, + }; + assert_eq!(FeedRecord::decode(&feed.encode()), Some(feed)); + } +} diff --git a/nodedb/src/control/change_stream/journal/mod.rs b/nodedb/src/control/change_stream/journal/mod.rs new file mode 100644 index 000000000..ae3adf2d8 --- /dev/null +++ b/nodedb/src/control/change_stream/journal/mod.rs @@ -0,0 +1,6 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod codec; +pub mod store; + +pub use store::ChangeJournal; diff --git a/nodedb/src/control/change_stream/journal/store.rs b/nodedb/src/control/change_stream/journal/store.rs new file mode 100644 index 000000000..255befe65 --- /dev/null +++ b/nodedb/src/control/change_stream/journal/store.rs @@ -0,0 +1,292 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The durable change-feed journal. +//! +//! The journal keeps what this node published on each partition, bounded per +//! partition, so the replay ring comes back after a restart and a cursor +//! inside retention resumes with no reset. +//! +//! - A data group's changes are journaled when the apply loop saves the +//! group's applied floor, before the floor: every entry at or below a +//! saved floor has its changes on disk. Entries above it are delivered +//! again after a restart and publish again. +//! - A Calvin transaction's changes are journaled before its applied marker +//! is durable. +//! - A forwarded run is journaled before the receiver acks it. +//! +//! Each partition keeps `after`, the position above which it holds every +//! event, and `through`, the position up to which it holds the partition. A +//! write whose `after` lies above the stored `through` is a hole: the +//! journal drops the partition's rows and restarts it there. + +use std::path::Path; + +use redb::{Database, ReadableDatabase, ReadableTable, TableDefinition}; + +use crate::control::change_stream::{ChangePartition, PositionedChange, SequencedChangeEvent}; +use crate::event::cdc::CdcOffset; + +use super::codec::{ + FeedRecord, StoredChange, decode_partition, partition_key, row_bounds, row_key, row_position, +}; + +const ROWS: TableDefinition<&[u8], &[u8]> = TableDefinition::new("change_feed_rows"); +const FEEDS: TableDefinition<&[u8], &[u8]> = TableDefinition::new("change_feed_feeds"); + +fn storage(detail: impl std::fmt::Display) -> crate::Error { + crate::Error::Storage { + engine: "change_stream".into(), + detail: format!("change-feed journal: {detail}"), + } +} + +/// One partition as the journal holds it. +#[derive(Debug, Clone)] +pub(crate) struct JournalFeed { + pub partition: ChangePartition, + pub after: CdcOffset, + pub through: CdcOffset, + /// In position order. + pub changes: Vec, +} + +/// The durable journal of the Control-Plane change feeds. +pub struct ChangeJournal { + db: Database, + /// Rows each partition keeps. + retain: u64, +} + +impl ChangeJournal { + /// Open or create the journal at `{dir}/change_feed.redb`, keeping + /// `retain` rows per partition. + pub fn open(dir: &Path, retain: usize) -> crate::Result { + std::fs::create_dir_all(dir) + .map_err(|e| storage(format!("create {}: {e}", dir.display())))?; + let path = dir.join("change_feed.redb"); + let db = Database::create(&path) + .map_err(|e| storage(format!("open {}: {e}", path.display())))?; + let txn = db.begin_write().map_err(storage)?; + { + txn.open_table(ROWS).map_err(storage)?; + txn.open_table(FEEDS).map_err(storage)?; + } + txn.commit().map_err(storage)?; + Ok(Self { + db, + retain: u64::try_from(retain.max(1)).unwrap_or(u64::MAX), + }) + } + + /// Journal `events` of `partition`, durably. The publisher holds every + /// event of the partition above `feed_after`, and has seen it up to + /// `through`. + pub fn persist( + &self, + partition: ChangePartition, + feed_after: CdcOffset, + through: CdcOffset, + events: &[SequencedChangeEvent], + ) -> crate::Result<()> { + let key = partition_key(partition); + let (first, last) = row_bounds(partition); + let txn = self.db.begin_write().map_err(storage)?; + { + let mut feeds = txn.open_table(FEEDS).map_err(storage)?; + let mut rows = txn.open_table(ROWS).map_err(storage)?; + let stored = feeds + .get(key.as_slice()) + .map_err(storage)? + .and_then(|bytes| FeedRecord::decode(bytes.value())); + let mut feed = match stored { + None => FeedRecord { + after: feed_after, + through: feed_after, + rows: 0, + }, + Some(stored) if feed_after <= stored.through => stored, + Some(_) => { + // A hole: no cursor below it can be served, so the rows + // below it go. + rows.retain_in(first.as_slice()..=last.as_slice(), |_, _| false) + .map_err(storage)?; + FeedRecord { + after: feed_after, + through: feed_after, + rows: 0, + } + } + }; + for event in events { + if event.position() <= feed.after { + continue; + } + let value = zerompk::to_msgpack_vec(&StoredChange::of(event)).map_err(storage)?; + let row = row_key(partition, event.position()); + if rows + .insert(row.as_slice(), value.as_slice()) + .map_err(storage)? + .is_none() + { + feed.rows += 1; + } + } + feed.through = feed.through.max(through); + if feed.rows > self.retain { + let excess = usize::try_from(feed.rows - self.retain).unwrap_or(usize::MAX); + let mut doomed = Vec::with_capacity(excess); + for entry in rows + .range(first.as_slice()..=last.as_slice()) + .map_err(storage)? + .take(excess) + { + let (row, _) = entry.map_err(storage)?; + doomed.push(row.value().to_vec()); + } + for row in doomed { + rows.remove(row.as_slice()).map_err(storage)?; + if let Some(position) = row_position(&row) { + feed.after = feed.after.max(position); + } + feed.rows -= 1; + } + } + feeds + .insert(key.as_slice(), feed.encode().as_slice()) + .map_err(storage)?; + } + txn.commit().map_err(storage) + } + + /// Every journaled partition with its rows. + pub(crate) fn load(&self) -> crate::Result> { + let txn = self.db.begin_read().map_err(storage)?; + let feeds = txn.open_table(FEEDS).map_err(storage)?; + let rows = txn.open_table(ROWS).map_err(storage)?; + let mut out = Vec::new(); + for entry in feeds.iter().map_err(storage)? { + let (key, value) = entry.map_err(storage)?; + let partition = decode_partition(key.value()) + .ok_or_else(|| storage("a feed key did not decode"))?; + let feed = FeedRecord::decode(value.value()) + .ok_or_else(|| storage("a feed record did not decode"))?; + let (first, last) = row_bounds(partition); + let mut changes = Vec::new(); + for row in rows + .range(first.as_slice()..=last.as_slice()) + .map_err(storage)? + { + let (row, value) = row.map_err(storage)?; + let position = + row_position(row.value()).ok_or_else(|| storage("a row key did not decode"))?; + let stored: StoredChange = zerompk::from_msgpack(value.value()).map_err(storage)?; + changes.push(stored.at(position)); + } + out.push(JournalFeed { + partition, + after: feed.after, + through: feed.through, + changes, + }); + } + Ok(out) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::change_stream::{ChangeEvent, ChangeOperation}; + use crate::types::{DatabaseId, Lsn, TenantId}; + + const GROUP: ChangePartition = ChangePartition::Group(2); + + fn event(index: u64) -> SequencedChangeEvent { + SequencedChangeEvent::new( + GROUP, + CdcOffset::data_event(0, index, 1), + CdcOffset::ZERO, + DatabaseId::DEFAULT, + ChangeEvent { + lsn: Lsn::new(index), + tenant_id: TenantId::new(1), + collection: "orders".into(), + document_id: nodedb_types::RowIdentity::from_user_key(format!("row-{index}")), + operation: ChangeOperation::Update, + timestamp_ms: index * 10, + after: None, + }, + ) + } + + #[test] + fn a_journaled_feed_loads_back_in_position_order() { + let dir = tempfile::tempdir().expect("tempdir"); + let journal = ChangeJournal::open(dir.path(), 8).expect("open"); + journal + .persist( + GROUP, + CdcOffset::whole_index(0), + CdcOffset::whole_index(2), + &[event(1), event(2)], + ) + .expect("persist"); + journal + .persist( + GROUP, + CdcOffset::whole_index(0), + CdcOffset::whole_index(5), + &[event(4)], + ) + .expect("persist"); + drop(journal); + + let journal = ChangeJournal::open(dir.path(), 8).expect("reopen"); + let feeds = journal.load().expect("load"); + assert_eq!(feeds.len(), 1); + let feed = &feeds[0]; + assert_eq!(feed.after, CdcOffset::whole_index(0)); + assert_eq!(feed.through, CdcOffset::whole_index(5)); + let positions: Vec = feed.changes.iter().map(|c| c.position).collect(); + assert_eq!( + positions, + vec![ + CdcOffset::data_event(0, 1, 1), + CdcOffset::data_event(0, 2, 1), + CdcOffset::data_event(0, 4, 1), + ] + ); + assert_eq!(feed.changes[1].event.document_id.as_str(), "row-2"); + assert_eq!(feed.changes[1].event.operation, ChangeOperation::Update); + assert_eq!(feed.changes[1].event.timestamp_ms, 20); + } + + #[test] + fn retention_and_holes_raise_the_journal_floor() { + let dir = tempfile::tempdir().expect("tempdir"); + let journal = ChangeJournal::open(dir.path(), 2).expect("open"); + journal + .persist( + GROUP, + CdcOffset::whole_index(0), + CdcOffset::whole_index(3), + &[event(1), event(2), event(3)], + ) + .expect("persist"); + let feed = &journal.load().expect("load")[0]; + assert_eq!(feed.changes.len(), 2, "retention keeps two rows"); + assert_eq!(feed.after, CdcOffset::data_event(0, 1, 1)); + + journal + .persist( + GROUP, + CdcOffset::whole_index(9), + CdcOffset::whole_index(10), + &[event(10)], + ) + .expect("persist past a hole"); + let feed = &journal.load().expect("load")[0]; + assert_eq!(feed.after, CdcOffset::whole_index(9)); + assert_eq!(feed.changes.len(), 1); + } +} diff --git a/nodedb/src/control/change_stream/mod.rs b/nodedb/src/control/change_stream/mod.rs index f0e2f8da3..667101373 100644 --- a/nodedb/src/control/change_stream/mod.rs +++ b/nodedb/src/control/change_stream/mod.rs @@ -1,10 +1,14 @@ // SPDX-License-Identifier: BUSL-1.1 +pub mod journal; pub mod live_set; pub mod stream; +pub use journal::ChangeJournal; pub use live_set::LiveSubscriptionSet; pub use stream::{ - ChangeCursor, ChangeEvent, ChangeOperation, ChangeStream, CursorParseError, ReplayError, - ReplaySnapshot, ReplayStart, SequencedChangeEvent, Subscription, broadcast_notify_to_cluster, + ChangeCursor, ChangeEvent, ChangeOperation, ChangePartition, ChangeStream, ChangeStreamError, + CursorParseError, CursorStep, ReplayError, ReplaySnapshot, ReplayStart, ReplayedChange, + SequencedChangeEvent, Subscription, }; +pub(crate) use stream::{ChangeRun, PositionedChange}; diff --git a/nodedb/src/control/change_stream/stream/bus.rs b/nodedb/src/control/change_stream/stream/bus.rs index 1a2bbb620..5cc496569 100644 --- a/nodedb/src/control/change_stream/stream/bus.rs +++ b/nodedb/src/control/change_stream/stream/bus.rs @@ -1,43 +1,52 @@ // SPDX-License-Identifier: BUSL-1.1 -use std::collections::VecDeque; -use std::sync::Arc; +//! The Control-Plane change stream: a bounded replay ring over partitioned +//! feeds, and the live broadcast that follows it. +//! +//! In a cluster every replica of a data group publishes the group's writes +//! as the apply loop settles them, in log order, at their log positions. +//! The group leader forwards the feed's new events to the nodes that do not +//! replicate the group (see [`super::fanout`]). Every node therefore holds +//! the full feed at the same positions, and a cursor from any node resumes +//! on any other. Each feed is journaled durably (see [`super::journaling`]), +//! so a cursor inside retention also resumes after a restart. + +use std::collections::BTreeMap; use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::{Arc, OnceLock}; use nodedb_types::RowIdentity; -use tracing::{debug, trace, warn}; +use tracing::debug; +use crate::control::change_stream::journal::ChangeJournal; +use crate::event::cdc::CdcOffset; +use crate::event::cross_shard::types::NotifyBroadcastMsg; use crate::types::{DatabaseId, Lsn, TenantId}; -use super::{ChangeCursor, ChangeEvent, ChangeOperation, SequencedChangeEvent, Subscription}; - -/// Replay start is either an acknowledged opaque cursor or an initial timestamp. -#[derive(Clone, Copy, Debug)] -pub enum ReplayStart { - Cursor(ChangeCursor), - Timestamp(u64), -} - -/// A consistent ring-buffer replay snapshot and its publication high-water mark. -#[derive(Clone, Debug)] -pub struct ReplaySnapshot { - pub events: Vec, - pub snapshot_cursor: ChangeCursor, -} +use super::fanout::ChangeFanout; +use super::ring::{ + AppendedRun, ChangeRun, PositionedChange, ReplayError, ReplayRing, ReplayScope, ReplaySnapshot, + ReplayStart, +}; +use super::{ChangeEvent, ChangeOperation, ChangePartition, SequencedChangeEvent, Subscription}; -/// A cursor cannot safely resume this in-memory stream; clients must reset. -#[derive(Clone, Copy, Debug, PartialEq, Eq)] -pub enum ReplayError { - Expired, -} +/// A settled entry's changes, staged by the write funnel until the apply +/// loop settles the entry in log order. +type StagedChanges = Vec<(DatabaseId, ChangeEvent)>; -struct ReplayState { - epoch: u128, - next_sequence: u64, - recent_changes: VecDeque, +pub(super) struct BusState { + pub(super) ring: ReplayRing, + /// `(group, log index)` to the entry's changes. Bounded by the apply + /// pipeline's window: every staged entry settles. + staged: BTreeMap<(u64, u64), StagedChanges>, + /// Per partition, the position through which this node forwarded it. + forwarded: BTreeMap, + /// Per data group, settled events above the group's saved applied + /// floor, which the journal takes when the floor rises past them. + pub(super) group_pending: BTreeMap>, } -/// Control-plane WAL change notification bus. +/// Control-plane change notification bus. pub struct ChangeStream { sender: tokio::sync::broadcast::Sender, legacy_sender: tokio::sync::broadcast::Sender, @@ -45,8 +54,11 @@ pub struct ChangeStream { active_subscriptions: Arc, events_published: AtomicU64, last_lsn: AtomicU64, - replay_state: std::sync::Mutex, - recent_capacity: usize, + state: std::sync::Mutex, + fanout: ChangeFanout, + /// The durable journal every feed is written to, once attached. + pub(super) journal: OnceLock>, + capacity: usize, } impl ChangeStream { @@ -61,15 +73,29 @@ impl ChangeStream { active_subscriptions: Arc::new(AtomicU64::new(0)), events_published: AtomicU64::new(0), last_lsn: AtomicU64::new(0), - replay_state: std::sync::Mutex::new(ReplayState { - epoch: new_epoch(), - next_sequence: 1, - recent_changes: VecDeque::with_capacity(capacity), + state: std::sync::Mutex::new(BusState { + ring: ReplayRing::new(capacity), + staged: BTreeMap::new(), + forwarded: BTreeMap::new(), + group_pending: BTreeMap::new(), }), - recent_capacity: capacity, + fanout: ChangeFanout::default(), + journal: OnceLock::new(), + capacity, } } + /// Events the replay ring holds. + pub fn capacity(&self) -> usize { + self.capacity + } + + pub(super) fn lock(&self) -> std::sync::MutexGuard<'_, BusState> { + self.state + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + } + pub fn subscribe( &self, collection_filter: Option, @@ -87,6 +113,8 @@ impl ChangeStream { self.subscribe_scoped(collection_filter, tenant_filter, Some(database_id)) } + /// The receivers and the start cursor are taken under the ring lock, so + /// the subscription receives exactly the events past its start cursor. fn subscribe_scoped( &self, collection_filter: Option, @@ -102,6 +130,7 @@ impl ChangeStream { ?database_filter, "change stream: new subscription" ); + let state = self.lock(); Subscription { id, receiver: self.legacy_sender.subscribe(), @@ -111,50 +140,176 @@ impl ChangeStream { sequenced_receiver: self.sender.subscribe(), database_filter, active_counter: Arc::clone(&self.active_subscriptions), + start: state.ring.head(), } } - /// Publish into the default database for legacy producer compatibility. - pub fn publish(&self, event: ChangeEvent) { - self.publish_in_database(DatabaseId::DEFAULT, event); + /// Hold the changes of the data-group entry at `(group_id, log_index)` + /// until the apply loop settles it. + pub(crate) fn stage( + &self, + group_id: u64, + log_index: u64, + database_id: DatabaseId, + events: Vec, + ) { + if events.is_empty() { + return; + } + let mut state = self.lock(); + state + .staged + .entry((group_id, log_index)) + .or_default() + .extend(events.into_iter().map(|event| (database_id, event))); } - /// Allocate sequence, append the replay ring, and publish under one mutex. - pub fn publish_in_database(&self, database_id: DatabaseId, event: ChangeEvent) { - self.last_lsn - .fetch_max(event.lsn.as_u64(), Ordering::Relaxed); - let mut state = self - .replay_state - .lock() - .unwrap_or_else(|poisoned| poisoned.into_inner()); - if state.next_sequence == 0 { - state.epoch = new_epoch(); - state.next_sequence = 1; - state.recent_changes.clear(); + /// Publish the staged changes of `group_id`'s entries `first..=last`, + /// which the apply loop settled in log order. Returns the run for the + /// nodes that do not replicate the group. + pub(crate) fn settle_group(&self, group_id: u64, first: u64, last: u64) -> AppendedRun { + let mut state = self.lock(); + let mut changes = Vec::new(); + let keys: Vec<(u64, u64)> = state + .staged + .range((group_id, 0)..=(group_id, last)) + .map(|(key, _)| *key) + .collect(); + for key in keys { + let Some(staged) = state.staged.remove(&key) else { + continue; + }; + // An entry below `first` settled in an earlier pass without + // this node ever publishing it; it is not part of this run. + if key.1 < first { + continue; + } + for (ordinal, (database_id, event)) in staged.into_iter().enumerate() { + changes.push(PositionedChange { + position: CdcOffset::data_event(0, key.1, ordinal as u64 + 1), + database_id, + event, + }); + } + } + let run = ChangeRun { + partition: ChangePartition::Group(group_id), + after: Some(CdcOffset::whole_index(first.saturating_sub(1))), + through: CdcOffset::whole_index(last), + changes, + }; + let appended = self.append_locked(&mut state, run); + self.queue_group(&mut state, group_id, &appended); + appended + } + + /// Record that a snapshot install covered `group_id`'s entries through + /// `last_included_index`. This node never applies them, so it holds none + /// of their events. A cursor below the install expires at once. Staged + /// changes of a covered entry never settle, so they are dropped. + pub(crate) fn install_group_floor(&self, group_id: u64, last_included_index: u64) { + let mut state = self.lock(); + state + .staged + .retain(|&(group, index), _| group != group_id || index > last_included_index); + state.ring.raise_floor( + ChangePartition::Group(group_id), + CdcOffset::whole_index(last_included_index), + ); + } + + /// Publish a run this node's own ordered feed produced, or a peer + /// forwarded, and journal it before the caller acknowledges it. + pub(crate) fn publish_run(&self, run: ChangeRun) -> AppendedRun { + let appended = { + let mut state = self.lock(); + self.append_locked(&mut state, run) + }; + self.journal_run(&appended); + appended + } + + fn append_locked(&self, state: &mut BusState, run: ChangeRun) -> AppendedRun { + let appended = state.ring.append(run); + for event in &appended.events { + self.last_lsn + .fetch_max(event.lsn.as_u64(), Ordering::Relaxed); + let _ = self.sender.send(event.clone()); + // The raw compatibility channel represents only the legacy + // default database. It carries no database identity and must not + // disclose non-default database events to legacy consumers. + if event.database_id() == DatabaseId::DEFAULT { + let _ = self.legacy_sender.send(event.event().clone()); + } + self.events_published.fetch_add(1, Ordering::Relaxed); } - let cursor = ChangeCursor::new(state.epoch, state.next_sequence); - state.next_sequence = state.next_sequence.checked_add(1).unwrap_or(0); - let sequenced = SequencedChangeEvent::new(cursor, database_id, event.clone()); - if state.recent_changes.len() == self.recent_capacity { - state.recent_changes.pop_front(); + appended + } + + /// Forward `partition`'s new events to the nodes that do not replicate + /// it, when this node leads it. + pub(crate) fn forward( + &self, + shared: &Arc, + partition: ChangePartition, + ) { + let targets = self.fanout.targets(shared, partition); + if targets.is_empty() { + return; } - state.recent_changes.push_back(sequenced.clone()); - let _ = self.sender.send(sequenced); - // The raw compatibility channel represents only the legacy default - // database. It carries no database identity and must not disclose - // non-default database events to legacy consumers. - if database_id == DatabaseId::DEFAULT { - let _ = self.legacy_sender.send(event); + self.fanout.ensure_heartbeat(shared); + if let Some(run) = self.outbound_run(partition, false) { + self.fanout.send(shared, &run, &targets); } - self.events_published.fetch_add(1, Ordering::Relaxed); } - pub fn publish_batch(&self, events: &[ChangeEvent]) { - for event in events { - self.publish(event.clone()); + /// Forward the position of every partition this node leads whose feed + /// advanced since it last forwarded it, events or not. + pub(crate) fn heartbeat(&self, shared: &crate::control::state::SharedState) { + let partitions = self.lock().ring.partitions(); + for partition in partitions { + let targets = self.fanout.targets(shared, partition); + if targets.is_empty() { + continue; + } + if let Some(run) = self.outbound_run(partition, true) { + self.fanout.send(shared, &run, &targets); + } } } + /// The run this node forwards for `partition`: every held event past the + /// position it last forwarded. `None` when the feed did not advance, or + /// when it gained no event and `heartbeat` is off: the next run carries + /// that stretch. + pub(crate) fn outbound_run( + &self, + partition: ChangePartition, + heartbeat: bool, + ) -> Option { + let mut state = self.lock(); + let (through, held_from) = state.ring.feed_bounds(partition)?; + // Continuity is vouched only from where this node holds every event. + let after = match state.forwarded.get(&partition) { + Some(&sent) if sent >= held_from => sent, + _ => held_from, + }; + if through <= after { + return None; + } + let events = state.ring.events_after(partition, after); + if events.is_empty() && !heartbeat { + return None; + } + state.forwarded.insert(partition, through); + Some(AppendedRun { + partition, + after, + through, + events, + }) + } + pub fn subscriber_count(&self) -> u64 { self.active_subscriptions.load(Ordering::Relaxed) } @@ -186,157 +341,86 @@ impl ChangeStream { start: ReplayStart, limit: usize, ) -> Result { - let state = self - .replay_state - .lock() - .unwrap_or_else(|poisoned| poisoned.into_inner()); - let snapshot_sequence = if state.next_sequence == 0 { - u64::MAX - } else { - state.next_sequence - 1 - }; - let snapshot_cursor = ChangeCursor::new(state.epoch, snapshot_sequence); - if let ReplayStart::Cursor(cursor) = start { - validate_cursor(&state, cursor)?; - } - let events = state - .recent_changes - .iter() - .filter(|event| event.database_id() == database_id) - .filter(|event| event.tenant_id == tenant_id) - .filter(|event| collection.is_none_or(|name| event.collection == name)) - .filter(|event| match start { - ReplayStart::Cursor(cursor) => event.cursor().is_after_in_same_epoch(cursor), - ReplayStart::Timestamp(timestamp) => event.timestamp_ms >= timestamp, - }) - .take(limit) - .cloned() - .collect(); - Ok(ReplaySnapshot { - events, - snapshot_cursor, - }) + let state = self.lock(); + state.ring.replay( + ReplayScope { + tenant_id, + database_id, + collection, + }, + &start, + limit, + ) } pub fn unsubscribe(&self) { self.active_subscriptions.fetch_sub(1, Ordering::Relaxed); } - pub fn deliver_remote_notify( - &self, - msg: &crate::event::cross_shard::types::NotifyBroadcastMsg, - ) { - let operation = match msg.operation.as_str() { - "INSERT" => ChangeOperation::Insert, - "UPDATE" => ChangeOperation::Update, - "DELETE" => ChangeOperation::Delete, - _ => ChangeOperation::Insert, + /// Append a run a peer that leads the partition forwarded. + pub fn deliver_remote_notify(&self, msg: &NotifyBroadcastMsg) { + let Some(partition) = msg.partition() else { + tracing::warn!( + kind = msg.partition_kind, + "change run names an unknown partition kind" + ); + return; }; - self.publish_in_database( - DatabaseId::new(msg.database_id), - ChangeEvent { - lsn: Lsn::new(msg.lsn), - tenant_id: TenantId::new(msg.tenant_id), - collection: msg.collection.clone(), - // The wire carries the identity as text; wrap it verbatim. - document_id: RowIdentity::from_user_key(msg.document_id.clone()), - operation, - timestamp_ms: msg.timestamp_ms, - after: None, - }, - ); + let changes = msg + .changes + .iter() + .map(|change| PositionedChange { + position: change.position, + database_id: DatabaseId::new(change.database_id), + event: ChangeEvent { + lsn: Lsn::new(change.lsn), + tenant_id: TenantId::new(change.tenant_id), + collection: change.collection.clone(), + // The wire carries the identity as text; wrap it verbatim. + document_id: RowIdentity::from_user_key(change.document_id.clone()), + operation: parse_operation(&change.operation), + timestamp_ms: change.timestamp_ms, + after: None, + }, + }) + .collect(); + self.publish_run(ChangeRun { + partition, + after: Some(msg.after), + through: msg.through, + changes, + }); } -} -fn validate_cursor(state: &ReplayState, cursor: ChangeCursor) -> Result<(), ReplayError> { - let current = if state.next_sequence == 0 { - u64::MAX - } else { - state.next_sequence - 1 - }; - if cursor.epoch() != state.epoch || cursor.sequence() > current { - return Err(ReplayError::Expired); - } - if let Some(oldest) = state.recent_changes.front() - && cursor.sequence().saturating_add(1) < oldest.cursor().sequence() - { - return Err(ReplayError::Expired); - } - if state.recent_changes.is_empty() && cursor.sequence() != current { - return Err(ReplayError::Expired); + /// Publish `events` as the data-group entry `(group_id, log_index)`, + /// the way a replica's apply stages the entry's changes and its apply + /// loop settles it. + #[cfg(test)] + pub(crate) fn settle_entry( + &self, + group_id: u64, + log_index: u64, + database_id: DatabaseId, + events: Vec, + ) { + self.stage(group_id, log_index, database_id, events); + self.settle_group(group_id, log_index, log_index); } - Ok(()) -} - -fn new_epoch() -> u128 { - uuid::Uuid::new_v4().as_u128() } -/// Broadcast a `ChangeEvent` to all peer nodes in the cluster. -pub fn broadcast_notify_to_cluster( - database_id: DatabaseId, - event: &ChangeEvent, - node_id: u64, - sequence: u64, - transport: &Arc, - topology: &Arc>, -) { - use crate::event::cross_shard::types::NotifyBroadcastMsg; - use nodedb_cluster::RaftRpc; - use nodedb_cluster::wire::{VShardEnvelope, VShardMessageType}; - let msg = NotifyBroadcastMsg { - source_node: node_id, - sequence, - tenant_id: event.tenant_id.as_u64(), - database_id: database_id.as_u64(), - collection: event.collection.clone(), - document_id: event.document_id.to_string(), - operation: event.operation.as_str().to_string(), - timestamp_ms: event.timestamp_ms, - lsn: event.lsn.as_u64(), - }; - let payload = match zerompk::to_msgpack_vec(&msg) { - Ok(payload) => payload, - Err(error) => { - warn!(error = %error, "failed to serialize NotifyBroadcast"); - return; - } - }; - let peer_ids: Vec = { - let topology = topology - .read() - .unwrap_or_else(|poisoned| poisoned.into_inner()); - topology - .active_nodes() - .iter() - .map(|node| node.node_id) - .filter(|id| *id != node_id) - .collect() - }; - trace!(peer_count = peer_ids.len(), database_id = database_id.as_u64(), collection = %event.collection, "broadcasting NOTIFY to cluster peers"); - for peer_id in peer_ids { - let envelope = VShardEnvelope::new( - VShardMessageType::NotifyBroadcast, - node_id, - peer_id, - 0, - payload.clone(), - ); - let transport = Arc::clone(transport); - tokio::spawn(async move { - if let Err(error) = transport - .send_rpc_oneway(peer_id, RaftRpc::VShardEnvelope(envelope.to_bytes())) - .await - { - trace!(peer = peer_id, error = %error, "NOTIFY broadcast to peer failed (best-effort)"); - } - }); +fn parse_operation(operation: &str) -> ChangeOperation { + match operation { + "UPDATE" => ChangeOperation::Update, + "DELETE" => ChangeOperation::Delete, + _ => ChangeOperation::Insert, } } #[cfg(test)] mod tests { use super::*; + use crate::control::change_stream::CursorStep; + fn event(lsn: u64, tenant: u64, document: &str) -> ChangeEvent { ChangeEvent { lsn: Lsn::new(lsn), @@ -348,75 +432,109 @@ mod tests { after: None, } } + + fn first_page(stream: &ChangeStream, limit: usize) -> ReplaySnapshot { + stream + .query_changes(TenantId::new(1), None, ReplayStart::Timestamp(0), limit) + .expect("replay") + } + + /// Publish each event as its own entry of data group 1, at consecutive + /// log indexes from 1. + fn settle_each(stream: &ChangeStream, events: Vec) { + for (index, event) in (1..).zip(events) { + stream.settle_entry(1, index, DatabaseId::DEFAULT, vec![event]); + } + } + #[test] - fn publication_sequence_not_lsn() { + fn publication_order_not_lsn() { let stream = ChangeStream::new(8); - stream.publish(event(102, 1, "first")); - stream.publish(event(101, 1, "second")); - let first = stream - .query_changes(TenantId::new(1), None, ReplayStart::Timestamp(0), 1) - .unwrap_or_else(|_| panic!()); + settle_each( + &stream, + vec![event(102, 1, "first"), event(101, 1, "second")], + ); + let first = first_page(&stream, 1); let next = stream .query_changes( TenantId::new(1), None, - ReplayStart::Cursor(first.events[0].cursor()), + ReplayStart::Cursor(first.events[0].cursor.clone()), 8, ) - .unwrap_or_else(|_| panic!()); + .expect("replay"); assert_eq!(next.events[0].document_id.as_str(), "second"); } + + /// One entry's events share its LSN and page one at a time. #[test] fn duplicate_lsn_events_paginate() { let stream = ChangeStream::new(8); - stream.publish(event(1, 1, "a")); - stream.publish(event(1, 1, "b")); - let first = stream - .query_changes(TenantId::new(1), None, ReplayStart::Timestamp(0), 1) - .unwrap_or_else(|_| panic!()); + stream.settle_entry( + 1, + 1, + DatabaseId::DEFAULT, + vec![event(1, 1, "a"), event(1, 1, "b")], + ); + let first = first_page(&stream, 1); let next = stream - .query_changes( - TenantId::new(1), - None, - ReplayStart::Cursor(first.events[0].cursor()), - 1, - ) - .unwrap_or_else(|_| panic!()); + .query_changes(TenantId::new(1), None, ReplayStart::Cursor(first.cursor), 1) + .expect("replay"); assert_eq!(next.events[0].document_id.as_str(), "b"); } + + /// A cursor whose event the ring evicted expires, as does one below a + /// hole in the feed. #[test] - fn evicted_and_wrong_epoch_cursors_expire() { + fn evicted_cursors_and_cursors_below_a_hole_expire() { let stream = ChangeStream::new(1); - stream.publish(event(1, 1, "a")); - let cursor = stream - .query_changes(TenantId::new(1), None, ReplayStart::Timestamp(0), 1) - .unwrap_or_else(|_| panic!()) - .events[0] - .cursor(); - stream.publish(event(2, 1, "b")); - stream.publish(event(3, 1, "c")); + stream.settle_entry(1, 1, DatabaseId::DEFAULT, vec![event(1, 1, "a")]); + let cursor = first_page(&stream, 1).events[0].cursor.clone(); + stream.settle_entry(1, 2, DatabaseId::DEFAULT, vec![event(2, 1, "b")]); + stream.settle_entry(1, 3, DatabaseId::DEFAULT, vec![event(3, 1, "c")]); assert!(matches!( stream.query_changes(TenantId::new(1), None, ReplayStart::Cursor(cursor), 1), Err(ReplayError::Expired) )); + + let holed = ChangeStream::new(8); + holed.settle_entry(1, 1, DatabaseId::DEFAULT, vec![event(1, 1, "a")]); + let below = first_page(&holed, 1).cursor; + // Entries 2..=4 never reached this node. + holed.settle_entry(1, 5, DatabaseId::DEFAULT, vec![event(5, 1, "e")]); assert!(matches!( - stream.query_changes( - TenantId::new(1), - None, - ReplayStart::Cursor(ChangeCursor::new(0, 0)), - 1 - ), + holed.query_changes(TenantId::new(1), None, ReplayStart::Cursor(below), 1), Err(ReplayError::Expired) )); } + + /// A snapshot install expires a cursor below it before any later entry + /// settles, and drops the staged changes of the entries it covers. + #[test] + fn a_cursor_below_a_snapshot_install_expires_at_once() { + let stream = ChangeStream::new(8); + stream.settle_entry(1, 1, DatabaseId::DEFAULT, vec![event(1, 1, "a")]); + let below = first_page(&stream, 1).cursor; + stream.stage(1, 3, DatabaseId::DEFAULT, vec![event(3, 1, "covered")]); + stream.install_group_floor(1, 4); + assert!(matches!( + stream.query_changes(TenantId::new(1), None, ReplayStart::Cursor(below), 1), + Err(ReplayError::Expired) + )); + stream.settle_group(1, 5, 5); + assert!( + first_page(&stream, 8) + .events + .iter() + .all(|event| event.document_id.as_str() != "covered") + ); + } + #[test] fn tenant_filter_precedes_limit() { let stream = ChangeStream::new(8); - stream.publish(event(1, 2, "other")); - stream.publish(event(2, 1, "mine")); - let result = stream - .query_changes(TenantId::new(1), None, ReplayStart::Timestamp(0), 1) - .unwrap_or_else(|_| panic!()); + settle_each(&stream, vec![event(1, 2, "other"), event(2, 1, "mine")]); + let result = first_page(&stream, 1); assert_eq!(result.events[0].document_id.as_str(), "mine"); } @@ -427,8 +545,8 @@ mod tests { let database_b = DatabaseId::new(1025); let mut subscription = stream.subscribe_in_database(Some("orders".into()), Some(TenantId::new(1)), database_a); - stream.publish_in_database(database_b, event(1, 1, "database-b")); - stream.publish_in_database(database_a, event(2, 1, "database-a")); + stream.settle_entry(1, 1, database_b, vec![event(1, 1, "database-b")]); + stream.settle_entry(1, 2, database_a, vec![event(2, 1, "database-a")]); let result = stream .query_changes_in_database( TenantId::new(1), @@ -437,37 +555,329 @@ mod tests { ReplayStart::Timestamp(0), 1, ) - .unwrap_or_else(|_| panic!()); + .expect("replay"); assert_eq!(result.events[0].document_id.as_str(), "database-a"); - let received = subscription.recv_sequenced().await.unwrap(); + let received = subscription.recv_sequenced().await.expect("event"); assert_eq!(received.document_id.as_str(), "database-a"); } + /// Two replicas settle one group's entries; a cursor from the first + /// resumes on the second at the next event, and each settled run + /// publishes the entry's changes once. #[test] - fn rotation_changes_epoch_without_cross_epoch_cursor_ordering() { - let stream = ChangeStream::new(8); - let old_epoch = 7; - { - let mut state = stream - .replay_state - .lock() - .unwrap_or_else(|poisoned| poisoned.into_inner()); - state.epoch = old_epoch; - state.next_sequence = u64::MAX; + fn replicas_publish_a_settled_entry_at_the_same_position() { + let replicas = [ChangeStream::new(16), ChangeStream::new(16)]; + for stream in &replicas { + stream.stage( + 7, + 10, + DatabaseId::DEFAULT, + vec![event(900, 1, "a"), event(900, 1, "b")], + ); + stream.stage(7, 12, DatabaseId::DEFAULT, vec![event(901, 1, "c")]); + stream.settle_group(7, 10, 12); + } + let page = first_page(&replicas[0], 2); + assert!(page.has_more); + let rest = replicas[1] + .query_changes(TenantId::new(1), None, ReplayStart::Cursor(page.cursor), 8) + .expect("the second replica holds the same feed"); + assert_eq!(rest.events.len(), 1); + assert_eq!(rest.events[0].document_id.as_str(), "c"); + assert_eq!(rest.events[0].position(), CdcOffset::data_event(0, 12, 1)); + assert_eq!(replicas[1].events_published(), 3); + } + + /// A node that does not replicate the group receives the leader's runs. + /// A run that repeats a stretch appends nothing, and a lost run makes + /// every cursor below it reset. + #[test] + fn forwarded_runs_deduplicate_and_report_a_lost_run() { + let leader = ChangeStream::new(16); + let outsider = ChangeStream::new(16); + let mut live = outsider.subscribe(None, Some(TenantId::new(1))); + let mut cursor = live.start_cursor().clone(); + let group = ChangePartition::Group(3); + let deliver = |run: &AppendedRun| { + outsider.deliver_remote_notify(&NotifyBroadcastMsg::from_run(1, run)) + }; + + leader.stage(3, 1, DatabaseId::DEFAULT, vec![event(1, 1, "a")]); + leader.settle_group(3, 1, 1); + let run = leader + .outbound_run(group, false) + .expect("a run with an event"); + deliver(&run); + deliver(&run); + let delivered = live.try_recv_sequenced().expect("first run"); + assert_eq!(cursor.accept(&delivered), CursorStep::Deliver); + assert!( + live.try_recv_sequenced().is_err(), + "the repeat appended nothing" + ); + + // The run carrying entry 2 never arrives. + leader.stage(3, 2, DatabaseId::DEFAULT, vec![event(2, 1, "lost")]); + leader.settle_group(3, 2, 2); + leader + .outbound_run(group, false) + .expect("a run with an event"); + leader.stage(3, 3, DatabaseId::DEFAULT, vec![event(3, 1, "c")]); + leader.settle_group(3, 3, 3); + deliver( + &leader + .outbound_run(group, false) + .expect("a run with an event"), + ); + let after_hole = live.try_recv_sequenced().expect("third run"); + assert_eq!(cursor.accept(&after_hole), CursorStep::Reset); + } + + /// Settled stretches with no event send nothing. The next run with an + /// event covers them, so the receiver sees no hole. An idle feed whose + /// position advanced goes out on a heartbeat, once. + #[test] + fn a_stretch_with_no_event_is_carried_by_the_next_run() { + let leader = ChangeStream::new(16); + let outsider = ChangeStream::new(16); + let mut live = outsider.subscribe(None, Some(TenantId::new(1))); + let mut cursor = live.start_cursor().clone(); + let group = ChangePartition::Group(5); + let deliver = |run: &AppendedRun| { + outsider.deliver_remote_notify(&NotifyBroadcastMsg::from_run(1, run)) + }; + + leader.stage(5, 1, DatabaseId::DEFAULT, vec![event(1, 1, "a")]); + leader.settle_group(5, 1, 1); + deliver( + &leader + .outbound_run(group, false) + .expect("a run with an event"), + ); + let first = live.try_recv_sequenced().expect("first event"); + assert_eq!(cursor.accept(&first), CursorStep::Deliver); + + // Entries 2..=9 carry no event: no message. + for index in 2..=9 { + leader.settle_group(5, index, index); + assert!(leader.outbound_run(group, false).is_none()); } - stream.publish(event(1, 1, "last-old-epoch")); - let old_cursor = stream - .query_changes(TenantId::new(1), None, ReplayStart::Timestamp(0), 1) - .unwrap_or_else(|_| panic!()) - .events[0] - .cursor(); - stream.publish(event(2, 1, "first-new-epoch")); - let new_cursor = stream - .query_changes(TenantId::new(1), None, ReplayStart::Timestamp(0), 1) - .unwrap_or_else(|_| panic!()) - .events[0] - .cursor(); - assert_ne!(old_cursor.epoch(), new_cursor.epoch()); - assert!(!new_cursor.is_after_in_same_epoch(old_cursor)); + leader.stage(5, 10, DatabaseId::DEFAULT, vec![event(10, 1, "b")]); + leader.settle_group(5, 10, 10); + let run = leader + .outbound_run(group, false) + .expect("a run with an event"); + assert_eq!(run.after, CdcOffset::whole_index(1)); + assert_eq!(run.events.len(), 1); + deliver(&run); + let next = live.try_recv_sequenced().expect("second event"); + assert_eq!(cursor.accept(&next), CursorStep::Deliver, "no false hole"); + + // An idle stretch goes out once on a heartbeat. + leader.settle_group(5, 11, 11); + let heartbeat = leader + .outbound_run(group, true) + .expect("the position advanced"); + assert!(heartbeat.events.is_empty()); + assert_eq!(heartbeat.through, CdcOffset::whole_index(11)); + deliver(&heartbeat); + assert!( + leader.outbound_run(group, true).is_none(), + "nothing new to say" + ); + } + + const TENANT_A: u64 = 10; + const TENANT_B: u64 = 20; + + fn tenant_event( + collection: &str, + document: &str, + tenant: u64, + lsn: u64, + timestamp_ms: u64, + ) -> ChangeEvent { + ChangeEvent { + lsn: Lsn::new(lsn), + tenant_id: TenantId::new(tenant), + collection: collection.into(), + document_id: RowIdentity::from_user_key(document), + operation: ChangeOperation::Insert, + timestamp_ms, + after: None, + } + } + + fn tenant_replay( + stream: &ChangeStream, + tenant: u64, + collection: &str, + limit: usize, + ) -> Vec { + stream + .query_changes( + TenantId::new(tenant), + Some(collection), + ReplayStart::Timestamp(0), + limit, + ) + .expect("timestamp replay cannot expire") + .events + } + + /// A tenant's replay returns only its own events on a shared collection. + #[test] + fn cdc_stream_isolated_between_tenants() { + let stream = ChangeStream::new(1024); + let _sub_b = stream.subscribe(Some("orders".into()), Some(TenantId::new(TENANT_B))); + settle_each( + &stream, + vec![ + tenant_event("orders", "order_1", TENANT_A, 1, 1000), + tenant_event("orders", "order_2", TENANT_B, 2, 2000), + ], + ); + + let a_events = tenant_replay(&stream, TENANT_A, "orders", 100); + let b_events = tenant_replay(&stream, TENANT_B, "orders", 100); + assert_eq!(a_events.len(), 1); + assert_eq!(a_events[0].tenant_id, TenantId::new(TENANT_A)); + assert_eq!(a_events[0].document_id.as_str(), "order_1"); + assert_eq!(b_events.len(), 1); + assert_eq!(b_events[0].tenant_id, TenantId::new(TENANT_B)); + assert_eq!(b_events[0].document_id.as_str(), "order_2"); + } + + /// Events of one millisecond page one at a time through their cursors. + #[test] + fn cdc_opaque_cursor_keeps_same_millisecond_events_pageable() { + let stream = ChangeStream::new(1024); + let tenant_id = TenantId::new(TENANT_A); + settle_each( + &stream, + ["first", "second", "third"] + .into_iter() + .zip(1..) + .map(|(document, lsn)| tenant_event("orders", document, TENANT_A, lsn, 1_000)) + .collect(), + ); + + let first_page = stream + .query_changes(tenant_id, Some("orders"), ReplayStart::Timestamp(0), 1) + .expect("timestamp replay cannot expire"); + assert_eq!(first_page.events.len(), 1); + assert_eq!(first_page.events[0].document_id.as_str(), "first"); + + let second_page = stream + .query_changes( + tenant_id, + Some("orders"), + ReplayStart::Cursor(first_page.events[0].cursor.clone()), + 1, + ) + .expect("fresh cursor must resume"); + assert_eq!(second_page.events.len(), 1); + assert_eq!(second_page.events[0].document_id.as_str(), "second"); + } + + /// A replay of one collection returns none of another's events. + #[test] + fn cdc_different_collections_isolated() { + let stream = ChangeStream::new(1024); + settle_each( + &stream, + vec![ + tenant_event("orders", "o1", TENANT_A, 1, 1000), + tenant_event("users", "u1", TENANT_A, 2, 2000), + ], + ); + let order_events = tenant_replay(&stream, TENANT_A, "orders", 100); + assert!(!order_events.is_empty()); + for event in &order_events { + assert_eq!(event.collection, "orders"); + } + } + + /// A subscription scoped to Tenant B never delivers Tenant A's events, + /// even when both tenants write to the same collection. + #[tokio::test] + async fn cdc_tenant_b_subscription_rejects_tenant_a_events() { + let stream = ChangeStream::new(1024); + let mut sub_b = stream.subscribe(Some("orders".into()), Some(TenantId::new(TENANT_B))); + let mut events: Vec = (0..5u64) + .map(|i| { + tenant_event( + "orders", + &format!("a_order_{i}"), + TENANT_A, + i + 1, + (i + 1) * 1000, + ) + }) + .collect(); + events.push(tenant_event("orders", "b_order_1", TENANT_B, 100, 10_000)); + settle_each(&stream, events); + + let received = + tokio::time::timeout(std::time::Duration::from_millis(500), sub_b.recv_filtered()) + .await + .expect("timed out waiting for Tenant B's event") + .expect("channel error"); + assert_eq!(received.tenant_id, TenantId::new(TENANT_B)); + assert_eq!(received.document_id.as_str(), "b_order_1"); + + let second = + tokio::time::timeout(std::time::Duration::from_millis(50), sub_b.recv_filtered()).await; + assert!( + second.is_err(), + "Tenant A's events must have been filtered from Tenant B's subscription" + ); + } + + /// A subscription with no tenant filter receives every tenant's events: + /// the filter is opt-in. + #[tokio::test] + async fn cdc_unfiltered_subscription_receives_all_tenants() { + let stream = ChangeStream::new(1024); + let mut sub_all = stream.subscribe(None, None); + settle_each( + &stream, + vec![ + tenant_event("events", "e_a", TENANT_A, 1, 1000), + tenant_event("events", "e_b", TENANT_B, 2, 2000), + ], + ); + + let mut tenant_ids = Vec::new(); + for _ in 0..2 { + let received = tokio::time::timeout( + std::time::Duration::from_millis(200), + sub_all.recv_filtered(), + ) + .await + .expect("timed out waiting for an event") + .expect("channel error"); + tenant_ids.push(received.tenant_id); + } + assert!(tenant_ids.contains(&TenantId::new(TENANT_A))); + assert!(tenant_ids.contains(&TenantId::new(TENANT_B))); + } + + /// `query_changes` filters by tenant before it applies the caller's limit. + #[test] + fn cdc_query_changes_scopes_the_requested_tenant_before_limit() { + let stream = ChangeStream::new(1024); + settle_each( + &stream, + vec![ + tenant_event("logs", "l_b_1", TENANT_B, 1, 1_000), + tenant_event("logs", "l_b_2", TENANT_B, 2, 1_000), + tenant_event("logs", "l_a", TENANT_A, 3, 1_000), + ], + ); + let a_events = tenant_replay(&stream, TENANT_A, "logs", 1); + assert_eq!(a_events.len(), 1); + assert_eq!(a_events[0].tenant_id, TenantId::new(TENANT_A)); + assert_eq!(a_events[0].document_id.as_str(), "l_a"); } } diff --git a/nodedb/src/control/change_stream/stream/cursor.rs b/nodedb/src/control/change_stream/stream/cursor.rs index 26bdc5729..db4236687 100644 --- a/nodedb/src/control/change_stream/stream/cursor.rs +++ b/nodedb/src/control/change_stream/stream/cursor.rs @@ -1,46 +1,128 @@ // SPDX-License-Identifier: BUSL-1.1 +//! Resumable positions in the Control-Plane change stream. +//! +//! The stream is a set of partitions. Each partition is one totally ordered +//! feed that every node numbers alike: +//! +//! - `Group(g)`: the writes of data group `g`, at their Raft log index. +//! - `Calvin(v)`: the Calvin transactions vShard `v` applied, at their +//! sequencer position. +//! +//! A cursor holds, per partition, the position of the last event it +//! consumed. A cursor taken on one node therefore resumes on any node that +//! holds the same feed. + +use std::collections::BTreeMap; use std::fmt; use std::str::FromStr; -const TOKEN_PREFIX: &str = "v1:"; -const EPOCH_HEX_LEN: usize = 32; -const MAX_TOKEN_LEN: usize = TOKEN_PREFIX.len() + EPOCH_HEX_LEN + 1 + 20; +use crate::event::cdc::CdcOffset; + +use super::SequencedChangeEvent; + +const TOKEN_PREFIX: &str = "v2:"; +/// Partitions one cursor holds at most: every data group and every vShard's +/// Calvin feed of a large cluster. +const MAX_ENTRIES: usize = 8192; +/// Upper bound on one entry's text: a tag, a 20-digit id, and three +/// 20-digit numbers with their separators. +const MAX_ENTRY_LEN: usize = 1 + 20 + 1 + 3 * 20 + 2; +const MAX_TOKEN_LEN: usize = TOKEN_PREFIX.len() + MAX_ENTRIES * (MAX_ENTRY_LEN + 1); + +/// One totally ordered feed of the change stream. +#[derive(Clone, Copy, Debug, PartialEq, Eq, PartialOrd, Ord, Hash)] +pub enum ChangePartition { + /// A data group's writes, at their Raft log index. + Group(u64), + /// A vShard's Calvin transactions, at their sequencer position. + Calvin(u32), +} -/// Opaque, versioned position in one ChangeStream publication epoch. -#[derive(Clone, Copy, Debug, PartialEq, Eq, Hash)] +/// What a consumer does with the next live event. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum CursorStep { + /// A new event: the cursor advanced past it. + Deliver, + /// An event the cursor already covers. + Skip, + /// The node's feed has a hole the cursor sits below. The consumer must + /// reset. + Reset, +} + +/// Opaque, versioned resume position across every partition. +#[derive(Clone, Debug, Default, PartialEq, Eq)] pub struct ChangeCursor { - epoch: u128, - sequence: u64, + positions: BTreeMap, } impl ChangeCursor { - pub(crate) const fn new(epoch: u128, sequence: u64) -> Self { - Self { epoch, sequence } + /// The last consumed position in `partition`, if any. + pub fn position(&self, partition: ChangePartition) -> Option { + self.positions.get(&partition).copied() } - pub(crate) const fn epoch(self) -> u128 { - self.epoch + pub fn is_empty(&self) -> bool { + self.positions.is_empty() } - pub(crate) const fn sequence(self) -> u64 { - self.sequence + pub(crate) fn entries(&self) -> impl Iterator + '_ { + self.positions.iter().map(|(p, o)| (*p, *o)) } - /// Whether two cursors belong to the same publication epoch. - pub const fn same_epoch(self, other: Self) -> bool { - self.epoch == other.epoch + /// Raise `partition` to `position`. A lower position leaves it as is. + pub(crate) fn raise(&mut self, partition: ChangePartition, position: CdcOffset) { + let held = self.positions.entry(partition).or_insert(position); + if *held < position { + *held = position; + } } - /// Compare sequences only after confirming the cursors share an epoch. - pub const fn is_after_in_same_epoch(self, other: Self) -> bool { - self.same_epoch(other) && self.sequence > other.sequence + /// Whether the cursor consumed `event` already. + pub fn covers(&self, event: &SequencedChangeEvent) -> bool { + self.position(event.partition()) + .is_some_and(|held| held >= event.position()) + } + + /// Advance past `event` when it is new. Reports a reset when the node's + /// feed of the event's partition has a hole above this cursor. + pub fn accept(&mut self, event: &SequencedChangeEvent) -> CursorStep { + let partition = event.partition(); + let position = event.position(); + let Some(held) = self.position(partition) else { + self.positions.insert(partition, position); + return CursorStep::Deliver; + }; + if held < event.floor() { + return CursorStep::Reset; + } + if held >= position { + return CursorStep::Skip; + } + self.positions.insert(partition, position); + CursorStep::Deliver } } impl fmt::Display for ChangeCursor { fn fmt(&self, formatter: &mut fmt::Formatter<'_>) -> fmt::Result { - write!(formatter, "v1:{:032x}:{}", self.epoch, self.sequence) + formatter.write_str(TOKEN_PREFIX)?; + for (index, (partition, position)) in self.positions.iter().enumerate() { + if index > 0 { + formatter.write_str(",")?; + } + match partition { + ChangePartition::Group(group) => write!(formatter, "g{group}")?, + ChangePartition::Calvin(vshard) => write!(formatter, "c{vshard}")?, + } + write!( + formatter, + "@{}.{}.{}", + position.epoch, position.index, position.sequence + )?; + } + Ok(()) } } @@ -64,46 +146,141 @@ impl FromStr for ChangeCursor { if token.len() > MAX_TOKEN_LEN { return Err(CursorParseError); } - let Some(rest) = token.strip_prefix(TOKEN_PREFIX) else { - return Err(CursorParseError); - }; - let mut parts = rest.split(':'); - let (Some(epoch), Some(sequence), None) = (parts.next(), parts.next(), parts.next()) else { - return Err(CursorParseError); - }; - if epoch.len() != EPOCH_HEX_LEN - || !epoch - .bytes() - .all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)) - || sequence.is_empty() - || sequence.len() > 20 - || (sequence.len() > 1 && sequence.starts_with('0')) - || !sequence.bytes().all(|byte| byte.is_ascii_digit()) - { - return Err(CursorParseError); + let rest = token.strip_prefix(TOKEN_PREFIX).ok_or(CursorParseError)?; + let mut positions = BTreeMap::new(); + if rest.is_empty() { + return Ok(Self { positions }); } - let epoch = u128::from_str_radix(epoch, 16).map_err(|_| CursorParseError)?; - let sequence = sequence.parse().map_err(|_| CursorParseError)?; - Ok(Self { epoch, sequence }) + for entry in rest.split(',') { + if positions.len() == MAX_ENTRIES { + return Err(CursorParseError); + } + let (partition, position) = parse_entry(entry)?; + if positions.insert(partition, position).is_some() { + return Err(CursorParseError); + } + } + Ok(Self { positions }) + } +} + +fn parse_entry(entry: &str) -> Result<(ChangePartition, CdcOffset), CursorParseError> { + let (partition, position) = entry.split_once('@').ok_or(CursorParseError)?; + let partition = match partition.as_bytes().first() { + Some(b'g') => ChangePartition::Group(number(&partition[1..])?), + Some(b'c') => ChangePartition::Calvin( + u32::try_from(number(&partition[1..])?).map_err(|_| CursorParseError)?, + ), + _ => return Err(CursorParseError), + }; + let mut parts = position.split('.'); + let (Some(epoch), Some(index), Some(sequence), None) = + (parts.next(), parts.next(), parts.next(), parts.next()) + else { + return Err(CursorParseError); + }; + Ok(( + partition, + CdcOffset::at(number(epoch)?, number(index)?, number(sequence)?), + )) +} + +/// A canonical decimal `u64`: digits only, no leading zero. +fn number(text: &str) -> Result { + if text.is_empty() + || text.len() > 20 + || (text.len() > 1 && text.starts_with('0')) + || !text.bytes().all(|byte| byte.is_ascii_digit()) + { + return Err(CursorParseError); } + text.parse().map_err(|_| CursorParseError) } #[cfg(test)] mod tests { use super::*; + use crate::control::change_stream::{ChangeEvent, ChangeOperation}; + use crate::types::{DatabaseId, Lsn, TenantId}; + + fn event( + partition: ChangePartition, + position: CdcOffset, + floor: CdcOffset, + ) -> SequencedChangeEvent { + SequencedChangeEvent::new( + partition, + position, + floor, + DatabaseId::DEFAULT, + ChangeEvent { + lsn: Lsn::new(1), + tenant_id: TenantId::new(1), + collection: "orders".into(), + document_id: nodedb_types::RowIdentity::from_user_key("a"), + operation: ChangeOperation::Insert, + timestamp_ms: 1, + after: None, + }, + ) + } #[test] - fn token_is_strict_and_bounded() { - let cursor = ChangeCursor::new(0xab, 42); - assert_eq!(cursor.to_string(), "v1:000000000000000000000000000000ab:42"); - assert!(cursor.to_string().parse::().is_ok()); + fn token_round_trips_and_is_strict() { + let mut cursor = ChangeCursor::default(); + cursor.raise(ChangePartition::Group(3), CdcOffset::data_event(0, 42, 1)); + cursor.raise( + ChangePartition::Calvin(7), + CdcOffset::data_event_in(0, 9, 2, 1), + ); + let token = cursor.to_string(); + assert!(token.starts_with("v2:")); + assert_eq!(token.parse::(), Ok(cursor)); + assert_eq!("v2:".parse::(), Ok(ChangeCursor::default())); for invalid in [ - "v1:AB:1", - "v1:ab:01", - "v2:000000000000000000000000000000ab:1", - "v1:000000000000000000000000000000ab:-1", + "v1:000000000000000000000000000000ab:1", + "v2:g01@0.1.2", + "v2:g1@0.1", + "v2:g1@0.1.2,g1@0.1.3", + "v2:x1@0.1.2", + "v2:l@0.1.2", + "v2:l7@0.1.2", + "v2:c99999999999@0.1.2", + "v2:g1@0.-1.2", ] { - assert!(invalid.parse::().is_err()); + assert!(invalid.parse::().is_err(), "{invalid}"); } } + + #[test] + fn accept_delivers_new_events_and_skips_covered_ones() { + let group = ChangePartition::Group(1); + let mut cursor = ChangeCursor::default(); + let first = event(group, CdcOffset::data_event(0, 5, 1), CdcOffset::ZERO); + let second = event(group, CdcOffset::data_event(0, 6, 1), CdcOffset::ZERO); + assert_eq!(cursor.accept(&first), CursorStep::Deliver); + assert_eq!(cursor.accept(&second), CursorStep::Deliver); + assert_eq!(cursor.accept(&first), CursorStep::Skip); + assert!(cursor.covers(&second)); + // Another partition's positions never compare with this one. + let other = event( + ChangePartition::Group(2), + CdcOffset::data_event(0, 1, 1), + CdcOffset::ZERO, + ); + assert_eq!(cursor.accept(&other), CursorStep::Deliver); + } + + #[test] + fn a_hole_above_the_cursor_resets() { + let group = ChangePartition::Group(1); + let mut cursor = ChangeCursor::default(); + cursor.raise(group, CdcOffset::whole_index(10)); + let past_hole = event( + group, + CdcOffset::data_event(0, 30, 1), + CdcOffset::whole_index(20), + ); + assert_eq!(cursor.accept(&past_hole), CursorStep::Reset); + } } diff --git a/nodedb/src/control/change_stream/stream/error.rs b/nodedb/src/control/change_stream/stream/error.rs new file mode 100644 index 000000000..6c4ba1e56 --- /dev/null +++ b/nodedb/src/control/change_stream/stream/error.rs @@ -0,0 +1,22 @@ +// SPDX-License-Identifier: BUSL-1.1 + +/// A write the change feed cannot carry. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub enum ChangeStreamError { + /// A write on `collection` yields change events but applies outside any + /// replicated entry. Every change event is published at its entry's log + /// position, so such a write has no position on any feed. + #[error( + "a write on '{collection}' yields change events but took an unreplicated route; every \ + such write must apply through its replicated entry" + )] + UnreplicatedChange { collection: String }, +} + +impl From for crate::Error { + fn from(error: ChangeStreamError) -> Self { + Self::Internal { + detail: error.to_string(), + } + } +} diff --git a/nodedb/src/control/change_stream/stream/fanout.rs b/nodedb/src/control/change_stream/stream/fanout.rs new file mode 100644 index 000000000..cef387379 --- /dev/null +++ b/nodedb/src/control/change_stream/stream/fanout.rs @@ -0,0 +1,216 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Forwarding change runs to the nodes that do not replicate a partition. +//! +//! Only the partition's leader forwards, per the routing table's leader hint. +//! A run goes out when the feed gained events. It covers the feed from the +//! position the leader last forwarded, so a feed stretch with no events costs +//! no message. An idle feed whose position advanced goes out on a heartbeat, +//! at most once per `change_feed_heartbeat_ms`. A receiver therefore sees a +//! continuous feed, and records a hole only for a run that never arrived. +//! +//! Each target node has one bounded queue drained by one task that awaits +//! every run's ack before it sends the next, so a node appends one leader's +//! runs in order. A run that cannot be queued is dropped. The target then +//! records a hole when the next run arrives, and every cursor below the hole +//! resets. + +use std::collections::HashMap; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{Arc, Mutex, Weak}; +use std::time::Duration; + +use tokio::sync::mpsc; +use tokio::sync::mpsc::error::TrySendError; +use tracing::{trace, warn}; + +use crate::control::state::SharedState; +use crate::event::cross_shard::types::{ + NOTIFY_PARTITION_CALVIN, NOTIFY_PARTITION_GROUP, NotifyBroadcastMsg, NotifyChange, +}; + +use super::ChangePartition; +use super::ring::AppendedRun; + +/// Runs queued for one target node before new runs are dropped. +const PEER_QUEUE: usize = 1024; + +#[derive(Default)] +pub(super) struct ChangeFanout { + peers: Mutex>>>, + heartbeat_started: AtomicBool, +} + +impl ChangeFanout { + /// The nodes this node forwards `partition` to: every active node that + /// does not replicate the partition's data group, when this node leads + /// the group. Empty otherwise. + pub fn targets(&self, shared: &SharedState, partition: ChangePartition) -> Vec { + let (Some(topology), Some(routing)) = (&shared.cluster_topology, &shared.cluster_routing) + else { + return Vec::new(); + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let group = match partition { + ChangePartition::Group(group) => group, + ChangePartition::Calvin(vshard) => match routing.group_for_vshard(vshard) { + Ok(group) => group, + Err(_) => return Vec::new(), + }, + }; + let Some(info) = routing.group_info(group) else { + return Vec::new(); + }; + if info.leader != shared.node_id { + return Vec::new(); + } + let topology = topology.read().unwrap_or_else(|p| p.into_inner()); + topology + .active_nodes() + .iter() + .map(|node| node.node_id) + .filter(|id| { + *id != shared.node_id && !info.members.contains(id) && !info.learners.contains(id) + }) + .collect() + } + + /// Queue `run` for every node in `targets`. + pub fn send(&self, shared: &SharedState, run: &AppendedRun, targets: &[u64]) { + let Some(transport) = &shared.cluster_transport else { + return; + }; + let msg = NotifyBroadcastMsg::from_run(shared.node_id, run); + let payload = match zerompk::to_msgpack_vec(&msg) { + Ok(payload) => payload, + Err(error) => { + warn!(%error, "encoding a change-stream run failed; peers record a hole"); + return; + } + }; + let Ok(runtime) = tokio::runtime::Handle::try_current() else { + return; + }; + let mut peers = self.peers.lock().unwrap_or_else(|p| p.into_inner()); + for &peer in targets { + let sender = peers.entry(peer).or_insert_with(|| { + spawn_peer(&runtime, Arc::clone(transport), shared.node_id, peer) + }); + match sender.try_send(payload.clone()) { + Ok(()) => {} + Err(TrySendError::Full(_)) => { + warn!( + peer, + "change-stream forward queue is full; the peer records a hole" + ); + } + Err(TrySendError::Closed(_)) => { + peers.remove(&peer); + } + } + } + } + + /// Start the heartbeat task once. It runs while `shared` lives. + pub fn ensure_heartbeat(&self, shared: &Arc) { + if self.heartbeat_started.swap(true, Ordering::AcqRel) { + return; + } + let Ok(runtime) = tokio::runtime::Handle::try_current() else { + self.heartbeat_started.store(false, Ordering::Release); + return; + }; + let interval = Duration::from_millis( + shared + .tuning + .cluster_transport + .change_feed_heartbeat_ms + .max(1), + ); + let weak: Weak = Arc::downgrade(shared); + runtime.spawn(async move { + let mut ticker = tokio::time::interval(interval); + ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + loop { + ticker.tick().await; + let Some(shared) = weak.upgrade() else { + return; + }; + shared.change_stream.heartbeat(&shared); + } + }); + } +} + +fn spawn_peer( + runtime: &tokio::runtime::Handle, + transport: Arc, + node_id: u64, + peer: u64, +) -> mpsc::Sender> { + use nodedb_cluster::RaftRpc; + use nodedb_cluster::wire::{VShardEnvelope, VShardMessageType}; + + let (sender, mut receiver) = mpsc::channel::>(PEER_QUEUE); + runtime.spawn(async move { + while let Some(payload) = receiver.recv().await { + let envelope = VShardEnvelope::new( + VShardMessageType::NotifyBroadcast, + node_id, + peer, + 0, + payload, + ); + // The ack returns once the peer appended the run. + if let Err(error) = transport + .send_rpc(peer, RaftRpc::VShardEnvelope(envelope.to_bytes())) + .await + { + trace!(peer, %error, "forwarding a change-stream run failed; the peer records a hole"); + } + } + }); + sender +} + +impl NotifyBroadcastMsg { + /// The wire form of `run`. + pub(crate) fn from_run(source_node: u64, run: &AppendedRun) -> Self { + let (partition_kind, partition_id) = match run.partition { + ChangePartition::Group(group) => (NOTIFY_PARTITION_GROUP, group), + ChangePartition::Calvin(vshard) => (NOTIFY_PARTITION_CALVIN, u64::from(vshard)), + }; + Self { + source_node, + partition_kind, + partition_id, + after: run.after, + through: run.through, + changes: run + .events + .iter() + .map(|event| NotifyChange { + position: event.position(), + tenant_id: event.tenant_id.as_u64(), + database_id: event.database_id().as_u64(), + collection: event.collection.clone(), + document_id: event.document_id.to_string(), + operation: event.operation.as_str().to_string(), + timestamp_ms: event.timestamp_ms, + lsn: event.lsn.as_u64(), + }) + .collect(), + } + } + + /// The partition the run belongs to, `None` for an unknown kind. + pub fn partition(&self) -> Option { + match self.partition_kind { + NOTIFY_PARTITION_GROUP => Some(ChangePartition::Group(self.partition_id)), + NOTIFY_PARTITION_CALVIN => u32::try_from(self.partition_id) + .ok() + .map(ChangePartition::Calvin), + _ => None, + } + } +} diff --git a/nodedb/src/control/change_stream/stream/journaling.rs b/nodedb/src/control/change_stream/stream/journaling.rs new file mode 100644 index 000000000..687042771 --- /dev/null +++ b/nodedb/src/control/change_stream/stream/journaling.rs @@ -0,0 +1,242 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The change stream's side of the durable journal: rebuild the ring at +//! boot, and journal each feed at the point its durability contract names. +//! See [`crate::control::change_stream::journal`] for the contract. + +use std::sync::Arc; + +use crate::control::change_stream::journal::ChangeJournal; +use crate::event::cdc::CdcOffset; + +use super::bus::{BusState, ChangeStream}; +use super::ring::{AppendedRun, ChangeRun, PositionedChange}; +use super::{ChangePartition, SequencedChangeEvent}; + +/// Events due for the journal, with the partition's continuity at the time. +pub(super) struct JournalBatch { + partition: ChangePartition, + feed_after: CdcOffset, + through: CdcOffset, + events: Vec, +} + +impl ChangeStream { + /// Rebuild the replay ring from `journal`, and journal every later + /// publish there. + /// + /// Events this process published before the journal attached, such as + /// entries the apply loop settled first, go back on top of the rebuilt + /// feeds and are journaled with the next batch of their feed. + pub fn attach_journal(&self, journal: Arc) -> crate::Result<()> { + let feeds = journal.load()?; + let early_runs = { + let mut state = self.lock(); + let (early_events, early_feeds) = state.ring.drain(); + for feed in feeds { + state.ring.append(ChangeRun { + partition: feed.partition, + after: Some(feed.after), + through: feed.through, + changes: feed.changes, + }); + } + let mut early_runs = Vec::new(); + for (partition, floor, through) in early_feeds { + let changes = early_events + .iter() + .filter(|event| event.partition() == partition) + .map(|event| PositionedChange { + position: event.position(), + database_id: event.database_id(), + event: event.event().clone(), + }) + .collect(); + let appended = state.ring.append(ChangeRun { + partition, + after: Some(floor), + through, + changes, + }); + match partition { + ChangePartition::Group(group_id) => state + .group_pending + .entry(group_id) + .or_default() + .extend(appended.events), + ChangePartition::Calvin(_) => early_runs.push(appended), + } + } + early_runs + }; + self.journal + .set(journal) + .map_err(|_| crate::Error::Internal { + detail: "the change-feed journal was already attached".into(), + })?; + for run in &early_runs { + self.journal_run(run); + } + Ok(()) + } + + /// Hold a settled run of a data group until the group's applied floor + /// covers it. + pub(super) fn queue_group(&self, state: &mut BusState, group_id: u64, run: &AppendedRun) { + if self.journal.get().is_none() || run.events.is_empty() { + return; + } + state + .group_pending + .entry(group_id) + .or_default() + .extend(run.events.iter().cloned()); + } + + /// Journal `group_id`'s changes of every entry at or below `floor`, the + /// applied floor the apply loop is about to save. Every entry the floor + /// covers then has its changes on disk. + pub(crate) fn journal_group(&self, group_id: u64, floor: u64) { + let through = CdcOffset::whole_index(floor); + let partition = ChangePartition::Group(group_id); + let batch = { + let mut state = self.lock(); + if self.journal.get().is_none() { + return; + } + let pending = state.group_pending.entry(group_id).or_default(); + let covered = pending.partition_point(|event| event.position() <= through); + let events: Vec = pending.drain(..covered).collect(); + let Some((feed_after, _)) = state.ring.feed_continuity(partition) else { + return; + }; + JournalBatch { + partition, + feed_after, + through, + events, + } + }; + self.persist(batch); + } + + /// Journal a run a Calvin transaction or a forwarding leader produced, + /// before its publisher acknowledges it. + pub(super) fn journal_run(&self, run: &AppendedRun) { + if self.journal.get().is_none() { + return; + } + let Some((feed_after, _)) = self.lock().ring.feed_continuity(run.partition) else { + return; + }; + self.persist(JournalBatch { + partition: run.partition, + feed_after, + through: run.through, + events: run.events.clone(), + }); + } + + /// Write `batch` to the journal. A lost batch shows as a hole in the + /// journal, and its cursors reset after a restart. + fn persist(&self, batch: JournalBatch) { + let Some(journal) = self.journal.get() else { + return; + }; + let JournalBatch { + partition, + feed_after, + through, + events, + } = batch; + if let Err(error) = journal.persist(partition, feed_after, through, &events) { + tracing::error!( + ?partition, + %error, + "journaling change-feed events failed" + ); + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::change_stream::{ChangeEvent, ChangeOperation, ReplayStart}; + use crate::types::{DatabaseId, Lsn, TenantId}; + use nodedb_types::RowIdentity; + + fn event(document: &str) -> ChangeEvent { + ChangeEvent { + lsn: Lsn::new(1), + tenant_id: TenantId::new(1), + collection: "orders".into(), + document_id: RowIdentity::from_user_key(document), + operation: ChangeOperation::Insert, + timestamp_ms: 1, + after: None, + } + } + + fn open(dir: &std::path::Path) -> ChangeStream { + let stream = ChangeStream::new(64); + let journal = Arc::new(ChangeJournal::open(dir, 64).expect("open journal")); + stream.attach_journal(journal).expect("attach"); + stream + } + + /// A cursor taken before a restart resumes after it at the next event: + /// the group's journaled changes come back. + #[test] + fn a_cursor_resumes_across_a_restart() { + let dir = tempfile::tempdir().expect("tempdir"); + let cursor = { + let stream = open(dir.path()); + for (index, document) in [(1, "g1"), (2, "g2"), (3, "g3"), (4, "g4")] { + stream.stage(4, index, DatabaseId::DEFAULT, vec![event(document)]); + } + stream.settle_group(4, 1, 4); + stream.journal_group(4, 4); + let page = stream + .query_changes(TenantId::new(1), None, ReplayStart::Timestamp(0), 2) + .expect("replay"); + page.cursor + }; + + let stream = open(dir.path()); + let resumed = stream + .query_changes(TenantId::new(1), None, ReplayStart::Cursor(cursor), 64) + .expect("replay"); + let documents: Vec<&str> = resumed + .events + .iter() + .map(|change| change.document_id.as_str()) + .collect(); + assert_eq!(documents, vec!["g3", "g4"]); + } + + /// Settled entries above the saved applied floor are not journaled, so + /// they are delivered again after a restart and publish again. A cursor + /// inside the journaled stretch resumes. + #[test] + fn entries_above_the_saved_floor_are_not_journaled() { + let dir = tempfile::tempdir().expect("tempdir"); + let cursor = { + let stream = open(dir.path()); + stream.stage(4, 1, DatabaseId::DEFAULT, vec![event("g1")]); + stream.stage(4, 2, DatabaseId::DEFAULT, vec![event("g2")]); + stream.settle_group(4, 1, 2); + stream.journal_group(4, 1); + stream + .query_changes(TenantId::new(1), None, ReplayStart::Timestamp(0), 1) + .expect("replay") + .cursor + }; + + let stream = open(dir.path()); + let resumed = stream + .query_changes(TenantId::new(1), None, ReplayStart::Cursor(cursor), 8) + .expect("the journaled stretch resumes"); + assert!(resumed.events.is_empty()); + } +} diff --git a/nodedb/src/control/change_stream/stream/mod.rs b/nodedb/src/control/change_stream/stream/mod.rs index 91c4fcf76..755d42a82 100644 --- a/nodedb/src/control/change_stream/stream/mod.rs +++ b/nodedb/src/control/change_stream/stream/mod.rs @@ -2,12 +2,17 @@ pub mod bus; pub mod cursor; +pub mod error; +pub mod fanout; +pub mod journaling; +pub mod ring; pub mod subscription; pub mod types; -pub use bus::{ - ChangeStream, ReplayError, ReplaySnapshot, ReplayStart, broadcast_notify_to_cluster, -}; -pub use cursor::{ChangeCursor, CursorParseError}; +pub use bus::ChangeStream; +pub use cursor::{ChangeCursor, ChangePartition, CursorParseError, CursorStep}; +pub use error::ChangeStreamError; +pub(crate) use ring::{ChangeRun, PositionedChange}; +pub use ring::{ReplayError, ReplaySnapshot, ReplayStart, ReplayedChange}; pub use subscription::Subscription; pub use types::{ChangeEvent, ChangeOperation, SequencedChangeEvent}; diff --git a/nodedb/src/control/change_stream/stream/ring.rs b/nodedb/src/control/change_stream/stream/ring.rs new file mode 100644 index 000000000..fe5b5084d --- /dev/null +++ b/nodedb/src/control/change_stream/stream/ring.rs @@ -0,0 +1,492 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The bounded replay ring behind the change stream, and what this node +//! holds of each partition's feed. +//! +//! A partition's events enter the ring in position order. The feed records +//! `through`, the position up to which this node saw the partition, and +//! `floor`, the position above which it holds every event. A run whose +//! publisher vouches continuity from a position above `through` leaves a +//! hole: the floor rises to that position, and a cursor below it must reset. + +use std::collections::{BTreeMap, VecDeque}; +use std::ops::Deref; + +use crate::event::cdc::CdcOffset; +use crate::types::{DatabaseId, TenantId}; + +use super::{ChangeCursor, ChangeEvent, ChangePartition, SequencedChangeEvent}; + +/// One change at its position, before it enters a ring. +#[derive(Debug, Clone)] +pub(crate) struct PositionedChange { + pub position: CdcOffset, + pub database_id: DatabaseId, + pub event: ChangeEvent, +} + +/// A publisher's contiguous stretch of one partition's feed. +#[derive(Debug, Clone)] +pub(crate) struct ChangeRun { + pub partition: ChangePartition, + /// The publisher holds every event of the partition in `(after, + /// through]`, and `changes` are all of them. `None` continues the + /// publisher's previous run of the partition. + pub after: Option, + pub through: CdcOffset, + /// In position order. + pub changes: Vec, +} + +/// What a ring took from a run: the events it did not hold yet, and the +/// stretch they cover. +#[derive(Debug, Clone)] +pub(crate) struct AppendedRun { + pub partition: ChangePartition, + /// The ring holds every event of the partition in `(after, through]`. + pub after: CdcOffset, + pub through: CdcOffset, + pub events: Vec, +} + +/// Replay starts after an acknowledged cursor or at a timestamp. +#[derive(Clone, Debug)] +pub enum ReplayStart { + Cursor(ChangeCursor), + Timestamp(u64), +} + +/// A replayed event and the cursor that resumes right after it. +#[derive(Clone, Debug)] +pub struct ReplayedChange { + pub event: SequencedChangeEvent, + pub cursor: ChangeCursor, +} + +impl Deref for ReplayedChange { + type Target = SequencedChangeEvent; + + fn deref(&self) -> &Self::Target { + &self.event + } +} + +/// A consistent replay of the ring. +#[derive(Clone, Debug)] +pub struct ReplaySnapshot { + pub events: Vec, + /// Resumes right after the replay: past every returned event and every + /// held event the replay passed over. + pub cursor: ChangeCursor, + /// The limit stopped the replay before the ring's end. + pub has_more: bool, +} + +/// A cursor this node cannot resume from; the client must reset. +#[derive(Clone, Copy, Debug, PartialEq, Eq)] +pub enum ReplayError { + Expired, +} + +/// Which events a replay returns. +#[derive(Clone, Copy, Debug)] +pub(crate) struct ReplayScope<'a> { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + pub collection: Option<&'a str>, +} + +impl ReplayScope<'_> { + fn matches(&self, event: &SequencedChangeEvent) -> bool { + event.database_id() == self.database_id + && event.tenant_id == self.tenant_id + && self.collection.is_none_or(|name| event.collection == name) + } +} + +#[derive(Debug, Clone, Copy)] +struct PartitionFeed { + through: CdcOffset, + floor: CdcOffset, + /// The highest position the ring evicted. + evicted: Option, +} + +pub(super) struct ReplayRing { + events: VecDeque, + feeds: BTreeMap, + capacity: usize, +} + +impl ReplayRing { + pub fn new(capacity: usize) -> Self { + Self { + events: VecDeque::with_capacity(capacity), + feeds: BTreeMap::new(), + capacity: capacity.max(1), + } + } + + /// Append the events of `run` the ring does not hold yet. + pub fn append(&mut self, run: ChangeRun) -> AppendedRun { + let ChangeRun { + partition, + after, + through, + changes, + } = run; + let prior = self.feeds.get(&partition).copied(); + let (start, floor) = match (prior, after) { + (Some(feed), None) => (feed.through, feed.floor), + (Some(feed), Some(after)) if after <= feed.through => (feed.through, feed.floor), + // A hole between what this node saw and what the run vouches. + (Some(_), Some(after)) | (None, Some(after)) => (after, after), + (None, None) => { + let start = changes + .first() + .map_or(through, |change| before(change.position)); + (start, start) + } + }; + let mut last = start; + let mut events = Vec::with_capacity(changes.len()); + for change in changes { + if change.position <= last { + continue; + } + last = change.position; + events.push(SequencedChangeEvent::new( + partition, + change.position, + floor, + change.database_id, + change.event, + )); + } + let through = through.max(last); + let evicted = prior.and_then(|feed| feed.evicted); + self.feeds.insert( + partition, + PartitionFeed { + through, + floor, + evicted, + }, + ); + for event in &events { + self.push(event.clone()); + } + AppendedRun { + partition, + after: start, + through, + events, + } + } + + fn push(&mut self, event: SequencedChangeEvent) { + if self.events.len() == self.capacity + && let Some(evicted) = self.events.pop_front() + && let Some(feed) = self.feeds.get_mut(&evicted.partition()) + { + feed.evicted = Some( + feed.evicted + .map_or(evicted.position(), |held| held.max(evicted.position())), + ); + } + self.events.push_back(event); + } + + /// Record that this node lacks every event of `partition` at or below + /// `floor`. + pub fn raise_floor(&mut self, partition: ChangePartition, floor: CdcOffset) { + let feed = self.feeds.entry(partition).or_insert(PartitionFeed { + through: floor, + floor, + evicted: None, + }); + feed.floor = feed.floor.max(floor); + feed.through = feed.through.max(floor); + } + + /// `(through, held_from)` of `partition`: the position up to which this + /// node saw it, and the position above which the ring holds every event + /// of it. + pub fn feed_bounds(&self, partition: ChangePartition) -> Option<(CdcOffset, CdcOffset)> { + let feed = self.feeds.get(&partition)?; + let held_from = feed + .evicted + .map_or(feed.floor, |evicted| evicted.max(feed.floor)); + Some((feed.through, held_from)) + } + + /// Every held event of `partition` above `after`, in position order. + pub fn events_after( + &self, + partition: ChangePartition, + after: CdcOffset, + ) -> Vec { + let mut events: Vec = Vec::new(); + for event in self.events.iter().rev() { + if event.partition() != partition { + continue; + } + // A partition's events sit in position order. + if event.position() <= after { + break; + } + events.push(event.clone()); + } + events.reverse(); + events + } + + /// `(floor, through)` of `partition`: the position above which this + /// node holds or held every event of it, and the position up to which it + /// saw it. + pub fn feed_continuity(&self, partition: ChangePartition) -> Option<(CdcOffset, CdcOffset)> { + let feed = self.feeds.get(&partition)?; + Some((feed.floor, feed.through)) + } + + /// Empty the ring: its events in ring order, and each partition's + /// `(partition, floor, through)`. + pub fn drain( + &mut self, + ) -> ( + Vec, + Vec<(ChangePartition, CdcOffset, CdcOffset)>, + ) { + let events = self.events.drain(..).collect(); + let feeds = std::mem::take(&mut self.feeds) + .into_iter() + .map(|(partition, feed)| (partition, feed.floor, feed.through)) + .collect(); + (events, feeds) + } + + /// Every partition this node saw. + pub fn partitions(&self) -> Vec { + self.feeds.keys().copied().collect() + } + + /// The cursor past every event this node saw. + pub fn head(&self) -> ChangeCursor { + let mut cursor = ChangeCursor::default(); + for (partition, feed) in &self.feeds { + cursor.raise(*partition, feed.through); + } + cursor + } + + /// Whether this node can resume `cursor` without a hole. A replicated + /// partition the node never saw has no hole to report. + pub fn validate(&self, cursor: &ChangeCursor) -> Result<(), ReplayError> { + for (partition, position) in cursor.entries() { + let Some(feed) = self.feeds.get(&partition) else { + continue; + }; + if position < feed.floor || feed.evicted.is_some_and(|evicted| position < evicted) { + return Err(ReplayError::Expired); + } + } + Ok(()) + } + + /// Replay the ring's events in `scope` from `start`, at most `limit`. + pub fn replay( + &self, + scope: ReplayScope<'_>, + start: &ReplayStart, + limit: usize, + ) -> Result { + let mut cursor = match start { + ReplayStart::Cursor(cursor) => { + self.validate(cursor)?; + cursor.clone() + } + ReplayStart::Timestamp(_) => ChangeCursor::default(), + }; + let mut events = Vec::new(); + let mut has_more = false; + for event in &self.events { + let eligible = scope.matches(event) + && match start { + ReplayStart::Cursor(start) => !start.covers(event), + ReplayStart::Timestamp(since) => event.timestamp_ms >= *since, + }; + if eligible && events.len() == limit { + has_more = true; + break; + } + cursor.raise(event.partition(), event.position()); + if eligible { + events.push(ReplayedChange { + event: event.clone(), + cursor: cursor.clone(), + }); + } + } + Ok(ReplaySnapshot { + events, + cursor, + has_more, + }) + } +} + +/// The largest position below `position`. +fn before(position: CdcOffset) -> CdcOffset { + if position.sequence > 0 { + CdcOffset::at(position.epoch, position.index, position.sequence - 1) + } else if position.index > 0 { + CdcOffset::at(position.epoch, position.index - 1, u64::MAX) + } else { + position + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::change_stream::ChangeOperation; + use crate::types::Lsn; + + const GROUP: ChangePartition = ChangePartition::Group(4); + + fn change(index: u64, ordinal: u64) -> PositionedChange { + PositionedChange { + position: CdcOffset::data_event(0, index, ordinal), + database_id: DatabaseId::DEFAULT, + event: ChangeEvent { + lsn: Lsn::new(index), + tenant_id: TenantId::new(1), + collection: "orders".into(), + document_id: nodedb_types::RowIdentity::from_user_key(format!("{index}.{ordinal}")), + operation: ChangeOperation::Insert, + timestamp_ms: index, + after: None, + }, + } + } + + fn run(after: u64, through: u64, changes: Vec) -> ChangeRun { + ChangeRun { + partition: GROUP, + after: Some(CdcOffset::whole_index(after)), + through: CdcOffset::whole_index(through), + changes, + } + } + + fn scope() -> ReplayScope<'static> { + ReplayScope { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + collection: None, + } + } + + fn at(cursor: &ChangeCursor) -> ReplayStart { + ReplayStart::Cursor(cursor.clone()) + } + + #[test] + fn a_repeated_run_appends_nothing() { + let mut ring = ReplayRing::new(16); + let first = ring.append(run(9, 11, vec![change(10, 1), change(11, 1)])); + assert_eq!(first.events.len(), 2); + // The same stretch from another replica after a leader change. + let again = ring.append(run( + 9, + 12, + vec![change(10, 1), change(11, 1), change(12, 1)], + )); + assert_eq!(again.events.len(), 1); + assert_eq!(again.after, CdcOffset::whole_index(11)); + assert_eq!(again.events[0].position(), CdcOffset::data_event(0, 12, 1)); + } + + #[test] + fn a_cursor_from_another_node_resumes_at_the_same_event() { + let mut leader = ReplayRing::new(16); + let mut follower = ReplayRing::new(16); + for ring in [&mut leader, &mut follower] { + ring.append(run(0, 3, vec![change(1, 1), change(2, 1), change(3, 1)])); + } + let page = leader + .replay(scope(), &ReplayStart::Timestamp(0), 2) + .expect("replay"); + assert!(page.has_more); + let resumed = follower + .replay(scope(), &at(&page.cursor), 16) + .expect("the follower holds the same feed"); + assert_eq!(resumed.events.len(), 1); + assert_eq!(resumed.events[0].position(), CdcOffset::data_event(0, 3, 1)); + } + + #[test] + fn a_hole_expires_every_cursor_below_it() { + let mut ring = ReplayRing::new(16); + ring.append(run(0, 2, vec![change(1, 1), change(2, 1)])); + let page = ring + .replay(scope(), &ReplayStart::Timestamp(0), 16) + .expect("replay"); + // Entries 3..=5 never reached this node. + let appended = ring.append(run(5, 6, vec![change(6, 1)])); + assert_eq!(appended.events[0].floor(), CdcOffset::whole_index(5)); + assert_eq!( + ring.replay(scope(), &at(&page.cursor), 16).err(), + Some(ReplayError::Expired) + ); + let mut past = ChangeCursor::default(); + past.raise(GROUP, CdcOffset::whole_index(5)); + assert_eq!( + ring.replay(scope(), &at(&past), 16) + .expect("replay") + .events + .len(), + 1 + ); + } + + #[test] + fn an_install_floor_and_eviction_expire_older_cursors() { + let mut ring = ReplayRing::new(2); + ring.append(run(0, 3, vec![change(1, 1), change(2, 1), change(3, 1)])); + let mut cursor = ChangeCursor::default(); + cursor.raise(GROUP, CdcOffset::whole_index(0)); + assert_eq!( + ring.replay(scope(), &at(&cursor), 16).err(), + Some(ReplayError::Expired), + "event 1 was evicted before this cursor consumed it" + ); + cursor.raise(GROUP, CdcOffset::data_event(0, 1, 1)); + assert!(ring.replay(scope(), &at(&cursor), 16).is_ok()); + ring.raise_floor(GROUP, CdcOffset::whole_index(8)); + assert_eq!( + ring.replay(scope(), &at(&cursor), 16).err(), + Some(ReplayError::Expired) + ); + } + + #[test] + fn a_timestamp_replay_resumes_past_the_events_it_skipped() { + let mut ring = ReplayRing::new(16); + ring.append(run(0, 3, vec![change(1, 1), change(2, 1), change(3, 1)])); + let page = ring + .replay(scope(), &ReplayStart::Timestamp(3), 16) + .expect("replay"); + assert_eq!(page.events.len(), 1); + assert_eq!( + page.cursor.position(GROUP), + Some(CdcOffset::data_event(0, 3, 1)) + ); + assert!( + ring.replay(scope(), &at(&page.cursor), 16) + .expect("replay") + .events + .is_empty() + ); + } +} diff --git a/nodedb/src/control/change_stream/stream/subscription.rs b/nodedb/src/control/change_stream/stream/subscription.rs index 6b740fc62..e63e10e59 100644 --- a/nodedb/src/control/change_stream/stream/subscription.rs +++ b/nodedb/src/control/change_stream/stream/subscription.rs @@ -5,7 +5,7 @@ use std::sync::atomic::{AtomicU64, Ordering}; use crate::types::{DatabaseId, TenantId}; -use super::{ChangeEvent, SequencedChangeEvent}; +use super::{ChangeCursor, ChangeEvent, SequencedChangeEvent}; /// A filtered, bounded change-stream subscription. pub struct Subscription { @@ -18,6 +18,9 @@ pub struct Subscription { pub(super) sequenced_receiver: tokio::sync::broadcast::Receiver, pub(super) database_filter: Option, pub(super) active_counter: Arc, + /// The stream's head when the subscription opened: it receives exactly + /// the events past this cursor. + pub(super) start: ChangeCursor, } impl Drop for Subscription { @@ -27,6 +30,11 @@ impl Drop for Subscription { } impl Subscription { + /// The cursor a consumer of this subscription starts from. + pub fn start_cursor(&self) -> &ChangeCursor { + &self.start + } + pub async fn recv_sequenced( &mut self, ) -> Result { diff --git a/nodedb/src/control/change_stream/stream/types.rs b/nodedb/src/control/change_stream/stream/types.rs index 4453f0c78..dcc0fc4ad 100644 --- a/nodedb/src/control/change_stream/stream/types.rs +++ b/nodedb/src/control/change_stream/stream/types.rs @@ -6,7 +6,9 @@ use nodedb_types::RowIdentity; use crate::types::{DatabaseId, Lsn, TenantId}; -use super::ChangeCursor; +use crate::event::cdc::CdcOffset; + +use super::ChangePartition; /// A single mutation event broadcast by the change stream. #[derive(Debug, Clone)] @@ -40,26 +42,45 @@ impl ChangeOperation { } } -/// A publication-ordered change event. The cursor is allocated atomically with -/// ring insertion and broadcast, rather than derived from the WAL LSN. +/// A change event at its position in its partition's feed. #[derive(Debug, Clone)] pub struct SequencedChangeEvent { - cursor: ChangeCursor, + partition: ChangePartition, + position: CdcOffset, + /// The node's feed of `partition` holds every event above this position. + /// A consumer whose cursor sits below it can have missed events. + floor: CdcOffset, database_id: DatabaseId, event: ChangeEvent, } impl SequencedChangeEvent { - pub(crate) fn new(cursor: ChangeCursor, database_id: DatabaseId, event: ChangeEvent) -> Self { + pub(crate) fn new( + partition: ChangePartition, + position: CdcOffset, + floor: CdcOffset, + database_id: DatabaseId, + event: ChangeEvent, + ) -> Self { Self { - cursor, + partition, + position, + floor, database_id, event, } } - pub fn cursor(&self) -> ChangeCursor { - self.cursor + pub fn partition(&self) -> ChangePartition { + self.partition + } + + pub fn position(&self) -> CdcOffset { + self.position + } + + pub fn floor(&self) -> CdcOffset { + self.floor } pub fn database_id(&self) -> DatabaseId { diff --git a/nodedb/src/control/checkpoint_archival.rs b/nodedb/src/control/checkpoint_archival.rs index c36ce4236..8113ffe2c 100644 --- a/nodedb/src/control/checkpoint_archival.rs +++ b/nodedb/src/control/checkpoint_archival.rs @@ -1,189 +1,170 @@ // SPDX-License-Identifier: BUSL-1.1 -//! WAL archival to cold storage, run by the checkpoint cycle before it -//! truncates. Archival bounds truncation: a segment cold storage did not -//! accept stays on local disk instead of being deleted unarchived. +//! The archived bound on WAL truncation. Truncation never deletes a segment +//! the archive does not hold. -use tracing::{debug, warn}; +use nodedb_wal::segment::SegmentMeta; use crate::wal::WalManager; +use crate::wal::archiver::{ArchiveCursor, WalArchiver}; /// Bound that truncates nothing: no WAL segment precedes LSN 0. const NO_TRUNCATION_BOUND: u64 = 0; -/// The LSN truncation must not pass, given the segments on disk and the -/// `first_lsn` of every segment whose archival failed. +/// The LSN truncation must not pass. /// -/// The bound is `checkpoint_lsn` when every eligible segment reached cold -/// storage, and the lowest failed `first_lsn` otherwise, so that segment and -/// everything after it survive. `None` segments means `list_segments` itself -/// failed: the eligible set is unknown, and unknown is never permissive. +/// `truncate_before(lsn)` deletes a segment when its successor starts at or +/// below `lsn`. A bound equal to the lowest unarchived segment's `first_lsn` +/// therefore keeps that segment and everything after it. +/// +/// - `None` segments: the WAL directory was unlistable. Unknown is never +/// permissive, so nothing is truncated. +/// - `None` cursor: the archive listing has not succeeded yet. Every local +/// segment counts as unarchived. fn archived_truncation_bound( - segment_first_lsns: Option<&[u64]>, - failed_first_lsns: &[u64], + segments: Option<&[SegmentMeta]>, + cursor: Option<&ArchiveCursor>, checkpoint_lsn: u64, ) -> u64 { - let Some(first_lsns) = segment_first_lsns else { + let Some(segments) = segments else { return NO_TRUNCATION_BOUND; }; - let mut bound = checkpoint_lsn; - for first_lsn in first_lsns { - if failed_first_lsns.contains(first_lsn) { - bound = bound.min(*first_lsn); - } - } - bound + let lowest_unarchived = match cursor { + Some(cursor) => cursor.lowest_unarchived(segments), + None => segments.first().map(|seg| seg.first_lsn), + }; + lowest_unarchived.map_or(checkpoint_lsn, |lsn| lsn.min(checkpoint_lsn)) } -/// Archive WAL segments that the upcoming truncation deletes, and return the -/// LSN that truncation must not pass. -/// -/// A segment is eligible for deletion (and therefore archival) when the segment -/// immediately following it has a `first_lsn <= checkpoint_lsn`. Each eligible -/// segment is uploaded before `truncate_before` deletes it, preserving a -/// continuous WAL archive in cold storage for point-in-time recovery. +/// The truncation bound after an archive pass. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct ArchivedBound { + /// The LSN truncation must not pass. + pub lsn: u64, + /// An unarchived sealed segment, or an unknown archive state, holds + /// truncation below the checkpoint. A bound at the active segment is not + /// held: the active segment is never deleted. + pub held: bool, +} + +/// Run an archive pass, then return the bound the upcoming truncation must +/// not pass. /// -/// A segment the archive did not accept holds truncation back at that segment. -/// The local WAL then grows until archival recovers. That is the intended -/// outcome: a full disk is loud and recoverable, an archive hole is silent and -/// permanent. -pub(crate) async fn archive_wal_segments_before_truncation( +/// A segment the archive did not accept holds truncation back at that +/// segment. The local WAL then grows until archival recovers. A full disk is +/// loud and recoverable. An archive hole is silent and permanent. +pub(crate) async fn archive_then_bound( + archiver: &mut WalArchiver, wal: &WalManager, checkpoint_lsn: u64, - cold: &crate::storage::cold::ColdStorage, -) -> u64 { - let segments = match wal.list_segments() { - Ok(s) => s, - Err(e) => { - warn!( - error = %e, - "WAL archival: segments unlistable — truncation holds until the next cycle" - ); - crate::diag::wal_archival_failed_truncation_held( - "list_segments", - Some(&e), - NO_TRUNCATION_BOUND, - ); - return archived_truncation_bound(None, &[], checkpoint_lsn); - } - }; - - // Determine which segments are eligible using the same logic as truncate_before: - // a segment is deletable when its successor's first_lsn <= checkpoint_lsn. - let mut failed_first_lsns: Vec = Vec::new(); - for seg in &segments { - let next_first_lsn = segments - .iter() - .find(|s| s.first_lsn > seg.first_lsn) - .map(|s| s.first_lsn) - .unwrap_or(u64::MAX); - - if next_first_lsn > checkpoint_lsn { - // Not eligible for deletion; skip. - continue; - } - - let segment_name = match seg.path.file_name().and_then(|n| n.to_str()) { - Some(n) => n.to_owned(), - None => { - warn!( - path = %seg.path.display(), - "WAL archival: segment path unnameable — segment stays on local disk" - ); - crate::diag::wal_archival_failed_truncation_held( - "segment_path", - None, - seg.first_lsn, - ); - failed_first_lsns.push(seg.first_lsn); - continue; - } - }; - - match cold.upload_wal_segment(&seg.path, &segment_name).await { - Ok(object_path) => { - debug!( - segment = %segment_name, - object_path = %object_path, - first_lsn = seg.first_lsn, - "WAL segment archived before truncation" - ); - } - Err(e) => { - warn!( - segment = %segment_name, - error = %e, - first_lsn = seg.first_lsn, - "WAL archival: upload failed — segment stays on local disk" - ); - crate::diag::wal_archival_failed_truncation_held("upload", Some(&e), seg.first_lsn); - failed_first_lsns.push(seg.first_lsn); - } - } +) -> ArchivedBound { + let snapshot = archiver.tick(wal).await; + let lsn = archived_truncation_bound( + snapshot.as_ref().map(|s| s.segments.as_slice()), + archiver.cursor(), + checkpoint_lsn, + ); + let active = snapshot.as_ref().map(|s| s.active_first_lsn); + ArchivedBound { + lsn, + held: lsn < checkpoint_lsn && Some(lsn) != active, } - - let first_lsns: Vec = segments.iter().map(|s| s.first_lsn).collect(); - archived_truncation_bound(Some(&first_lsns), &failed_first_lsns, checkpoint_lsn) } #[cfg(test)] mod tests { + use std::path::Path; + + use nodedb_wal::segment::{discover_segments, segment_path, truncate_segments}; + use super::*; - /// Archival that fully succeeded leaves truncation exactly where the - /// checkpoint put it. + fn seg(first_lsn: u64) -> SegmentMeta { + SegmentMeta { + path: segment_path(Path::new("/nonexistent"), first_lsn), + first_lsn, + file_size: 1, + } + } + + fn archived(lsns: &[u64]) -> ArchiveCursor { + let mut cursor = ArchiveCursor::default(); + for lsn in lsns { + cursor.mark_archived(*lsn); + } + cursor + } + + /// With every sealed segment archived, the active segment is the bound. + /// It is never deleted anyway, so the checkpoint alone decides. #[test] - fn every_upload_succeeded_does_not_lower_the_checkpoint_lsn() { + fn fully_archived_sealed_set_bounds_at_the_active_segment() { + let segments = [seg(10), seg(20), seg(30)]; + let cursor = archived(&[10, 20]); assert_eq!( - archived_truncation_bound(Some(&[10, 20, 30]), &[], 900), - 900 + archived_truncation_bound(Some(&segments), Some(&cursor), 900), + 30 ); } - /// A failed upload keeps its own segment and every later one on disk: - /// deleting them would leave a hole the archive can never fill. + /// An unarchived middle segment keeps itself and every later one. #[test] - fn middle_segment_upload_failure_bounds_truncation_at_that_segment() { + fn unarchived_middle_segment_bounds_truncation_at_that_segment() { + let segments = [seg(10), seg(20), seg(30)]; + let cursor = archived(&[10, 30]); assert_eq!( - archived_truncation_bound(Some(&[10, 20, 30]), &[20], 900), + archived_truncation_bound(Some(&segments), Some(&cursor), 900), 20 ); } - /// The lowest failed segment wins, so a later success cannot raise the - /// bound past a gap. + /// A checkpoint below the archived bound stays the binding floor. #[test] - fn lowest_failed_segment_wins_over_a_later_one() { + fn checkpoint_below_the_archived_bound_wins() { + let segments = [seg(10), seg(20), seg(30)]; + let cursor = archived(&[10, 20]); assert_eq!( - archived_truncation_bound(Some(&[10, 20, 30]), &[30, 20], 900), - 20 + archived_truncation_bound(Some(&segments), Some(&cursor), 15), + 15 ); } - /// The first eligible segment failing truncates nothing: no segment - /// precedes it. + /// Before the archive listing succeeds, the lowest local segment is the + /// bound, so nothing is deleted. #[test] - fn first_segment_upload_failure_truncates_nothing() { - assert_eq!( - archived_truncation_bound(Some(&[10, 20, 30]), &[10], 900), - 10 - ); + fn unrecovered_archiver_truncates_nothing() { + let segments = [seg(10), seg(20), seg(30)]; + assert_eq!(archived_truncation_bound(Some(&segments), None, 900), 10); } - /// An unlistable WAL directory hides which segments are eligible, so the - /// cycle archives nothing and truncates nothing. + /// An unlistable WAL directory hides which segments exist. #[test] fn list_segments_failure_truncates_nothing() { - assert_eq!(archived_truncation_bound(None, &[], 900), 0); + assert_eq!(archived_truncation_bound(None, None, 900), 0); } - /// A failed segment above the checkpoint LSN was never eligible for - /// deletion, so it cannot lower the bound. + /// The bound applied to a real WAL directory: the checkpoint allows + /// deleting every sealed segment, but the unarchived one survives, and so + /// does everything after it. #[test] - fn failure_above_the_checkpoint_lsn_does_not_lower_the_bound() { - assert_eq!( - archived_truncation_bound(Some(&[10, 950]), &[950], 900), - 900 - ); + fn held_archive_bound_stops_truncation() { + let dir = tempfile::tempdir().unwrap(); + for lsn in [10u64, 20, 30, 40] { + std::fs::write(segment_path(dir.path(), lsn), b"segment").unwrap(); + } + let segments = discover_segments(dir.path()).unwrap(); + let cursor = archived(&[10]); + + let bound = archived_truncation_bound(Some(&segments), Some(&cursor), 1_000); + assert_eq!(bound, 20); + let result = truncate_segments(dir.path(), bound, 40).unwrap(); + assert_eq!(result.segments_deleted, 1); + + let left: Vec = discover_segments(dir.path()) + .unwrap() + .iter() + .map(|s| s.first_lsn) + .collect(); + assert_eq!(left, vec![20, 30, 40]); } } diff --git a/nodedb/src/control/checkpoint_manager.rs b/nodedb/src/control/checkpoint_manager.rs index 310049ba5..18a2f4652 100644 --- a/nodedb/src/control/checkpoint_manager.rs +++ b/nodedb/src/control/checkpoint_manager.rs @@ -23,9 +23,10 @@ //! them, so a truncation past a lagging consumer silently drops every CDC //! row, trigger fire and streaming-MV update in the gap. //! 5. A `RecordType::Checkpoint` WAL record is written at the global LSN. -//! 6. Eligible segments are archived to cold storage when it is configured. -//! A segment the archive did not accept bounds the truncation point, so -//! no segment is deleted before it is safely in cold storage. +//! 6. When cold storage is configured, the WAL archiver uploads every sealed +//! segment the archive does not hold yet. The lowest unarchived segment +//! bounds the truncation point, so no segment is deleted before it is in +//! cold storage. The archiver also runs on its own shorter tick. //! 7. `WalManager::truncate_before()` deletes old WAL segments. //! //! ## Frequency @@ -40,31 +41,73 @@ use tracing::{debug, info, warn}; use crate::bridge::dispatch::Dispatcher; use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Status}; -use crate::control::checkpoint_archival::archive_wal_segments_before_truncation; +use crate::control::checkpoint_archival::archive_then_bound; use crate::control::request_tracker::RequestTracker; use crate::types::{DatabaseId, Lsn, ReadConsistency, RequestId, TenantId, TraceId, VShardId}; use crate::wal::WalManager; +use crate::wal::archiver::WalArchiver; use nodedb_physical::physical_plan::MetaOp; /// Monotonic counter for checkpoint request IDs. /// Uses a high base to avoid collision with session-generated request IDs. static CHECKPOINT_REQUEST_COUNTER: AtomicU64 = AtomicU64::new(0xFFFF_0000_0000_0000); -/// Configuration for the checkpoint manager. +/// Configuration for the checkpoint manager. Both intervals are non-zero by +/// construction: a zero period makes the tick timer panic. #[derive(Debug, Clone)] pub struct CheckpointManagerConfig { + interval: Duration, + core_timeout: Duration, + archive_interval: Duration, +} + +impl CheckpointManagerConfig { + /// Fails with `Error::Config` naming the setting when an interval is zero. + pub fn new( + interval: Duration, + core_timeout: Duration, + archive_interval: Duration, + ) -> crate::Result { + for (value, field) in [ + (interval, "checkpoint.interval_secs"), + (archive_interval, "checkpoint.wal_archive_interval_secs"), + ] { + if value.is_zero() { + return Err(crate::Error::Config { + detail: format!("invalid value '0' for {field}: expected a positive integer"), + }); + } + } + Ok(Self { + interval, + core_timeout, + archive_interval, + }) + } + /// Interval between checkpoint cycles. - pub interval: Duration, + pub fn interval(&self) -> Duration { + self.interval + } /// Timeout for individual core checkpoint responses. - pub core_timeout: Duration, + pub fn core_timeout(&self) -> Duration { + self.core_timeout + } + + /// Interval between WAL archive passes. Bounds the archive lag of a + /// sealed segment. + pub fn archive_interval(&self) -> Duration { + self.archive_interval + } } impl Default for CheckpointManagerConfig { fn default() -> Self { Self { - interval: Duration::from_secs(300), // 5 minutes + interval: Duration::from_secs(300), core_timeout: Duration::from_secs(30), + archive_interval: Duration::from_secs(10), } } } @@ -73,9 +116,9 @@ impl Default for CheckpointManagerConfig { /// /// Returns `None` (defer truncation) unless EVERY core reported a fresh /// flush LSN. A core that failed to dispatch or missed its response -/// deadline may still hold acknowledged-but-unflushed records below the -/// reporting cores' minimum LSN; truncating there would delete them and -/// lose the writes on restart. Also returns `None` when the minimum is 0 +/// deadline can still hold acknowledged-but-unflushed records below the +/// reporting cores' minimum LSN; truncating there deletes them and +/// loses the writes on restart. Also returns `None` when the minimum is 0 /// (no writes yet, nothing to truncate). /// /// ## Why the Event Plane is a floor too @@ -85,10 +128,10 @@ impl Default for CheckpointManagerConfig { /// watermark — and that watermark is flushed lazily, so it always trails the /// engines. Once a segment below a consumer's watermark is unlinked, that /// suffix is gone: every CDC row, trigger fire, and streaming-MV update for the -/// acknowledged writes in it is unrecoverable. Replay now refuses a request +/// acknowledged writes in it is unrecoverable. Replay refuses a request /// below the retained floor rather than returning the shorter suffix that -/// survives, so the loss is loud instead of silent — but refusing is only an -/// alarm. This floor is what keeps it from happening. +/// survives, so the loss is loud instead of silent. Refusing is only an +/// alarm. This floor keeps the loss from happening. /// /// `event_watermarks` is therefore folded in exactly as each core's engine LSN /// is, with the same conservatism: one entry per core, and a core that has @@ -125,8 +168,9 @@ pub struct CheckpointCycleInputs<'a> { pub num_cores: usize, /// Per-core response deadline. pub timeout: Duration, - /// When configured, segments are archived before they are unlinked. - pub cold_storage: Option>, + /// Present when cold storage is configured. Truncation never passes the + /// lowest segment it has not archived. + pub archiver: Option<&'a mut WalArchiver>, /// When present, the tombstone set is GC'd to the new truncation point. pub catalog: Option<&'a crate::control::security::catalog::SystemCatalog>, /// The applied state of every Calvin scheduler on this node. Saved in @@ -156,11 +200,11 @@ fn save_calvin_applied( } /// Run one checkpoint cycle: dispatch checkpoint to all cores, collect LSNs, -/// write checkpoint record, archive eligible WAL segments to cold storage (if +/// write checkpoint record, archive sealed WAL segments to cold storage (if /// configured), then truncate the WAL. /// /// Returns the global checkpoint LSN (min across all cores), or `None` if -/// the checkpoint could not be completed (e.g., a core didn't respond). +/// the checkpoint did not complete (e.g., a core didn't respond). pub async fn run_checkpoint_cycle(inputs: CheckpointCycleInputs<'_>) -> Option { let CheckpointCycleInputs { dispatcher, @@ -169,7 +213,7 @@ pub async fn run_checkpoint_cycle(inputs: CheckpointCycleInputs<'_>) -> Option) -> Option) -> Option = match watermark_store.load_all(num_cores) { @@ -288,9 +333,9 @@ pub async fn run_checkpoint_cycle(inputs: CheckpointCycleInputs<'_>) -> Option) -> Option) -> Option {} Err(e) => { // Non-fatal: a stale tombstone row is replay-safe, - // it just wastes redb space until the next pass. + // it only wastes redb space until the next pass. warn!( error = %e, truncate_lsn, @@ -460,6 +505,21 @@ mod tests { vec![u64::MAX; num_cores] } + #[test] + fn zero_intervals_are_config_errors_not_panics() { + let s = Duration::from_secs(1); + let zero = Duration::ZERO; + for (interval, archive, field) in [ + (zero, s, "checkpoint.interval_secs"), + (s, zero, "checkpoint.wal_archive_interval_secs"), + ] { + let err = CheckpointManagerConfig::new(interval, s, archive).unwrap_err(); + assert!(matches!(err, crate::Error::Config { .. }), "{err}"); + assert!(err.to_string().contains(field), "{err}"); + } + assert!(CheckpointManagerConfig::new(s, s, s).is_ok()); + } + #[test] fn all_cores_reported_distinct_lsns_returns_min() { assert_eq!( @@ -470,7 +530,7 @@ mod tests { #[test] fn one_core_missing_defers_even_though_responders_have_a_min() { - // Truncating at 5 would delete the missing core's unflushed records. + // Truncating at 5 deletes the missing core's unflushed records. assert_eq!(checkpoint_truncation_lsn(&[5, 10], &ahead(3), 3), None); } @@ -496,7 +556,7 @@ mod tests { /// The Event Plane binds truncation when it trails the engines. Its /// consumers recover ONLY from the WAL above their persisted watermark, so - /// truncating at the engine minimum would drop every CDC row, trigger fire + /// truncating at the engine minimum drops every CDC row, trigger fire /// and MV update in between — unrecoverably, whether or not replay notices. #[test] fn event_plane_behind_the_engines_clamps_truncation_to_it() { diff --git a/nodedb/src/control/checkpoint_task.rs b/nodedb/src/control/checkpoint_task.rs index 638e037e3..029ee6269 100644 --- a/nodedb/src/control/checkpoint_task.rs +++ b/nodedb/src/control/checkpoint_task.rs @@ -1,7 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 -//! The checkpoint manager's background task: a periodic cycle plus one final -//! cycle at shutdown. +//! The checkpoint manager's background task: a periodic cycle, a shorter WAL +//! archive tick when cold storage is configured, and one final cycle at +//! shutdown. use std::sync::Arc; @@ -11,11 +12,14 @@ use super::checkpoint_manager::{ CheckpointCycleInputs, CheckpointManagerConfig, run_checkpoint_cycle, }; use super::startup::StartupPhase; +use crate::wal::archiver::WalArchiver; /// Spawn the checkpoint manager as a background Tokio task. /// /// Runs `run_checkpoint_cycle` at the configured interval, and one final cycle -/// when the shutdown bus enters `DrainingControlPlane`. +/// when the shutdown bus enters `DrainingControlPlane`. The task owns the WAL +/// archiver: it archives every sealed segment each archive tick, and each +/// checkpoint cycle archives again before it truncates. /// /// The phase is load-bearing. The final cycle dispatches a checkpoint request /// to every Data Plane core and waits for the answers, so it must complete @@ -44,7 +48,7 @@ pub fn spawn_checkpoint_task( warn!(%error, "checkpoint manager not started: startup did not complete"); }), // The WAL on disk is intact, so restart replay covers what a final - // cycle would have made redundant. + // cycle makes redundant. _ = guard.await_signal() => { info!("shutdown before startup completed: no final checkpoint"); Err(()) @@ -55,14 +59,33 @@ pub fn spawn_checkpoint_task( return; } info!( - interval_secs = config.interval.as_secs(), + interval_secs = config.interval().as_secs(), + archive_interval_secs = config.archive_interval().as_secs(), "checkpoint manager started" ); + let mut archiver = shared.cold_storage.clone().map(|cold| { + WalArchiver::new( + shared.node_id, + shared.data_dir.clone(), + cold, + shared.system_metrics.clone(), + ) + }); + let mut checkpoint_tick = delayed_interval(config.interval()); + let mut archive_tick = delayed_interval(config.archive_interval()); + loop { let mut draining = false; tokio::select! { - _ = tokio::time::sleep(config.interval) => {} + _ = checkpoint_tick.tick() => {} + _ = archive_tick.tick(), if archiver.is_some() => { + if let Some(archiver) = archiver.as_mut() { + archiver.tick(&shared.wal).await; + archiver.sweep_stale_markers().await; + } + continue; + } _ = guard.await_signal() => { draining = true; } } @@ -79,7 +102,7 @@ pub fn spawn_checkpoint_task( watermark_store: &watermark_store, num_cores, timeout: budget, - cold_storage: shared.cold_storage.clone(), + archiver: archiver.as_mut(), catalog: Some(shared.credentials.catalog()), calvin_mirrors: Some(shared.authorization_fence.calvin_mirrors()), }); @@ -101,8 +124,8 @@ pub fn spawn_checkpoint_task( wal: &shared.wal, watermark_store: &watermark_store, num_cores, - timeout: config.core_timeout, - cold_storage: shared.cold_storage.clone(), + timeout: config.core_timeout(), + archiver: archiver.as_mut(), catalog: Some(shared.credentials.catalog()), calvin_mirrors: Some(shared.authorization_fence.calvin_mirrors()), }) @@ -110,3 +133,11 @@ pub fn spawn_checkpoint_task( } }) } + +/// An interval whose first tick is one `period` away, and which delays rather +/// than bursts after a slow cycle. +fn delayed_interval(period: std::time::Duration) -> tokio::time::Interval { + let mut interval = tokio::time::interval_at(tokio::time::Instant::now() + period, period); + interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + interval +} diff --git a/nodedb/src/control/clone/catalog_copy.rs b/nodedb/src/control/clone/catalog_copy.rs index 189226219..97cb96bfa 100644 --- a/nodedb/src/control/clone/catalog_copy.rs +++ b/nodedb/src/control/clone/catalog_copy.rs @@ -408,7 +408,7 @@ mod tests { /// Seed one source collection carrying a vector index, a vector model row, /// and column statistics. fn seed_source(catalog: &SystemCatalog) { - let mut coll = StoredCollection::new(TENANT, "chunks", "cloner"); + let mut coll = StoredCollection::stamped_for_test(TENANT, "chunks", "cloner"); coll.database_id = SOURCE; catalog .put_collection(SOURCE, &coll) @@ -428,6 +428,7 @@ mod tests { pq_m: 0, ivf_cells: 0, ivf_nprobe: 0, + modification_hlc: nodedb_types::Hlc::ZERO, }) .expect("seed vector index params"); @@ -796,6 +797,7 @@ mod tests { name: "db_terms".into(), terms: vec!["database".into(), "db".into()], created_at: 5, + modification_hlc: nodedb_types::Hlc::ZERO, }) .expect("seed synonym group"); catalog diff --git a/nodedb/src/control/clone/copyup.rs b/nodedb/src/control/clone/copyup.rs index ebcde51b9..6ea105f7f 100644 --- a/nodedb/src/control/clone/copyup.rs +++ b/nodedb/src/control/clone/copyup.rs @@ -6,19 +6,27 @@ //! `Shadowed` clone, this module performs the copy-up: //! //! 1. Allocate a fresh target surrogate. -//! 2. Write the source row to target with the fresh surrogate. -//! 3. Record `(target_collection, source_surrogate) → target_surrogate` -//! in the `clone_copyups` catalog table. +//! 2. Write the source row to the target shard's owner, through its +//! replicated write path, so every replica holds it. +//! 3. Record `(target_collection, source_surrogate) → target_surrogate` in +//! the `clone_copyups` table of every node, through the metadata log. //! -//! Steps 2-3 are performed inside the existing WAL group-commit boundary. - -use std::time::Duration; +//! The target row goes first. A clone read suppresses every source row that +//! has a mapping, so a mapping without its target row hides the row +//! entirely. A failure after the target write instead leaves an exact copy of +//! the source row with no mapping. A clone read also suppresses a source row +//! whose primary key the target holds, so the row still reads once, from the +//! target. A retry of the statement assigns the same target surrogate, +//! rewrites the same row, and then writes the mapping, so it converges. The +//! materializer's insert-if-absent skips the existing target row, and +//! materialization then drops the source side. use nodedb_types::{DatabaseId, Surrogate, TenantId}; -use crate::bridge::envelope::{Priority, Request, Status}; +use crate::control::catalog_entry::CatalogEntry; +use crate::control::maintenance::clone_materializer::dispatch_to_owner; +use crate::control::planner::sql_plan_convert::convert::db_qualified; use crate::control::state::SharedState; -use crate::types::{ReadConsistency, RequestId, TraceId}; use nodedb_physical::physical_plan::{DocumentOp, KvOp, PhysicalPlan}; /// Parameters for a KV copy-up operation. @@ -34,9 +42,14 @@ pub struct KvCopyUpParams<'a> { pub source_value_bytes: Vec, } -/// Perform a KV copy-up: write `source_value_bytes` into target KV storage -/// under `kv_key`, making the row available for subsequent FieldSet or Delete -/// operations in the clone. +/// Perform a KV copy-up: write `source_value_bytes` into the target shard +/// under `kv_key` on every replica, making the row available for subsequent +/// FieldSet or Delete operations in the clone. +/// +/// A KV copy-up records no mapping: a clone read hides a source key the +/// target holds, and the caller's tombstone follows. A failure before the +/// tombstone leaves a target row that already shadows the source, and a retry +/// rewrites it. pub async fn perform_kv_clone_copyup(params: KvCopyUpParams<'_>) -> crate::Result<()> { let KvCopyUpParams { state, @@ -48,15 +61,18 @@ pub async fn perform_kv_clone_copyup(params: KvCopyUpParams<'_>) -> crate::Resul } = params; let target_key = nodedb_types::CollectionKey::from_bare(target_db_id, target_collection); - - // Allocate a surrogate for the target KV row. - let surrogate = state - .surrogate_assigner - .assign(target_key, tenant_id, &kv_key) - .map_err(|e| crate::Error::Storage { - engine: "clone_kv_copyup".into(), - detail: format!("surrogate alloc failed: {e}"), - })?; + let surrogate = crate::control::server::surrogate_exchange::assign_surrogate_routed( + state, + target_key, + tenant_id, + &kv_key, + crate::types::TraceId::ZERO, + ) + .await + .map_err(|e| crate::Error::Storage { + engine: "clone_kv_copyup".into(), + detail: format!("surrogate alloc failed: {e}"), + })?; let put_plan = PhysicalPlan::Kv(KvOp::Put { collection: nodedb_types::QualifiedCollection::new(target_db_id, target_collection), @@ -68,59 +84,18 @@ pub async fn perform_kv_clone_copyup(params: KvCopyUpParams<'_>) -> crate::Resul rls_filters: Vec::new(), provenance: None, }); - - let vshard_id = target_key.vshard(); - let req_id = RequestId::new( - state - .request_id_counter - .fetch_add(1, std::sync::atomic::Ordering::Relaxed), - ); - - let deadline_secs = state.tuning.network.default_deadline_secs; - let deadline_dur = Duration::from_secs(deadline_secs); - let req = Request { - request_id: req_id, + dispatch_to_owner( + state, tenant_id, - vshard_id, - database_id: target_db_id, - plan: put_plan, - deadline: std::time::Instant::now() + deadline_dur, - priority: Priority::Normal, - trace_id: TraceId::ZERO, - consistency: ReadConsistency::Strong, - idempotency_key: None, - event_source: crate::event::EventSource::User, - user_roles: Vec::new(), - user_id: None, - statement_digest: None, - txn_id: None, - wal_lsn: None, - resolved_now_ms: None, - admission: crate::bridge::envelope::Admission::Exempt( - crate::bridge::envelope::ExemptReason::AlreadyOrdered, - ), - }; - - let mut rx = state.tracker.register(req_id); - match state.dispatcher.lock() { - Ok(mut d) => d.dispatch(req)?, - Err(p) => p.into_inner().dispatch(req)?, - }; - - let resp = tokio::time::timeout(deadline_dur, rx.recv()) - .await - .map_err(|_| crate::Error::DeadlineExceeded { request_id: req_id })? - .ok_or(crate::Error::Dispatch { - detail: "clone_kv_copyup: response channel closed".into(), - })?; - - if resp.status != Status::Ok { - return Err(crate::Error::Storage { - engine: "clone_kv_copyup".into(), - detail: format!("Data Plane returned error status {:?}", resp.status), - }); - } - + target_db_id, + &db_qualified(target_db_id, target_collection), + put_plan, + ) + .await + .map_err(|e| crate::Error::Storage { + engine: "clone_kv_copyup".into(), + detail: format!("target write failed: {e}"), + })?; Ok(()) } @@ -139,8 +114,9 @@ pub struct CopyUpParams<'a> { pub source_row_bytes: Vec, } -/// Perform a copy-up: write `source_row_bytes` into target storage with a -/// fresh surrogate and record the mapping in `clone_copyups`. +/// Perform a copy-up: write `source_row_bytes` into the target shard with a +/// fresh surrogate on every replica, then record the mapping in +/// `clone_copyups` on every node. /// /// Returns the fresh target surrogate so the caller can apply the pending /// UPDATE to it. @@ -157,60 +133,31 @@ pub async fn perform_clone_copyup(params: CopyUpParams<'_>) -> crate::Result super::identity::carry_identity(&coll, source_row_bytes, &source_doc_id), + None => source_row_bytes, }; - - // Write the source row into target storage using PointPut. let put_plan = PhysicalPlan::Document(DocumentOp::PointPut { collection: nodedb_types::QualifiedCollection::new(target_db_id, target_collection), document_id: source_doc_id.clone(), - value: source_row_bytes.clone(), + value, surrogate: target_surrogate, pk_bytes: source_doc_id.as_bytes().to_vec(), // A copy-up is internal plumbing behind the caller's own statement; it @@ -219,70 +166,54 @@ pub async fn perform_clone_copyup(params: CopyUpParams<'_>) -> crate::Result d.dispatch(req), - Err(p) => p.into_inner().dispatch(req), + // Keyed by the TARGET collection: every reader looks the mapping up + // under the clone it belongs to. + let row = CopyupRow { + database_id: target_db_id.as_u64(), + tenant_id: tenant_id.as_u64(), + collection: target_collection.to_string(), + source_surrogate: source_surrogate.as_u32(), }; - if let Err(e) = dispatch_outcome { - rollback_mapping("dispatch failed"); - return Err(e); - } + super::cow_entry::replicate_async(state, &row.put(target_surrogate)) + .await + .map_err(|e| crate::Error::Storage { + engine: "clone_copyup".into(), + detail: format!( + "mapping of document '{source_doc_id}' failed after its target write: {e}" + ), + })?; + Ok(target_surrogate) +} - let resp = match tokio::time::timeout(deadline_dur, rx.recv()).await { - Err(_) => { - rollback_mapping("deadline exceeded"); - return Err(crate::Error::DeadlineExceeded { request_id: req_id }); - } - Ok(None) => { - rollback_mapping("response channel closed"); - return Err(crate::Error::Dispatch { - detail: "clone_copyup: response channel closed".into(), - }); - } - Ok(Some(r)) => r, - }; +/// One copy-up mapping row, before it becomes a catalog entry. +struct CopyupRow { + database_id: u64, + tenant_id: u64, + collection: String, + source_surrogate: u32, +} - if resp.status != Status::Ok { - rollback_mapping("data plane returned non-Ok status"); - return Err(crate::Error::Storage { - engine: "clone_copyup".into(), - detail: format!("Data Plane returned error status {:?}", resp.status), - }); +impl CopyupRow { + fn put(self, target_surrogate: Surrogate) -> CatalogEntry { + CatalogEntry::PutCloneCopyup { + database_id: self.database_id, + tenant_id: self.tenant_id, + collection: self.collection, + source_surrogate: self.source_surrogate, + target_surrogate: target_surrogate.as_u32(), + } } - - Ok(target_surrogate) } diff --git a/nodedb/src/control/clone/cow_entry.rs b/nodedb/src/control/clone/cow_entry.rs new file mode 100644 index 000000000..29f1e8d76 --- /dev/null +++ b/nodedb/src/control/clone/cow_entry.rs @@ -0,0 +1,20 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Replicate one copy-on-write catalog row through the metadata log. +//! +//! Every node applies the entry, so a clone read on any node sees the same +//! copy-ups and tombstones, and the materializer on one node sees them all. + +use crate::control::catalog_entry::CatalogEntry; +use crate::control::metadata_proposer::propose_catalog_entry_async; +use crate::control::state::SharedState; + +/// Propose `entry` and wait until this node applied it, on any runtime +/// flavor. Inside an open transaction the entry commits with it. +pub(crate) async fn replicate_async( + state: &SharedState, + entry: &CatalogEntry, +) -> crate::Result<()> { + propose_catalog_entry_async(state, entry).await?; + Ok(()) +} diff --git a/nodedb/src/control/clone/identity.rs b/nodedb/src/control/clone/identity.rs new file mode 100644 index 000000000..6053891cc --- /dev/null +++ b/nodedb/src/control/clone/identity.rs @@ -0,0 +1,62 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A clone copy keeps its row's identity. +//! +//! A schemaless row with no `id` in its body takes its identity from its +//! surrogate. A copy into the clone target lands under a new surrogate, so +//! without `id` in the body the copy shows a new identity, and a clone +//! read cannot match it to its source row by primary key. The copy writes +//! the source identity into `id` instead. A collection that declares its +//! primary key already carries the key in the body. A strict row carries its +//! identity in `_rowid`, and a hash-chained row carries `id` from its first +//! write, so the copy leaves both unchanged. + +use nodedb_query::msgpack_scan; +use nodedb_types::DEFAULT_IDENTITY_COLUMN; + +use crate::control::security::catalog::StoredCollection; + +/// `body` as the clone target of `coll` stores it, for the row whose identity +/// is `identity`. +pub(crate) fn carry_identity(coll: &StoredCollection, body: Vec, identity: &str) -> Vec { + if coll.declared_primary_key.is_some() + || !coll.collection_type.is_schemaless() + || msgpack_scan::map_header(&body, 0).is_none() + { + return body; + } + // Leaves a body that already carries `id` unchanged. + msgpack_scan::inject_str_field(&body, DEFAULT_IDENTITY_COLUMN, identity) +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::{RowIdentity, StorageKey, Surrogate}; + + /// The copy of a minted row under a new surrogate keeps the source + /// identity. + #[test] + fn minted_row_copy_keeps_its_identity() { + let coll = StoredCollection::new(1, "docs", "admin"); + assert!(coll.collection_type.is_schemaless()); + let body = msgpack_scan::build_str_map(&[("content", "a")]); + let source_key = StorageKey::for_surrogate(Surrogate::new(123)); + let identity = RowIdentity::of_stored_row(&body, None, source_key).into_string(); + assert_eq!(identity, "123"); + + let copied = carry_identity(&coll, body, &identity); + let target_key = StorageKey::for_surrogate(Surrogate::new(456)); + assert_eq!( + RowIdentity::of_stored_row(&copied, None, target_key).as_str(), + "123" + ); + } + + #[test] + fn body_with_id_is_unchanged() { + let coll = StoredCollection::new(1, "docs", "admin"); + let body = msgpack_scan::build_str_map(&[("id", "d1"), ("content", "a")]); + assert_eq!(carry_identity(&coll, body.clone(), "d1"), body); + } +} diff --git a/nodedb/src/control/clone/lsn_resolve.rs b/nodedb/src/control/clone/lsn_resolve.rs index f533e4d33..bee2d50dd 100644 --- a/nodedb/src/control/clone/lsn_resolve.rs +++ b/nodedb/src/control/clone/lsn_resolve.rs @@ -1,21 +1,36 @@ // SPDX-License-Identifier: BUSL-1.1 -//! LSN ↔ wall-clock millisecond resolution for the clone CoW resolver. +//! Wall-clock millisecond → LSN resolution for the clone CoW resolver. //! -//! Converts a user-supplied `AS OF SYSTEM TIME ` value to the closest -//! WAL LSN using the anchor map held by `SharedState`. When the map is -//! empty (no anchors replayed yet) the WAL frontier is used as a safe -//! approximation — the same behaviour as the original clone handler. +//! Resolves a user-supplied `AS OF SYSTEM TIME ` value to the WAL state +//! committed by then, from the WAL's time anchors. LSNs here are exclusive +//! bounds, as `wal.next_lsn()` is. -use nodedb_types::Lsn; +use nodedb_types::{Lsn, LsnTimeError}; use crate::control::state::SharedState; -/// Resolve wall-clock milliseconds to the nearest LSN. +/// The exclusive LSN bound of the state committed by millisecond `wall_ms`. /// -/// Delegates to [`SharedState::ms_to_lsn`] which performs binary-search -/// interpolation across the anchor map, falling back to `wal.next_lsn()` -/// when the map is empty. -pub fn wall_ms_to_lsn(state: &SharedState, wall_ms: i64) -> Lsn { +/// Fails when `wall_ms` is before the oldest retained time anchor. See +/// [`SharedState::ms_to_lsn`]. +pub fn wall_ms_to_lsn(state: &SharedState, wall_ms: i64) -> Result { state.ms_to_lsn(wall_ms) } + +/// System time, in wall ms, at which a read through a clone cuts its source: +/// the commit time of the source state below `as_of_lsn`. +/// +/// `None` for a non-bitemporal source. It keeps only current versions, so a +/// system-time cut hides every row. A clone collection carries its +/// source's `bitemporal` flag. +pub fn source_as_of_ms( + state: &SharedState, + source_bitemporal: bool, + as_of_lsn: Lsn, +) -> Option { + if !source_bitemporal { + return None; + } + state.ms_to_lsn_inverse(as_of_lsn) +} diff --git a/nodedb/src/control/clone/materialize_freeze.rs b/nodedb/src/control/clone/materialize_freeze.rs deleted file mode 100644 index 1bf84f009..000000000 --- a/nodedb/src/control/clone/materialize_freeze.rs +++ /dev/null @@ -1,148 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Source-database freeze registry for clone materialization. -//! -//! While a clone materializer is copying rows from a source database, that -//! source must not accept new user writes. For MVCC-capable engines -//! (Document, Columnar) the `system_as_of_ms` filter at the scan layer -//! enforces the as-of semantic, but for the KV engine — which has no MVCC — -//! a concurrent write to the source would slip into the target copy. -//! -//! `MaterializeFreezeRegistry` solves this with a reference-counted freeze: -//! -//! * `freeze(db_id)` — increments the refcount for `db_id` and returns an -//! RAII [`FreezeGuard`] whose [`Drop`] decrements it. -//! * `is_frozen(db_id)` — fast read-only check used by the dispatch gate. -//! -//! Two concurrent materializers sweeping different clones of the same source -//! nest correctly: the first `freeze()` inserts with count 1; the second -//! increments to 2. The source is unfrozen only when the last guard drops -//! and the count returns to 0. -//! -//! All operations are wait-free under the read path: `is_frozen` holds the -//! read-lock only long enough to look up one entry. - -use std::collections::HashMap; -use std::sync::{Arc, RwLock}; - -use nodedb_types::id::DatabaseId; - -/// Database-level freeze registry. -/// -/// All methods are `Send + Sync` and safe for concurrent use from multiple -/// Tokio tasks. -pub struct MaterializeFreezeRegistry { - /// Maps `database_id → active freeze count`. - inner: RwLock>, -} - -impl MaterializeFreezeRegistry { - /// Create a new, empty registry. - pub fn new() -> Arc { - Arc::new(Self { - inner: RwLock::new(HashMap::new()), - }) - } - - /// Freeze `db_id` for the duration of the returned guard. - /// - /// Multiple concurrent calls with the same `db_id` are allowed; the - /// database stays frozen until every guard is dropped. - pub fn freeze(self: &Arc, db_id: DatabaseId) -> FreezeGuard { - { - let mut map = self.inner.write().unwrap_or_else(|p| p.into_inner()); - let count = map.entry(db_id).or_insert(0); - *count += 1; - } - FreezeGuard { - registry: Arc::clone(self), - db_id, - } - } - - /// Returns `true` if `db_id` currently has at least one active freeze. - /// - /// Called on the hot write path — read lock is held only for the duration - /// of the `contains_key` check. - pub fn is_frozen(&self, db_id: DatabaseId) -> bool { - let map = self.inner.read().unwrap_or_else(|p| p.into_inner()); - map.get(&db_id).copied().unwrap_or(0) > 0 - } - - /// Decrement the refcount for `db_id`, removing the entry when it hits 0. - /// - /// Called exclusively by [`FreezeGuard::drop`]. - fn release(&self, db_id: DatabaseId) { - let mut map = self.inner.write().unwrap_or_else(|p| p.into_inner()); - if let Some(count) = map.get_mut(&db_id) { - if *count <= 1 { - map.remove(&db_id); - } else { - *count -= 1; - } - } - } -} - -impl Default for MaterializeFreezeRegistry { - fn default() -> Self { - Self { - inner: RwLock::new(HashMap::new()), - } - } -} - -/// RAII guard that holds a freeze for one database. -/// -/// Dropping this guard releases the freeze (decrements the refcount). A panic -/// between `freeze()` and the natural drop point is safe: Rust's stack -/// unwinding calls `Drop`, so the registry is always cleaned up. -pub struct FreezeGuard { - registry: Arc, - db_id: DatabaseId, -} - -impl Drop for FreezeGuard { - fn drop(&mut self) { - self.registry.release(self.db_id); - } -} - -#[cfg(test)] -mod tests { - use super::*; - - fn db(id: u64) -> DatabaseId { - DatabaseId::new(id) - } - - #[test] - fn freeze_and_release_single() { - let reg = MaterializeFreezeRegistry::new(); - assert!(!reg.is_frozen(db(1))); - let guard = reg.freeze(db(1)); - assert!(reg.is_frozen(db(1))); - drop(guard); - assert!(!reg.is_frozen(db(1))); - } - - #[test] - fn nested_freeze_releases_on_last_drop() { - let reg = MaterializeFreezeRegistry::new(); - let g1 = reg.freeze(db(2)); - let g2 = reg.freeze(db(2)); - assert!(reg.is_frozen(db(2))); - drop(g1); - assert!(reg.is_frozen(db(2)), "still frozen after first drop"); - drop(g2); - assert!(!reg.is_frozen(db(2)), "unfrozen after last drop"); - } - - #[test] - fn different_databases_independent() { - let reg = MaterializeFreezeRegistry::new(); - let _g1 = reg.freeze(db(3)); - assert!(reg.is_frozen(db(3))); - assert!(!reg.is_frozen(db(4))); - } -} diff --git a/nodedb/src/control/clone/mod.rs b/nodedb/src/control/clone/mod.rs index db0e5bb1c..5defbef07 100644 --- a/nodedb/src/control/clone/mod.rs +++ b/nodedb/src/control/clone/mod.rs @@ -7,8 +7,9 @@ pub mod catalog_copy; pub mod copyup; +pub(crate) mod cow_entry; +pub(crate) mod identity; pub mod lsn_resolve; -pub mod materialize_freeze; pub mod metadata; pub mod resolver; pub mod tombstone; @@ -16,7 +17,6 @@ pub mod tombstone; pub use catalog_copy::copy_database_metadata; pub use copyup::{KvCopyUpParams, perform_clone_copyup, perform_kv_clone_copyup}; pub use lsn_resolve::wall_ms_to_lsn; -pub use materialize_freeze::{FreezeGuard, MaterializeFreezeRegistry}; pub use metadata::ClonePredicatesNote; pub use resolver::{CloneReadParams, resolve_read}; pub use tombstone::{KvTombstoneParams, perform_clone_tombstone, perform_kv_clone_tombstone}; diff --git a/nodedb/src/control/clone/resolver/filter.rs b/nodedb/src/control/clone/resolver/filter.rs index 4a3bdd225..bb5aae4db 100644 --- a/nodedb/src/control/clone/resolver/filter.rs +++ b/nodedb/src/control/clone/resolver/filter.rs @@ -5,9 +5,9 @@ /// Filter tombstoned source surrogates from response bytes. /// /// Given the raw msgpack payload from a source scan, returns a filtered -/// payload that excludes rows whose surrogates are in `tombstoned`. -/// If `tombstoned` is empty this is a no-op and returns `None` (caller -/// keeps the original bytes). +/// payload that excludes rows whose surrogates are in `tombstoned`. Returns +/// the payload unchanged when nothing is filtered, and `None` only when a +/// non-empty payload is not a msgpack array. pub fn filter_tombstoned_rows( payload: &[u8], tombstoned: &std::collections::HashSet, @@ -15,7 +15,7 @@ pub fn filter_tombstoned_rows( use nodedb_query::msgpack_scan; if tombstoned.is_empty() || payload.is_empty() { - return None; + return Some(payload.to_vec()); } let (count, mut offset) = msgpack_scan::array_header(payload, 0)?; @@ -38,7 +38,7 @@ pub fn filter_tombstoned_rows( } if kept.len() == count { - return None; // nothing filtered + return Some(payload.to_vec()); } Some(encode_msgpack_array(&kept)) diff --git a/nodedb/src/control/clone/resolver/mod.rs b/nodedb/src/control/clone/resolver/mod.rs index 53b0dc73e..dc263ca5d 100644 --- a/nodedb/src/control/clone/resolver/mod.rs +++ b/nodedb/src/control/clone/resolver/mod.rs @@ -11,6 +11,7 @@ pub mod filter; pub mod refusal; pub mod resolve; pub mod rewrite; +pub mod rewrite_engine; pub use filter::filter_tombstoned_rows; pub use refusal::SourceRewrite; diff --git a/nodedb/src/control/clone/resolver/resolve.rs b/nodedb/src/control/clone/resolver/resolve.rs index bdafd5372..0cf369073 100644 --- a/nodedb/src/control/clone/resolver/resolve.rs +++ b/nodedb/src/control/clone/resolver/resolve.rs @@ -3,7 +3,7 @@ //! `resolve_read`: walk the clone chain for ONE physical task and build its //! source-side twins. -use nodedb_types::{CloneStatus, Lsn, TenantId}; +use nodedb_types::{CloneOrigin, CloneStatus, DatabaseId, Lsn, TenantId}; use crate::control::server::shared::plan_util::extract_collection; use crate::control::state::SharedState; @@ -29,7 +29,7 @@ pub enum ResolveOutcome { /// `target_task` plus every source-side task the chain walk produced. Augmented { /// Boxed: `PhysicalTask` embeds `PhysicalPlan`, the crate's largest - /// enum, which would otherwise blow this variant's size far past + /// enum, which inline blows this variant's size far past /// `PreDatesClone`'s. target_task: Box, source_tasks: Vec, @@ -44,43 +44,27 @@ pub enum ResolveOutcome { /// /// Returns `None` when the collection has no clone origin (fast path: zero /// overhead). Returns `Some(ResolveOutcome)` when resolution is required. -pub fn resolve_read( +pub async fn resolve_read( state: &SharedState, task: PhysicalTask, tenant_id: TenantId, params: &CloneReadParams, ) -> crate::Result> { let db_id = task.database_id; - let catalog = state.credentials.catalog(); // The shared extractor sees through the `Exchange` / `PostProcess` // wrappers the converter puts over every sharded read; a clone-local - // copy would drift out of sync and misread those as "not a clone". + // copy drifts out of sync and misreads those as "not a clone". let Some(raw_coll) = extract_collection(&task.plan) else { return Ok(None); }; // Strip the database prefix that db_qualified() prepends, e.g. "1/users" → "users". let coll_name = super::rewrite::strip_db_prefix(db_id, raw_coll); - // Look up the stored collection descriptor. - let Some(desc) = catalog - .get_collection(db_id, tenant_id.as_u64(), coll_name) - .map_err(|e| crate::Error::Storage { - engine: "catalog".into(), - detail: format!("clone resolver: get_collection failed: {e}"), - })? - else { - return Ok(None); - }; - // Short-circuit: not a clone or fully materialized. - let Some(ref origin) = desc.cloned_from else { + let Some(live) = live_clone(state, db_id, tenant_id, coll_name, "get_collection")? else { return Ok(None); }; - match desc.clone_status { - CloneStatus::Materialized => return Ok(None), - CloneStatus::Shadowed | CloneStatus::Materializing { .. } => {} - } // UNION DISTINCT/INTERSECT/EXCEPT dedup/subtract this task's response // against siblings' by exact row match — unsound without proven parity. @@ -94,127 +78,171 @@ pub fn resolve_read( } // Bitemporal correctness: check if T_lsn < clone_created_at. - if params.query_lsn < origin.clone_created_at { + if params.query_lsn < live.origin.clone_created_at { return Ok(Some(ResolveOutcome::PreDatesClone( - ClonePredicatesNote::new(params.query_lsn, origin.clone_created_at), + ClonePredicatesNote::new(params.query_lsn, live.origin.clone_created_at), ))); } - // Compute effective source LSN: min(T_lsn, as_of_lsn). - let effective_source_lsn = if params.query_lsn > origin.as_of_lsn { - origin.as_of_lsn - } else { - params.query_lsn + let first = ChainLevel { + db_id, + coll_name: coll_name.to_string(), + live, }; + let source_tasks = walk_clone_chain(state, &task, tenant_id, params.query_lsn, first).await?; + let target_collection_key = + crate::control::planner::sql_plan_convert::convert::db_qualified(db_id, coll_name); - // Convert effective_source_lsn to wall-ms for the engine. - let effective_source_ms = state.ms_to_lsn_inverse(effective_source_lsn); + Ok(Some(ResolveOutcome::Augmented { + target_task: Box::new(task), + source_tasks, + target_collection_key, + note: None, + })) +} - // Walk source-side tasks up the clone chain until `cloned_from = None` - // or `Materialized`. `MAX_CLONE_DEPTH` bounds the chain at create time; - // the loop still caps at 8 as a guard against catalog corruption. - let mut source_tasks: Vec = Vec::new(); +/// A clone collection that is not yet materialized. +struct LiveClone { + origin: CloneOrigin, + bitemporal: bool, +} - // Current "target" level for this iteration. - let mut cur_db_id = db_id; - let mut cur_coll_name_owned = coll_name.to_string(); - let mut cur_origin = origin.clone(); - let mut cur_effective_ms = effective_source_ms; +/// One level of the clone chain: the clone a rewrite reads through. +struct ChainLevel { + db_id: DatabaseId, + coll_name: String, + live: LiveClone, +} - // Template for the next rewrite; after each level, updated to the task - // just pushed so the next iteration rewrites the correct per-level - // qualified name rather than the original target task. - let mut prev_level_tasks: Vec = vec![task.clone()]; +/// The collection `name` as a clone that is not yet materialized. `None` +/// for a missing collection, a collection that is no clone, and a +/// materialized clone. `lookup` names the lookup in a catalog error. +fn live_clone( + state: &SharedState, + db_id: DatabaseId, + tenant_id: TenantId, + name: &str, + lookup: &str, +) -> crate::Result> { + let desc = state + .credentials + .catalog() + .get_collection(db_id, tenant_id.as_u64(), name) + .map_err(|e| crate::Error::Storage { + engine: "catalog".into(), + detail: format!("clone resolver: {lookup} failed: {e}"), + })?; + let Some(desc) = desc else { + return Ok(None); + }; + // A materialized clone holds all its data itself. + match desc.clone_status { + CloneStatus::Materialized => return Ok(None), + CloneStatus::Shadowed | CloneStatus::Materializing { .. } => {} + } + let bitemporal = desc.bitemporal; + Ok(desc + .cloned_from + .map(|origin| LiveClone { origin, bitemporal })) +} +/// Walk source-side tasks up the clone chain until `cloned_from = None` or +/// `Materialized`. `MAX_CLONE_DEPTH` bounds the chain at create time; the +/// walk still caps at 8 levels as a guard against catalog corruption. +async fn walk_clone_chain( + state: &SharedState, + task: &PhysicalTask, + tenant_id: TenantId, + query_lsn: Lsn, + first: ChainLevel, +) -> crate::Result> { const MAX_WALK: u32 = 8; - let mut depth = 0u32; - - loop { - if depth >= MAX_WALK { - break; - } - depth += 1; - - let src_db_id = cur_origin.source_database; - let src_coll_name = cur_origin.source_collection.as_str(); - let cur_coll_str = cur_coll_name_owned.as_str(); - - let mut this_level_tasks: Vec = Vec::new(); - - for level_task in &prev_level_tasks { - // An unsupported read shape over the cloned collection returns an - // error here, not `NoSourceTask` — it propagates to the client - // instead of quietly producing a target-only answer. - let rewritten = rewrite_plan_for_source(super::rewrite::RewriteForSourceParams { - plan: &level_task.plan, - target_db_id: cur_db_id, - source_db_id: src_db_id, - tenant_id, - target_coll: cur_coll_str, - source_coll: src_coll_name, - effective_source_ms: cur_effective_ms, - kv_surrogate_ceiling: cur_origin.kv_surrogate_ceiling, - state, - })?; - let SourceRewrite::Task(source_plan) = rewritten else { - continue; - }; - let source_vshard = - nodedb_types::CollectionKey::from_bare(src_db_id, src_coll_name).vshard(); - this_level_tasks.push(PhysicalTask { - tenant_id, - vshard_id: source_vshard, - database_id: src_db_id, - plan: *source_plan, - post_set_op: PostSetOp::None, - txn_id: None, - }); - } - + let mut source_tasks: Vec = Vec::new(); + // Template for the next rewrite. After each level it holds the tasks + // pushed, so the next level rewrites the correct per-level qualified name + // rather than the original target task. + let mut prev_level_tasks: Vec = vec![task.clone()]; + let mut level = first; + for _ in 0..MAX_WALK { + let this_level_tasks = + rewrite_level(state, &prev_level_tasks, &level, tenant_id, query_lsn).await?; source_tasks.extend(this_level_tasks.iter().cloned()); prev_level_tasks = this_level_tasks; - // Check whether `src_db_id / src_coll_name` is itself a clone so we - // can continue the walk. - let ancestor_desc = catalog - .get_collection(src_db_id, tenant_id.as_u64(), src_coll_name) - .map_err(|e| crate::Error::Storage { - engine: "catalog".into(), - detail: format!("clone resolver: ancestor get_collection failed: {e}"), - })?; - - let Some(ancestor) = ancestor_desc else { break }; - - // Materialized ancestor — data is fully self-contained; stop. - match ancestor.clone_status { - CloneStatus::Materialized => break, - CloneStatus::Shadowed | CloneStatus::Materializing { .. } => {} - } - - let Some(ancestor_origin) = ancestor.cloned_from else { + // The walk continues while the source is itself a live clone. + let src_db_id = level.live.origin.source_database; + let src_coll_name = level.live.origin.source_collection.clone(); + let Some(ancestor) = live_clone( + state, + src_db_id, + tenant_id, + &src_coll_name, + "ancestor get_collection", + )? + else { break; }; - - // Compute effective LSN for this ancestor level. - let ancestor_effective_lsn = if params.query_lsn > ancestor_origin.as_of_lsn { - ancestor_origin.as_of_lsn - } else { - params.query_lsn + level = ChainLevel { + db_id: src_db_id, + coll_name: src_coll_name, + live: ancestor, }; - cur_effective_ms = state.ms_to_lsn_inverse(ancestor_effective_lsn); - - cur_db_id = src_db_id; - cur_coll_name_owned = src_coll_name.to_string(); - cur_origin = ancestor_origin; } + Ok(source_tasks) +} - let target_collection_key = - crate::control::planner::sql_plan_convert::convert::db_qualified(db_id, coll_name); - - Ok(Some(ResolveOutcome::Augmented { - target_task: Box::new(task), - source_tasks, - target_collection_key, - note: None, - })) +/// Rewrite each task of the previous level into a task on this level's +/// source collection. +async fn rewrite_level( + state: &SharedState, + prev_level_tasks: &[PhysicalTask], + level: &ChainLevel, + tenant_id: TenantId, + query_lsn: Lsn, +) -> crate::Result> { + let origin = &level.live.origin; + let src_db_id = origin.source_database; + let src_coll_name = origin.source_collection.as_str(); + // Effective source LSN: min(T_lsn, as_of_lsn), as wall-ms for the engine. + let effective_source_lsn = if query_lsn > origin.as_of_lsn { + origin.as_of_lsn + } else { + query_lsn + }; + let effective_source_ms = crate::control::clone::lsn_resolve::source_as_of_ms( + state, + level.live.bitemporal, + effective_source_lsn, + ); + + let mut tasks: Vec = Vec::new(); + for level_task in prev_level_tasks { + // An unsupported read shape over the cloned collection returns an + // error here, not `NoSourceTask` — it propagates to the client + // instead of quietly producing a target-only answer. + let rewritten = rewrite_plan_for_source(super::rewrite::RewriteForSourceParams { + plan: &level_task.plan, + target_db_id: level.db_id, + source_db_id: src_db_id, + tenant_id, + target_coll: &level.coll_name, + source_coll: src_coll_name, + effective_source_ms, + kv_surrogate_ceiling: origin.kv_surrogate_ceiling, + state, + }) + .await?; + let SourceRewrite::Task(source_plan) = rewritten else { + continue; + }; + tasks.push(PhysicalTask { + tenant_id, + vshard_id: nodedb_types::CollectionKey::from_bare(src_db_id, src_coll_name).vshard(), + database_id: src_db_id, + plan: *source_plan, + post_set_op: PostSetOp::None, + txn_id: None, + }); + } + Ok(tasks) } diff --git a/nodedb/src/control/clone/resolver/rewrite.rs b/nodedb/src/control/clone/resolver/rewrite.rs index 779e12498..80eaff0c0 100644 --- a/nodedb/src/control/clone/resolver/rewrite.rs +++ b/nodedb/src/control/clone/resolver/rewrite.rs @@ -4,15 +4,15 @@ //! read at the effective source LSN. use nodedb_types::DatabaseId; +use nodedb_types::QualifiedCollection; use nodedb_types::TenantId; use crate::control::state::SharedState; -use nodedb_physical::physical_plan::{ - ColumnarOp, DocumentOp, ExchangeOp, KvOp, PhysicalPlan, QueryOp, SetOpKind, TimeseriesOp, -}; +use nodedb_physical::physical_plan::{ExchangeOp, PhysicalPlan, QueryOp, SetOpKind}; use nodedb_types::SystemTimeScope; use super::refusal::{SourceRewrite, plan_reads_cloned_collection, refuse_clone_read_shape}; +use super::rewrite_engine::{rewrite_columnar, rewrite_document, rewrite_kv, rewrite_timeseries}; /// Compute the source-side system-time selection for a clone scan rewrite. /// @@ -70,7 +70,9 @@ pub struct RewriteForSourceParams<'a> { /// /// A read that names the cloned collection but has no rewrite is refused with a /// typed error; see [`SourceRewrite`]. -pub fn rewrite_plan_for_source(params: RewriteForSourceParams<'_>) -> crate::Result { +pub async fn rewrite_plan_for_source( + params: RewriteForSourceParams<'_>, +) -> crate::Result { let RewriteForSourceParams { plan, target_db_id, @@ -82,38 +84,116 @@ pub fn rewrite_plan_for_source(params: RewriteForSourceParams<'_>) -> crate::Res kv_surrogate_ceiling, state, } = params; - let target_qualified = nodedb_types::QualifiedCollection::new(target_db_id, target_coll); - let source_qualified = nodedb_types::QualifiedCollection::new(source_db_id, source_coll); + let ctx = RewriteCtx { + source_db_id, + tenant_id, + target_coll, + source_coll, + target_qualified: QualifiedCollection::new(target_db_id, target_coll), + source_qualified: QualifiedCollection::new(source_db_id, source_coll), + effective_source_ms, + kv_surrogate_ceiling, + state, + }; + rewrite(plan, &ctx).await +} + +/// The inputs every rewrite stage shares. +pub(super) struct RewriteCtx<'a> { + pub source_db_id: DatabaseId, + pub tenant_id: TenantId, + pub target_coll: &'a str, + pub source_coll: &'a str, + pub target_qualified: QualifiedCollection, + pub source_qualified: QualifiedCollection, + pub effective_source_ms: Option, + pub kv_surrogate_ceiling: Option, + pub state: &'a SharedState, +} + +impl RewriteCtx<'_> { + /// Whether `collection` is the cloned collection. + pub(super) fn is_target(&self, collection: &QualifiedCollection) -> bool { + collection == &self.target_qualified + } + + /// The source-side system-time selection for a plan's own selection. + pub(super) fn system_time( + &self, + plan_scope: SystemTimeScope, + ) -> crate::Result { + rewrite_system_time(self.effective_source_ms, plan_scope) + } +} +/// Rewrite one plan node through the stage for its plan family. +async fn rewrite(plan: &PhysicalPlan, ctx: &RewriteCtx<'_>) -> crate::Result { match plan { - // Structural wrappers, matched before the engine arms: the converter - // wraps every sharded read in `Exchange{Gather}` / `PostProcess`. - // Recursing and re-wrapping with the same mode makes the source-side - // task fan and gather exactly like the target-side one. - PhysicalPlan::Query(QueryOp::Exchange(op)) => { - let rewritten = rewrite_plan_for_source(RewriteForSourceParams { - plan: &op.child, - target_db_id, - source_db_id, - tenant_id, - target_coll, - source_coll, - effective_source_ms, - kv_surrogate_ceiling, - state, - })?; - Ok(match rewritten { - SourceRewrite::Task(child) => { - SourceRewrite::task(PhysicalPlan::Query(QueryOp::Exchange(ExchangeOp { - child, - mode: op.mode.clone(), - }))) - } - SourceRewrite::NoSourceTask => SourceRewrite::NoSourceTask, - }) - } + PhysicalPlan::Query(op) => rewrite_query(plan, op, ctx).await, + PhysicalPlan::Document(op) => rewrite_document(plan, op, ctx).await, + PhysicalPlan::Kv(op) => rewrite_kv(plan, op, ctx), + PhysicalPlan::Columnar(op) => rewrite_columnar(plan, op, ctx), + PhysicalPlan::Timeseries(op) => rewrite_timeseries(plan, op, ctx), + // No proven rewrite for these families. Every top-level variant is + // enumerated so a new engine forces a decision here. + PhysicalPlan::Vector(_) + | PhysicalPlan::Graph(_) + | PhysicalPlan::Text(_) + | PhysicalPlan::Spatial(_) + | PhysicalPlan::Crdt(_) + | PhysicalPlan::Meta(_) + | PhysicalPlan::Array(_) + | PhysicalPlan::ClusterArray(_) + | PhysicalPlan::ClusterEvent(_) => refuse_or_skip(plan, ctx), + } +} + +/// The default for a plan with no rewrite. A read that names the cloned +/// collection (aggregates, joins, vector/text/graph/spatial searches) is +/// refused. Everything else passes through untouched. +pub(super) fn refuse_or_skip( + plan: &PhysicalPlan, + ctx: &RewriteCtx<'_>, +) -> crate::Result { + if plan_reads_cloned_collection(plan, ctx.target_qualified.as_str()) { + return Err(refuse_clone_read_shape(plan, ctx.target_coll)); + } + Ok(SourceRewrite::NoSourceTask) +} + +/// Put a rewritten child back under the wrapper node it came from. +fn rewrap( + rewritten: SourceRewrite, + wrap: impl FnOnce(Box) -> PhysicalPlan, +) -> SourceRewrite { + match rewritten { + SourceRewrite::Task(child) => SourceRewrite::task(wrap(child)), + SourceRewrite::NoSourceTask => SourceRewrite::NoSourceTask, + } +} - PhysicalPlan::Query(QueryOp::PostProcess { +/// Query plans. The structural wrappers recurse into their inputs. Every +/// other query op takes the default. +async fn rewrite_query( + plan: &PhysicalPlan, + op: &QueryOp, + ctx: &RewriteCtx<'_>, +) -> crate::Result { + match op { + // The converter wraps every sharded read in `Exchange{Gather}` / + // `PostProcess`. Recursing and re-wrapping with the same mode makes + // the source-side task fan and gather exactly like the target-side + // one. + QueryOp::Exchange(exchange) => { + let rewritten = Box::pin(rewrite(&exchange.child, ctx)).await?; + Ok(rewrap(rewritten, |child| { + PhysicalPlan::Query(QueryOp::Exchange(ExchangeOp { + child, + mode: exchange.mode.clone(), + })) + })) + } + QueryOp::PostProcess { input, filters, projection, @@ -123,325 +203,66 @@ pub fn rewrite_plan_for_source(params: RewriteForSourceParams<'_>) -> crate::Res limit, offset, distinct, - }) => { - let rewritten = rewrite_plan_for_source(RewriteForSourceParams { - plan: input, - target_db_id, - source_db_id, - tenant_id, - target_coll, - source_coll, - effective_source_ms, - kv_surrogate_ceiling, - state, - })?; - Ok(match rewritten { - SourceRewrite::Task(child) => { - SourceRewrite::task(PhysicalPlan::Query(QueryOp::PostProcess { - input: child, - filters: filters.clone(), - projection: projection.clone(), - computed_columns: computed_columns.clone(), - window_functions: window_functions.clone(), - sort_keys: sort_keys.clone(), - limit: *limit, - offset: *offset, - distinct: *distinct, - })) - } - SourceRewrite::NoSourceTask => SourceRewrite::NoSourceTask, - }) - } - - // SetOp: rewrite every branch. Branches that do not read the cloned - // collection yield no source task and are dropped, so the source-side - // node carries only the rows the target side is missing. That is - // sound for `UNION ALL` (the merge appends). Any other kind dedups or - // subtracts by exact row match against the target rows, which is - // unsound across an unmaterialized clone; refuse it the same way the - // task-level `post_set_op` refusal does. - PhysicalPlan::Query(QueryOp::SetOp { inputs, op }) => { - let mut rewritten_inputs = Vec::with_capacity(inputs.len()); - for input in inputs { - let rewritten = rewrite_plan_for_source(RewriteForSourceParams { - plan: input, - target_db_id, - source_db_id, - tenant_id, - target_coll, - source_coll, - effective_source_ms, - kv_surrogate_ceiling, - state, - })?; - if let SourceRewrite::Task(child) = rewritten { - rewritten_inputs.push(*child); - } - } - if rewritten_inputs.is_empty() { - return Ok(SourceRewrite::NoSourceTask); - } - match op { - SetOpKind::UnionAll => { - Ok(SourceRewrite::task(PhysicalPlan::Query(QueryOp::SetOp { - inputs: rewritten_inputs, - op: SetOpKind::UnionAll, - }))) - } - SetOpKind::UnionDistinct - | SetOpKind::Intersect - | SetOpKind::IntersectAll - | SetOpKind::Except - | SetOpKind::ExceptAll => Err(crate::Error::PlanError { - detail: format!( - "a set operation over '{target_coll}' cannot be read through an \ - unmaterialized clone; run ALTER DATABASE MATERIALIZE first" - ), - }), - } - } - - PhysicalPlan::Document(DocumentOp::Scan { - collection, - limit, - offset, - sort_keys, - filters, - distinct, - projection, - computed_columns, - window_functions, - system_time, - valid_at_ms, - prefilter, - }) if collection == &target_qualified => { - let system_time = rewrite_system_time(effective_source_ms, *system_time)?; - Ok(SourceRewrite::task(PhysicalPlan::Document( - DocumentOp::Scan { - collection: source_qualified, - limit: *limit, - offset: *offset, - sort_keys: sort_keys.clone(), + } => { + let rewritten = Box::pin(rewrite(input, ctx)).await?; + Ok(rewrap(rewritten, |child| { + PhysicalPlan::Query(QueryOp::PostProcess { + input: child, filters: filters.clone(), - distinct: *distinct, projection: projection.clone(), computed_columns: computed_columns.clone(), window_functions: window_functions.clone(), - system_time, - valid_at_ms: *valid_at_ms, - prefilter: prefilter.clone(), - }, - ))) - } - - PhysicalPlan::Document(DocumentOp::PointGet { - collection, - document_id, - surrogate: _, - pk_bytes, - rls_filters, - system_time, - valid_at_ms, - }) if collection == &target_qualified => { - // The target surrogate is invalid in the source database — each - // maintains its own pk→surrogate mapping. No binding means the - // row never existed there; skip rather than use a sentinel. - // Lookup errors are also treated as "skip" (visible in the - // assigner's own metrics/logs instead). - let system_time = rewrite_system_time(effective_source_ms, *system_time)?; - let Some(source_surrogate) = state - .surrogate_assigner - .lookup( - nodedb_types::CollectionKey::from_bare(source_db_id, source_coll), - tenant_id, - pk_bytes, - ) - .ok() - .flatten() - else { - return Ok(SourceRewrite::NoSourceTask); - }; - Ok(SourceRewrite::task(PhysicalPlan::Document( - DocumentOp::PointGet { - collection: source_qualified, - document_id: document_id.clone(), - surrogate: source_surrogate, - pk_bytes: pk_bytes.clone(), - rls_filters: rls_filters.clone(), - system_time, - valid_at_ms: *valid_at_ms, - }, - ))) - } - - PhysicalPlan::Document(DocumentOp::IndexedFetch { - collection, - path, - value, - filters, - projection, - limit, - offset, - }) if collection == &target_qualified => Ok(SourceRewrite::task(PhysicalPlan::Document( - DocumentOp::IndexedFetch { - collection: source_qualified, - path: path.clone(), - value: value.clone(), - filters: filters.clone(), - projection: projection.clone(), - limit: *limit, - offset: *offset, - }, - ))), - - PhysicalPlan::Kv(KvOp::Scan { - collection, - cursor, - count, - filters, - projection, - computed_columns, - match_pattern, - sort_keys, - // The original target-side scan never carries a ceiling - // (clones-of-clones still funnel through here per-level); - // the resolver overrides it for source delegation below. - surrogate_ceiling: _, - }) if collection == &target_qualified => { - Ok(SourceRewrite::task(PhysicalPlan::Kv(KvOp::Scan { - collection: source_qualified, - cursor: cursor.clone(), - count: *count, - filters: filters.clone(), - projection: projection.clone(), - computed_columns: computed_columns.clone(), - match_pattern: match_pattern.clone(), - sort_keys: sort_keys.clone(), - surrogate_ceiling: kv_surrogate_ceiling, - }))) - } - - PhysicalPlan::Kv(KvOp::Get { - collection, - key, - rls_filters, - surrogate_ceiling: _, - }) if collection == &target_qualified => { - Ok(SourceRewrite::task(PhysicalPlan::Kv(KvOp::Get { - collection: source_qualified, - key: key.clone(), - rls_filters: rls_filters.clone(), - surrogate_ceiling: kv_surrogate_ceiling, - }))) - } - - PhysicalPlan::Columnar(ColumnarOp::Scan { - collection, - projection, - limit, - filters, - rls_filters, - sort_keys, - system_time, - valid_at_ms, - prefilter, - computed_columns, - }) if collection == &target_qualified => { - let system_time = rewrite_system_time(effective_source_ms, *system_time)?; - Ok(SourceRewrite::task(PhysicalPlan::Columnar( - ColumnarOp::Scan { - collection: source_qualified, - projection: projection.clone(), - limit: *limit, - filters: filters.clone(), - rls_filters: rls_filters.clone(), sort_keys: sort_keys.clone(), - system_time, - valid_at_ms: *valid_at_ms, - prefilter: prefilter.clone(), - computed_columns: computed_columns.clone(), - }, - ))) - } - - // A bucketing or aggregating timeseries scan is the same unsound - // concatenation as `Query::Aggregate`: target and source payloads are - // appended, so a bucket present on both sides comes back twice and - // every sum/avg over the union is wrong. Only a plain scan reads - // through; the aggregating form is refused. - PhysicalPlan::Timeseries(TimeseriesOp::Scan { - collection, - bucket_interval_ms, - group_by, - aggregates, - .. - }) if collection == &target_qualified - && (!group_by.is_empty() || !aggregates.is_empty() || *bucket_interval_ms != 0) => - { - Err(refuse_clone_read_shape(plan, target_coll)) - } - - PhysicalPlan::Timeseries(TimeseriesOp::Scan { - collection, - time_range, - projection, - limit, - filters, - sort_keys, - bucket_interval_ms, - group_by, - aggregates, - gap_fill, - computed_columns, - rls_filters, - system_time, - valid_at_ms, - }) if collection == &target_qualified => { - let system_time = rewrite_system_time(effective_source_ms, *system_time)?; - Ok(SourceRewrite::task(PhysicalPlan::Timeseries( - TimeseriesOp::Scan { - collection: source_qualified, - time_range: *time_range, - projection: projection.clone(), limit: *limit, - filters: filters.clone(), - sort_keys: sort_keys.clone(), - bucket_interval_ms: *bucket_interval_ms, - group_by: group_by.clone(), - aggregates: aggregates.clone(), - gap_fill: gap_fill.clone(), - computed_columns: computed_columns.clone(), - rls_filters: rls_filters.clone(), - system_time, - valid_at_ms: *valid_at_ms, - }, - ))) + offset: *offset, + distinct: *distinct, + }) + })) } + QueryOp::SetOp { inputs, op } => rewrite_set_op(inputs, op, ctx).await, + _ => refuse_or_skip(plan, ctx), + } +} - // DEFAULT: refuse any READ naming the cloned collection (aggregates, - // joins, vector/text/graph/spatial searches — no proven rewrite); - // allow everything else through untouched. Every top-level variant - // is enumerated so a new engine forces a decision here. - PhysicalPlan::Document(_) - | PhysicalPlan::Kv(_) - | PhysicalPlan::Vector(_) - | PhysicalPlan::Graph(_) - | PhysicalPlan::Text(_) - | PhysicalPlan::Columnar(_) - | PhysicalPlan::Timeseries(_) - | PhysicalPlan::Spatial(_) - | PhysicalPlan::Crdt(_) - | PhysicalPlan::Query(_) - | PhysicalPlan::Meta(_) - | PhysicalPlan::Array(_) - | PhysicalPlan::ClusterArray(_) - | PhysicalPlan::ClusterEvent(_) => { - if plan_reads_cloned_collection(plan, target_qualified.as_str()) { - return Err(refuse_clone_read_shape(plan, target_coll)); - } - Ok(SourceRewrite::NoSourceTask) +/// Rewrite every branch of a set operation. +/// +/// Branches that do not read the cloned collection yield no source task and +/// are dropped, so the source-side node carries only the rows the target side +/// is missing. That is sound for `UNION ALL` (the merge appends). Any other +/// kind dedups or subtracts by exact row match against the target rows, which +/// is unsound across an unmaterialized clone. It is refused the same way the +/// task-level `post_set_op` refusal does. +async fn rewrite_set_op( + inputs: &[PhysicalPlan], + op: &SetOpKind, + ctx: &RewriteCtx<'_>, +) -> crate::Result { + let mut rewritten_inputs = Vec::with_capacity(inputs.len()); + for input in inputs { + if let SourceRewrite::Task(child) = Box::pin(rewrite(input, ctx)).await? { + rewritten_inputs.push(*child); } } + if rewritten_inputs.is_empty() { + return Ok(SourceRewrite::NoSourceTask); + } + match op { + SetOpKind::UnionAll => Ok(SourceRewrite::task(PhysicalPlan::Query(QueryOp::SetOp { + inputs: rewritten_inputs, + op: SetOpKind::UnionAll, + }))), + SetOpKind::UnionDistinct + | SetOpKind::Intersect + | SetOpKind::IntersectAll + | SetOpKind::Except + | SetOpKind::ExceptAll => Err(crate::Error::PlanError { + detail: format!( + "a set operation over '{}' cannot be read through an \ + unmaterialized clone; run ALTER DATABASE MATERIALIZE first", + ctx.target_coll + ), + }), + } } /// Strip the `"/"` prefix added by `db_qualified()`, returning the @@ -627,6 +448,7 @@ mod tests { options: GraphTraversalOptions::default(), bm25_query: None, bm25_field: None, + stage: nodedb_physical::physical_plan::RagStage::Local, }), ), ] @@ -659,8 +481,8 @@ mod tests { /// The default arm refuses a plan when it is classified `Read` AND its /// collection is extractable. Both inputs must hold for every sharded - /// source, or an unrewritable read would fall through to `NoSourceTask` and - /// answer from the target alone. + /// source, or an unrewritable read falls through to `NoSourceTask` and + /// answers from the target alone. #[test] fn every_sharded_source_is_a_classified_read() { for (name, plan) in sharded_source_plans() { diff --git a/nodedb/src/control/clone/resolver/rewrite_engine.rs b/nodedb/src/control/clone/resolver/rewrite_engine.rs new file mode 100644 index 000000000..9e7935cc4 --- /dev/null +++ b/nodedb/src/control/clone/resolver/rewrite_engine.rs @@ -0,0 +1,264 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-engine rewrite stages: a read of the cloned collection becomes the +//! same read of the source collection. + +use nodedb_physical::physical_plan::{ColumnarOp, DocumentOp, KvOp, PhysicalPlan, TimeseriesOp}; + +use super::refusal::{SourceRewrite, refuse_clone_read_shape}; +use super::rewrite::{RewriteCtx, refuse_or_skip}; + +/// Document plans. A scan, point get, or indexed fetch of the cloned +/// collection reads the source collection instead. +pub(super) async fn rewrite_document( + plan: &PhysicalPlan, + op: &DocumentOp, + ctx: &RewriteCtx<'_>, +) -> crate::Result { + match op { + DocumentOp::Scan { + collection, + limit, + offset, + sort_keys, + filters, + distinct, + projection, + computed_columns, + window_functions, + system_time, + valid_at_ms, + prefilter, + } if ctx.is_target(collection) => Ok(SourceRewrite::task(PhysicalPlan::Document( + DocumentOp::Scan { + collection: ctx.source_qualified.clone(), + limit: *limit, + offset: *offset, + sort_keys: sort_keys.clone(), + filters: filters.clone(), + distinct: *distinct, + projection: projection.clone(), + computed_columns: computed_columns.clone(), + window_functions: window_functions.clone(), + system_time: ctx.system_time(*system_time)?, + valid_at_ms: *valid_at_ms, + prefilter: prefilter.clone(), + }, + ))), + + DocumentOp::PointGet { + collection, + document_id, + surrogate: _, + pk_bytes, + rls_filters, + system_time, + valid_at_ms, + } if ctx.is_target(collection) => { + // The target surrogate is invalid in the source database — each + // maintains its own pk→surrogate mapping. The source collection's + // home answers the binding through the async routed exchange. No + // binding means the row never existed there; skip rather than use + // a sentinel. Lookup errors are also treated as "skip" (visible in + // the exchange's own logs instead). + let system_time = ctx.system_time(*system_time)?; + let Some(source_surrogate) = + crate::control::server::surrogate_exchange::lookup_surrogate_routed( + ctx.state, + nodedb_types::CollectionKey::from_bare(ctx.source_db_id, ctx.source_coll), + ctx.tenant_id, + pk_bytes, + crate::types::TraceId::ZERO, + ) + .await + .ok() + .flatten() + else { + return Ok(SourceRewrite::NoSourceTask); + }; + Ok(SourceRewrite::task(PhysicalPlan::Document( + DocumentOp::PointGet { + collection: ctx.source_qualified.clone(), + document_id: document_id.clone(), + surrogate: Some(source_surrogate), + pk_bytes: pk_bytes.clone(), + rls_filters: rls_filters.clone(), + system_time, + valid_at_ms: *valid_at_ms, + }, + ))) + } + + DocumentOp::IndexedFetch { + collection, + path, + value, + filters, + projection, + limit, + offset, + } if ctx.is_target(collection) => Ok(SourceRewrite::task(PhysicalPlan::Document( + DocumentOp::IndexedFetch { + collection: ctx.source_qualified.clone(), + path: path.clone(), + value: value.clone(), + filters: filters.clone(), + projection: projection.clone(), + limit: *limit, + offset: *offset, + }, + ))), + + _ => refuse_or_skip(plan, ctx), + } +} + +/// KV plans. A scan or get of the cloned collection reads the source +/// collection under the clone's surrogate ceiling. +pub(super) fn rewrite_kv( + plan: &PhysicalPlan, + op: &KvOp, + ctx: &RewriteCtx<'_>, +) -> crate::Result { + match op { + KvOp::Scan { + collection, + cursor, + count, + filters, + projection, + computed_columns, + match_pattern, + sort_keys, + // The original target-side scan never carries a ceiling + // (clones-of-clones still funnel through here per-level); + // the resolver overrides it for source delegation below. + surrogate_ceiling: _, + } if ctx.is_target(collection) => Ok(SourceRewrite::task(PhysicalPlan::Kv(KvOp::Scan { + collection: ctx.source_qualified.clone(), + cursor: cursor.clone(), + count: *count, + filters: filters.clone(), + projection: projection.clone(), + computed_columns: computed_columns.clone(), + match_pattern: match_pattern.clone(), + sort_keys: sort_keys.clone(), + surrogate_ceiling: ctx.kv_surrogate_ceiling, + }))), + + KvOp::Get { + collection, + key, + rls_filters, + surrogate_ceiling: _, + } if ctx.is_target(collection) => Ok(SourceRewrite::task(PhysicalPlan::Kv(KvOp::Get { + collection: ctx.source_qualified.clone(), + key: key.clone(), + rls_filters: rls_filters.clone(), + surrogate_ceiling: ctx.kv_surrogate_ceiling, + }))), + + _ => refuse_or_skip(plan, ctx), + } +} + +/// Columnar plans. A scan of the cloned collection reads the source +/// collection instead. +pub(super) fn rewrite_columnar( + plan: &PhysicalPlan, + op: &ColumnarOp, + ctx: &RewriteCtx<'_>, +) -> crate::Result { + match op { + ColumnarOp::Scan { + collection, + projection, + limit, + filters, + rls_filters, + sort_keys, + system_time, + valid_at_ms, + prefilter, + computed_columns, + } if ctx.is_target(collection) => Ok(SourceRewrite::task(PhysicalPlan::Columnar( + ColumnarOp::Scan { + collection: ctx.source_qualified.clone(), + projection: projection.clone(), + limit: *limit, + filters: filters.clone(), + rls_filters: rls_filters.clone(), + sort_keys: sort_keys.clone(), + system_time: ctx.system_time(*system_time)?, + valid_at_ms: *valid_at_ms, + prefilter: prefilter.clone(), + computed_columns: computed_columns.clone(), + }, + ))), + + _ => refuse_or_skip(plan, ctx), + } +} + +/// Timeseries plans. A plain scan of the cloned collection reads the source +/// collection instead. +/// +/// A bucketing or aggregating scan is the same unsound concatenation as +/// `Query::Aggregate`: target and source payloads are appended, so a bucket +/// present on both sides comes back twice and every sum/avg over the union is +/// wrong. The aggregating form is refused. +pub(super) fn rewrite_timeseries( + plan: &PhysicalPlan, + op: &TimeseriesOp, + ctx: &RewriteCtx<'_>, +) -> crate::Result { + match op { + TimeseriesOp::Scan { + collection, + bucket_interval_ms, + group_by, + aggregates, + .. + } if ctx.is_target(collection) + && (!group_by.is_empty() || !aggregates.is_empty() || *bucket_interval_ms != 0) => + { + Err(refuse_clone_read_shape(plan, ctx.target_coll)) + } + + TimeseriesOp::Scan { + collection, + time_range, + projection, + limit, + filters, + sort_keys, + bucket_interval_ms, + group_by, + aggregates, + gap_fill, + computed_columns, + rls_filters, + system_time, + valid_at_ms, + } if ctx.is_target(collection) => Ok(SourceRewrite::task(PhysicalPlan::Timeseries( + TimeseriesOp::Scan { + collection: ctx.source_qualified.clone(), + time_range: *time_range, + projection: projection.clone(), + limit: *limit, + filters: filters.clone(), + sort_keys: sort_keys.clone(), + bucket_interval_ms: *bucket_interval_ms, + group_by: group_by.clone(), + aggregates: aggregates.clone(), + gap_fill: gap_fill.clone(), + computed_columns: computed_columns.clone(), + rls_filters: rls_filters.clone(), + system_time: ctx.system_time(*system_time)?, + valid_at_ms: *valid_at_ms, + }, + ))), + + _ => refuse_or_skip(plan, ctx), + } +} diff --git a/nodedb/src/control/clone/tombstone.rs b/nodedb/src/control/clone/tombstone.rs index 6d1ad61de..78527f7f4 100644 --- a/nodedb/src/control/clone/tombstone.rs +++ b/nodedb/src/control/clone/tombstone.rs @@ -3,18 +3,19 @@ //! Tombstone write helper for cloned collections. //! //! When a DELETE targets a row that exists only in the source of a `Shadowed` -//! clone, this module records a tombstone in `_system.clone_tombstones`. The -//! read path consults this table before falling back to source storage, so -//! subsequent reads correctly return "not found." +//! clone, this module records a tombstone in `_system.clone_tombstones` on +//! every node. The read path consults this table before falling back to +//! source storage, so subsequent reads return "not found." -use nodedb_types::{DatabaseId, Surrogate}; +use nodedb_types::{DatabaseId, Surrogate, TenantId}; -use crate::control::planner::sql_plan_convert::convert::db_qualified; +use crate::control::catalog_entry::CatalogEntry; use crate::control::state::SharedState; /// Parameters for a tombstone write. pub struct TombstoneParams<'a> { pub state: &'a SharedState, + pub tenant_id: TenantId, /// The database ID of the clone (target). pub target_db_id: DatabaseId, /// The plain collection name (not db_qualified). @@ -26,6 +27,7 @@ pub struct TombstoneParams<'a> { /// Parameters for a KV tombstone write. pub struct KvTombstoneParams<'a> { pub state: &'a SharedState, + pub tenant_id: TenantId, /// The database ID of the clone (target). pub target_db_id: DatabaseId, /// The plain collection name (not db_qualified). @@ -34,50 +36,51 @@ pub struct KvTombstoneParams<'a> { pub kv_key: String, } -/// Record a KV tombstone for `kv_key` in `target_collection`. +/// Record a KV tombstone for `kv_key` in `target_collection` on every node. /// /// After this call, the clone read path will exclude the source row with this /// KV key from scan results, even though the row still exists in the source. -pub fn perform_kv_clone_tombstone(params: KvTombstoneParams<'_>) -> crate::Result<()> { +pub async fn perform_kv_clone_tombstone(params: KvTombstoneParams<'_>) -> crate::Result<()> { let KvTombstoneParams { state, + tenant_id, target_db_id, target_collection, kv_key, } = params; - - let catalog = state.credentials.catalog(); - - let target_coll_qualified = db_qualified(target_db_id, target_collection); - - catalog - .put_kv_clone_tombstone(&target_coll_qualified, &kv_key) - .map_err(|e| crate::Error::Storage { - engine: "clone_kv_tombstone".into(), - detail: format!("put_kv_clone_tombstone failed: {e}"), - }) + super::cow_entry::replicate_async( + state, + &CatalogEntry::PutKvCloneTombstone { + database_id: target_db_id.as_u64(), + tenant_id: tenant_id.as_u64(), + collection: target_collection.to_string(), + kv_key, + }, + ) + .await } -/// Record a tombstone for `source_surrogate` in `target_collection`. +/// Record a tombstone for `source_surrogate` in `target_collection` on every +/// node. /// /// After this call, the clone read path will return "not found" for this /// surrogate, even though the row still exists in the source database. -pub fn perform_clone_tombstone(params: TombstoneParams<'_>) -> crate::Result<()> { +pub async fn perform_clone_tombstone(params: TombstoneParams<'_>) -> crate::Result<()> { let TombstoneParams { state, + tenant_id, target_db_id, target_collection, source_surrogate, } = params; - - let catalog = state.credentials.catalog(); - - let target_coll_qualified = db_qualified(target_db_id, target_collection); - - catalog - .put_clone_tombstone(&target_coll_qualified, source_surrogate.as_u32()) - .map_err(|e| crate::Error::Storage { - engine: "clone_tombstone".into(), - detail: format!("put_clone_tombstone failed: {e}"), - }) + super::cow_entry::replicate_async( + state, + &CatalogEntry::PutCloneTombstone { + database_id: target_db_id.as_u64(), + tenant_id: tenant_id.as_u64(), + collection: target_collection.to_string(), + source_surrogate: source_surrogate.as_u32(), + }, + ) + .await } diff --git a/nodedb/src/control/cluster/array_cluster_helpers.rs b/nodedb/src/control/cluster/array_cluster_helpers.rs index 482467674..bbd41789b 100644 --- a/nodedb/src/control/cluster/array_cluster_helpers.rs +++ b/nodedb/src/control/cluster/array_cluster_helpers.rs @@ -117,10 +117,12 @@ pub(super) fn cluster_err(e: ClusterError) -> Error { vshard_id, expected_owner_node, } => match expected_owner_node { + // `WrongOwner` names the owner without a term. Some(leader_node) => Error::NotLeader { vshard_id: crate::types::VShardId::new(vshard_id), leader_node, leader_addr: String::new(), + leader_term: 0, }, None => Error::NoLeader { vshard_id: crate::types::VShardId::new(vshard_id), @@ -163,6 +165,7 @@ pub(super) fn cluster_err(e: ClusterError) -> Error { | ClusterError::SpatialGather(_) | ClusterError::Bm25Gather(_) | ClusterError::TsGather(_) + | ClusterError::ShufflePush(_) | ClusterError::RemoteUntyped { .. }) => Error::Internal { detail: format!("array cluster: {other}"), }, diff --git a/nodedb/src/control/cluster/array_executor/executor.rs b/nodedb/src/control/cluster/array_executor/executor.rs index de04c03fc..b75c53d2b 100644 --- a/nodedb/src/control/cluster/array_executor/executor.rs +++ b/nodedb/src/control/cluster/array_executor/executor.rs @@ -11,6 +11,10 @@ use nodedb_cluster::rpc_codec::DataPlaneErrorCode; use super::refusal::execution_error; use crate::bridge::envelope::{Priority, Request, Response}; +use crate::control::cluster::linearizable_read::{ + confirm_linearizable_read, groups_of_vshards, statement_read_deadline, +}; +use crate::control::server::shared::write_admission::plan_is_write; use crate::control::state::SharedState; use crate::event::types::EventSource; use crate::types::{ReadConsistency, RequestId, TraceId, TxnId, VShardId}; @@ -50,17 +54,32 @@ impl DataPlaneArrayExecutor { plan: PhysicalPlan, txn_id: Option, ) -> Result { + // Array reads have no weaker consistency: a shard read confirms its + // group on this node before it reads. + if !plan_is_write(&plan) { + let groups = groups_of_vshards(&self.state, [local_vshard_id.as_u32()]) + .map_err(|e| execution_error("array read confirmation", e))?; + confirm_linearizable_read(&self.state, &groups, statement_read_deadline(&self.state)) + .await + .map_err(|e| execution_error("array read confirmation", e))?; + } let request_id = self.state.next_request_id(); let request = local_request(request_id, array_id, local_vshard_id, plan, txn_id); let mut rx = self.state.tracker.register(request_id); - let dispatch_result = match self.state.dispatcher.lock() { - Ok(mut d) => d.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - }; - - // A dispatch refusal, such as a capacity limit, keeps its own class. + // A shard fan-out sends one request per shard, which can exceed the + // tenant's in-flight cap: each request waits for a freed slot until + // its deadline instead of refusing the statement. + let dispatch_result = crate::control::server::dispatch_utils::dispatch_when_capacity_frees( + &self.state, + request, + None, + ) + .await; + + // A dispatch refusal, such as a capacity limit past the deadline, + // keeps its own class. if let Err(e) = dispatch_result { return Err(execution_error("array executor dispatch", e)); } @@ -114,6 +133,7 @@ fn local_request( txn_id, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: crate::bridge::envelope::Admission::Exempt( crate::bridge::envelope::ExemptReason::AlreadyOrdered, ), @@ -137,14 +157,14 @@ mod tests { assert_eq!(array_id.tenant_id, default_scope_array.tenant_id); assert_ne!(array_id.database_id, default_scope_array.database_id); + let vshard_id = VShardId::new(19); let plan = PhysicalPlan::Array(ArrayOp::Delete { array_id: array_id.clone(), coords_msgpack: Vec::new(), wal_lsn: 0, provenance: None, + vshard_id: vshard_id.as_u32(), }); - - let vshard_id = VShardId::new(19); let replicable = ReplicableWrite::decide_for_replication(&plan) .expect("array write plan carries no live RLS predicate"); let entry = to_replicated_entry(tenant_id, database_id, vshard_id, &replicable) @@ -193,6 +213,7 @@ mod tests { coords_msgpack: Vec::new(), wal_lsn: 0, provenance: None, + vshard_id: vshard_id.as_u32(), }); let read_request = local_request(RequestId::new(8), &array_id, vshard_id, read, None); diff --git a/nodedb/src/control/cluster/array_executor/refusal.rs b/nodedb/src/control/cluster/array_executor/refusal.rs index d3d37ff5c..af94eae15 100644 --- a/nodedb/src/control/cluster/array_executor/refusal.rs +++ b/nodedb/src/control/cluster/array_executor/refusal.rs @@ -92,9 +92,7 @@ pub(super) fn execution_error(context: &str, error: crate::Error) -> ClusterErro | crate::Error::CrdtApplyForbiddenInTransaction | crate::Error::NotInTransactionBlock { .. } | crate::Error::CrdtAdmissionTimeout { .. } - | crate::Error::FanOutExceeded { .. } | crate::Error::CrossCollectionNotColocated { .. } - | crate::Error::SourceFrozen { .. } | crate::Error::CloneWriteRequiresMaterialize { .. } | crate::Error::BadRequest { .. } | crate::Error::BackupTenantMismatch { .. } @@ -113,10 +111,14 @@ pub(super) fn execution_error(context: &str, error: crate::Error) -> ClusterErro | crate::Error::InvalidLimitValue { .. } | crate::Error::RetryableSchemaChanged { .. } | crate::Error::RetryableLeaderChange { .. } + | crate::Error::CommittedResultUnavailable { .. } + | crate::Error::ProposalOutcomeUnknown { .. } | crate::Error::GroupQuorumUnavailable { .. } | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::BackupCaptureMoved { .. } | crate::Error::MetadataLeaderUnavailable | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::LinearizableReadRefused { .. } | crate::Error::ExecutionLimitExceeded { .. } | crate::Error::LimitExceeded { .. } | crate::Error::Wal(_) @@ -134,12 +136,15 @@ pub(super) fn execution_error(context: &str, error: crate::Error) -> ClusterErro | crate::Error::Encryption { .. } | crate::Error::Bridge { .. } | crate::Error::VersionCompat { .. } + | crate::Error::RestoreTargetNotEmpty { .. } + | crate::Error::RestoreVerificationFailed { .. } | crate::Error::Internal { .. } | crate::Error::Shaping(_) | crate::Error::Ddl(_) | crate::Error::RemoteTyped { .. } | crate::Error::DescriptorVersionAnomaly { .. } | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CollectionUnstamped { .. } | crate::Error::CatalogIntegrityViolation { .. } | crate::Error::Promql(_) | crate::Error::DependentObjectsExist { .. } @@ -240,6 +245,7 @@ mod tests { vshard_id: crate::types::VShardId::new(9), leader_node: 4, leader_addr: "10.0.0.4:9000".into(), + leader_term: 2, }; assert!(matches!( execution_error("array put raft propose", error), diff --git a/nodedb/src/control/cluster/array_executor/write.rs b/nodedb/src/control/cluster/array_executor/write.rs index 6d2982a89..1bd825663 100644 --- a/nodedb/src/control/cluster/array_executor/write.rs +++ b/nodedb/src/control/cluster/array_executor/write.rs @@ -2,10 +2,9 @@ //! Write handlers for [`DataPlaneArrayExecutor`] — put and delete. //! -//! Run on the shard OWNER after coordinator RPC-routing. Multi-node: proposed -//! to the owning shard's data Raft group as `ReplicatedWrite::ArrayCellPut` / -//! `ArrayCellDelete`. Single-node: applied via the Control-Plane write funnel, -//! whose redo record is the array engine's only durability (it's a memtable). +//! Run on the shard OWNER after coordinator RPC-routing. Proposed to the +//! owning shard's data Raft group as `ReplicatedWrite::ArrayCellPut` / +//! `ArrayCellDelete`, on a one-node cluster too. use nodedb_array::types::ArrayId; use nodedb_cluster::distributed_array::ArrayShardWriteOutcome; @@ -14,12 +13,9 @@ use nodedb_cluster::error::{ClusterError, Result}; use super::cells::flatten_blob_vec; use super::executor::DataPlaneArrayExecutor; -use super::refusal::{execution_error, refusal_error}; -use crate::control::server::dispatch_utils::{ - ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, submit_write, -}; +use super::refusal::execution_error; use crate::control::server::shared::sql::staging_predicates::require_affected_count; -use crate::types::{TraceId, VShardId}; +use crate::types::VShardId; use nodedb_physical::physical_plan::{ArrayOp, PhysicalPlan}; impl DataPlaneArrayExecutor { @@ -48,9 +44,10 @@ impl DataPlaneArrayExecutor { cells_msgpack, wal_lsn: req.wal_lsn, provenance: None, + vshard_id: local_vshard_id, }); - self.propose_or_dispatch(&array_id, local_vshard_id, plan, req.wal_lsn, "array put") + self.propose(&array_id, local_vshard_id, plan, req.wal_lsn, "array put") .await } @@ -79,9 +76,10 @@ impl DataPlaneArrayExecutor { coords_msgpack, wal_lsn: req.wal_lsn, provenance: None, + vshard_id: local_vshard_id, }); - self.propose_or_dispatch( + self.propose( &array_id, local_vshard_id, plan, @@ -91,13 +89,12 @@ impl DataPlaneArrayExecutor { .await } - /// Replicate `plan` to the owning shard's data Raft group when a proposer - /// exists; otherwise apply it locally through the write funnel. Returns - /// the `applied_lsn` the coordinator acks with, plus the real cell count - /// the Data Plane handler's `{"inserted"|"deleted": n}` response reports. - /// Both branches use the vShard from the validated RPC envelope so all - /// paths select the same Data Plane core. - async fn propose_or_dispatch( + /// Replicate `plan` to the owning shard's data Raft group. Returns the + /// `applied_lsn` the coordinator acks with, plus the real cell count the + /// Data Plane handler's `{"inserted"|"deleted": n}` response reports. + /// The entry carries the vShard from the validated RPC envelope, so every + /// replica selects the same Data Plane core. + async fn propose( &self, array_id: &ArrayId, local_vshard_id: u32, @@ -105,128 +102,42 @@ impl DataPlaneArrayExecutor { wal_lsn: u64, op_label: &str, ) -> Result { - if let Some(proposer) = self.state.async_raft_proposer() { - let replicable = - crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan) - .map_err(|e| ClusterError::Storage { - detail: format!("{op_label}: {e}"), - })?; - let entry = crate::control::wal_replication::to_replicated_entry( - array_id.tenant_id, - array_id.database_id, - VShardId::new(local_vshard_id), - &replicable, - ) - .map_err(|e| ClusterError::Storage { - detail: format!("{op_label}: {e}"), - })? - .ok_or_else(|| ClusterError::Storage { - detail: format!("{op_label}: plan is not encodable as a replicated entry"), - })?; - - let (apply_payload, _write_version) = - crate::control::wal_replication::propose_replicated_entry( - &self.state, - proposer, - entry, - ) - .await - .map_err(|e| execution_error(&format!("{op_label} raft propose"), e))?; - let affected = - require_affected_count(&apply_payload).map_err(|e| ClusterError::Storage { + let proposer = self + .state + .async_raft_proposer() + .map_err(|e| execution_error(&format!("{op_label} raft propose"), e))?; + let replicable = + crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan) + .map_err(|e| ClusterError::Storage { detail: format!("{op_label}: {e}"), })?; - // No LSN exists to report here: each replica mints its own redo, - // none authoritative. `wal_lsn` is echoed back verbatim — what - // the coordinator sent, not a claim about what was recorded. - return Ok(ArrayShardWriteOutcome { - applied_lsn: wal_lsn, - affected, - }); - } - - // Single-node: this node's WAL is the write's only durability. The - // funnel appends the redo, stamps the minted LSN into the plan, and - // fsyncs before this ack returns. Ordering is decided HERE by the gate. - let outcome: SubmitOutcome = submit_write( - &self.state, - single_node_submit(array_id, VShardId::new(local_vshard_id), plan), + let entry = crate::control::wal_replication::to_replicated_entry( + array_id.tenant_id, + array_id.database_id, + VShardId::new(local_vshard_id), + &replicable, ) - .await - .map_err(|e| execution_error(op_label, e))?; - - if outcome.response.status == crate::bridge::envelope::Status::Error { - return Err(refusal_error(op_label, &outcome.response)); - } + .map_err(|e| ClusterError::Storage { + detail: format!("{op_label}: {e}"), + })? + .ok_or_else(|| ClusterError::Storage { + detail: format!("{op_label}: plan is not encodable as a replicated entry"), + })?; - // Ack with the LSN the funnel actually minted. `None` would mean the - // funnel classified this as appending nothing — a wiring bug. - let applied_lsn = - outcome - .wal_lsn - .map(|lsn| lsn.as_u64()) - .ok_or_else(|| ClusterError::Storage { - detail: format!( - "{op_label}: applied with no WAL redo record — write is not durable" - ), - })?; - let affected = require_affected_count(outcome.response.payload.as_ref()).map_err(|e| { - ClusterError::Storage { + let (apply_payload, _write_version) = + crate::control::wal_replication::propose_replicated_entry(&self.state, proposer, entry) + .await + .map_err(|e| execution_error(&format!("{op_label} raft propose"), e))?; + let affected = + require_affected_count(&apply_payload).map_err(|e| ClusterError::Storage { detail: format!("{op_label}: {e}"), - } - })?; + })?; + // No LSN exists to report here: each replica mints its own redo, + // none authoritative. `wal_lsn` is echoed back verbatim — what + // the coordinator sent, not a claim about what was recorded. Ok(ArrayShardWriteOutcome { - applied_lsn, + applied_lsn: wal_lsn, affected, }) } } - -/// Build the single-node write-funnel request using the envelope's validated -/// vShard. The funnel carries this through to both the bridge request and the -/// WAL record, whose replay uses the same vShard-to-core mapping. -fn single_node_submit( - array_id: &ArrayId, - local_vshard_id: VShardId, - plan: PhysicalPlan, -) -> SubmitWrite { - SubmitWrite { - tenant_id: array_id.tenant_id, - database_id: array_id.database_id, - vshard_id: local_vshard_id, - plan, - trace_id: TraceId::generate(), - event_source: crate::event::EventSource::User, - txn_id: None, - user_id: None, - durability: WalDurability::AppendHere { - now_override: None, - apply_key: 0, - commit_hlc: None, - }, - ordering: WriteOrdering::Gate, - change_feed: ChangeFeedOwner::Funnel, - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::types::TenantId; - - #[test] - fn single_node_write_preserves_nonzero_vshard() { - let array_id = ArrayId::new(TenantId::new(41), "measurements"); - let vshard_id = VShardId::new(19); - let plan = PhysicalPlan::Array(ArrayOp::Delete { - array_id: array_id.clone(), - coords_msgpack: Vec::new(), - wal_lsn: 0, - provenance: None, - }); - - let request = single_node_submit(&array_id, vshard_id, plan); - - assert_eq!(request.vshard_id, vshard_id); - } -} diff --git a/nodedb/src/control/cluster/boot_restore.rs b/nodedb/src/control/cluster/boot_restore.rs deleted file mode 100644 index 43b86eb5c..000000000 --- a/nodedb/src/control/cluster/boot_restore.rs +++ /dev/null @@ -1,106 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Follower boot-restore of persisted Raft snapshots. -//! -//! On node startup the committed `.snap` files left under -//! `/recv_snapshots/` by the install-snapshot RECEIVE path survive -//! across restarts (the startup orphan sweep only removes `.partial` files). -//! Those `.snap` files hold the engine state for the log prefix that Raft -//! compacted away after applying the snapshot — once the leader compacts its -//! log past the snapshot index, that prefix can NEVER be re-replayed. The -//! persisted snapshot is therefore the only source for that state, and it must -//! be re-installed into the Data Plane BEFORE the apply loop replays the -//! post-snapshot log tail. Otherwise a restarted follower comes up missing all -//! pre-snapshot rows even though full scans of the tail succeed. -//! -//! This mirrors the in-band [`crate::control::cluster::snapshot_applier`] -//! RECEIVE path: it reuses the same [`DataPlaneSnapshotApplier`] so an apply -//! failure is FATAL (a node must not come up with partially-restored state). - -use std::path::Path; - -use tracing::{info, warn}; - -use crate::control::cluster::snapshot_applier::DataPlaneSnapshotApplier; -use nodedb_cluster::SnapshotApplier; - -/// Re-install every persisted `.snap` snapshot under -/// `/recv_snapshots/` into the local Data Plane via `applier`. -/// -/// Returns the number of group snapshots applied. Snapshots are applied in -/// ascending `group_id` order for deterministic boot behaviour. A `.snap` file -/// whose stem is not a `u64` group id, or whose body is empty, is logged and -/// skipped (it cannot name a real group / has nothing to restore). An apply -/// failure is FATAL and returned to the caller: the node must not finish boot -/// with missing engine state. -pub async fn restore_persisted_snapshots( - data_dir: &Path, - applier: &DataPlaneSnapshotApplier, -) -> crate::Result { - let recv_dir = data_dir.join("recv_snapshots"); - // No receive directory means this node never received a snapshot — nothing - // to restore. - if !recv_dir.exists() { - return Ok(0); - } - - // Collect `.snap` paths paired with their parsed group id, then sort by - // group id so the apply order is stable across boots. - let mut snaps: Vec<(u64, std::path::PathBuf)> = Vec::new(); - for entry in std::fs::read_dir(&recv_dir)? { - let entry = entry?; - let path = entry.path(); - if path.extension().and_then(|e| e.to_str()) != Some("snap") { - continue; - } - let group_id = match path - .file_stem() - .and_then(|s| s.to_str()) - .and_then(|s| s.parse::().ok()) - { - Some(id) => id, - None => { - warn!( - path = %path.display(), - "boot-restore: skipping snapshot with non-numeric group id stem" - ); - continue; - } - }; - snaps.push((group_id, path)); - } - snaps.sort_by_key(|(group_id, _)| *group_id); - - let mut applied = 0usize; - for (group_id, path) in snaps { - let bytes = std::fs::read(&path)?; - if bytes.is_empty() { - warn!( - group_id, - path = %path.display(), - "boot-restore: skipping empty persisted snapshot" - ); - continue; - } - - // An apply failure here is FATAL: the same rationale as the in-band - // RECEIVE path — a node must not come up with partially-restored - // engine state. The applier's boxed error is mapped onto the crate's - // typed error. - applier - .apply_snapshot(group_id, &bytes) - .await - .map_err(|e| crate::Error::Internal { - detail: format!("boot-restore: apply group {group_id} snapshot: {e}"), - })?; - - info!( - group_id, - bytes = bytes.len(), - "boot-restore: re-installed persisted group snapshot" - ); - applied += 1; - } - - Ok(applied) -} diff --git a/nodedb/src/control/cluster/bootstrap_listener.rs b/nodedb/src/control/cluster/bootstrap_listener.rs index 5d3d6b00b..140b9f82d 100644 --- a/nodedb/src/control/cluster/bootstrap_listener.rs +++ b/nodedb/src/control/cluster/bootstrap_listener.rs @@ -29,7 +29,9 @@ type HostRaftLoop = nodedb_cluster::RaftLoop< crate::control::LocalPlanExecutor, >; -/// Async metadata proposer used only by durable bootstrap token state. +/// Async metadata proposer used only by durable bootstrap token state. Every +/// entry it proposes carries the leader's stamp, as the host's other +/// proposals do. pub(crate) struct BootstrapMetadataProposer { raft_loop: Weak, watchers: Arc, @@ -57,7 +59,7 @@ impl nodedb_cluster::decommission::MetadataProposer for BootstrapMetadataPropose detail: format!("bootstrap metadata entry: {error}"), })?; let index = raft_loop - .propose_to_metadata_group_via_leader(bytes) + .propose_stamped_to_metadata_group_via_leader(bytes) .await?; let watchers = Arc::clone(&self.watchers); let outcome = tokio::task::spawn_blocking(move || { @@ -281,7 +283,7 @@ impl BootstrapHandler for HostBootstrapHandler { .mark_consumed(&hash, remote_addr, lease, epoch_ms(), recovery_bundle) .await { - // The proposal result is indeterminate: it may commit after the + // The proposal result is indeterminate: it can commit after the // local apply wait times out. Keep the bounded preauthorization // so a later retry can decrypt the Raft-persisted bundle and use // the exact certificate whose token became Consumed. @@ -318,9 +320,9 @@ impl BootstrapHandler for HostBootstrapHandler { remote_addr: SocketAddr, ) -> std::pin::Pin + Send + 'a>> { Box::pin(async move { - // Credential bytes may already be observable when the delivery ACK + // Credential bytes can already be observable when the delivery ACK // is lost. Keep the bounded enrollment authorization alive until - // topology commit or certificate expiry; revoking here would + // topology commit or certificate expiry; revoking here will // strand a joiner whose token is already durably consumed. tracing::warn!(%remote_addr, "bootstrap delivery ACK missing; retaining bounded enrollment authorization"); }) diff --git a/nodedb/src/control/cluster/ca_trust.rs b/nodedb/src/control/cluster/ca_trust.rs new file mode 100644 index 000000000..b0bcc91c7 --- /dev/null +++ b/nodedb/src/control/cluster/ca_trust.rs @@ -0,0 +1,142 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The overlap CA trust set under `tls/ca.d/`: one PEM file per CA, named by +//! its fingerprint. + +use std::collections::HashSet; +use std::fs; +use std::path::{Path, PathBuf}; + +use nodedb_cluster::transport::pki_types::CertificateDer; + +use super::tls::{CA_TRUST_DIR, read_single_cert, write_pem_cert}; + +/// Load every PEM-encoded CA certificate from `tls_dir/ca.d/*.crt`, +/// sorted by filename for deterministic output. Missing directory is +/// treated as "no overlap CAs" and returns an empty vec. +pub(crate) fn load_extra_cas(tls_dir: &Path) -> crate::Result>> { + let dir = tls_dir.join(CA_TRUST_DIR); + if !dir.exists() { + return Ok(Vec::new()); + } + let listing = fs::read_dir(&dir).map_err(|e| crate::Error::Config { + detail: format!("read ca.d {}: {e}", dir.display()), + })?; + let entries = crt_paths(&dir, listing.map(|r| r.map(|e| e.path())))?; + let mut out = Vec::with_capacity(entries.len()); + for p in entries { + out.push(read_single_cert(&p)?); + } + Ok(out) +} + +/// The `.crt` paths of a `ca.d` listing, sorted by file name. +/// +/// An entry the listing cannot read fails the load. Skipping it will +/// silently drop a trusted CA. +fn crt_paths( + dir: &Path, + listing: impl Iterator>, +) -> crate::Result> { + let mut paths = Vec::new(); + for entry in listing { + let path = entry.map_err(|e| crate::Error::Config { + detail: format!("read ca.d entry in {}: {e}", dir.display()), + })?; + if path.extension().and_then(|s| s.to_str()) == Some("crt") { + paths.push(path); + } + } + paths.sort(); + Ok(paths) +} + +/// Write a PEM-encoded CA cert into `tls_dir/ca.d/.crt`. +/// Called by the production applier when a `CaTrustChange { add: ... }` +/// entry commits. +pub fn write_trusted_ca(tls_dir: &Path, ca_der: &[u8]) -> crate::Result<[u8; 32]> { + let dir = tls_dir.join(CA_TRUST_DIR); + fs::create_dir_all(&dir).map_err(|e| crate::Error::Config { + detail: format!("create ca.d dir {}: {e}", dir.display()), + })?; + let cert = CertificateDer::from(ca_der.to_vec()); + let fp = nodedb_cluster::ca_fingerprint(&cert); + let name = format!("{}.crt", nodedb_cluster::ca_fingerprint_hex(&fp)); + write_pem_cert(&dir, &name, ca_der)?; + Ok(fp) +} + +/// Delete the overlap-CA file identified by `fp` from `tls_dir/ca.d/`. +/// No-op (and returns `Ok(())`) when the file isn't present — applier +/// behaviour must be idempotent across re-apply and snapshot replay. +pub fn remove_trusted_ca(tls_dir: &Path, fp: &[u8; 32]) -> crate::Result<()> { + let dir = tls_dir.join(CA_TRUST_DIR); + let path = dir.join(format!("{}.crt", nodedb_cluster::ca_fingerprint_hex(fp))); + match fs::remove_file(&path) { + Ok(()) => Ok(()), + Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(()), + Err(e) => Err(crate::Error::Config { + detail: format!("remove ca.d entry {}: {e}", path.display()), + }), + } +} + +/// DER of every CA in `tls_dir/ca.d/`, sorted by file name. +pub fn trusted_ca_ders(tls_dir: &Path) -> crate::Result>> { + Ok(load_extra_cas(tls_dir)? + .into_iter() + .map(|cert| cert.as_ref().to_vec()) + .collect()) +} + +/// Make `tls_dir/ca.d/` hold exactly the CAs in `ders`: write each, then +/// remove every other CA file. +pub fn replace_trusted_cas(tls_dir: &Path, ders: &[Vec]) -> crate::Result<()> { + let mut keep: HashSet<[u8; 32]> = HashSet::with_capacity(ders.len()); + for der in ders { + keep.insert(write_trusted_ca(tls_dir, der)?); + } + for der in trusted_ca_ders(tls_dir)? { + let fp = nodedb_cluster::ca_fingerprint(&CertificateDer::from(der)); + if !keep.contains(&fp) { + remove_trusted_ca(tls_dir, &fp)?; + } + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// A listing entry that fails to read fails the load with a config + /// error naming the directory. + #[test] + fn unreadable_entry_fails_the_load() { + let dir = Path::new("/etc/nodedb/tls/ca.d"); + let listing = vec![ + Ok(dir.join("a.crt")), + Err(std::io::Error::other("entry vanished")), + Ok(dir.join("b.crt")), + ]; + let err = crt_paths(dir, listing.into_iter()).expect_err("an unreadable entry must fail"); + let crate::Error::Config { detail } = err else { + panic!("expected a config error, got {err:?}"); + }; + assert!(detail.contains("/etc/nodedb/tls/ca.d"), "detail: {detail}"); + assert!(detail.contains("entry vanished"), "detail: {detail}"); + } + + /// Only `.crt` entries load, sorted by file name. + #[test] + fn crt_entries_load_sorted() { + let dir = Path::new("/tls/ca.d"); + let listing = vec![ + Ok(dir.join("b.crt")), + Ok(dir.join("notes.txt")), + Ok(dir.join("a.crt")), + ]; + let paths = crt_paths(dir, listing.into_iter()).expect("readable listing"); + assert_eq!(paths, vec![dir.join("a.crt"), dir.join("b.crt")]); + } +} diff --git a/nodedb/src/control/cluster/calvin/executor/ollp/orchestrator.rs b/nodedb/src/control/cluster/calvin/executor/ollp/orchestrator.rs index 8914d0a59..83277474b 100644 --- a/nodedb/src/control/cluster/calvin/executor/ollp/orchestrator.rs +++ b/nodedb/src/control/cluster/calvin/executor/ollp/orchestrator.rs @@ -181,6 +181,9 @@ impl OllpOrchestrator { // `submit_with_retry` call from the SQL layer with a fresh `tx_builder` // closure; the budget + circuit checks below run on every attempt // because they are stored on `self` and persist across calls. + // A bare inbox carries single-entry classes only: a multi-part class + // needs a coordinator to stream its parts, which the routed submit + // of `submit_with_retry_via` runs. let Some(tx_class) = self.check_and_build(predicate_class, tenant_id, tx_builder)? else { return Ok(None); }; @@ -343,11 +346,11 @@ mod tests { #[test] fn tenant_budget_exceeded_at_1000_per_min() { let mut bucket = RateBucket::new(1000, Duration::from_secs(60)); - // First 1000 should NOT exceed. + // First 1000 must NOT exceed. for _ in 0..1000 { assert!(!bucket.record_and_check()); } - // 1001st should exceed. + // 1001st must exceed. assert!(bucket.record_and_check()); } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs b/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs index dc84ac9b5..82ed6f408 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/applied_mirror.rs @@ -128,6 +128,15 @@ impl AppliedMirrors { .collect() } + /// Drop the mirror of `vshard_id`: this node left the vShard's group and + /// holds none of its applied positions. + pub fn remove(&self, vshard_id: u32) { + self.by_vshard + .lock() + .unwrap_or_else(|p| p.into_inner()) + .remove(&vshard_id); + } + /// The mirror of `vshard_id`, when this node runs its scheduler. pub fn get(&self, vshard_id: u32) -> Option> { self.by_vshard diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs index 41bf6fcfd..d02acd9a6 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/catch_up.rs @@ -6,7 +6,7 @@ //! per-vShard scheduler channels with a bounded `try_send`. On a Full/Closed //! channel the input is DROPPED (only bookkept, never blocking — `apply` shares //! its call stack with every Raft group and must not stall node-wide -//! heartbeats). A dropped input would otherwise permanently diverge this +//! heartbeats). A dropped input will otherwise permanently diverge this //! replica's lock table from its peers, since the lock table is a local //! projection every replica rebuilds from the byte-identical sequencer Raft log. //! @@ -94,28 +94,33 @@ impl Scheduler { // 2. MultiRaft-lock scope: read the committed sequencer log window. No // SM lock is held here (see the lock-discipline note above). + // The read runs under the MultiRaft lock alone; the result is handled + // after the lock drops. + let read = self + .multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .read_committed_entries(SEQUENCER_GROUP_ID, lo, end); let entries = { - let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); - match mr.read_committed_entries(SEQUENCER_GROUP_ID, lo, end) { + match read { Ok(entries) => entries, Err(nodedb_cluster::error::ClusterError::Raft( nodedb_raft::RaftError::LogCompacted { .. }, )) => { - // The armed index has been compacted below the retained log. - // The sequencer-group compaction hold-down (floored at - // `min_catch_up_from`) is meant to make this unreachable for - // an armed catch-up; if it is nonetheless hit (e.g. a - // snapshot-install resync that already subsumes this index), - // no replay is owed. Escalate non-silently, CLEAR the entry - // to avoid an infinite retry against a permanently-compacted - // index, and return. + // The armed index is below the retained log: an input + // this replica never applied is gone, and only a + // data-group snapshot brings the vShard's Calvin state + // back. The compaction floor keeps every armed index, so + // the log reached this node compacted already: a + // sequencer snapshot installed under the armed index. self.metrics.record_catch_up_log_compacted(); tracing::error!( vshard = self.vshard_id, lo, - "calvin catch-up: sequencer log compacted below armed index; \ - state is snapshot-covered" + "calvin catch-up: the sequencer log no longer holds the armed index; \ + the vShard's data group takes a snapshot" ); + self.lose_calvin_base(); self.sequencer_state_machine .lock() .unwrap_or_else(|p| p.into_inner()) @@ -173,8 +178,14 @@ impl Scheduler { // next drain resumes at the first input not yet processed. let mut replayed: u64 = 0; let mut resume_from: Option = None; + // The replayed transactions, which the sequencer log keeps until + // this scheduler made them durable. + let mut received: Vec<(u64, u64, u32)> = Vec::new(); let mut feed = inputs.into_iter().peekable(); - while let Some((_, input)) = feed.next() { + while let Some((index, input)) = feed.next() { + if let SchedulerInput::Txn(txn) = &input { + received.push((index, txn.epoch, txn.position)); + } self.process_scheduler_input(input); replayed += 1; if self.intake_closure().is_some() { @@ -186,10 +197,13 @@ impl Scheduler { let resume_from = resume_from.or(end.checked_add(1).filter(|&next| next <= hi)); { - let sm = self + let mut sm = self .sequencer_state_machine .lock() .unwrap_or_else(|p| p.into_inner()); + for (index, epoch, position) in received { + sm.note_replayed_txn(index, self.vshard_id, epoch, position); + } match resume_from { // Stopped early: re-arm exactly at the first unprocessed input's // index. Clearing below it first lets the min-collapse arm move @@ -278,10 +292,11 @@ mod tests { // it through `replay_epochs_for_vshard`, and feeds each input through // `process_scheduler_input` — and assert the dropped input's effect lands in // the scheduler's lock table. This proves the whole mechanism closes the - // fan-out gap, not just each half. + // fan-out gap, not only each half. /// Ensure the sequencer Raft group exists on this scheduler's `MultiRaft` and - /// that this single node is its leader, so proposals commit immediately. + /// that this single node is its leader, so a proposal commits once its disk + /// holds it. fn ensure_sequencer_leader(scheduler: &Scheduler) { let mut mr = scheduler .multi_raft @@ -312,7 +327,8 @@ mod tests { /// Encode `batch` and propose it to the committed sequencer Raft log, returning /// its committed Raft index and the encoded bytes (reused to drive `apply`, so /// the index handed to `apply` is the SAME real committed index the drain will - /// read back). Single-voter groups commit on propose. + /// read back). A single-voter group commits once its disk holds the entry, + /// on the next tick. fn commit_epoch_batch(scheduler: &Scheduler, batch: EpochBatch) -> (u64, Vec) { let bytes = zerompk::to_msgpack_vec(&SequencerEntry::EpochBatch { batch }) .expect("encode epoch batch"); @@ -321,8 +337,12 @@ mod tests { .multi_raft .lock() .unwrap_or_else(|p| p.into_inner()); - mr.propose_to_group(nodedb_cluster::calvin::SEQUENCER_GROUP_ID, bytes.clone()) - .expect("propose epoch batch to sequencer group") + let index = mr + .propose_to_group(nodedb_cluster::calvin::SEQUENCER_GROUP_ID, bytes.clone()) + .expect("propose epoch batch to sequencer group"); + mr.wait_all_durable_blocking(); + mr.tick().expect("tick"); + index }; (index, bytes) } @@ -351,7 +371,7 @@ mod tests { ) { let (full_tx, full_rx) = tokio::sync::mpsc::channel(1); full_tx - .try_send(SchedulerInput::Txn(fill.clone())) + .try_send(SchedulerInput::Txn(Box::new(fill.clone()))) .expect("pre-fill the capacity-1 channel"); // Keep the receiver alive for the duration of `apply` so the sender reports // Full rather than Closed (either records a catch-up index, but Full is the @@ -495,7 +515,7 @@ mod tests { // Deliver epoch 1 LIVE: it acquires, blocks behind the sentinel, and is now // in-flight (its (epoch, position) sits in `blocked`). It was NOT dropped. - scheduler.process_scheduler_input(SchedulerInput::Txn(txn1.clone())); + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(txn1.clone()))); let live_owner = TxnId::new(1, 0); assert!( scheduler.blocked.contains_key(&live_owner), diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs index 11c773826..6ca3b632a 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_redo.rs @@ -8,11 +8,10 @@ //! drives the resolve step: dispatch `MetaOp::CalvinResolve` to reconstitute //! the staged post-images as one replayable `RedoRecord`, WAL-append that //! record (restoring restart durability for this vShard's slice of the -//! commit), then hand off to `dispatch_commit_resolution` for the flush that -//! `finish_resolved_commit` / `commit_apply_tail` complete. +//! commit), then queue the flush for its sequencer-order turn +//! ([`super::flush_turn`]); `finish_resolved_commit` / `commit_apply_tail` +//! complete it. -use super::super::types::CommitState; -use super::commit_resolution_dispatch::CommitResolution; use super::deferred::{DispatchOutcome, DispatchStep}; use super::halt::{HaltReason, HaltStep, error_response_text}; use super::scheduler::Scheduler; @@ -26,10 +25,12 @@ use nodedb_physical::physical_plan::meta::MetaOp; impl Scheduler { /// Handle the `MetaOp::CalvinResolve` response: decode the resolved - /// `RedoRecord`, WAL-append it (unless its op set is empty), then dispatch - /// the flush that installs it, stamped with that record's LSN. + /// `RedoRecord`, attach the transaction's messages on the participant that + /// carries them, WAL-append it (unless it holds no op and no message), + /// then dispatch the flush that installs it, stamped with that record's + /// LSN. /// - /// The verdict is already COMMIT, so a skipped resolve would tear the + /// The verdict is already COMMIT, so a skipped resolve will tear the /// committed txn on this replica. A non-`Ok` response, a decode failure, /// or a WAL-append failure halts the scheduler: the txn keeps its /// `pending` entry and locks, and its position stays unapplied. @@ -61,7 +62,7 @@ impl Scheduler { } }; let Some(pending) = self.pending.get(&txn_id) else { - // Txn state was reclaimed out from under us (should not happen — + // Txn state was reclaimed out from under us (must not happen — // locks are held until `on_txn_complete`); complete defensively. self.metrics.record_completed(); self.on_txn_complete(txn_id); @@ -69,6 +70,11 @@ impl Scheduler { }; let tenant_id = pending.txn.tx_class.tenant_id; let database_id = pending.txn.tx_class.database_id; + let event_source = + super::request::slice_event_source(&pending.txn.tx_class, &pending.flush_scope); + let commit_hlc = self + .cut_floors + .commit_hlc(pending.txn.epoch, pending.txn.epoch_system_ms); // The stamp carries what the slice folds, so the live install and // restart replay fold at this record's LSN. redo.calvin_stamp = Some(CalvinStamp { @@ -78,10 +84,80 @@ impl Scheduler { collections: pending.flush_scope.collections.clone(), sum_targets: pending.flush_scope.sum_targets.clone(), }); + // The participants whose records carry the messages and the applied + // key. A class whose write vShards cannot be derived halts the resolve. + let tx_class = &pending.txn.tx_class; + let homes = tx_class + .publish_vshard() + .and_then(|publish| Ok((publish, tx_class.applied_key_home()?))); + let (publish_home, applied_key_home) = match homes { + Ok(homes) => homes, + Err(e) => { + self.halt_apply( + txn_id, + HaltReason::ResolveFailed, + HaltStep::Resolve, + format!("Calvin transaction write vShards underivable: {e}"), + ); + return; + } + }; + // One participant's record carries the messages the transaction's + // trigger bodies published, so they commit once, with its writes. + if publish_home == Some(self.vshard_id) { + match crate::wal::RedoPublish::decode_all(&pending.txn.tx_class.publishes) { + Ok(mut publishes) => { + // Named by the transaction's sequencer position on its + // vShard's Calvin partition, the same on every replica. + crate::wal::RedoPublish::stamp_all( + &mut publishes, + crate::wal::PublishPosition { + partition: crate::event::cdc::position::calvin_partition( + self.vshard_id, + ), + epoch: 0, + index: txn_id.epoch, + base: u64::from(txn_id.position), + }, + ); + redo.publishes = publishes; + } + Err(e) => { + self.halt_apply( + txn_id, + HaltReason::ResolveFailed, + HaltStep::Resolve, + format!("Calvin transaction publishes decode failed: {e}"), + ); + return; + } + } + } + // One participant's record carries the dedup key of the cross-shard + // request the transaction applies, so the key is durable exactly when + // the request's writes are. + if applied_key_home == Some(self.vshard_id) { + match zerompk::from_msgpack::( + &pending.txn.tx_class.applied_key, + ) { + Ok(key) => redo.cross_shard_applied = Some(key), + Err(e) => { + self.halt_apply( + txn_id, + HaltReason::ResolveFailed, + HaltStep::Resolve, + format!("Calvin transaction applied key decode failed: {e}"), + ); + return; + } + } + } + let installs_nothing = + redo.ops.is_empty() && redo.publishes.is_empty() && redo.cross_shard_applied.is_none(); // The flush installs these exact bytes, the payload of the record // appended below. - let redo_bytes = if redo.ops.is_empty() { + let redo_bytes = if installs_nothing { Vec::new() } else { match redo.to_bytes() { @@ -100,14 +176,18 @@ impl Scheduler { // The record's outcome-floor window opens before the append. It stays // with the pending txn until the flush completes. - let (redo_lsn, redo_records) = if redo.ops.is_empty() { + let (redo_lsn, redo_records) = if installs_nothing { (None, None) } else { let records = MintedRecords::open(&self.shared.outcome_floor); + // The flush reports no rows beyond the record's own: restart + // replay folds the sum targets from the record's stamp. The record + // is therefore whole at append, and no part follows it. let appended = records .appender(&self.shared.wal, crate::wal::manager::NO_APPLY_KEY) - .with_event_source(super::request::CALVIN_EVENT_SOURCE) - .append_transaction_redo( + .with_event_source(event_source) + .with_commit_hlc(commit_hlc) + .append_whole_transaction_redo( tenant_id, VShardId::new(self.vshard_id), database_id, @@ -118,6 +198,19 @@ impl Scheduler { // The txn committed, so its redo record is never // cancelled. The flush closes it from its outcome. records.mark_sent(); + // Before the flush, so the position exists before any of + // the txn's change events reach the Event Plane. + self.shared.cdc_router.positions().record_calvin( + lsn.as_u64(), + crate::event::cdc::position::CalvinPosition { + sequencer_epoch: txn_id.epoch, + position: txn_id.position, + }, + ); + self.shared + .cdc_router + .positions() + .record_commit_hlc(lsn.as_u64(), commit_hlc); (Some(lsn), Some(records)) } Err(e) => { @@ -136,24 +229,21 @@ impl Scheduler { }; if let Some(pending) = self.pending.get_mut(&txn_id) { pending.redo_records = redo_records; + // The commit publishes the rows its redo installs, as a + // data-group `TransactionRedo` apply does. A rolled-back + // transaction never resolves, so it publishes nothing. + pending.change_sets = if redo_bytes.is_empty() { + Vec::new() + } else { + vec![crate::control::server::dispatch_utils::redo_change_set( + &redo_bytes, + )] + }; pending.flush_scope.redo = redo_bytes; } - // A flush refused at capacity is parked for re-send. The txn awaits its - // flush response either way, so the state below is the same. - if let DispatchOutcome::Failed(error) = - self.dispatch_commit_resolution(txn_id, CommitResolution::Flush { redo_lsn }) - { - self.fail_dispatch_step(txn_id, DispatchStep::Flush, error); - return; - } - - if let Some(pending) = self.pending.get_mut(&txn_id) { - pending.commit_state = Some(CommitState::AwaitingResolve { - committed: true, - redo_lsn, - }); - } + // The flush runs in sequencer order, once every lower txn finished. + self.queue_flush(txn_id, redo_lsn); } /// Dispatch `MetaOp::CalvinResolve` to this vShard's core, registering a @@ -205,6 +295,7 @@ pub(in crate::control::cluster::calvin::scheduler::driver::core) fn missing_pend #[cfg(test)] mod tests { + use super::super::super::types::CommitState; use super::*; use crate::bridge::envelope::{ErrorCode, Payload}; use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ @@ -212,7 +303,7 @@ mod tests { }; /// A resolve that returns an error under a COMMIT verdict holds the txn - /// unapplied and halts: skipping it would tear the committed txn. + /// unapplied and halts: skipping it will tear the committed txn. #[tokio::test] async fn resolve_error_response_holds_committed_txn_unapplied() { let txn_id = TxnId::new(8, 0); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs index 3eacfeceb..228f0a329 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolution_dispatch.rs @@ -8,6 +8,7 @@ use nodedb_physical::physical_plan::meta::MetaOp; use super::commit_redo::missing_pending_error; use super::deferred::{DispatchOutcome, DispatchStep}; use super::scheduler::Scheduler; +use crate::bridge::dispatch::JournalGroup; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; use crate::types::Lsn; @@ -31,23 +32,46 @@ impl Scheduler { /// A capacity refusal returns [`DispatchOutcome::Deferred`]: the flush or /// drop is parked for re-send and the txn stays in flight. A txn with no /// `pending` entry returns [`DispatchOutcome::Failed`]. The flush takes its - /// collections and sum targets from the scope derived at stage time. + /// collections and sum targets from the scope derived at stage time. A + /// flush whose gates a halt released takes them back first, and fails when + /// a collection it writes no longer holds its planned incarnation. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn dispatch_commit_resolution( &mut self, txn_id: TxnId, resolution: CommitResolution, ) -> DispatchOutcome { + // A flush writes the collections: it holds their gates while it runs. + if matches!(resolution, CommitResolution::Flush { .. }) + && let Err(error) = self.regate(txn_id) + { + return DispatchOutcome::Failed(error); + } let Some(pending) = self.pending.get_mut(&txn_id) else { return DispatchOutcome::Failed(missing_pending_error(txn_id)); }; let tenant_id = pending.txn.tx_class.tenant_id; let database_id = pending.txn.tx_class.database_id; + let event_source = + super::request::slice_event_source(&pending.txn.tx_class, &pending.flush_scope); + let commit_hlc = self + .cut_floors + .commit_hlc(pending.txn.epoch, pending.txn.epoch_system_ms); let epoch = txn_id.epoch; let position = txn_id.position; - let (plan, step, wal_lsn) = match resolution { + let (plan, step, wal_lsn, journal) = match resolution { CommitResolution::Flush { redo_lsn } => { pending.flush_scope.sends = pending.flush_scope.sends.saturating_add(1); let scope = &pending.flush_scope; + // The core stores the flush's append inputs beside its + // effects for boot. The record is whole at append, so no part + // follows it. + let journal = redo_lsn.map(|origin| JournalGroup { + origin, + collection: scope.collections.first().cloned().unwrap_or_default(), + apply_key: crate::wal::manager::NO_APPLY_KEY, + commit_hlc: Some(commit_hlc), + change_position: None, + }); ( PhysicalPlan::Meta(MetaOp::CalvinFlush { epoch, @@ -58,16 +82,23 @@ impl Scheduler { }), DispatchStep::Flush, redo_lsn, + journal, ) } CommitResolution::Drop => ( PhysicalPlan::Meta(MetaOp::CalvinDrop { epoch, position }), DispatchStep::Drop, None, + None, ), }; let request_id = self.next_request_id(); - let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, wal_lsn); - self.dispatch_sequenced(txn_id, step, request) + let mut request = + self.build_exempt_request(request_id, tenant_id, database_id, plan, wal_lsn); + // The flush emits the transaction's events, under its own source and + // dated by its commit HLC. + request.event_source = event_source; + request.commit_hlc = Some(commit_hlc); + self.dispatch_sequenced_journalled(txn_id, step, request, journal) } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs index 48a491c9a..940ba2974 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/apply_tail.rs @@ -76,7 +76,11 @@ impl Scheduler { let completed = if committed { self.commit_apply_tail(txn_id, response, redo_lsn).await } else { - self.propose_sequencer_entry(txn_id, SchedulerProposal::CompletionAck); + // A dropped txn installed nothing, so its ack reports nothing. + self.propose_sequencer_entry( + txn_id, + SchedulerProposal::CompletionAck { result: Vec::new() }, + ); true }; // `false` means the commit tail halted the scheduler: the txn stays @@ -167,13 +171,22 @@ impl Scheduler { // it for a COMMIT tag anyway), and the plain-write siblings do not // conflict. Only a genuine cross-shard RETURNING union — two // participants each carrying RETURNING rows — records `Conflict`. - // Results travel via this in-process sidecar only — never the sequencer - // Raft log. + // The primary slice's answer, its RETURNING rows or its affected + // count, and the timeseries install counts also ride the replicated + // CompletionAck, so a coordinator on a node with no replica of this + // vShard reads them. let (has_primary_write, has_returning) = self .pending .get(&txn_id) .map(|p| (p.has_primary_write, p.has_returning)) .unwrap_or((false, false)); + let ack_result = self.ack_result_of( + &response, + crate::control::state::AckSlice { + primary_write: has_primary_write, + returning: has_returning, + }, + ); if has_primary_write { use std::collections::hash_map::Entry; @@ -182,13 +195,8 @@ impl Scheduler { use crate::control::state::CalvinApplyResult; let key = nodedb_cluster::calvin::TxnId::new(txn_id.epoch, txn_id.position); - let mut results = self - .shared - .calvin - .apply_results - .lock() - .unwrap_or_else(|p| p.into_inner()); - match results.entry(key) { + let vshard_id = self.vshard_id; + self.shared.calvin.apply_results.deposit_with(key, |entry| match entry { Entry::Vacant(slot) => { slot.insert(CalvinApplyResult::Single { response, @@ -218,7 +226,7 @@ impl Scheduler { tracing::error!( epoch = txn_id.epoch, position = txn_id.position, - vshard = self.vshard_id, + vshard = vshard_id, "two RETURNING-bearing participants for one Calvin txn — cross-shard \ RETURNING union unsupported" ); @@ -234,9 +242,12 @@ impl Scheduler { // Incoming is a plain write; keep the existing entry — a // multi-collection cross-shard COMMIT coalesces (the // coordinator discards it for a COMMIT tag anyway). + // The timeseries install counts of both merge, so the + // COMMIT reports every participant's apply count. + merge_install_counts(slot.get_mut(), &response); } } - } + }); } let applied_lsn = match redo_lsn { // The TransactionRedo record already durably marks this apply — the @@ -251,6 +262,7 @@ impl Scheduler { .shared .wal .appender(crate::wal::manager::NO_APPLY_KEY) + .with_commit_hlc(self.txn_commit_hlc(txn_id)) .append_calvin_applied( crate::types::VShardId::new(self.vshard_id), txn_id.epoch, @@ -277,17 +289,52 @@ impl Scheduler { }; let Some(lsn) = applied_lsn else { // The apply cannot be acknowledged without a durable participant - // LSN: CDC and write-version consumers would otherwise observe a + // LSN: CDC and write-version consumers will otherwise observe a // successful commit with no authoritative ordering point. The // scheduler halted above. return false; }; + // The record is whole at append: the flush reports no rows, so no + // part follows it. A record split over the WAL record limit is + // durable once its last continuation is. + let durable_through = self + .pending + .get(&txn_id) + .and_then(|pending| pending.redo_records.as_ref()) + .and_then(|records| records.last_lsn()) + .map_or(lsn, |last| last.max(lsn)); + // Control change-stream events are distinct from Data-Plane + // WriteEvents. Every replica publishes the rows this participant's + // redo installs at the transaction's sequencer position, which every + // replica shares; the vShard's leader forwards them to the nodes + // that do not replicate it. The publish journals them durably before + // the applied marker's fsync below, so a restart that finds the + // position applied also finds its changes. + if let Some(pending) = self.pending.get_mut(&txn_id) { + let calvin = crate::control::server::dispatch_utils::CalvinApply { + tenant_id: pending.txn.tx_class.tenant_id, + database_id: pending.txn.tx_class.database_id, + vshard: self.vshard_id, + sequencer_epoch: txn_id.epoch, + position: txn_id.position, + commit_hlc: self + .cut_floors + .commit_hlc(pending.txn.epoch, pending.txn.epoch_system_ms), + }; + let change_sets = std::mem::take(&mut pending.change_sets); + crate::control::server::dispatch_utils::publish_calvin_change_sets( + &self.shared, + calvin, + change_sets, + lsn, + ); + } // The record at `lsn` is this position's only applied marker. An // append only buffers it, so it is durable before the mark and the - // ack: a restart that lost it would take the position for unapplied, + // ack: a restart that lost it will take the position for unapplied, // run the transaction again, and never settle its ack. The wait joins // the WAL group commit, so concurrent Calvin commits share one fsync. - if let Err(e) = self.shared.wal.wait_durable(lsn).await { + if let Err(e) = self.shared.wal.wait_durable(durable_through).await { self.halt_apply( txn_id, HaltReason::WalAppendFailed, @@ -296,35 +343,107 @@ impl Scheduler { ); return false; } - // Control change-stream events are distinct from Data-Plane - // WriteEvents. Publish the participant-local logical manifests once, - // from the data-group leader, at the authoritative committed LSN. - if self.is_group_leader() - && let Some(pending) = self.pending.get_mut(&txn_id) - { - let tenant_id = pending.txn.tx_class.tenant_id; - let database_id = pending.txn.tx_class.database_id; - for change_set in std::mem::take(&mut pending.change_sets) { - crate::control::server::dispatch_utils::publish_change_set_with_lsn( - &self.shared, - tenant_id, - database_id, - change_set, - lsn, - ); - } + if let Some(origin) = redo_lsn { + self.note_flush_settled(origin); + self.record_applied_key(txn_id); } // The commit's mark lands before the ack, as a write through the // funnel records its mark before its response returns. self.record_calvin_write_mark(txn_id); - self.propose_sequencer_entry(txn_id, SchedulerProposal::CompletionAck); + self.propose_sequencer_entry( + txn_id, + SchedulerProposal::CompletionAck { result: ack_result }, + ); true } } +impl Scheduler { + /// Record the dedup key this participant's redo record carries, once the + /// record is durable and installed. Every replica records it, as a + /// Raft-applied redo's key is, so a request or trigger body sent again to + /// any replica applies once. + fn record_applied_key(&self, txn_id: TxnId) { + let Some(pending) = self.pending.get(&txn_id) else { + return; + }; + let tx_class = &pending.txn.tx_class; + if tx_class.applied_key.is_empty() + || tx_class.applied_key_home().ok().flatten() != Some(self.vshard_id) + { + return; + } + let Some(dedup) = self.shared.cross_shard_dedup.get() else { + return; + }; + match zerompk::from_msgpack::(&tx_class.applied_key) { + Ok(key) => { + if let Err(error) = dedup.record_applied(&key) { + tracing::error!( + vshard_id = self.vshard_id, + origin = %key.origin, + %error, + "calvin: dedup key held in memory only; the WAL restores it on restart" + ); + } + } + Err(error) => tracing::error!( + vshard_id = self.vshard_id, + %error, + "calvin: the applied key does not decode; it is not recorded" + ), + } + } + + /// The report the flush `response` owes the coordinator, encoded for the + /// replicated `CompletionAck`: the apply's timeseries install counts, and + /// the rows of a `returning` slice. The rows are bounded by the query + /// result limit and by the ack's frame budget, whichever is lower. + fn ack_result_of( + &self, + response: &Response, + slice: crate::control::state::AckSlice, + ) -> Vec { + let limit = self + .shared + .tuning + .network + .max_query_result_bytes + .min(crate::control::state::ACK_ROWS_FRAME_BUDGET); + crate::control::state::CalvinAckResult::of(response, slice, limit) + .to_bytes() + .unwrap_or_else(|error| { + tracing::warn!( + vshard_id = self.vshard_id, + %error, + "calvin: the completion ack carries no report" + ); + Vec::new() + }) + } +} + /// The response the statement drains. A flush whose install succeeded but /// whose reply failed to render answers `Ok` with the render error in /// `error_code`. The statement reports that error. +/// Fold the timeseries install counts `incoming` answers into the plain +/// result `held`. A RETURNING result keeps its rows untouched. +fn merge_install_counts(held: &mut crate::control::state::CalvinApplyResult, incoming: &Response) { + let crate::control::state::CalvinApplyResult::Single { + response, + has_returning: false, + } = held + else { + return; + }; + if let Some(merged) = crate::engine::timeseries::install_counts::merge_count_payloads( + response.payload.as_bytes(), + incoming.payload.as_bytes(), + ) { + response.payload = crate::bridge::envelope::Payload::from_vec(merged); + } +} + fn statement_reply(mut response: Response) -> Response { if response.status == Status::Ok && response.error_code.is_some() { response.status = Status::Error; @@ -337,7 +456,7 @@ fn statement_reply(mut response: Response) -> Response { mod tests { use super::*; use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ - error_response, scheduler_with_pending, + error_response, scheduler_with_pending, staged_response, }; use crate::control::cluster::calvin::scheduler::driver::types::CommitState; @@ -347,8 +466,60 @@ mod tests { }) } + fn flushed_with_counts(collection: &str, accepted: u64, rejected: u64) -> Response { + use crate::engine::timeseries::install_counts::{TsInstallCount, TsInstallCounts}; + let mut response = staged_response(Status::Ok, None); + response.payload = crate::bridge::envelope::Payload::from_vec( + TsInstallCounts::new(vec![TsInstallCount { + collection: collection.into(), + accepted, + rejected, + }]) + .to_bytes() + .expect("encode install counts"), + ); + response + } + + /// Two plain participants of one Calvin transaction each installed a + /// timeseries batch. The coordinator's result carries both installs' + /// apply counts, so COMMIT reports every participant's rejected rows. A + /// RETURNING result keeps its rows. + #[test] + fn plain_participants_merge_their_install_counts() { + use crate::control::state::CalvinApplyResult; + use crate::engine::timeseries::install_counts::TsInstallCounts; + let mut held = CalvinApplyResult::Single { + response: flushed_with_counts("cpu", 2, 1), + has_returning: false, + }; + merge_install_counts(&mut held, &flushed_with_counts("mem", 3, 2)); + let CalvinApplyResult::Single { response, .. } = &held else { + panic!("a merged result stays single"); + }; + let totals = TsInstallCounts::from_payload(response.payload.as_bytes()) + .expect("install counts") + .by_collection(); + assert_eq!(totals.get("cpu"), Some(&(2, 1))); + assert_eq!(totals.get("mem"), Some(&(3, 2))); + + let rows = crate::bridge::envelope::Payload::from_vec(vec![0x90]); + let mut returning = CalvinApplyResult::Single { + response: Response { + payload: rows, + ..flushed_with_counts("cpu", 1, 0) + }, + has_returning: true, + }; + merge_install_counts(&mut returning, &flushed_with_counts("mem", 1, 1)); + let CalvinApplyResult::Single { response, .. } = &returning else { + panic!("a RETURNING result stays single"); + }; + assert_eq!(response.payload.as_bytes(), &[0x90]); + } + /// A flush that returns an error under a COMMIT verdict holds the txn - /// unapplied and halts: a second flush would apply nothing. + /// unapplied and halts: a second flush will apply nothing. #[tokio::test] async fn flush_error_response_holds_committed_txn_unapplied() { let txn_id = TxnId::new(9, 2); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/flush_parts.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/flush_parts.rs new file mode 100644 index 000000000..7a830fc08 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/flush_parts.rs @@ -0,0 +1,28 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The stored write set of a committed Calvin flush. +//! +//! A committed Calvin record is whole at append: its flush reports no rows, +//! so no part follows it. The core still runs the flush journalled, which +//! stores the flush's append inputs beside its effects for boot. Once the +//! record is durable, the core drops what it stored. + +use crate::control::cluster::calvin::scheduler::driver::core::scheduler::Scheduler; +use crate::types::{Lsn, VShardId}; + +impl Scheduler { + /// Note that the record at `origin` is durable, so the core drops the + /// write set it stored. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn note_flush_settled( + &self, + origin: Lsn, + ) { + let vshard_id = VShardId::new(self.vshard_id); + match self.shared.dispatcher.lock() { + Ok(mut d) => d.note_write_set_settled(vshard_id, origin), + Err(poisoned) => poisoned + .into_inner() + .note_write_set_settled(vshard_id, origin), + } + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/mod.rs index be4a38649..69168ac82 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/mod.rs @@ -13,5 +13,6 @@ //! drop. mod apply_tail; +mod flush_parts; mod verdict; mod vote; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs index e396e1261..2168b5c19 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/verdict.rs @@ -67,6 +67,23 @@ impl Scheduler { return; } + // A slice that truncates rows resolves at its turn, once every lower + // txn of this vShard finished. + if committed + && let Some(pending) = self.pending.get_mut(&txn_id) + && pending.flush_scope.resolve_at_turn + { + pending.commit_state = Some(CommitState::AwaitingResolveTurn); + pending.verdict_deadline = None; + self.shared + .calvin + .counters + .commits_flushed + .fetch_add(1, Ordering::Relaxed); + self.pump_flush_turn(); + return; + } + let (outcome, step) = if committed { // Resolve the staged post-images into a replayable `RedoRecord` // first; the redo is WAL-appended (in `finish_redo_resolve`) before @@ -142,7 +159,7 @@ impl Scheduler { /// emit a stall metric + warning, and re-arm the deadline so the warning is /// rate-limited rather than per-iteration. It NEVER releases locks and NEVER /// unilaterally aborts: a participant cannot know whether a peer already - /// flushed a COMMIT, so aborting one side while a peer committed would tear + /// flushed a COMMIT, so aborting one side while a peer committed will tear /// the transaction. The verdict is guaranteed to arrive eventually — a /// post-failover leader re-aggregates the replicated votes (seeded on every /// replica) into the same verdict — so waiting is always the safe action. @@ -245,6 +262,10 @@ mod tests { version: 1, ops: Vec::new(), calvin_stamp: None, + cross_shard_applied: None, + row_sources: Vec::new(), + publishes: Vec::new(), + row_changes: Vec::new(), }; let mut response = staged_response(Status::Ok, None); response.payload = Payload::from_vec(redo.to_bytes().expect("encode empty redo record")); @@ -403,7 +424,7 @@ mod tests { .insert(txn_id, staged_pending(make_sequenced_txn(14, 2), txn_id)); // Local staging votes only park their own staged slices; neither the - // affirmative nor the failed participant may resolve or drop unilaterally. + // affirmative nor the failed participant can resolve or drop unilaterally. first_scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Ok, Some(true))); second_scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Error, None)); for (scheduler, data_side) in [ @@ -428,18 +449,18 @@ mod tests { registry.note_vote( txn, 9, - ParticipantVote::Abort(Some(AbortReason::SerializationConflict)), + ParticipantVote::Abort(AbortReason::SerializationConflict), ); assert_eq!( registry.drain_unproposed_verdicts(), vec![( txn, - VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)) + VerdictOutcome::Abort(AbortReason::SerializationConflict) )] ); registry.note_verdict( txn, - VerdictOutcome::Abort(Some(AbortReason::SerializationConflict)), + VerdictOutcome::Abort(AbortReason::SerializationConflict), ); assert_eq!(registry.verdict(txn), Some(false)); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs index be5738a4e..2a837fe53 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/commit_resolve/vote.rs @@ -42,27 +42,46 @@ impl Scheduler { staged_response: &Response, ) { // A staged error is always an abort vote. Only successful staged - // responses may use `None` for the dependent-read path; accepting an - // error-plus-None as commit would let a failed participant flush after + // responses can use `None` for the dependent-read path; accepting an + // error-plus-None as commit will let a failed participant flush after // its peers received a global commit verdict. - let vote = staged_commit_vote(staged_response); + let superseded = self + .pending + .get(&txn_id) + .is_some_and(|pending| pending.superseded); + let vote = if superseded { + StagedVote::CollectionSuperseded + } else { + match staged_commit_vote(staged_response) { + // A read another node served is numbered in that node's WAL: + // this node's versions cannot show it still current. + StagedVote::Commit if self.validates_read_served_elsewhere(txn_id) => { + StagedVote::SerializationConflict + } + vote => vote, + } + }; // Durably propose this participant's commit vote via the sequencer - // Raft group, leader-guarded like `OllpMismatch`: only the data-group - // leader ran read-set validation, so only a leader's vote is + // Raft group, leader-guarded: only the data-group leader ran + // read-set validation, so only a leader's vote is // authoritative. The sequencer aggregates every participant's vote into // the single global verdict this txn parks on below. An abort travels as // `AbortVote` so its cause survives to the coordinator. The vote stays // owed until the tally holds it, so a refused or dropped proposal is // proposed again rather than lost. - if self.is_group_leader() { - self.propose_sequencer_entry( - txn_id, - SchedulerProposal::Vote { - abort: vote.abort_reason(), - }, - ); - } + // + // A follower owes an abort vote and proposes it only if it comes to + // lead before any vote of this vShard applies: the previous leader + // lost its leadership, or its scheduler cannot run. A follower ran + // none of the leader-only checks, so abort is the only vote it can + // cast. The txn fails with a retryable conflict and never hangs. + let abort = if self.is_group_leader() { + vote.abort_reason() + } else { + Some(nodedb_cluster::calvin::AbortReason::SerializationConflict) + }; + self.propose_sequencer_entry(txn_id, SchedulerProposal::Vote { abort }); if vote == StagedVote::SerializationConflict { // The staged slice's read-set was no longer current: observe it, the @@ -88,15 +107,25 @@ impl Scheduler { match self.pending.get_mut(&txn_id) { Some(pending) => { pending.commit_state = Some(CommitState::AwaitingVerdict); - pending.stage_error = (vote == StagedVote::ParticipantError) - .then(|| error_response_text("stage", staged_response)); + // A superseded slice cannot flush here: its collection is gone. + pending.stage_error = match vote { + StagedVote::ParticipantError | StagedVote::PredictionDrift => { + Some(error_response_text("stage", staged_response)) + } + StagedVote::CollectionSuperseded => Some( + "a collection the transaction names no longer holds its planned \ + incarnation" + .to_owned(), + ), + StagedVote::Commit | StagedVote::SerializationConflict => None, + }; // no-determinism: local stall-warning deadline only; the global replicated verdict, not this wall-clock, decides commit/abort. pending.verdict_deadline = Some(Instant::now() + self.config.verdict_stall_warn()); } None => return, } - // PROBE on park (correctness backstop): the verdict may already be + // PROBE on park (correctness backstop): the verdict can already be // durable — on replay, or a push that raced ahead of this park. Resume // immediately if so; the double-resume guard in `resume_on_verdict` // makes a later duplicate push/probe a no-op. @@ -107,4 +136,20 @@ impl Scheduler { self.resume_on_verdict(txn_id, verdict); } } + + /// Whether a read this vShard validates for `txn_id` was served by a node + /// other than this one. Its `read_lsn` is a position in that node's WAL. + fn validates_read_served_elsewhere(&self, txn_id: TxnId) -> bool { + let Some(pending) = self.pending.get(&txn_id) else { + return false; + }; + let tx_class = &pending.txn.tx_class; + tx_class.versioned_reads.iter().any(|entry| { + super::super::routing::versioned_read_homes_on( + entry, + tx_class.database_id, + self.vshard_id, + ) && entry.served_by != self.shared.node_id + }) + } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs index 02f9eb0fb..aee1145f2 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/completion_route.rs @@ -63,7 +63,7 @@ impl Scheduler { let commit_state = self.pending.get(&txn_id).and_then(|p| p.commit_state); // OLLP mismatch: the active executor detected predicate drift and returned - // OllpRetryRequired without writing. The retry loop is now COORDINATOR-owned + // OllpRetryRequired without writing. The retry loop is COORDINATOR-owned // (`run_dependent_with_retry`): the scheduler must NOT re-submit a stale // prediction. Instead it (1) releases the aborted attempt's locks and // (2) signals the coordinator's completion waiter via the registry so it @@ -76,11 +76,10 @@ impl Scheduler { // A mismatch is a normal OLLP retry signal, not a failure: the executor // correctly detected predicate drift and declined to write. Count it as // a received executor response, but NOT as an executor error or infra - // abort — those would inflate failure metrics on every routine retry. + // abort — those will inflate failure metrics on every routine retry. self.metrics.record_completed(); // (1) Release the aborted attempt's locks and clean up pending state. - // This fixes the lock-leak: the old `schedule_ollp_retry` re-submitted - // without ever releasing the aborted attempt's locks. + // A retry that left the aborted attempt's locks held will leak them. self.on_txn_complete(txn_id); // OLLP mismatch broadcast is LEADER-ONLY. The optimistic-lock // verification runs only on the data-group leader (the data plane @@ -88,9 +87,9 @@ impl Scheduler { // bulk-DML handlers), so by construction only a leader's executor can // return `OllpRetryRequired`. This guard is defense-in-depth: a // non-leader scheduler must never broadcast a mismatch — a lagging - // follower could otherwise poison an attempt the leader already + // follower can otherwise poison an attempt the leader already // completed, exhausting retries on a static dataset. A non-leader - // simply releases locks (done above) and returns; the leader owns the + // releases locks (done above) and returns; the leader owns the // single mismatch signal that the completion registry observes. if !self.is_group_leader() { tracing::debug!( @@ -137,12 +136,24 @@ impl Scheduler { .await; return; } + Some(CommitState::AwaitingFlushTurn { .. } | CommitState::AwaitingResolveTurn) => { + // A txn waiting for its flush or resolve turn has no request + // out. + tracing::warn!( + vshard_id = self.vshard_id, + request_id = request_id.as_u64(), + epoch = txn_id.epoch, + position = txn_id.position, + "calvin: executor completion for a txn waiting for its flush turn; ignoring" + ); + return; + } Some(CommitState::AwaitingVerdict) => { // A parked txn dispatched no executor request (it is waiting on // the cross-shard verdict, not the Data Plane), so no response // bridge is outstanding for it — a completion here indicates a // logic error, not a real Data-Plane reply. Do NOT run the commit - // tail or release locks (that could tear the transaction); + // tail or release locks (that can tear the transaction); // remain parked and let the verdict path resume it. tracing::warn!( vshard_id = self.vshard_id, diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/cut_marker.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/cut_marker.rs index d93be00c0..00565b04e 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/cut_marker.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/cut_marker.rs @@ -45,6 +45,21 @@ impl Scheduler { self.cut_floors.fold(self.applied.fully_applied_epoch()); } + /// The commit HLC of the transaction `txn_id`: the instant every WAL + /// record of its install carries. A transaction whose state is gone takes + /// the node's HLC now. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn txn_commit_hlc( + &self, + txn_id: TxnId, + ) -> u64 { + match self.pending.get(&txn_id) { + Some(pending) => self + .cut_floors + .commit_hlc(pending.txn.epoch, pending.txn.epoch_system_ms), + None => self.shared.hlc_clock.now().wall_ns, + } + } + /// Record the commit HLC of the committed transaction `txn_id` on its /// tenant's observed write high-water, before its `CompletionAck` lets /// the coordinator acknowledge the COMMIT. @@ -52,6 +67,11 @@ impl Scheduler { /// Only a slice that carries a primary user data write records it. The /// implicit edge cleanup that dual-homes alongside one writes no user row /// of its own. + /// + /// A transaction that re-issues a RESTORE raises the tenant's restore + /// mark under its restore id instead, so a retry of that RESTORE finds + /// only its own marks. Its commit HLC still folds into this node's clock, + /// so a later backup here stamps a newer watermark. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn record_calvin_write_mark( &self, txn_id: TxnId, @@ -64,12 +84,19 @@ impl Scheduler { } let tx_class = &pending.txn.tx_class; let tenant_id = tx_class.tenant_id.as_u64(); + let restore_id = tx_class.restore_id; let collection = tx_class.write_set.0.first().map(|set| set.collection()); let commit_hlc = self .cut_floors .commit_hlc(pending.txn.epoch, pending.txn.epoch_system_ms); - self.shared - .advance_tenant_write_hlc(tenant_id, commit_hlc, "calvin flush", collection); + if restore_id == 0 { + self.shared + .advance_tenant_write_hlc(tenant_id, commit_hlc, "calvin flush", collection); + } else { + self.shared + .hlc_clock + .update(nodedb_types::Hlc::new(commit_hlc, 0)); + } // The durable mark, in the data group that homes this vShard, lands // before the ack too: RESTORE's guard reads it on every node, after a @@ -87,13 +114,17 @@ impl Scheduler { } }; let marks = &self.shared.tenant_marks; - marks.raise( - group_id, - tenant_id, - commit_hlc, - MarkSite::CalvinFlush, - collection, - ); + if restore_id == 0 { + marks.raise( + group_id, + tenant_id, + commit_hlc, + MarkSite::CalvinFlush, + collection, + ); + } else { + marks.raise_restore(group_id, tenant_id, commit_hlc, collection, restore_id); + } if let Err(error) = marks.persist(self.shared.credentials.catalog()) { // The mark stays pending; the next persist on this node writes it. tracing::error!( diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs index 4173b8b7f..a42623686 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/deferred.rs @@ -18,7 +18,7 @@ use nodedb_physical::physical_plan::meta::MetaOp; use super::halt::{HaltReason, HaltStep}; use super::scheduler::Scheduler; -use crate::bridge::dispatch::DispatchRefusal; +use crate::bridge::dispatch::{DispatchRefusal, JournalGroup}; use crate::bridge::envelope::Request; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; use crate::types::RequestId; @@ -57,6 +57,8 @@ pub(in crate::control::cluster::calvin::scheduler::driver::core) struct Deferred txn_id: TxnId, step: DispatchStep, request: Request, + /// The record group the request's write set journals into. + journal: Option, } /// FIFO of requests refused at capacity, in refusal order. @@ -85,7 +87,19 @@ impl Scheduler { step: DispatchStep, request: Request, ) -> DispatchOutcome { - match self.send_once(txn_id, step, request) { + self.dispatch_sequenced_journalled(txn_id, step, request, None) + } + + /// [`Self::dispatch_sequenced`] for a request whose write set journals + /// into `journal`. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn dispatch_sequenced_journalled( + &mut self, + txn_id: TxnId, + step: DispatchStep, + request: Request, + journal: Option, + ) -> DispatchOutcome { + match self.send_once(txn_id, step, request, journal) { Attempt::Sent => DispatchOutcome::Sent, Attempt::Capacity(parked) => { self.deferred.push_back(*parked); @@ -112,6 +126,14 @@ impl Scheduler { self.has_deferred_dispatch() && !self.is_apply_halted() } + /// Whether a refused request of `txn_id` waits for capacity. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn has_parked_dispatch( + &self, + txn_id: TxnId, + ) -> bool { + self.deferred.iter().any(|parked| parked.txn_id == txn_id) + } + /// Number of refused requests waiting for capacity. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn deferred_dispatch_len( &self, @@ -132,6 +154,7 @@ impl Scheduler { txn_id, step, mut request, + journal, } = parked; if step != DispatchStep::WriteVersionRecord && !self.pending.contains_key(&txn_id) { tracing::error!( @@ -144,7 +167,7 @@ impl Scheduler { continue; } self.refresh_deferred_request(step, &mut request); - match self.send_once(txn_id, step, request) { + match self.send_once(txn_id, step, request, journal) { Attempt::Sent => {} Attempt::Capacity(parked) => { self.deferred.push_front(*parked); @@ -201,7 +224,13 @@ impl Scheduler { } /// One send attempt: register, dispatch, and cancel on refusal. - fn send_once(&mut self, txn_id: TxnId, step: DispatchStep, request: Request) -> Attempt { + fn send_once( + &mut self, + txn_id: TxnId, + step: DispatchStep, + request: Request, + journal: Option, + ) -> Attempt { // A crash test holds one collection's flush here: the redo record is // appended and no core holds the flush. The flush waits in the // re-send queue, as at capacity, so the scheduler keeps running. @@ -217,13 +246,16 @@ impl Scheduler { txn_id, step, request, + journal, })); } let request_id = request.request_id; let resp_rx = self.shared.tracker.register(request_id); let result = match self.shared.dispatcher.lock() { - Ok(mut dispatcher) => dispatcher.try_dispatch(request), - Err(poisoned) => poisoned.into_inner().try_dispatch(request), + Ok(mut dispatcher) => dispatcher.try_dispatch_journalled(request, journal.clone()), + Err(poisoned) => poisoned + .into_inner() + .try_dispatch_journalled(request, journal.clone()), }; let refusal = match result { Ok(()) => { @@ -249,6 +281,7 @@ impl Scheduler { txn_id, step, request, + journal, })) } other => Attempt::Failed(other), @@ -341,8 +374,10 @@ mod tests { begin_data_plane_drain(&scheduler.shared); let txn_id = TxnId::new(3, 0); - scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); - scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(4, 0))); + scheduler + .process_scheduler_input(SchedulerInput::Txn(Box::new(make_validate_only_txn(3, 0)))); + scheduler + .process_scheduler_input(SchedulerInput::Txn(Box::new(make_validate_only_txn(4, 0)))); assert!( !scheduler.applied.is_applied(3, 0), @@ -377,7 +412,8 @@ mod tests { let shared = Arc::clone(&scheduler.shared); fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); let txn_id = TxnId::new(3, 0); - scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + scheduler + .process_scheduler_input(SchedulerInput::Txn(Box::new(make_validate_only_txn(3, 0)))); assert!(scheduler.has_deferred_dispatch(), "the stage parks"); begin_data_plane_drain(&shared); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs index 2726df689..27058bed9 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/active_dispatch.rs @@ -14,8 +14,7 @@ use nodedb_physical::physical_plan::meta::MetaOp; use super::super::deferred::{DispatchOutcome, DispatchStep}; use super::super::scheduler::Scheduler; use super::primary_write::{ - participant_change_sets, plans_have_primary_write, plans_have_returning, - txn_has_non_derived_write, + plans_have_primary_write, plans_have_returning, txn_has_non_derived_write, }; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; @@ -50,7 +49,12 @@ impl Scheduler { return; } }; - let has_non_derived_write = txn_has_non_derived_write(&plans); + // An assembled multi-part txn carries only its local tasks. The + // whole-transaction fact comes from its manifest. + let has_non_derived_write = match &txn.tx_class.multi_part { + Some(manifest) => manifest.user_write, + None => txn_has_non_derived_write(&plans), + }; let mut plans = match self.local_calvin_plans(plans, txn.tx_class.database_id, epoch, position) { Ok(p) if !p.is_empty() => p, @@ -97,7 +101,6 @@ impl Scheduler { } let has_primary_write = plans_have_primary_write(&plans, has_non_derived_write); let has_returning = plans_have_returning(&plans); - let change_sets = participant_change_sets(&plans, tenant_id, self.vshard_id); let flush_scope = super::super::super::types::FlushScope::of_plans(&plans); let plan = PhysicalPlan::Meta(MetaOp::CalvinExecuteActive { epoch, @@ -117,6 +120,8 @@ impl Scheduler { // The txn enters `pending` before the dispatch, so a stage refused at // capacity stays in flight with its locks until the re-send. + // Every replica checks the transaction's collection incarnations. + let (superseded, gates) = self.check_incarnations(&txn.tx_class); self.pending.insert( txn_id, super::super::super::types::PendingTxn { @@ -126,7 +131,8 @@ impl Scheduler { dispatch_time: Instant::now(), has_primary_write, has_returning, - change_sets, + // The commit's resolved redo fills them. + change_sets: Vec::new(), // The dependent-read active path STAGES (leader-verify OLLP + // buffer, no base apply); its response drives the same // resolve → redo → flush as the static path, for @@ -139,6 +145,11 @@ impl Scheduler { // Set once a committed txn appends its redo record. redo_records: None, flush_scope, + superseded, + gates, + ungated: false, + // Taken when the flush dispatches. + install_permit: None, }, ); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/incarnation.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/incarnation.rs new file mode 100644 index 000000000..8ad904fb2 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/incarnation.rs @@ -0,0 +1,339 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The incarnation check a Calvin transaction passes when a replica stages it. +//! +//! The coordinator stamps the incarnation of every collection the transaction +//! names. Each replica checks all of them, not only its own slice's, against +//! its committed catalog under the collections' gates, held shared until the +//! transaction leaves `pending`. A mismatch makes the leader's vote an abort, +//! and the sequencer's global verdict then aborts every participant alike: no +//! part of the transaction applies. A follower with a mismatch under a COMMIT +//! verdict halts rather than flush into a recreated collection. +//! +//! A halt releases every gate no flush in flight needs, so a purge never waits +//! on a halted scheduler. A txn whose gates a halt released checks again, and +//! takes them back, before its flush. + +use std::collections::BTreeSet; + +use nodedb_cluster::calvin::types::TxClass; +use nodedb_types::{CollectionKey, Hlc}; +use tracing::error; + +use super::super::super::types::CommitState; +use super::super::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; +use crate::control::write_gate::{self, GateKey, SharedGate}; + +impl Scheduler { + /// Hold the gate of every collection `tx_class` names, and report whether + /// one no longer holds its planned incarnation on this replica. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn check_incarnations( + &self, + tx_class: &TxClass, + ) -> (bool, Vec) { + let tenant_id = tx_class.tenant_id.as_u64(); + let database_id = tx_class.database_id; + let keys: BTreeSet = tx_class + .incarnations + .iter() + .map(|named| { + let key = CollectionKey::from_qualified_str(database_id, &named.collection) + .unwrap_or_else(|_| CollectionKey::from_bare(database_id, &named.collection)); + GateKey::Collection { + database_id: key.database_id().as_u64(), + tenant_id, + name: key.name().to_string(), + } + }) + .collect(); + let mut superseded = false; + let mut gates = Vec::with_capacity(keys.len()); + for key in keys { + // A reclaim holds a gate exclusive only after the catalog dropped + // the collection's row, so its incarnation is gone. + match write_gate::try_shared(key) { + Some(gate) => gates.push(gate), + None => superseded = true, + } + } + let catalog = self.shared.credentials.catalog(); + for named in &tx_class.incarnations { + if named.incarnation == Hlc::ZERO { + continue; + } + match catalog.holds_incarnation( + database_id, + tenant_id, + &named.collection, + named.incarnation, + ) { + Ok(true) => {} + Ok(false) => superseded = true, + Err(e) => { + error!( + vshard_id = self.vshard_id, + collection = %named.collection, + error = %e, + "calvin scheduler: incarnation check could not read the catalog; \ + voting abort" + ); + superseded = true; + } + } + } + (superseded, gates) + } + + /// Release the gates a halt leaves nothing to protect. + /// + /// A halted scheduler never re-sends a parked request, and `stuck` never + /// applies before restart. Only a committed flush the Data Plane already + /// holds can still write, and it keeps its gates until it answers. Every + /// other txn gives its gates back, so a purge never waits on this + /// scheduler. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn release_gates_on_halt( + &mut self, + stuck: TxnId, + ) { + let release: Vec = self + .pending + .iter() + .filter(|(txn_id, pending)| { + **txn_id == stuck + || !writes_in_flight(pending.commit_state) + || self.has_parked_dispatch(**txn_id) + }) + .map(|(txn_id, _)| *txn_id) + .collect(); + for txn_id in release { + if let Some(pending) = self.pending.get_mut(&txn_id) + && !pending.ungated + { + pending.gates.clear(); + pending.ungated = true; + } + } + } + + /// Take back the gates of `txn_id` before its flush, when a halt released + /// them. Fails when a collection the txn names no longer holds its planned + /// incarnation: the flush cannot apply here. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn regate( + &mut self, + txn_id: TxnId, + ) -> crate::Result<()> { + let Some(tx_class) = self + .pending + .get(&txn_id) + .filter(|pending| pending.ungated) + .map(|pending| pending.txn.tx_class.clone()) + else { + return Ok(()); + }; + let (superseded, gates) = self.check_incarnations(&tx_class); + if superseded { + return Err(crate::Error::Internal { + detail: format!( + "calvin txn ({}, {}) committed, but a collection it writes was purged \ + after the halt released its gates", + txn_id.epoch, txn_id.position + ), + }); + } + if let Some(pending) = self.pending.get_mut(&txn_id) { + pending.gates = gates; + pending.ungated = false; + } + Ok(()) + } +} + +/// Whether a txn in `state` can have a write in the Data Plane's hands: a +/// committed flush, or a direct apply. +fn writes_in_flight(state: Option) -> bool { + matches!( + state, + None | Some(CommitState::AwaitingResolve { + committed: true, + .. + }) + ) +} + +#[cfg(test)] +mod tests { + use nodedb_cluster::calvin::types::CalvinIncarnation; + use nodedb_types::DatabaseId; + + use super::super::super::test_support::{ + make_sequenced_txn, scheduler_with_pending, staged_response, + }; + use super::*; + use crate::bridge::envelope::Status; + use crate::control::cluster::calvin::scheduler::driver::types::CommitState; + use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + use crate::control::security::catalog::StoredCollection; + + fn named(collection: &str, incarnation: Hlc) -> CalvinIncarnation { + CalvinIncarnation { + collection: collection.into(), + incarnation, + } + } + + /// A transaction staged against `orders` before a purge and a same-name + /// create is superseded on this replica, though `ledger` is unchanged. The + /// leader's vote is an abort, which the global verdict applies to every + /// participant, `ledger`'s included. A replica that staged it under a + /// COMMIT verdict halts instead of flushing. + #[test] + fn a_txn_staged_before_a_recreate_is_superseded_everywhere() { + let txn_id = TxnId::new(3, 0); + let (mut scheduler, _dir) = scheduler_with_pending(txn_id, CommitState::Staged); + let catalog = scheduler.shared.credentials.catalog(); + let incarnation = |name: &str| { + catalog + .get_committed_collection(DatabaseId::DEFAULT, 1, name) + .expect("read") + .expect("row") + .incarnation + }; + for name in ["orders", "ledger"] { + catalog + .put_collection( + DatabaseId::DEFAULT, + &StoredCollection::stamped_for_test(1, name, "admin"), + ) + .expect("create"); + } + let planned = incarnation("orders"); + let ledger = incarnation("ledger"); + + let mut tx_class = make_sequenced_txn(3, 0).tx_class; + tx_class.set_incarnations(vec![named("orders", planned), named("ledger", ledger)]); + let (superseded, gates) = scheduler.check_incarnations(&tx_class); + assert!( + !superseded, + "both collections hold their planned incarnations" + ); + assert_eq!(gates.len(), 2); + drop(gates); + + catalog + .delete_collection(DatabaseId::DEFAULT, 1, "orders") + .expect("purge"); + catalog + .put_collection( + DatabaseId::DEFAULT, + &StoredCollection::stamped_for_test(1, "orders", "admin"), + ) + .expect("recreate"); + assert_ne!(incarnation("orders"), planned); + let (superseded, _gates) = scheduler.check_incarnations(&tx_class); + assert!( + superseded, + "the recreated collection is not the planned one" + ); + + // The staged slice votes abort, whatever its own read-set says, and + // cannot flush under a COMMIT verdict. + if let Some(pending) = scheduler.pending.get_mut(&txn_id) { + pending.superseded = true; + } + scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Ok, Some(true))); + let pending = scheduler.pending.get(&txn_id).expect("still parked"); + assert_eq!(pending.commit_state, Some(CommitState::AwaitingVerdict)); + assert!(pending.stage_error.is_some(), "a COMMIT verdict halts it"); + } + + /// A reclaim in progress holds the gate exclusive: the check does not + /// wait on it and reports the collection superseded. + #[tokio::test] + async fn a_reclaim_in_progress_supersedes_without_waiting() { + let txn_id = TxnId::new(4, 0); + let (scheduler, _dir) = scheduler_with_pending(txn_id, CommitState::Staged); + let reclaim = write_gate::exclusive(GateKey::Collection { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: "purging".into(), + }) + .await; + let mut tx_class = make_sequenced_txn(4, 0).tx_class; + tx_class.set_incarnations(vec![named("purging", Hlc::ZERO)]); + let (superseded, gates) = scheduler.check_incarnations(&tx_class); + assert!(superseded); + assert!(gates.is_empty()); + drop(reclaim); + } + + /// A replica that halts on a txn it never staged gives back the gates the + /// txn held: a purge of its collection proceeds. The txn checks again + /// before a flush and fails once the collection was recreated. + #[tokio::test] + async fn a_halt_releases_the_gates_and_a_later_flush_checks_again() { + let txn_id = TxnId::new(5, 0); + let (mut scheduler, _dir) = scheduler_with_pending(txn_id, CommitState::AwaitingVerdict); + let shared = std::sync::Arc::clone(&scheduler.shared); + let catalog = shared.credentials.catalog(); + catalog + .put_collection( + DatabaseId::DEFAULT, + &StoredCollection::stamped_for_test(1, "orders", "admin"), + ) + .expect("create"); + let planned = catalog + .get_committed_collection(DatabaseId::DEFAULT, 1, "orders") + .expect("read") + .expect("row") + .incarnation; + let mut tx_class = make_sequenced_txn(5, 0).tx_class; + tx_class.set_incarnations(vec![named("orders", planned)]); + let (superseded, gates) = scheduler.check_incarnations(&tx_class); + assert!(!superseded); + if let Some(pending) = scheduler.pending.get_mut(&txn_id) { + pending.txn.tx_class = tx_class; + pending.gates = gates; + pending.stage_error = Some("stage refused on this replica".into()); + } + + scheduler.resume_on_verdict(txn_id, true); + assert!(scheduler.is_apply_halted()); + let pending = scheduler.pending.get(&txn_id).expect("held unapplied"); + assert!(pending.gates.is_empty(), "the halt released the gates"); + assert!(pending.ungated); + + let key = GateKey::Collection { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: "orders".into(), + }; + let reclaim = tokio::time::timeout( + std::time::Duration::from_secs(5), + write_gate::exclusive(key), + ) + .await + .expect("a purge never waits on a halted scheduler"); + drop(reclaim); + + scheduler.regate(txn_id).expect("nothing changed yet"); + let pending = scheduler.pending.get(&txn_id).expect("pending"); + assert_eq!(pending.gates.len(), 1, "the check took the gate back"); + assert!(!pending.ungated); + + scheduler.release_gates_on_halt(txn_id); + catalog + .delete_collection(DatabaseId::DEFAULT, 1, "orders") + .expect("purge"); + catalog + .put_collection( + DatabaseId::DEFAULT, + &StoredCollection::stamped_for_test(1, "orders", "admin"), + ) + .expect("recreate"); + assert!( + scheduler.regate(txn_id).is_err(), + "a flush never lands in the recreated collection" + ); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/mod.rs index 5dc967ae2..303ce0094 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/mod.rs @@ -4,5 +4,6 @@ mod active_dispatch; mod bind_identities; +mod incarnation; mod primary_write; mod static_dispatch; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/primary_write.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/primary_write.rs index 054c12dc2..7e9cfca34 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/primary_write.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/primary_write.rs @@ -1,49 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Per-slice write classification: primary-write / RETURNING / change-set -//! predicates shared by the static and active dispatch paths. +//! Per-slice write classification: primary-write and RETURNING predicates +//! shared by the static and active dispatch paths. use nodedb_physical::physical_plan::PhysicalPlan; -use crate::types::VShardId; - -/// Whether this vShard's slice carries a PRIMARY user data write — the write -/// whose applied `Response` (affected-count + any RETURNING rows) the -/// coordinator surfaces. -/// -/// A primary write is a Document / KV / Vector / Timeseries / Columnar / Array -/// write — NOT the implicit graph-edge cleanup (`EdgePut` / `EdgeDelete`) that -/// dual-homes alongside a document delete/update. For a single-collection user -/// DML (plus its implicit edges) exactly ONE participant carries the primary -/// write, so only it deposits the applied `Response` into the coordinator's -/// sidecar and the edge participants never clobber the entry. -/// -/// This gate subsumes the RETURNING case (a RETURNING write IS a primary write, -/// so its rows are still deposited) while ALSO carrying the affected-count of a -/// plain (non-RETURNING) write — which a RETURNING-only gate dropped, making a -/// routed plain write report zero rows affected. -pub(super) fn participant_change_sets( - plans: &[PhysicalPlan], - tenant_id: crate::types::TenantId, - vshard_id: u32, -) -> Vec { - plans - .iter() - .filter(|plan| match plan { - // Edge plans are dual-homed; only the source participant publishes - // the one logical Control-Plane event. - PhysicalPlan::Graph( - nodedb_physical::physical_plan::GraphOp::EdgePut { src_id, .. } - | nodedb_physical::physical_plan::GraphOp::EdgeDelete { src_id, .. }, - ) => VShardId::from_key(src_id.as_bytes()).as_u32() == vshard_id, - _ => true, - }) - .map(|plan| { - crate::control::server::dispatch_utils::extract_write_change_set(plan, tenant_id) - }) - .collect() -} - /// Whether the transaction carries any write that is NOT a derived side /// effect (an implicit graph edge, a cross-shard balance delta). Decided over /// the FULL plan set before it is sliced per vShard, because a slice alone diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs index db07d33dd..22b531f89 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/dispatch/static_dispatch.rs @@ -16,12 +16,20 @@ use super::super::owed::SchedulerProposal; use super::super::routing::PlanRouting; use super::super::scheduler::Scheduler; use super::primary_write::{ - participant_change_sets, plans_have_primary_write, plans_have_returning, - txn_has_non_derived_write, + plans_have_primary_write, plans_have_returning, txn_has_non_derived_write, }; use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; use crate::types::DatabaseId; +/// This vShard's slice of a transaction's write plans, with the indexes into +/// it of the plans a trigger body buffered. +struct LocalSlice { + plans: Vec, + body_plans: Vec, + /// Every local plan is a body's, and a client plan homes elsewhere. + body_only: bool, +} + impl Scheduler { /// Whether THIS node is currently the leader of the data-group owning this /// scheduler's vshard. @@ -51,8 +59,13 @@ impl Scheduler { /// waiter immediately with the reason instead of leaving it to burn the /// full deadline and report a generic timeout. Mirrors the OllpMismatch /// broadcast in `handle_executor_response`. Shared by `dispatch_txn` and - /// `dispatch_active_txn`. - pub(super) fn propose_routing_failure(&mut self, txn_id: TxnId, err: &crate::Error) { + /// `dispatch_active_txn`, and by a multi-part transaction whose part + /// fails to decode. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn propose_routing_failure( + &mut self, + txn_id: TxnId, + err: &crate::Error, + ) { self.propose_sequencer_entry( txn_id, SchedulerProposal::RoutingFailed { @@ -145,7 +158,20 @@ impl Scheduler { return; } }; - let has_non_derived_write = txn_has_non_derived_write(&plans); + // An assembled multi-part txn carries only its local tasks. The + // whole-transaction facts come from its manifest. + let has_non_derived_write = match &txn.tx_class.multi_part { + Some(manifest) => manifest.user_write, + None => txn_has_non_derived_write(&plans), + }; + let local_body_plans = + self.local_body_plans(&plans, txn.tx_class.database_id, &txn.tx_class.body_plans); + let client_wrote = match &txn.tx_class.multi_part { + Some(manifest) => manifest.client_write, + None => (0..plans.len()).any(|index| { + u32::try_from(index).map_or(true, |index| !txn.tx_class.body_plans.contains(&index)) + }), + }; let mut local = match self.local_calvin_plans(plans, txn.tx_class.database_id, epoch, position) { Ok(p) => p, @@ -202,16 +228,52 @@ impl Scheduler { // identical static path so each casts a real commit/abort Vote through // stage -> resolve -> verdict. The validate-only task stages no plans; // its response carries only the read-set vote. + let body_only = client_wrote && !local.is_empty() && local_body_plans.len() == local.len(); self.dispatch_calvin_static( txn, txn_id, lock_owner, tenant_id, - local, + LocalSlice { + plans: local, + body_plans: local_body_plans, + body_only, + }, has_non_derived_write, ); } + /// Indexes into this vShard's local slice of the plans a trigger body + /// buffered, from their indexes into the transaction's `plans`. Walks the + /// same homing `local_calvin_plans` does, so the two slices align. + fn local_body_plans( + &self, + plans: &[PhysicalPlan], + database_id: DatabaseId, + body_plans: &[u32], + ) -> Vec { + if body_plans.is_empty() { + return Vec::new(); + } + let mut local_index: u32 = 0; + let mut local_body = Vec::new(); + for (index, plan) in plans.iter().enumerate() { + let PlanRouting::Vshards(vshards) = + super::super::routing::plan_vshard_in_database(plan, database_id) + else { + continue; + }; + if !vshards.iter().any(|v| v.as_u32() == self.vshard_id) { + continue; + } + if u32::try_from(index).is_ok_and(|index| body_plans.contains(&index)) { + local_body.push(local_index); + } + local_index = local_index.saturating_add(1); + } + local_body + } + /// Park the txn in `pending` as `Staged`, then build and dispatch its /// `CalvinExecuteStatic` task. /// @@ -230,9 +292,14 @@ impl Scheduler { txn_id: TxnId, lock_owner: TxnId, tenant_id: crate::types::TenantId, - plans: Vec, + slice: LocalSlice, has_non_derived_write: bool, ) { + let LocalSlice { + plans, + body_plans, + body_only, + } = slice; // The apply-slot identity (used in the CalvinExecuteStatic task and // error logs) is exactly `txn_id`; deriving it here keeps the two in // lockstep instead of passing the pair redundantly. @@ -241,8 +308,8 @@ impl Scheduler { let request_id = self.next_request_id(); let has_primary_write = plans_have_primary_write(&plans, has_non_derived_write); let has_returning = plans_have_returning(&plans); - let change_sets = participant_change_sets(&plans, tenant_id, self.vshard_id); - let flush_scope = super::super::super::types::FlushScope::of_plans(&plans); + let mut flush_scope = super::super::super::types::FlushScope::of_plans(&plans); + flush_scope.body_only = body_only; let database_id = txn.tx_class.database_id; let plan = PhysicalPlan::Meta(MetaOp::CalvinExecuteStatic { epoch, @@ -255,13 +322,19 @@ impl Scheduler { // participant can check, at apply, whether its slice of the reads was // still current. Empty for pure-write / autocommit transactions. versioned_reads: txn.tx_class.versioned_reads.as_slice().to_vec(), + body_plans, }); // Calvin allocates the CalvinApplied WAL LSN post-apply (in the // scheduler's response handler), so no committed LSN is known at // dispatch time to stamp here. - let request = self.build_exempt_request(request_id, tenant_id, database_id, plan, None); + let mut request = self.build_exempt_request(request_id, tenant_id, database_id, plan, None); + // The stage tags the rows it stages by the slice's source. + request.event_source = + super::super::request::slice_event_source(&txn.tx_class, &flush_scope); + // Every replica checks the transaction's collection incarnations. + let (superseded, gates) = self.check_incarnations(&txn.tx_class); self.pending.insert( txn_id, super::super::super::types::PendingTxn { @@ -271,7 +344,8 @@ impl Scheduler { dispatch_time: Instant::now(), has_primary_write, has_returning, - change_sets, + // The commit's resolved redo fills them. + change_sets: Vec::new(), // This dispatch STAGES the txn (validate + buffer, no apply); // its response carries the local commit vote that drives the // subsequent flush-or-drop. @@ -282,6 +356,11 @@ impl Scheduler { // Set once a committed txn appends its redo record. redo_records: None, flush_scope, + superseded, + gates, + ungated: false, + // Taken when the flush dispatches. + install_permit: None, }, ); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/flush_turn.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/flush_turn.rs new file mode 100644 index 000000000..82ed6d1ca --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/flush_turn.rs @@ -0,0 +1,231 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Sequencer-order flushes. +//! +//! A vShard's transactions stage, vote, and resolve in any order: their +//! responses and verdicts arrive in any order. Their flushes run one at a +//! time, in `(epoch, position)` order. A committed transaction's flush +//! dispatches once no lower transaction of this vShard is unfinished. +//! +//! One core applies every request of a vShard, in dispatch order. So the core +//! installs committed transactions, and emits their change events, in +//! sequencer order on every replica. The commit tail, which publishes the +//! transaction's Control-Plane changes, runs in the same order. A Calvin +//! change feed therefore never delivers a lower position after a higher one. +//! +//! Every wait points at a lower transaction: the scheduler takes input, and +//! grants locks, in sequencer order. So the turn order adds no wait cycle. + +use super::super::types::CommitState; +use super::commit_resolution_dispatch::CommitResolution; +use super::deferred::{DispatchOutcome, DispatchStep}; +use super::install_gate::FlushAdmission; +use super::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + +impl Scheduler { + /// Queue the committed flush of `txn_id` for its turn, then dispatch it + /// when its turn came. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn queue_flush( + &mut self, + txn_id: TxnId, + redo_lsn: Option, + ) { + if let Some(pending) = self.pending.get_mut(&txn_id) { + pending.commit_state = Some(CommitState::AwaitingFlushTurn { redo_lsn }); + } + self.pump_flush_turn(); + } + + /// Dispatch the flush of the lowest unfinished transaction when it waits + /// for its turn. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn pump_flush_turn(&mut self) { + let Some(lowest) = self.lowest_unfinished() else { + return; + }; + let resolve_turn = self + .pending + .get(&lowest) + .is_some_and(|pending| pending.commit_state == Some(CommitState::AwaitingResolveTurn)); + if resolve_turn { + self.dispatch_resolve_turn(lowest); + return; + } + let Some(redo_lsn) = + self.pending + .get(&lowest) + .and_then(|pending| match pending.commit_state { + Some(CommitState::AwaitingFlushTurn { redo_lsn }) => Some(redo_lsn), + _ => None, + }) + else { + return; + }; + // The flush holds its data group's apply gate shared until its + // position is marked applied (see `super::install_gate`). + let permit = match self.admit_flush() { + FlushAdmission::Go(permit) => permit, + FlushAdmission::Wait | FlushAdmission::Retired => return, + }; + if let Some(pending) = self.pending.get_mut(&lowest) { + pending.install_permit = permit; + } + // A flush refused at capacity is parked for re-send. The txn awaits its + // flush response either way, so the state below is the same. + if let DispatchOutcome::Failed(error) = + self.dispatch_commit_resolution(lowest, CommitResolution::Flush { redo_lsn }) + { + self.fail_dispatch_step(lowest, DispatchStep::Flush, error); + return; + } + if let Some(pending) = self.pending.get_mut(&lowest) { + pending.commit_state = Some(CommitState::AwaitingResolve { + committed: true, + redo_lsn, + }); + } + } + + /// Dispatch the resolve of `txn_id`, whose turn came. Its flush then + /// follows at once: it stays the lowest unfinished txn. + fn dispatch_resolve_turn(&mut self, txn_id: TxnId) { + if let DispatchOutcome::Failed(error) = self.dispatch_calvin_resolve(txn_id) { + self.fail_dispatch_step(txn_id, DispatchStep::Resolve, error); + return; + } + if let Some(pending) = self.pending.get_mut(&txn_id) { + pending.commit_state = Some(CommitState::AwaitingRedoResolve); + } + } + + /// The lowest transaction of this vShard that has not finished: in + /// flight, blocked on a lock, or waiting at a dependent-read barrier. + fn lowest_unfinished(&self) -> Option { + let pending = self.pending.keys().next().copied(); + let barrier = self.dependent_barrier.keys().next().copied(); + let blocked = self + .blocked + .values() + .map(|blocked| TxnId::new(blocked.txn.epoch, blocked.txn.position)) + .min(); + // A multi-part txn holding its locks while its parts arrive is + // unfinished too: every later flush waits for it. + let awaiting_parts = self.parts.lowest_awaiting(); + [pending, barrier, blocked, awaiting_parts] + .into_iter() + .flatten() + .min() + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + build_test_scheduler_with_data_side, make_sequenced_txn, staged_pending, + }; + use nodedb_cluster::calvin::CalvinCompletionRegistry; + + /// A higher committed txn waits for the lower one; the lower one's + /// completion gives it its turn. + #[tokio::test] + async fn a_higher_flush_waits_for_every_lower_txn() { + let low = TxnId::new(5, 0); + let high = TxnId::new(5, 1); + let (mut scheduler, _dir, _data_side) = + build_test_scheduler_with_data_side(7, CalvinCompletionRegistry::new_detached()); + for txn_id in [low, high] { + scheduler.pending.insert( + txn_id, + staged_pending(make_sequenced_txn(txn_id.epoch, txn_id.position), txn_id), + ); + } + + scheduler.queue_flush(high, None); + assert_eq!( + scheduler.pending.get(&high).and_then(|p| p.commit_state), + Some(CommitState::AwaitingFlushTurn { redo_lsn: None }), + "the lower txn is unfinished, so the higher flush waits" + ); + + scheduler.pending.remove(&low); + scheduler.pump_flush_turn(); + assert_eq!( + scheduler.pending.get(&high).and_then(|p| p.commit_state), + Some(CommitState::AwaitingResolve { + committed: true, + redo_lsn: None, + }), + "with the lower txn finished, the higher flush dispatches" + ); + } + + /// A committed TRUNCATE of rows resolves only at its turn: while a lower + /// txn of this vShard is unfinished, its resolve waits, so it reads every + /// row sequenced before it. + #[tokio::test] + async fn a_truncate_resolves_only_at_its_turn() { + let low = TxnId::new(5, 0); + let truncate = TxnId::new(5, 1); + let (mut scheduler, _dir, _data_side) = + build_test_scheduler_with_data_side(7, CalvinCompletionRegistry::new_detached()); + for txn_id in [low, truncate] { + scheduler.pending.insert( + txn_id, + staged_pending(make_sequenced_txn(txn_id.epoch, txn_id.position), txn_id), + ); + } + if let Some(pending) = scheduler.pending.get_mut(&truncate) { + pending.flush_scope.resolve_at_turn = true; + pending.commit_state = Some(CommitState::AwaitingResolveTurn); + } + + scheduler.pump_flush_turn(); + assert_eq!( + scheduler + .pending + .get(&truncate) + .and_then(|p| p.commit_state), + Some(CommitState::AwaitingResolveTurn), + "the lower txn is unfinished, so the resolve waits" + ); + + scheduler.pending.remove(&low); + scheduler.pump_flush_turn(); + assert_eq!( + scheduler + .pending + .get(&truncate) + .and_then(|p| p.commit_state), + Some(CommitState::AwaitingRedoResolve), + "with the lower txn finished, the resolve dispatches" + ); + } + + /// Every truncate of rows reads a whole collection at resolve; a point + /// write does not. A TRUNCATE's edge share records a cut, which reads no + /// stored edge, so it resolves without waiting for its turn. + #[test] + fn a_truncate_slice_resolves_at_its_turn() { + use nodedb_physical::physical_plan::{DocumentOp, GraphOp, PhysicalPlan}; + let collection = nodedb_types::QualifiedCollection::from_stored("c".to_string()); + let truncate = PhysicalPlan::Document(DocumentOp::Truncate { + collection: collection.clone(), + restart_identity: false, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }); + let share = PhysicalPlan::Graph(GraphOp::TruncateEdges { + collection, + vshard: 3, + }); + use crate::control::cluster::calvin::scheduler::driver::types::FlushScope; + assert!(FlushScope::of_plans(std::slice::from_ref(&truncate)).resolve_at_turn); + assert!(!FlushScope::of_plans(std::slice::from_ref(&share)).resolve_at_turn); + assert!( + FlushScope::of_plans(&[truncate, share]).resolve_at_turn, + "the rows' vShard holds both, and its rows' truncate waits" + ); + assert!(!FlushScope::of_plans(&[]).resolve_at_turn); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs index aaeb8e559..33b24843c 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/halt.rs @@ -10,6 +10,9 @@ //! //! A halted scheduler: //! - keeps the stuck txn unapplied, with its locks and its `pending` entry; +//! - releases the collection gates of every txn with no flush in flight, so a +//! purge never waits on it. Such a txn checks its incarnations again, and +//! takes its gates back, before a later flush; //! - closes intake with [`IntakeClosure::ApplyHalted`]. The fan-out then drops //! new input and arms catch-up, which keeps the sequencer log retained; //! - stops the deferred re-send and the catch-up drain; @@ -56,6 +59,9 @@ pub(in crate::control::cluster::calvin::scheduler::driver::core) enum HaltReason IdentityBindFailed, /// A redo or `CalvinApplied` WAL append failed. WalAppendFailed, + /// The metadata group left this node before its apply reached a held + /// txn's floor. + MetadataGroupGone, } impl HaltReason { @@ -70,6 +76,7 @@ impl HaltReason { Self::LocalStageFailed => apply_halt_reason::LOCAL_STAGE_FAILED, Self::IdentityBindFailed => apply_halt_reason::IDENTITY_BIND_FAILED, Self::WalAppendFailed => apply_halt_reason::WAL_APPEND_FAILED, + Self::MetadataGroupGone => apply_halt_reason::METADATA_GROUP_GONE, } } @@ -100,6 +107,8 @@ pub(in crate::control::cluster::calvin::scheduler::driver::core) enum HaltStep { AppliedMarker, /// The surrogate identity binding before the stage dispatch. IdentityBind, + /// The wait for this node's metadata apply to reach the txn's floor. + MetadataHold, } impl HaltStep { @@ -116,6 +125,7 @@ impl HaltStep { Self::RedoAppend => "redo_append", Self::AppliedMarker => "applied_marker", Self::IdentityBind => "identity_bind", + Self::MetadataHold => "metadata_hold", } } @@ -125,7 +135,10 @@ impl HaltStep { ) -> Self { match state { Some(CommitState::Staged | CommitState::AwaitingVerdict) => Self::Stage, - Some(CommitState::AwaitingRedoResolve) => Self::Resolve, + Some(CommitState::AwaitingRedoResolve | CommitState::AwaitingResolveTurn) => { + Self::Resolve + } + Some(CommitState::AwaitingFlushTurn { .. }) => Self::Flush, Some(CommitState::AwaitingResolve { committed: true, .. }) => Self::Flush, @@ -223,6 +236,7 @@ impl Scheduler { "calvin scheduler: apply halted; holding another txn unapplied" ); } + self.release_gates_on_halt(txn_id); return; } @@ -269,6 +283,7 @@ impl Scheduler { .record(report.clone()); } self.halt.first = Some(ApplyHalt { reason, report }); + self.release_gates_on_halt(txn_id); } /// Whether the node shuts down: the shutdown watch fired, or the diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/install_gate.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/install_gate.rs new file mode 100644 index 000000000..da4feba01 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/install_gate.rs @@ -0,0 +1,226 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The scheduler's admission of a flush against its data group's snapshots. +//! +//! A flush installs a committed transaction's slice on the vShard's core. +//! The data group's apply gate orders it against the group's snapshots: +//! +//! - A snapshot capture on this node holds the gate exclusive. A flush holds +//! it shared from its dispatch until its position is marked applied, so +//! the capture reads storage and applied positions that agree. +//! - A snapshot install on this node holds it exclusive too, and replaces +//! the vShard's base. A scheduler started under an older base installs +//! nothing more: the next scheduler reconcile starts it again from the +//! installed state. +//! +//! A flush the gate refuses waits for the gate's release and is tried again. + +use tokio::sync::watch; + +use super::scheduler::Scheduler; +use crate::control::security::auth_fence::cluster::group_of_vshard; +use crate::control::state::SharedState; + +/// What the gate says to one flush. +pub(in crate::control::cluster::calvin::scheduler::driver::core) enum FlushAdmission { + /// Dispatch, holding the permit until the position is marked applied. + /// `None` when no apply gate is wired here. + Go(Option), + /// A snapshot holds the gate: try again once it releases. + Wait, + /// A snapshot install replaced this vShard's base since this scheduler + /// started: install nothing. + Retired, +} + +/// The scheduler's view of its data group's apply gate. +pub(in crate::control::cluster::calvin::scheduler::driver::core) struct InstallGate { + /// The vShard's base generation this scheduler started under. + generation: u64, + /// Changes each time a snapshot releases a group's gate. + released: Option>, + /// Whether a flush waits for a release. + waiting: bool, +} + +impl InstallGate { + /// The gate of a scheduler starting for `vshard_id` now. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn new( + shared: &SharedState, + vshard_id: u32, + ) -> Self { + Self { + generation: shared.calvin.bases.generation(vshard_id), + released: shared + .raft_apply_gates + .get() + .map(|gates| gates.subscribe_released()), + waiting: false, + } + } + + /// Admit one flush of `vshard_id`. + fn admit(&mut self, shared: &SharedState, vshard_id: u32) -> FlushAdmission { + if shared.calvin.bases.generation(vshard_id) != self.generation { + return FlushAdmission::Retired; + } + let Some(gates) = shared.raft_apply_gates.get() else { + return FlushAdmission::Go(None); + }; + let Ok(group_id) = group_of_vshard(shared, vshard_id) else { + // No routing table holds the vShard: no snapshot of it runs. + return FlushAdmission::Go(None); + }; + // Marked as seen before the try, so a release after it still wakes + // the wait. + if let Some(released) = self.released.as_mut() { + drop(released.borrow_and_update()); + } + match gates.try_apply(group_id) { + Some(permit) => { + self.waiting = false; + FlushAdmission::Go(Some(permit)) + } + None => { + self.waiting = true; + FlushAdmission::Wait + } + } + } + + /// Whether a flush waits for the gate's release. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn is_waiting(&self) -> bool { + self.waiting + } + + /// Resolves once a snapshot released a group's gate since the last + /// admission. + pub(in crate::control::cluster::calvin::scheduler::driver::core) async fn released(&mut self) { + match self.released.as_mut() { + Some(released) => { + if released.changed().await.is_err() { + // The gates are gone: no snapshot holds one again. + self.waiting = false; + } + } + None => std::future::pending().await, + } + } +} + +impl Scheduler { + /// Admit the flush of the lowest unfinished transaction. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn admit_flush( + &mut self, + ) -> FlushAdmission { + let vshard_id = self.vshard_id; + self.install_gate.admit(&self.shared, vshard_id) + } + + /// Record that this replica's Calvin state of the vShard has a hole, and + /// make its data-group replica refuse log entries until a snapshot + /// brings the state back. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn lose_calvin_base(&self) { + if let Err(error) = self + .shared + .calvin + .bases + .record_lost(self.shared.credentials.catalog(), self.vshard_id) + { + tracing::error!( + vshard_id = self.vshard_id, + %error, + "calvin: the vShard's lost base did not persist" + ); + } + match group_of_vshard(&self.shared, self.vshard_id) { + Ok(group_id) => { + self.multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .set_snapshot_required(group_id, true); + } + Err(error) => tracing::error!( + vshard_id = self.vshard_id, + %error, + "calvin: no data group for the vShard; its replica cannot ask for a snapshot" + ), + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::{Arc, RwLock}; + use std::time::Duration; + + use nodedb_cluster::{GroupApplyGates, RoutingTable}; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::wal::WalManager; + + /// A snapshot capture holds the group's gate exclusive. A flush that + /// starts meanwhile waits, and a capture that starts while a flush is + /// out waits until the flush marked its position and dropped its + /// permit. So every capture sees each vShard's flushes either whole or + /// not at all, on every core. A snapshot install that replaced the + /// vShard's base stops the scheduler started before it. + #[tokio::test] + async fn a_capture_never_sees_a_flush_half_applied() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = + Arc::new(WalManager::open_for_testing(&dir.path().join("gate.wal")).expect("wal")); + let (dispatcher, _sides) = Dispatcher::new(1, 64); + let mut state = SharedState::new(dispatcher, wal).expect("shared state"); + Arc::get_mut(&mut state) + .expect("sole owner of fresh state") + .cluster_routing = Some(Arc::new(RwLock::new(RoutingTable::uniform(2, &[1], 1)))); + let gates = Arc::new(GroupApplyGates::new()); + let vshard_id = 0; + let group_id = group_of_vshard(&state, vshard_id).expect("the vShard has a group"); + gates.mount(group_id); + assert!(state.raft_apply_gates.set(Arc::clone(&gates)).is_ok()); + + let mut gate = InstallGate::new(&state, vshard_id); + let capture = gates.install(group_id).await; + assert!(matches!( + gate.admit(&state, vshard_id), + FlushAdmission::Wait + )); + assert!(gate.is_waiting()); + drop(capture); + tokio::time::timeout(Duration::from_secs(5), gate.released()) + .await + .expect("the capture's release wakes the waiting flush"); + + let FlushAdmission::Go(Some(permit)) = gate.admit(&state, vshard_id) else { + panic!("an open gate admits the flush with a permit"); + }; + assert!(!gate.is_waiting()); + let next_capture = { + let gates = Arc::clone(&gates); + tokio::spawn(async move { drop(gates.install(group_id).await) }) + }; + tokio::time::sleep(Duration::from_millis(50)).await; + assert!( + !next_capture.is_finished(), + "a capture waits for the flush in flight" + ); + drop(permit); + tokio::time::timeout(Duration::from_secs(5), next_capture) + .await + .expect("the capture starts once the flush finished") + .expect("capture task"); + + state + .calvin + .bases + .record_snapshot(state.credentials.catalog(), &[vshard_id], 9) + .expect("record the installed base"); + assert!(matches!( + gate.admit(&state, vshard_id), + FlushAdmission::Retired + )); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs index f87b16717..06cbf77b9 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/intake.rs @@ -14,6 +14,10 @@ //! path then drops the next input and arms catch-up for this vShard, and the //! drain replays it from the committed log once the gate opens. //! +//! A gate closed by the backlog bound keeps one lane open: the parts of a +//! txn waiting for them, and the releases a blocked txn waits on, still +//! reach the scheduler. See [`super::parts_lane`]. +//! //! The gate changes only when inputs are read, never the order they are //! processed in, so replicas stay deterministic. //! @@ -34,6 +38,10 @@ pub(in crate::control::cluster::calvin::scheduler::driver::core) enum IntakeClos /// The in-flight backlog is at its bound, and some of it progresses /// without new input. BacklogFull, + /// A sequenced txn waits for this node's metadata apply to reach the + /// catalog its coordinator planned it against. Later input waits behind + /// it, so the processing order stays the log order. + MetadataCatchUp, } impl IntakeClosure { @@ -43,6 +51,7 @@ impl IntakeClosure { Self::ApplyHalted => intake_closure_reason::APPLY_HALTED, Self::DeferredDispatch => intake_closure_reason::DEFERRED_DISPATCH, Self::BacklogFull => intake_closure_reason::BACKLOG_FULL, + Self::MetadataCatchUp => intake_closure_reason::METADATA_CATCH_UP, } } } @@ -54,11 +63,14 @@ pub(in crate::control::cluster::calvin::scheduler::driver::core) struct IntakeGa } impl Scheduler { - /// Pending, blocked, and dependent-barrier txns. + /// Pending, blocked, dependent-barrier, and parts-awaiting txns. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn inflight_backlog( &self, ) -> usize { - self.pending.len() + self.blocked.len() + self.dependent_barrier.len() + self.pending.len() + + self.blocked.len() + + self.dependent_barrier.len() + + self.parts.awaiting_len() } /// Why intake must stay closed, or `None` when the gate is open. @@ -76,8 +88,16 @@ impl Scheduler { if self.has_deferred_dispatch() { return Some(IntakeClosure::DeferredDispatch); } - let drains_without_input = !self.pending.is_empty() || !self.dependent_barrier.is_empty(); - if drains_without_input && self.inflight_backlog() >= self.config.max_inflight_backlog { + if self.metadata_hold.is_some() { + return Some(IntakeClosure::MetadataCatchUp); + } + // A multi-part txn waiting for its parts finishes on input too. The + // closed gate still lets its parts through, on the backlog lane (see + // `super::parts_lane`), so the backlog stays bounded and it drains. + let drains_behind_the_gate = !self.pending.is_empty() + || !self.dependent_barrier.is_empty() + || self.parts.awaiting_len() > 0; + if drains_behind_the_gate && self.inflight_backlog() >= self.config.max_inflight_backlog { return Some(IntakeClosure::BacklogFull); } None @@ -218,7 +238,8 @@ mod tests { build_test_scheduler_with_data_side(test_coll_vshard(), registry); let shared = Arc::clone(&scheduler.shared); let fillers = fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); - scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + scheduler + .process_scheduler_input(SchedulerInput::Txn(Box::new(make_validate_only_txn(3, 0)))); assert!( scheduler.has_deferred_dispatch(), "the stage dispatch defers" @@ -228,7 +249,7 @@ mod tests { let running = spawn_scheduler_loop(scheduler); running .input_tx() - .send(SchedulerInput::Txn(make_validate_only_txn(4, 0))) + .send(SchedulerInput::Txn(Box::new(make_validate_only_txn(4, 0)))) .await .expect("the loop's input channel is open"); @@ -254,8 +275,9 @@ mod tests { ); } - /// With the backlog at the configured bound, the run loop leaves a new - /// input unread. + /// With the backlog at the configured bound, the run loop processes no + /// new txn. The backlog lane can read the channel, but it holds a txn + /// unprocessed, so the backlog stays at the bound. #[tokio::test] async fn full_backlog_holds_new_input() { let registry = CalvinCompletionRegistry::new_detached(); @@ -271,17 +293,13 @@ mod tests { let running = spawn_scheduler_loop(scheduler); running .input_tx() - .send(SchedulerInput::Txn(make_validate_only_txn(4, 0))) + .send(SchedulerInput::Txn(Box::new(make_validate_only_txn(4, 0)))) .await .expect("the loop's input channel is open"); - let read_at_bound = inputs_consumed_within(running.input_tx(), HOLD_WAIT).await; + tokio::time::sleep(HOLD_WAIT).await; running.stop().await; - assert!( - !read_at_bound, - "the loop must not read input while the backlog is at its bound" - ); assert_eq!(metrics.intake_gate_closed.load(Ordering::Relaxed), 1); assert_eq!(metrics.intake_backlog.load(Ordering::Relaxed), 1); assert_eq!( @@ -311,7 +329,7 @@ mod tests { let running = spawn_scheduler_loop(scheduler); running .input_tx() - .send(SchedulerInput::Txn(make_validate_only_txn(4, 0))) + .send(SchedulerInput::Txn(Box::new(make_validate_only_txn(4, 0)))) .await .expect("the loop's input channel is open"); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/metadata_hold.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/metadata_hold.rs new file mode 100644 index 000000000..8ffb1cff1 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/metadata_hold.rs @@ -0,0 +1,224 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Hold a sequenced transaction until this node's catalog reached the one +//! its coordinator planned it against. +//! +//! The sequencer group and the metadata group apply on independent loops. +//! Without this hold, a replica can run a transaction that writes a +//! collection before it applied that collection's creation: the write lands +//! on an unregistered collection, and the creation's storage clear then +//! erases it. The coordinator stamps its applied metadata index on the class +//! (`TxClass::metadata_floor`). +//! +//! A held transaction closes the intake gate, so every later input waits +//! behind it. The processing order stays the log order on every replica. +//! +//! The wait has no deadline, like the data-group hold: applying the txn +//! earlier breaks the order the hold keeps. It ends without applying only when +//! the metadata group left this node. That cause is local to this replica, so +//! no replica can abort the txn in a way the others reach too. The scheduler +//! halts instead and never marks the position applied: the txn replays from +//! the sequencer log on the next boot, and the other replicas apply it as +//! usual. The originator's completion comes from the sequencer group, which +//! this replica's halt does not block. + +use std::time::{Duration, Instant}; + +use nodedb_cluster::METADATA_GROUP_ID; +use nodedb_cluster::calvin::types::SequencedTxn; +use tracing::warn; + +use super::halt::{HaltReason, HaltStep}; +use super::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + +/// How often the run loop re-checks a held transaction's floor. +pub(in crate::control::cluster::calvin::scheduler::driver::core) const HOLD_POLL: Duration = + Duration::from_millis(10); + +/// How often a hold that still waits logs a warning. +const HOLD_WARN_EVERY: Duration = Duration::from_secs(5); + +/// A sequenced transaction waiting for this node's metadata apply. +pub(in crate::control::cluster::calvin::scheduler::driver::core) struct MetadataHold { + txn: SequencedTxn, + // no-determinism: only paces the warning log. + warned_at: Instant, +} + +impl Scheduler { + /// Whether this node's metadata apply is below `txn`'s floor. + fn metadata_floor_unreached(&self, txn: &SequencedTxn) -> bool { + let floor = txn.tx_class.metadata_floor; + floor != 0 + && self + .shared + .applied_index_watcher(METADATA_GROUP_ID) + .current() + < floor + } + + /// Process `txn` now, or hold it until this node's metadata apply + /// reaches its floor. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn process_or_hold_for_metadata( + &mut self, + txn: SequencedTxn, + ) { + if self.metadata_floor_unreached(&txn) { + self.metadata_hold = Some(MetadataHold { + txn, + // no-determinism: only paces the warning log. + warned_at: Instant::now(), + }); + return; + } + self.process_new_txn(txn); + } + + /// Process the held transaction once its floor is reached. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn resume_metadata_hold( + &mut self, + ) { + let Some(mut hold) = self.metadata_hold.take() else { + return; + }; + if !self.metadata_floor_unreached(&hold.txn) { + self.process_new_txn(hold.txn); + return; + } + let watcher = self.shared.applied_index_watcher(METADATA_GROUP_ID); + if watcher.is_closed() { + let floor = hold.txn.tx_class.metadata_floor; + self.halt_apply( + TxnId::new(hold.txn.epoch, hold.txn.position), + HaltReason::MetadataGroupGone, + HaltStep::MetadataHold, + format!( + "the metadata group left this node at applied index {} before it \ + reached the txn's floor {floor}; the txn stays unapplied and replays \ + on the next boot", + watcher.current() + ), + ); + return; + } + if hold.warned_at.elapsed() >= HOLD_WARN_EVERY { + warn!( + vshard_id = self.vshard_id, + epoch = hold.txn.epoch, + position = hold.txn.position, + metadata_floor = hold.txn.tx_class.metadata_floor, + metadata_applied = self + .shared + .applied_index_watcher(METADATA_GROUP_ID) + .current(), + "a sequenced txn waits for this node's metadata apply to reach the catalog \ + its coordinator planned it against" + ); + // no-determinism: only paces the warning log. + hold.warned_at = Instant::now(); + } + self.metadata_hold = Some(hold); + } + + /// Whether a transaction waits for this node's metadata apply. + pub fn holds_for_metadata(&self) -> bool { + self.metadata_hold.is_some() + } +} + +#[cfg(test)] +mod tests { + use nodedb_cluster::calvin::types::SchedulerInput; + + use super::super::intake::IntakeClosure; + use super::super::test_support::{build_test_scheduler, make_sequenced_txn}; + use super::*; + use crate::control::cluster::calvin::scheduler::lock_manager::{AcquireOutcome, TxnId}; + + /// A txn planned against a catalog this node has not applied yet waits, + /// and closes intake behind it. Once the metadata apply reaches its + /// floor, the txn runs. The metadata watcher bumps only after the + /// applier returned, so the collection's creation, storage clear + /// included, is done by then. + #[tokio::test] + async fn a_txn_waits_for_its_metadata_floor_and_runs_once_reached() { + let (mut scheduler, _dir) = build_test_scheduler(0); + let watcher = scheduler.shared.applied_index_watcher(METADATA_GROUP_ID); + let floor = watcher.current() + 3; + let mut txn = make_sequenced_txn(1, 0); + txn.tx_class.metadata_floor = floor; + + // A conflicting holder on the txn's key: once it runs, it blocks + // instead of dispatching to a Data Plane this test does not run. + let keys = crate::control::cluster::calvin::scheduler::driver::helpers::expand_rw_set(&txn); + { + let mut lm = scheduler + .lock_manager + .lock() + .unwrap_or_else(|p| p.into_inner()); + assert_eq!( + lm.acquire(TxnId::new(u64::MAX, 0), keys), + AcquireOutcome::Ready + ); + } + + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(txn))); + assert!(scheduler.holds_for_metadata()); + assert_eq!( + scheduler.intake_closure(), + Some(IntakeClosure::MetadataCatchUp), + "later input waits behind the held txn" + ); + assert!(scheduler.blocked.is_empty() && scheduler.pending.is_empty()); + + scheduler.resume_metadata_hold(); + assert!( + scheduler.holds_for_metadata(), + "the txn keeps waiting below its floor" + ); + + watcher.bump(floor); + scheduler.resume_metadata_hold(); + assert!(!scheduler.holds_for_metadata()); + assert!( + scheduler.blocked.contains_key(&TxnId::new(1, 0)), + "the released txn runs" + ); + assert_eq!(scheduler.intake_closure(), None); + } + + /// A metadata group that left this node ends the hold: the scheduler + /// halts with the txn unapplied, and intake stays closed. + #[tokio::test] + async fn a_gone_metadata_group_halts_the_held_txn() { + let (mut scheduler, _dir) = build_test_scheduler(0); + let watcher = scheduler.shared.applied_index_watcher(METADATA_GROUP_ID); + let mut txn = make_sequenced_txn(1, 0); + txn.tx_class.metadata_floor = watcher.current() + 3; + + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(txn))); + assert!(scheduler.holds_for_metadata()); + + watcher.close(); + scheduler.resume_metadata_hold(); + assert!(!scheduler.holds_for_metadata(), "the hold ended"); + assert!(scheduler.is_apply_halted()); + let halt = scheduler.apply_halt().expect("the scheduler halted"); + assert_eq!(halt.reason, HaltReason::MetadataGroupGone); + assert_eq!(scheduler.intake_closure(), Some(IntakeClosure::ApplyHalted)); + assert!( + scheduler.blocked.is_empty() && scheduler.pending.is_empty(), + "the txn never ran on this replica" + ); + } + + /// A class with no floor never waits. + #[tokio::test] + async fn a_txn_without_a_floor_runs_at_once() { + let (scheduler, _dir) = build_test_scheduler(0); + let txn = make_sequenced_txn(1, 0); + assert_eq!(txn.tx_class.metadata_floor, 0); + assert!(!scheduler.metadata_floor_unreached(&txn)); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs index 8fd8be522..aaad71d68 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/mod.rs @@ -22,9 +22,14 @@ pub mod completion_route; mod cut_marker; pub mod deferred; pub mod dispatch; +pub mod flush_turn; pub mod halt; +mod install_gate; pub mod intake; +mod metadata_hold; pub mod owed; +pub mod parts; +pub mod parts_lane; pub mod process; pub mod propose; pub mod read_result; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/entry.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/entry.rs index 31453d838..51f485d9a 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/entry.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/entry.rs @@ -21,8 +21,10 @@ use crate::control::cluster::calvin::scheduler::metrics::sequencer_propose_kind; pub enum SchedulerProposal { /// This vShard's commit vote. `Some(reason)` is an abort vote. Vote { abort: Option }, - /// This vShard applied or dropped the txn. - CompletionAck, + /// This vShard applied or dropped the txn. `result` is the apply result + /// the coordinator reads: the timeseries install counts of the apply. + /// Empty when there are none. + CompletionAck { result: Vec }, /// The active executor saw its OLLP prediction drift. OllpMismatch, /// The txn's local plan routing failed for good. @@ -34,21 +36,21 @@ impl SchedulerProposal { pub fn kind(&self) -> OwedKind { match self { Self::Vote { .. } => OwedKind::Vote, - Self::CompletionAck => OwedKind::CompletionAck, + Self::CompletionAck { .. } => OwedKind::CompletionAck, Self::OllpMismatch => OwedKind::OllpMismatch, Self::RoutingFailed { .. } => OwedKind::RoutingFailed, } } - /// The sequencer entry for `txn_id` proposed by `vshard`. - pub fn entry(&self, txn_id: TxnId, vshard: u32) -> SequencerEntry { + /// The sequencer entry for `txn_id` proposed by `vshard`'s scheduler on + /// node `node_id`. + pub fn entry(&self, txn_id: TxnId, vshard: u32, node_id: u64) -> SequencerEntry { let (epoch, position) = (txn_id.epoch, txn_id.position); match self { Self::Vote { abort: None } => SequencerEntry::Vote { epoch, position, vshard, - commit: true, }, Self::Vote { abort: Some(reason), @@ -58,10 +60,12 @@ impl SchedulerProposal { vshard, reason: *reason, }, - Self::CompletionAck => SequencerEntry::CompletionAck { + Self::CompletionAck { result } => SequencerEntry::CompletionAck { epoch, position, vshard_id: vshard, + result: result.clone(), + from_node: node_id, }, Self::OllpMismatch => SequencerEntry::OllpMismatch { epoch, position }, Self::RoutingFailed { detail } => SequencerEntry::TxnRoutingFailed { @@ -98,6 +102,11 @@ impl OwedKind { } } + /// Whether only the vShard's data-group leader proposes this kind. + pub fn leader_only(self) -> bool { + matches!(self, Self::Vote | Self::CompletionAck) + } + /// Whether this node applied the entry of this kind. /// /// `progress` is the registry's view of the txn for the scheduler's @@ -219,19 +228,18 @@ mod tests { fn a_vote_proposal_carries_its_abort_reason() { let txn = TxnId::new(3, 4); assert!(matches!( - SchedulerProposal::Vote { abort: None }.entry(txn, 9), + SchedulerProposal::Vote { abort: None }.entry(txn, 9, 1), SequencerEntry::Vote { epoch: 3, position: 4, - vshard: 9, - commit: true + vshard: 9 } )); assert!(matches!( SchedulerProposal::Vote { abort: Some(AbortReason::SerializationConflict) } - .entry(txn, 9), + .entry(txn, 9, 1), SequencerEntry::AbortVote { epoch: 3, position: 4, diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs index cf31a06c3..7ddf13a87 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/owed/retry.rs @@ -30,7 +30,8 @@ impl Scheduler { proposal: SchedulerProposal, ) { let kind = proposal.kind(); - let bytes = match zerompk::to_msgpack_vec(&proposal.entry(txn_id, self.vshard_id)) { + let entry = proposal.entry(txn_id, self.vshard_id, self.shared.node_id); + let bytes = match zerompk::to_msgpack_vec(&entry) { Ok(bytes) => bytes, Err(e) => { error!( @@ -45,13 +46,14 @@ impl Scheduler { } }; let entry_seen = self.participant_progress(txn_id).is_some(); - let in_flight = propose_owed( - self.sequencer_proposer.as_ref(), - self.vshard_id, - txn_id, - kind, - bytes.clone(), - ); + let in_flight = self.may_propose(kind) + && propose_owed( + self.sequencer_proposer.as_ref(), + self.vshard_id, + txn_id, + kind, + bytes.clone(), + ); self.owed.insert( (txn_id, kind), OwedEntry { @@ -70,6 +72,7 @@ impl Scheduler { pub(in crate::control::cluster::calvin::scheduler::driver::core) fn retry_owed_sequencer_entries( &mut self, ) { + let leads = self.is_group_leader(); let registry = &self.registry; let vshard_id = self.vshard_id; self.owed.retain(|(txn_id, kind), owed| { @@ -79,6 +82,10 @@ impl Scheduler { }); for ((txn_id, kind), owed) in self.owed.iter_mut() { + if kind.leader_only() && !leads { + // Stays owed: this replica proposes it if it comes to lead. + continue; + } if owed.in_flight { owed.in_flight = false; continue; @@ -95,6 +102,18 @@ impl Scheduler { } } + /// Whether this scheduler can propose an entry of `kind` now. + /// + /// A vote and a `CompletionAck` answer for the vShard on every node: the + /// first one the sequencer log holds counts. So only the vShard's + /// data-group leader proposes them. A node that left the group leads none + /// of it, so an entry from its stale state never enters the log. Every + /// other replica keeps the entry owed and proposes it if it comes to lead + /// before one applies. Which entry counts then depends on the log alone. + fn may_propose(&self, kind: OwedKind) -> bool { + !kind.leader_only() || self.is_group_leader() + } + /// The registry's view of `txn_id` for this scheduler's vShard. fn participant_progress(&self, txn_id: TxnId) -> Option { self.registry @@ -184,7 +203,6 @@ mod tests { epoch: 20, position: 1, vshard: VSHARD, - commit: true, }] ); @@ -229,6 +247,8 @@ mod tests { ); let proposer = CapturingProposer::failing_first(1); scheduler.sequencer_proposer = proposer.clone(); + // Only the vShard's group leader proposes the ack. + elect_data_group_leader(&scheduler); scheduler.registry.seed_expected(cluster_txn_id(txn_id), 2); scheduler @@ -244,6 +264,8 @@ mod tests { epoch: 21, position: 0, vshard_id: VSHARD, + result: Vec::new(), + from_node: scheduler.shared.node_id, }] ); scheduler.retry_owed_sequencer_entries(); @@ -274,12 +296,16 @@ mod tests { let (mut scheduler, _dir) = build_test_scheduler(VSHARD); let proposer = CapturingProposer::failing_first(1); scheduler.sequencer_proposer = proposer.clone(); + elect_data_group_leader(&scheduler); let txn_id = TxnId::new(22, 0); let _outcome = scheduler .registry .register_completion(cluster_txn_id(txn_id), 1); - scheduler.propose_sequencer_entry(txn_id, SchedulerProposal::CompletionAck); + scheduler.propose_sequencer_entry( + txn_id, + SchedulerProposal::CompletionAck { result: Vec::new() }, + ); scheduler .registry .note_completion_ack(cluster_txn_id(txn_id), VSHARD); @@ -290,6 +316,59 @@ mod tests { assert!(scheduler.owed.is_empty()); } + /// A replica that does not lead the vShard's group proposes no ack. It + /// keeps the ack owed, so it proposes it if it comes to lead first. + #[tokio::test] + async fn a_replica_that_does_not_lead_proposes_no_ack() { + let (mut scheduler, _dir) = build_test_scheduler(VSHARD); + let proposer = CapturingProposer::accepting(); + scheduler.sequencer_proposer = proposer.clone(); + let txn_id = TxnId::new(25, 0); + + scheduler.propose_sequencer_entry( + txn_id, + SchedulerProposal::CompletionAck { result: Vec::new() }, + ); + scheduler.retry_owed_sequencer_entries(); + assert_eq!(proposer.attempt_count(), 0); + assert_eq!(scheduler.owed.len(), 1); + + elect_data_group_leader(&scheduler); + scheduler.retry_owed_sequencer_entries(); + assert_eq!(proposer.attempt_count(), 1); + } + + /// A follower that staged a txn owes an abort vote. It proposes the vote + /// only once it leads the vShard's group and no vote has applied, so the + /// txn ends in a retryable abort instead of waiting on a lost leader. + #[tokio::test] + async fn a_follower_that_comes_to_lead_votes_abort_for_its_staged_txn() { + let (mut scheduler, _dir) = build_test_scheduler(VSHARD); + let proposer = CapturingProposer::accepting(); + scheduler.sequencer_proposer = proposer.clone(); + let txn_id = TxnId::new(26, 0); + scheduler.registry.seed_expected(cluster_txn_id(txn_id), 2); + scheduler + .pending + .insert(txn_id, staged_pending(make_sequenced_txn(26, 0), txn_id)); + + scheduler.resolve_staged_commit(txn_id, &staged_response(Status::Ok, Some(true))); + scheduler.retry_owed_sequencer_entries(); + assert_eq!(proposer.attempt_count(), 0, "a follower casts no vote"); + + elect_data_group_leader(&scheduler); + scheduler.retry_owed_sequencer_entries(); + assert_eq!( + proposer.accepted(), + vec![SequencerEntry::AbortVote { + epoch: 26, + position: 0, + vshard: VSHARD, + reason: nodedb_cluster::calvin::AbortReason::SerializationConflict, + }] + ); + } + /// A txn this node never seeded keeps its entry owed: a missing registry /// entry that was never seen does not settle it. #[tokio::test] diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/parts.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/parts.rs new file mode 100644 index 000000000..c0601d7f3 --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/parts.rs @@ -0,0 +1,749 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Multi-part transactions on this vShard's scheduler. +//! +//! A multi-part transaction's header arrives as an ordinary sequenced txn, +//! with its full read and write sets and no plans. The scheduler takes its +//! locks at the header's position, as for any txn, so lock order stays the +//! sequence order. Its plans arrive later as parts, each targeting the +//! vShards its tasks route to. +//! +//! - At the header, the scheduler opens an assembly: how many parts target +//! this vShard, per the header's manifest. +//! - Each part is kept by its index in the transaction. A part delivered +//! again is ignored. Parts can arrive in any order. +//! - Once every part arrived, the scheduler reads them in index order. Each +//! part decodes once, and only the tasks that route here are kept, with +//! their index in the whole transaction. +//! - A task too large for one part travels as a run of chunk parts with +//! consecutive indexes. Every chunk targets the task's homes, so a home +//! receives the whole run. The task decodes from its joined chunks. +//! - The assembly reads only the parts' contents, never their arrival +//! order. A part that cannot be read therefore fails the transaction on +//! every replica alike. +//! - A txn whose locks are granted before its parts arrived waits in +//! `awaiting`, holding its locks. It counts as unfinished, so every later +//! txn's flush on this vShard waits for it. +//! - Once every part that targets this vShard arrived, the txn stages as a +//! single-entry txn whose plans are its local tasks. Its manifest keeps +//! the whole-transaction facts the dispatch reads. +//! - An abandoned txn stages nothing. It releases its locks and completes +//! when its locks are granted, or at once when it already holds them. + +use std::collections::BTreeMap; +use std::sync::Arc; + +use nodedb_cluster::calvin::types::{SequencedTxn, TaskChunk, TxnIdWire}; +use nodedb_physical::physical_plan::PhysicalPlan; +use tracing::{error, warn}; + +use super::routing::{PlanRouting, plan_vshard_in_database}; +use super::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + +/// One part as it arrived: its first task, its bytes, and its chunk. +#[derive(Debug)] +struct ReceivedPart { + first_task: u32, + plans: Arc>, + chunk: Option, +} + +/// The parts one multi-part txn owes this vShard, and the ones that arrived. +#[derive(Debug)] +struct Assembly { + database_id: crate::types::DatabaseId, + /// How many parts target this vShard. + expected: u32, + /// Every part that arrived, by its index in the txn. + received: BTreeMap, + /// The sequencer abandoned the txn: it stages nothing. + abandoned: bool, +} + +impl Assembly { + fn is_complete(&self) -> bool { + self.received.len() >= self.expected as usize + } +} + +/// A task whose chunks are being joined. +#[derive(Debug)] +struct SpanningTask { + task: u32, + bytes: Vec, + total_len: u64, +} + +/// Reads an assembly's parts, in index order, into its local tasks. +#[derive(Debug)] +struct Assembler { + database_id: crate::types::DatabaseId, + /// `(task index in the whole txn, plan)` of every local task so far. + local: Vec<(u32, PhysicalPlan)>, + /// The task whose chunks are being joined. + spanning: Option, + /// Why a part cannot be read. The txn fails at dispatch. + failed: Option, +} + +impl Assembler { + fn fail(&mut self, detail: impl FnOnce() -> String) { + self.failed.get_or_insert_with(detail); + } + + /// Keep `plan`, the txn's task `task` from part `index`, when it routes + /// to `vshard_id`. + fn keep_if_local(&mut self, index: u32, task: u32, plan: PhysicalPlan, vshard_id: u32) { + match plan_vshard_in_database(&plan, self.database_id) { + PlanRouting::Vshards(vshards) => { + if vshards.iter().any(|v| v.as_u32() == vshard_id) { + self.local.push((task, plan)); + } + } + PlanRouting::ControlPlaneOnly => { + self.fail(|| format!("part {index} task {task} is a control-plane-only plan")); + } + PlanRouting::NotAWrite => { + self.fail(|| format!("part {index} task {task} is not a write")); + } + PlanRouting::Unroutable(reason) => { + self.fail(|| format!("part {index} task {task} is unroutable: {reason}")); + } + } + } + + /// Take part `index`, a batch of whole tasks from task `first_task`. + fn take_tasks(&mut self, index: u32, first_task: u32, plans: &[u8], vshard_id: u32) { + if let Some(spanning) = &self.spanning { + let task = spanning.task; + self.fail(|| format!("part {index} interrupts the chunks of task {task}")); + return; + } + match super::super::helpers::decode_plans(plans) { + Ok(decoded) => { + for (task, plan) in (first_task..).zip(decoded) { + self.keep_if_local(index, task, plan, vshard_id); + } + } + Err(error) => self.fail(|| format!("part {index} does not decode: {error}")), + } + } + + /// Take part `index`, one chunk of task `task`. + fn take_chunk( + &mut self, + index: u32, + task: u32, + bytes: &[u8], + chunk: TaskChunk, + vshard_id: u32, + ) { + let continues = match &self.spanning { + None => chunk.offset == 0, + Some(spanning) => { + spanning.task == task + && spanning.bytes.len() as u64 == chunk.offset + && spanning.total_len == chunk.total_len + } + }; + if !continues { + self.fail(|| format!("part {index} does not continue the chunks of task {task}")); + self.spanning = None; + return; + } + let spanning = self.spanning.get_or_insert_with(|| SpanningTask { + task, + bytes: Vec::new(), + total_len: chunk.total_len, + }); + spanning.bytes.extend_from_slice(bytes); + let len = spanning.bytes.len() as u64; + if len < spanning.total_len { + return; + } + let Some(whole) = self.spanning.take() else { + return; + }; + if len > whole.total_len { + self.fail(|| format!("task {task} chunks run past its {} bytes", whole.total_len)); + return; + } + match nodedb_physical::physical_plan::wire::decode(&whole.bytes) { + Ok(plan) => self.keep_if_local(index, task, plan, vshard_id), + Err(error) => self.fail(|| format!("task {task} does not decode: {error}")), + } + } +} + +/// A txn holding its locks while its parts arrive. +#[derive(Debug)] +struct AwaitingParts { + txn: SequencedTxn, + lock_owner: TxnId, +} + +/// The multi-part txns this vShard participates in and has not staged. +#[derive(Debug, Default)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) struct PartsState { + assemblies: BTreeMap, + awaiting: BTreeMap, + /// The inputs a closed gate still takes (see [`super::parts_lane`]). + pub(in crate::control::cluster::calvin::scheduler::driver::core) lane: + super::parts_lane::PartsLane, +} + +impl PartsState { + /// Whether `txn_id` holds its locks and waits for parts. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn is_awaiting( + &self, + txn_id: TxnId, + ) -> bool { + self.awaiting.contains_key(&txn_id) + } + + /// How many txns hold their locks and wait for parts. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn awaiting_len( + &self, + ) -> usize { + self.awaiting.len() + } + + /// Whether the assembly of `txn` is open: its header was taken here and + /// its parts are not all assembled. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn has_assembly( + &self, + txn: TxnId, + ) -> bool { + self.assemblies.contains_key(&txn) + } + + /// The lowest txn that holds its locks and waits for parts. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn lowest_awaiting( + &self, + ) -> Option { + self.awaiting.keys().next().copied() + } +} + +impl Scheduler { + /// Open the assembly of `txn` when it is a multi-part txn this vShard + /// has not seen. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn open_assembly( + &mut self, + txn: &SequencedTxn, + ) { + let Some(manifest) = &txn.tx_class.multi_part else { + return; + }; + let txn_id = TxnId::new(txn.epoch, txn.position); + let vshard_id = self.vshard_id; + self.parts + .assemblies + .entry(txn_id) + .or_insert_with(|| Assembly { + database_id: txn.tx_class.database_id, + expected: manifest.parts_for(vshard_id), + received: BTreeMap::new(), + abandoned: false, + }); + } + + /// Pass a ready txn to the dispatch, or hold it until its parts arrived. + /// + /// A multi-part txn with every part that targets this vShard is + /// assembled here. One still missing parts waits with its locks. An + /// abandoned one, and one whose part cannot be read, completes here. + /// + /// Returns the txn to dispatch: a single-entry txn, or a multi-part one + /// with every part that targets this vShard assembled. `None` means + /// nothing dispatches: the txn waits for parts, or it completed. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn gate_on_parts( + &mut self, + txn: SequencedTxn, + txn_id: TxnId, + lock_owner: TxnId, + ) -> Option { + if txn.tx_class.multi_part.is_none() { + return Some(txn); + } + // Intake opens the assembly before the locks, and only this gate or + // an abandonment of a txn holding its locks closes it. + let (abandoned, missing) = match self.parts.assemblies.get(&txn_id) { + Some(assembly) => (assembly.abandoned, !assembly.is_complete()), + None => (false, false), + }; + if abandoned { + self.parts.assemblies.remove(&txn_id); + self.on_unpending_txn_complete(txn_id, lock_owner); + return None; + } + if missing { + self.parts + .awaiting + .insert(txn_id, AwaitingParts { txn, lock_owner }); + return None; + } + let vshard_id = self.vshard_id; + let assembled = match self.parts.assemblies.remove(&txn_id) { + Some(assembly) => assemble(txn, assembly, vshard_id), + None => Err("the txn reached dispatch with no assembly of its parts".to_owned()), + }; + match assembled { + Ok(txn) => Some(txn), + Err(detail) => { + let error = crate::Error::Internal { + detail: format!( + "calvin multi-part txn {}/{} for vshard {}: {detail}", + txn_id.epoch, txn_id.position, self.vshard_id + ), + }; + error!(vshard_id = self.vshard_id, %error, "calvin scheduler: part unreadable"); + self.propose_routing_failure(txn_id, &error); + self.on_unpending_txn_complete(txn_id, lock_owner); + None + } + } + } + + /// Receive part `index` of `txn`, whose first task is the txn's task + /// `first_task`. `chunk` is set on a part that holds one byte range of + /// that task. The part is kept by its index, so any delivery order + /// assembles alike. A part this vShard already holds, or of a txn it + /// holds no assembly for, is ignored. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn receive_part( + &mut self, + txn: TxnIdWire, + index: u32, + first_task: u32, + plans: Arc>, + chunk: Option, + ) { + let txn_id = TxnId::from(txn); + let Some(assembly) = self.parts.assemblies.get_mut(&txn_id) else { + return; + }; + // A catch-up replay delivers the parts that follow an abandonment. + if assembly.abandoned || assembly.is_complete() || assembly.received.contains_key(&index) { + return; + } + assembly.received.insert( + index, + ReceivedPart { + first_task, + plans, + chunk, + }, + ); + let complete = assembly.is_complete(); + if complete && let Some(awaiting) = self.parts.awaiting.remove(&txn_id) { + self.dispatch_or_barrier(awaiting.txn, txn_id, awaiting.lock_owner); + } + } + + /// The sequencer abandoned `txn`: it lost parts. A txn holding its locks + /// completes at once. One still waiting for locks completes when granted. + /// + /// An abandonment that follows every part targeting this vShard is + /// ignored. The sequencer applies an abandonment only while the txn is + /// open, and a catch-up replay delivers one it ignored. Here the verdict + /// decides instead: the registry holds `Abort(PartsLost)` when the + /// sequencer applied the abandonment, and the staged txn reads it. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn receive_parts_abandoned( + &mut self, + txn: TxnIdWire, + ) { + let txn_id = TxnId::from(txn); + let Some(assembly) = self.parts.assemblies.get_mut(&txn_id) else { + return; + }; + if assembly.is_complete() { + return; + } + assembly.abandoned = true; + warn!( + vshard_id = self.vshard_id, + epoch = txn_id.epoch, + position = txn_id.position, + "calvin scheduler: multi-part txn abandoned before its parts arrived" + ); + if let Some(awaiting) = self.parts.awaiting.remove(&txn_id) { + self.parts.assemblies.remove(&txn_id); + self.on_unpending_txn_complete(txn_id, awaiting.lock_owner); + } + } +} + +/// The single-entry form of `txn`: its local tasks as `plans`, in their +/// order in the whole txn, and `body_plans` as indexes into them. The +/// manifest stays for the whole-transaction facts. +/// +/// Reads the parts in index order, so the result depends only on their +/// contents. Every replica that holds the same parts assembles the same +/// plans, or fails alike. +fn assemble( + mut txn: SequencedTxn, + assembly: Assembly, + vshard_id: u32, +) -> Result { + let mut assembler = Assembler { + database_id: assembly.database_id, + local: Vec::new(), + spanning: None, + failed: None, + }; + for (index, part) in &assembly.received { + match part.chunk { + Some(chunk) => { + assembler.take_chunk(*index, part.first_task, &part.plans, chunk, vshard_id); + } + None => assembler.take_tasks(*index, part.first_task, &part.plans, vshard_id), + } + } + if let Some(failed) = assembler.failed { + return Err(failed); + } + if let Some(spanning) = assembler.spanning { + return Err(format!( + "task {} stopped at byte {} of {}", + spanning.task, + spanning.bytes.len(), + spanning.total_len + )); + } + let mut local = assembler.local; + local.sort_by_key(|(task, _)| *task); + let body_plans: Vec = (0u32..) + .zip(&local) + .filter(|(_, (task, _))| txn.tx_class.body_plans.contains(task)) + .map(|(local_index, _)| local_index) + .collect(); + let plans: Vec<&PhysicalPlan> = local.iter().map(|(_, plan)| plan).collect(); + txn.tx_class.plans = + zerompk::to_msgpack_vec(&plans).map_err(|e| format!("local plans do not encode: {e}"))?; + txn.tx_class.body_plans = body_plans; + Ok(txn) +} + +#[cfg(test)] +mod tests { + use nodedb_cluster::calvin::CalvinCompletionRegistry; + use nodedb_cluster::calvin::types::{ + MultiPartPlans, PartStreamId, SchedulerInput, VShardParts, + }; + use nodedb_physical::physical_plan::DocumentOp; + use nodedb_physical::physical_plan::wire as plan_wire; + use nodedb_types::QualifiedCollection; + + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + build_test_scheduler_with_data_side, make_local_write_txn, test_coll_vshard, + }; + use crate::types::DatabaseId; + + fn truncate_plan() -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::Truncate { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "test_coll"), + restart_identity: false, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }) + } + + /// A single-entry truncate of `test_coll` at `(epoch, position)`. Each + /// test's epoch holds two txns on this vShard, so the epoch folds only + /// once both completed. + fn plain_write(epoch: u64, position: u32) -> SequencedTxn { + let mut txn = make_local_write_txn(epoch, position); + txn.epoch_vshard_txn_count = 2; + txn + } + + /// The header of a txn of `parts` parts carrying `tasks` tasks, every + /// part targeting the `test_coll` vShard. + fn header(epoch: u64, position: u32, parts: u32, tasks: u32) -> SequencedTxn { + let mut txn = plain_write(epoch, position); + txn.tx_class.plans = Vec::new(); + txn.tx_class.multi_part = Some(MultiPartPlans { + stream: PartStreamId { node: 1, seq: 1 }, + part_count: parts, + total_tasks: tasks, + user_write: true, + client_write: true, + per_vshard: vec![VShardParts { + vshard: test_coll_vshard(), + parts, + }], + }); + txn + } + + /// The header of a two-part txn: one truncate task per part. + fn two_part_header(epoch: u64, position: u32) -> SequencedTxn { + header(epoch, position, 2, 2) + } + + fn part(epoch: u64, position: u32, index: u32, plans: Vec) -> SchedulerInput { + SchedulerInput::TxnPart { + txn: TxnIdWire { epoch, position }, + index, + first_task: index, + plans: Arc::new(plans), + chunk: None, + } + } + + fn chunk_part( + epoch: u64, + index: u32, + task: u32, + bytes: &[u8], + offset: u64, + total_len: u64, + ) -> SchedulerInput { + SchedulerInput::TxnPart { + txn: TxnIdWire { epoch, position: 0 }, + index, + first_task: task, + plans: Arc::new(bytes.to_vec()), + chunk: Some(TaskChunk { offset, total_len }), + } + } + + fn one_truncate() -> Vec { + plan_wire::encode_batch(&vec![truncate_plan()]).expect("encode one truncate") + } + + /// A txn whose locks are granted before its parts holds them and waits. + /// It stages once the last part that targets this vShard arrived, with + /// every local task of every part. + #[tokio::test] + async fn a_multi_part_txn_stages_once_every_part_arrived() { + let (mut scheduler, _dir, _data_side) = build_test_scheduler_with_data_side( + test_coll_vshard(), + CalvinCompletionRegistry::new_detached(), + ); + let txn_id = TxnId::new(3, 0); + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(two_part_header(3, 0)))); + assert!( + scheduler.parts.is_awaiting(txn_id), + "locks held, parts missing" + ); + assert!(!scheduler.pending.contains_key(&txn_id)); + + scheduler.process_scheduler_input(part(3, 0, 0, one_truncate())); + assert!( + scheduler.parts.is_awaiting(txn_id), + "one part still missing" + ); + + scheduler.process_scheduler_input(part(3, 0, 1, one_truncate())); + assert!(!scheduler.parts.is_awaiting(txn_id)); + let pending = scheduler.pending.get(&txn_id).expect("staged"); + let plans = plan_wire::decode_batch(&pending.txn.tx_class.plans).expect("decode"); + assert_eq!(plans.len(), 2, "both parts' tasks stage together"); + } + + /// A task split over two chunk parts decodes whole once both arrived, + /// and stages beside the whole task of the part before it. + #[tokio::test] + async fn a_task_split_over_chunk_parts_stages_whole() { + let (mut scheduler, _dir, _data_side) = build_test_scheduler_with_data_side( + test_coll_vshard(), + CalvinCompletionRegistry::new_detached(), + ); + let txn_id = TxnId::new(7, 0); + let task = zerompk::to_msgpack_vec(&truncate_plan()).expect("encode one task"); + let half = task.len() / 2; + let total = task.len() as u64; + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(header(7, 0, 3, 2)))); + scheduler.process_scheduler_input(part(7, 0, 0, one_truncate())); + scheduler.process_scheduler_input(chunk_part(7, 1, 1, &task[..half], 0, total)); + assert!(scheduler.parts.is_awaiting(txn_id), "the task is half here"); + scheduler.process_scheduler_input(chunk_part(7, 2, 1, &task[half..], half as u64, total)); + + let pending = scheduler.pending.get(&txn_id).expect("staged"); + let plans = plan_wire::decode_batch(&pending.txn.tx_class.plans).expect("decode"); + assert_eq!(plans, vec![truncate_plan(), truncate_plan()]); + } + + /// Chunk parts that arrive out of order, and again, assemble exactly as + /// in order: the assembly reads parts by index, never by arrival. A + /// replica whose live channel and catch-up replay interleave its parts + /// stages the same plans as every other replica. + #[tokio::test] + async fn chunk_parts_out_of_order_and_repeated_stage_whole() { + let (mut scheduler, _dir, _data_side) = build_test_scheduler_with_data_side( + test_coll_vshard(), + CalvinCompletionRegistry::new_detached(), + ); + let txn_id = TxnId::new(9, 0); + let task = zerompk::to_msgpack_vec(&truncate_plan()).expect("encode one task"); + let third = task.len() / 3; + let total = task.len() as u64; + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(header(9, 0, 3, 1)))); + scheduler.process_scheduler_input(chunk_part( + 9, + 1, + 0, + &task[third..2 * third], + third as u64, + total, + )); + scheduler.process_scheduler_input(chunk_part(9, 0, 0, &task[..third], 0, total)); + scheduler.process_scheduler_input(chunk_part( + 9, + 1, + 0, + &task[third..2 * third], + third as u64, + total, + )); + assert!(scheduler.parts.is_awaiting(txn_id), "one chunk missing"); + scheduler.process_scheduler_input(chunk_part( + 9, + 2, + 0, + &task[2 * third..], + 2 * third as u64, + total, + )); + + let pending = scheduler.pending.get(&txn_id).expect("staged"); + let plans = plan_wire::decode_batch(&pending.txn.tx_class.plans).expect("decode"); + assert_eq!(plans, vec![truncate_plan()]); + } + + /// A chunk that skips bytes fails the txn: nothing stages. + #[tokio::test] + async fn a_chunk_that_skips_bytes_stages_nothing() { + let (mut scheduler, _dir, _data_side) = build_test_scheduler_with_data_side( + test_coll_vshard(), + CalvinCompletionRegistry::new_detached(), + ); + let txn_id = TxnId::new(8, 0); + let task = zerompk::to_msgpack_vec(&truncate_plan()).expect("encode one task"); + let total = task.len() as u64; + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(header(8, 0, 2, 1)))); + scheduler.process_scheduler_input(chunk_part(8, 0, 0, &task[..2], 0, total)); + scheduler.process_scheduler_input(chunk_part(8, 1, 0, &task[3..], 3, total)); + + assert!(!scheduler.pending.contains_key(&txn_id), "nothing staged"); + assert!(scheduler.applied.is_applied(8, 0), "the txn completed"); + } + + /// A late part that does not decode fails the txn before it stages: the + /// earlier part's task never stages, and the txn's locks are released. + #[tokio::test] + async fn a_late_part_that_fails_stages_nothing_of_the_earlier_parts() { + let (mut scheduler, _dir, _data_side) = build_test_scheduler_with_data_side( + test_coll_vshard(), + CalvinCompletionRegistry::new_detached(), + ); + let txn_id = TxnId::new(3, 0); + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(two_part_header(3, 0)))); + scheduler.process_scheduler_input(part(3, 0, 0, one_truncate())); + scheduler.process_scheduler_input(part(3, 0, 1, vec![0xc1, 0xff, 0x00])); + + assert!(!scheduler.pending.contains_key(&txn_id), "nothing staged"); + assert!(!scheduler.parts.is_awaiting(txn_id)); + assert!(scheduler.applied.is_applied(3, 0), "the txn completed"); + + // Its locks are free: the next txn on the same key stages at once. + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(plain_write(3, 1)))); + assert!(scheduler.pending.contains_key(&TxnId::new(3, 1))); + } + + /// An abandoned txn holding its locks completes at once and frees them. + #[tokio::test] + async fn an_abandoned_txn_holding_its_locks_completes_and_frees_them() { + let (mut scheduler, _dir, _data_side) = build_test_scheduler_with_data_side( + test_coll_vshard(), + CalvinCompletionRegistry::new_detached(), + ); + let txn_id = TxnId::new(4, 0); + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(two_part_header(4, 0)))); + scheduler.process_scheduler_input(part(4, 0, 0, one_truncate())); + scheduler.process_scheduler_input(SchedulerInput::PartsAbandoned { + txn: TxnIdWire { + epoch: 4, + position: 0, + }, + }); + + assert!(!scheduler.parts.is_awaiting(txn_id)); + assert!(!scheduler.pending.contains_key(&txn_id)); + assert!(scheduler.applied.is_applied(4, 0)); + // A part that arrives after the abandonment is ignored. + scheduler.process_scheduler_input(part(4, 0, 1, one_truncate())); + assert!(!scheduler.pending.contains_key(&txn_id)); + + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(plain_write(4, 1)))); + assert!(scheduler.pending.contains_key(&TxnId::new(4, 1))); + } + + /// An abandonment after every part that targets this vShard is ignored: + /// a catch-up replay delivers one the sequencer ignored. A part replayed + /// after an abandonment the sequencer applied is ignored too. + #[tokio::test] + async fn an_abandonment_after_the_last_local_part_leaves_the_txn_to_its_verdict() { + let (mut scheduler, _dir, _data_side) = build_test_scheduler_with_data_side( + test_coll_vshard(), + CalvinCompletionRegistry::new_detached(), + ); + let holder = TxnId::new(6, 0); + let multi = TxnId::new(6, 1); + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(plain_write(6, 0)))); + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(two_part_header(6, 1)))); + scheduler.process_scheduler_input(part(6, 1, 0, one_truncate())); + scheduler.process_scheduler_input(part(6, 1, 1, one_truncate())); + scheduler.process_scheduler_input(SchedulerInput::PartsAbandoned { + txn: TxnIdWire { + epoch: 6, + position: 1, + }, + }); + + scheduler.pending.remove(&holder); + scheduler.on_unpending_txn_complete(holder, holder); + assert!( + scheduler.pending.contains_key(&multi), + "stages with its parts" + ); + } + + /// An abandoned txn still waiting for its locks completes when they are + /// granted, and stages nothing. + #[tokio::test] + async fn an_abandoned_txn_waiting_for_locks_completes_when_granted() { + let (mut scheduler, _dir, _data_side) = build_test_scheduler_with_data_side( + test_coll_vshard(), + CalvinCompletionRegistry::new_detached(), + ); + let holder = TxnId::new(5, 0); + let multi = TxnId::new(5, 1); + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(plain_write(5, 0)))); + assert!(scheduler.pending.contains_key(&holder)); + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(two_part_header(5, 1)))); + assert!( + scheduler.blocked.contains_key(&multi), + "waits for the holder's locks" + ); + scheduler.process_scheduler_input(SchedulerInput::PartsAbandoned { + txn: TxnIdWire { + epoch: 5, + position: 1, + }, + }); + assert!( + scheduler.blocked.contains_key(&multi), + "still waits for its locks" + ); + + scheduler.pending.remove(&holder); + scheduler.on_unpending_txn_complete(holder, holder); + assert!(!scheduler.blocked.contains_key(&multi)); + assert!(!scheduler.pending.contains_key(&multi), "stages nothing"); + assert!(scheduler.applied.is_applied(5, 1)); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/parts_lane.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/parts_lane.rs new file mode 100644 index 000000000..041c34aec --- /dev/null +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/parts_lane.rs @@ -0,0 +1,295 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The backlog lane: the inputs a scheduler still takes while its backlog +//! bound holds the intake gate closed. +//! +//! A closed gate processes no new txn, so the backlog stays bounded. Some +//! of the backlog, though, finishes only on input: +//! +//! - A multi-part txn that holds its locks finishes only once its parts +//! arrive, and every later flush on this vShard waits for it. +//! - A blocked txn below it can wait on a reservation release. +//! +//! While the gate is closed for the backlog, the lane lets exactly those +//! inputs through, and holds every other input back: +//! +//! 1. The scheduler arms its own catch-up past the last applied sequencer +//! entry. The sequencer then sends it nothing live, so its channel holds +//! a finite run of inputs, and the log holds the rest. +//! 2. It reads the whole channel. A part or an abandonment of a txn whose +//! assembly is open, and a release no held input depends on, is +//! processed at once. Every other input joins the held queue, in +//! arrival order. The queue never outgrows the channel. +//! 3. Once the channel is empty, it scans the committed log from the armed +//! catch-up start, one window per pass, and processes the same kinds of +//! input it finds there. The catch-up drain replays that range again +//! later, in full. A part delivered twice is ignored, and a release +//! applies once. +//! +//! When the gate opens, the held queue is processed first, in order, before +//! the channel and the catch-up drain. No input is lost or reordered, except +//! the lane's inputs, which commute with every held input: +//! +//! - A part only fills its txn's assembly, which reads parts by index. An +//! abandonment only ends its txn. Neither touches another txn. +//! - A release frees locks. Lock queues grant in arrival order, so a +//! release before a held acquire leaves the same holders once the acquire +//! is processed. A release is held back when a held input names its +//! owner, so it never overtakes the acquire it releases. +//! +//! No deadlock follows. A txn waiting for parts holds its locks, so the +//! parts and any abandonment of it follow its header in the log. They sit +//! in the channel or in the log past the armed start, and the lane reaches +//! both. So does a release a blocked txn waits on. The txn then stages, +//! flushes, and the backlog behind it drains, which opens the gate. + +use std::collections::VecDeque; + +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use nodedb_cluster::calvin::types::SchedulerInput; + +use super::intake::IntakeClosure; +use super::scheduler::Scheduler; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; + +/// What the backlog lane holds between passes. +#[derive(Debug, Default)] +pub(in crate::control::cluster::calvin::scheduler::driver::core) struct PartsLane { + /// Inputs read while the gate was closed and not yet processed, in + /// arrival order. + held: VecDeque, + /// The next sequencer log index the lane's scan reads. + scan_next: Option, +} + +#[cfg(test)] +impl PartsLane { + /// How many inputs the lane holds. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn held_len(&self) -> usize { + self.held.len() + } +} + +impl Scheduler { + /// Run the intake gate for one loop pass, and return whether the run + /// loop can read new input. + /// + /// An open gate first processes the held inputs, in order, re-checking + /// the gate after each one. A gate closed for the backlog runs the lane. + /// The run loop reads new input only once nothing is held. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn pass_intake_lane( + &mut self, + ) -> bool { + loop { + if !self.refresh_intake_gate() { + if self.intake_closure() == Some(IntakeClosure::BacklogFull) { + self.run_backlog_lane(); + } + return false; + } + let Some(input) = self.parts.lane.held.pop_front() else { + // Nothing held: the lane's scan starts afresh next time. + self.parts.lane.scan_next = None; + return true; + }; + self.process_scheduler_input(input); + } + } + + /// One pass of the backlog lane: arm the catch-up, read the channel, and + /// scan one window of the log. + fn run_backlog_lane(&mut self) { + let armed_from = self + .sequencer_state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .arm_catch_up_past_applied(self.vshard_id); + // An empty or closed channel ends the read. A closed channel ends the + // run loop once the gate opens. + while let Ok(input) = self.receiver.try_recv() { + self.take_in_lane(input); + } + let from = self + .parts + .lane + .scan_next + .map_or(armed_from, |next| next.max(armed_from)); + self.parts.lane.scan_next = Some(self.scan_log_for_lane(from)); + } + + /// Process `input` now when the lane lets it through, else hold it. + fn take_in_lane(&mut self, input: SchedulerInput) { + if self.passes_lane(&input) { + self.process_scheduler_input(input); + } else { + self.parts.lane.held.push_back(input); + } + } + + /// Whether the lane processes `input` while the gate is closed. + /// + /// - A part or an abandonment of a txn whose assembly is open. + /// - A release whose owner no held input names: no held reservation of + /// it, and no held txn that locks as it. + fn passes_lane(&self, input: &SchedulerInput) -> bool { + match input { + SchedulerInput::TxnPart { txn, .. } | SchedulerInput::PartsAbandoned { txn } => { + self.parts.has_assembly(TxnId::from(*txn)) + } + SchedulerInput::Release { owner, .. } => { + !self.parts.lane.held.iter().any(|held| match held { + SchedulerInput::Reserve { + owner: reserved, .. + } => reserved == owner, + SchedulerInput::Txn(txn) => txn.lock_owner.as_ref() == Some(owner), + _ => false, + }) + } + SchedulerInput::Txn(_) + | SchedulerInput::Reserve { .. } + | SchedulerInput::CutMarker { .. } => false, + } + } + + /// Scan one window of the committed sequencer log from `from` for the + /// lane's inputs, and return the next index to scan. + /// + /// A failed read leaves `from` for the next pass. The compaction + /// hold-down keeps every index at or past the armed catch-up start. + fn scan_log_for_lane(&mut self, from: u64) -> u64 { + let hi = self + .sequencer_state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .current_committed_index(); + let Some(hi) = hi.filter(|&hi| hi >= from) else { + return from; + }; + let window = self.config.catch_up_window.max(1); + let end = from.saturating_add(window - 1).min(hi); + let entries = { + let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + match mr.read_committed_entries(SEQUENCER_GROUP_ID, from, end) { + Ok(entries) => entries, + Err(error) => { + tracing::warn!( + vshard = self.vshard_id, + from, + end, + %error, + "calvin backlog lane: failed to read committed sequencer entries" + ); + return from; + } + } + }; + let inputs = self + .sequencer_state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .replay_epochs_for_vshard(&entries, self.vshard_id, 0, u64::MAX); + for input in inputs { + if self.passes_lane(&input) { + self.process_scheduler_input(input); + } + } + end.saturating_add(1) + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_cluster::calvin::CalvinCompletionRegistry; + use nodedb_cluster::calvin::types::{ + MultiPartPlans, PartStreamId, SequencedTxn, TxnIdWire, VShardParts, + }; + use nodedb_physical::physical_plan::wire as plan_wire; + use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; + use nodedb_types::QualifiedCollection; + + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + build_test_scheduler_with_data_side, make_local_write_txn, test_coll_vshard, + }; + use crate::types::DatabaseId; + + fn truncate_plan() -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::Truncate { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "test_coll"), + restart_identity: false, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }) + } + + /// The header of a one-part txn at `(epoch, 0)` targeting the + /// `test_coll` vShard. + fn one_part_header(epoch: u64) -> SequencedTxn { + let mut txn = make_local_write_txn(epoch, 0); + txn.epoch_vshard_txn_count = 1; + txn.tx_class.plans = Vec::new(); + txn.tx_class.multi_part = Some(MultiPartPlans { + stream: PartStreamId { node: 1, seq: 1 }, + part_count: 1, + total_tasks: 1, + user_write: true, + client_write: true, + per_vshard: vec![VShardParts { + vshard: test_coll_vshard(), + parts: 1, + }], + }); + txn + } + + /// With the gate closed at the backlog bound, a txn waiting for its + /// part still receives it through the lane, and stages. A new txn read + /// with it is held, unprocessed, until the gate opens. + #[tokio::test] + async fn a_closed_gate_lets_an_awaited_part_through_and_holds_new_txns() { + let (mut scheduler, _dir, _data_side) = build_test_scheduler_with_data_side( + test_coll_vshard(), + CalvinCompletionRegistry::new_detached(), + ); + let (input_tx, input_rx) = tokio::sync::mpsc::channel(8); + scheduler.receiver = input_rx; + scheduler.config.max_inflight_backlog = 1; + let awaiting = TxnId::new(3, 0); + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(one_part_header(3)))); + assert!(scheduler.parts.is_awaiting(awaiting)); + assert_eq!( + scheduler.intake_closure(), + Some(IntakeClosure::BacklogFull), + "a txn waiting for parts counts against the bound" + ); + + let later = make_local_write_txn(4, 0); + input_tx + .try_send(SchedulerInput::Txn(Box::new(later))) + .expect("room on the channel"); + let plans = plan_wire::encode_batch(&vec![truncate_plan()]).expect("encode"); + input_tx + .try_send(SchedulerInput::TxnPart { + txn: TxnIdWire { + epoch: 3, + position: 0, + }, + index: 0, + first_task: 0, + plans: Arc::new(plans), + chunk: None, + }) + .expect("room on the channel"); + + assert!(!scheduler.pass_intake_lane(), "the gate stays closed"); + assert!( + !scheduler.parts.is_awaiting(awaiting), + "the part got through the lane" + ); + assert!(scheduler.pending.contains_key(&awaiting), "it staged"); + assert_eq!(scheduler.parts.lane.held_len(), 1, "the new txn is held"); + assert!(!scheduler.pending.contains_key(&TxnId::new(4, 0))); + } +} diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs index 9392b872e..d8b9e8818 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/process.rs @@ -15,7 +15,7 @@ use super::super::barrier::PendingDependentBarrier; use super::scheduler::Scheduler; use crate::control::cluster::calvin::scheduler::lock_manager::{AcquireOutcome, TxnId}; -/// Epochs a read reservation may live before the scheduler reaps it as orphaned. +/// Epochs a read reservation can live before the scheduler reaps it as orphaned. /// At the default 20ms epoch tick this is ~5s of wall-clock — far longer than any /// real think-time between reservation install and commit, yet short enough that a /// crashed coordinator's reservation is reclaimed promptly. Expressed in epochs @@ -50,22 +50,35 @@ impl Scheduler { } match input { - SchedulerInput::Txn(txn) => self.process_new_txn(txn), + SchedulerInput::Txn(txn) => self.process_or_hold_for_metadata(*txn), SchedulerInput::Reserve { owner, key } => self.install_reservation(owner, key), SchedulerInput::Release { owner, reason } => self.release_reservation(owner, reason), SchedulerInput::CutMarker { hlc } => self.receive_cut_marker(hlc), + SchedulerInput::TxnPart { + txn, + index, + first_task, + plans, + chunk, + } => self.receive_part(txn, index, first_task, plans, chunk), + SchedulerInput::PartsAbandoned { txn } => self.receive_parts_abandoned(txn), } } /// The replicated epoch an input is stamped with — the monotonic logical /// clock the lease reap advances on. A cut marker carries no epoch, so it - /// reports `0`, which never advances the clock. + /// reports `0`, which never advances the clock. A part and an abandonment + /// report `0` too: they name their header's epoch, which this scheduler + /// already saw, and a replay delivers an abandonment to vShards the live + /// fan-out skips. fn input_epoch(input: &SchedulerInput) -> u64 { match input { SchedulerInput::Txn(txn) => txn.epoch, SchedulerInput::Reserve { owner, .. } => owner.epoch, SchedulerInput::Release { owner, .. } => owner.epoch, - SchedulerInput::CutMarker { .. } => 0, + SchedulerInput::CutMarker { .. } + | SchedulerInput::TxnPart { .. } + | SchedulerInput::PartsAbandoned { .. } => 0, } } @@ -134,12 +147,12 @@ impl Scheduler { // Exact per-position skip: never re-apply a position that already // committed (its CalvinApplied marker is durable), and never re-run a // whole epoch that has fully folded into the watermark. Re-running an - // applied position would re-fire its side effects — this gate IS the + // applied position will re-fire its side effects — this gate IS the // exactly-once mechanism. Skipping a whole epoch on its first completing // position (the previous per-epoch gate) dropped every other position of // that epoch across a restart: a torn transaction. if self.applied.is_applied(txn.epoch, txn.position) { - // Learning the count for an already-applied position may complete a + // Learning the count for an already-applied position can complete a // historical epoch's applied set (during restart re-fan-out), folding // it into the watermark and pruning its tail — bounding memory. if let Some(watermark) = self.applied.advance() { @@ -151,7 +164,7 @@ impl Scheduler { // In-flight guard (catch-up-replay idempotency). Skip a txn that is // already in-flight on this scheduler — dispatched-and-awaiting-response // (`pending`), blocked on locks (`blocked`), or parked on a dependent-read - // barrier (`dependent_barrier`). Re-running any of these would dispatch a + // barrier (`dependent_barrier`). Re-running any of these will dispatch a // SECOND copy and double-execute the transaction. // // This is a strict NO-OP for LIVE inputs: the sequencer delivers each @@ -160,15 +173,19 @@ impl Scheduler { // when the catch-up drain replays a committed log range that overlaps an // input already delivered live and still in-flight — the exact overlap // the drain cannot avoid (it replays from the earliest dropped index - // forward, which may re-cover inputs that were NOT dropped). Reserve / + // forward, which can re-cover inputs that were NOT dropped). Reserve / // Release replay is already idempotent in the lock manager, so only Txn // needs this guard. if self.pending.contains_key(&txn_id) || self.blocked.contains_key(&lock_owner) || self.dependent_barrier.contains_key(&txn_id) + || self.parts.is_awaiting(txn_id) { return; } + // A multi-part txn collects the parts that target this vShard from + // here on, whether its locks are granted now or later. + self.open_assembly(&txn); let keys = super::super::helpers::expand_rw_set(&txn); let keys_count = keys.len(); @@ -211,6 +228,10 @@ impl Scheduler { txn_id: TxnId, lock_owner: TxnId, ) { + // A multi-part txn dispatches only once its parts arrived. + let Some(txn) = self.gate_on_parts(txn, txn_id, lock_owner) else { + return; + }; let is_dependent = txn.tx_class.dependent_reads.is_some(); if is_dependent { self.insert_dependent_barrier(txn, txn_id, lock_owner); @@ -307,6 +328,8 @@ impl Scheduler { if let Some(watermark) = folded { self.publish_watermark(watermark); } + // A finished txn can give the next committed flush its turn. + self.pump_flush_turn(); } /// Dispatch transactions that a `LockManager::release` promoted to holder. @@ -342,7 +365,7 @@ impl Scheduler { for waiter_id in promoted { let Some(blocked) = self.blocked.get(&waiter_id) else { // A promotion can only name a waiter this scheduler enqueued - // (its key set lives in `blocked`), so a miss should not + // (its key set lives in `blocked`), so a miss must not // happen. Skip defensively rather than panic — the txn holds // no dispatch state here to act on. tracing::debug!( @@ -413,7 +436,8 @@ mod tests { fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); let watermark_before = shared.calvin.last_applied_epoch.load(Ordering::Acquire); - scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + scheduler + .process_scheduler_input(SchedulerInput::Txn(Box::new(make_validate_only_txn(3, 0)))); assert!( !scheduler.applied.is_applied(3, 0), @@ -436,8 +460,10 @@ mod tests { let shared = Arc::clone(&scheduler.shared); fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); - scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); - scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(4, 0))); + scheduler + .process_scheduler_input(SchedulerInput::Txn(Box::new(make_validate_only_txn(3, 0)))); + scheduler + .process_scheduler_input(SchedulerInput::Txn(Box::new(make_validate_only_txn(4, 0)))); assert!( scheduler.blocked.contains_key(&TxnId::new(4, 0)), @@ -455,7 +481,8 @@ mod tests { fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); let tracked_before = shared.tracker.in_flight(); - scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + scheduler + .process_scheduler_input(SchedulerInput::Txn(Box::new(make_validate_only_txn(3, 0)))); assert_eq!( shared.tracker.in_flight(), @@ -474,7 +501,8 @@ mod tests { let shared = Arc::clone(&scheduler.shared); let fillers = fill_tenant_inflight(&shared, &mut data_side, TenantId::new(1)); - scheduler.process_scheduler_input(SchedulerInput::Txn(make_validate_only_txn(3, 0))); + scheduler + .process_scheduler_input(SchedulerInput::Txn(Box::new(make_validate_only_txn(3, 0)))); let running = spawn_scheduler_loop(scheduler); release_filler(&shared, &mut data_side, fillers[0]); @@ -522,7 +550,7 @@ mod tests { // exactly what `drain_catch_up` does for a dropped-then-recovered input that // overlaps an already-in-flight live one. The in-flight guard must turn it // into a no-op: no second dispatch, no duplicate in-flight entry. - scheduler.process_scheduler_input(SchedulerInput::Txn(txn)); + scheduler.process_scheduler_input(SchedulerInput::Txn(Box::new(txn))); let dispatched_after = scheduler.metrics.dispatch_count.load(Ordering::Relaxed); assert_eq!( diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/propose.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/propose.rs index 9d58bcb3d..58c329dc5 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/propose.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/propose.rs @@ -50,7 +50,7 @@ pub async fn propose_calvin_read_result( values: values_payload, }, ); - let data = entry.to_bytes(); + let data = entry.encode()?; raft_proposer(vshard_id, data)?; Ok(()) } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/read_result.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/read_result.rs index 4a5a11539..41a0c0093 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/read_result.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/read_result.rs @@ -19,7 +19,7 @@ impl Scheduler { let barrier = match self.dependent_barrier.get_mut(&txn_id) { Some(b) => b, None => { - // No barrier for this txn — may have already timed out or + // No barrier for this txn — can have already timed out or // been dispatched. Log and ignore. warn!( vshard_id = self.vshard_id, @@ -40,10 +40,9 @@ impl Scheduler { } // All passive results in — remove barrier and dispatch active. - let barrier = self - .dependent_barrier - .remove(&txn_id) - .expect("barrier just confirmed present"); + let Some(barrier) = self.dependent_barrier.remove(&txn_id) else { + return; + }; let injected_reads = barrier.assemble_injected_reads(); let lock_owner = barrier.lock_owner; diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs index 44bd7b658..95d79d73d 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/request.rs @@ -6,15 +6,41 @@ use std::time::{Duration, Instant}; use super::scheduler::Scheduler; use crate::bridge::envelope::{Admission, ExemptReason, Priority, Request}; +use crate::control::cluster::calvin::scheduler::lock_manager::TxnId; use crate::types::{DatabaseId, Lsn, ReadConsistency, RequestId, TenantId, VShardId}; use nodedb_physical::physical_plan::PhysicalPlan; -/// The event source every Calvin sub-operation runs with. The redo record a -/// committed Calvin transaction appends carries the same source, so WAL -/// replay rebuilds the events its flush emits. +/// The event source a Calvin sub-operation runs with when its transaction +/// names none: a client transaction's. pub(in crate::control::cluster::calvin::scheduler::driver::core) const CALVIN_EVENT_SOURCE: crate::event::EventSource = crate::event::EventSource::User; +/// The event source a sequenced transaction's writes carry. Its redo record +/// and its flush both carry it, so WAL replay rebuilds the events the flush +/// emits, and a server-run body's writes fire no triggers. +pub(in crate::control::cluster::calvin::scheduler::driver::core) fn txn_event_source( + tx_class: &nodedb_cluster::calvin::types::TxClass, +) -> crate::event::EventSource { + crate::event::EventSource::from_wal_code(tx_class.event_source).unwrap_or(CALVIN_EVENT_SOURCE) +} + +/// The event source this vShard's slice of a sequenced transaction commits +/// under. A slice only trigger bodies wrote, in a transaction a client also +/// wrote, takes the source a body's row takes in one overlay. Every other +/// slice takes the transaction's source. The stage, the resolve, the redo +/// record and the flush all carry it. +pub(in crate::control::cluster::calvin::scheduler::driver::core) fn slice_event_source( + tx_class: &nodedb_cluster::calvin::types::TxClass, + flush_scope: &super::super::types::FlushScope, +) -> crate::event::EventSource { + let source = txn_event_source(tx_class); + if flush_scope.body_only { + source.committed_row_override(crate::event::EventSource::Trigger) + } else { + source + } +} + impl Scheduler { /// Builds a `Request` for an already-sequenced Calvin sub-operation. /// @@ -50,6 +76,7 @@ impl Scheduler { txn_id: None, wal_lsn, resolved_now_ms: None, + commit_hlc: None, admission: Admission::Exempt(ExemptReason::AlreadyOrdered), } } @@ -65,4 +92,31 @@ impl Scheduler { self.config.epoch_duration_ms * u64::from(self.config.txn_deadline_multiplier), ) } + + /// Spawn a bridge task that awaits a single executor response and forwards + /// it to the scheduler's fan-in completion channel. + /// + /// The bridge task is cancel-safe: it holds only a cloned sender and the + /// per-request receiver. Dropping the scheduler's `completion_rx` causes + /// the bridge's `send` to fail silently, which is fine on shutdown. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn spawn_response_bridge( + &self, + txn_id: TxnId, + request_id: RequestId, + mut response_rx: crate::control::ResponseReceiver, + ) { + let tx = self.completion_tx.clone(); + tokio::spawn(async move { + let result = response_rx.recv().await; + // Ignore send error: scheduler has shut down. + let _ = tx.send((txn_id, request_id, result)).await; + }); + } + + /// Allocate a fresh request ID for a dispatch. + pub(in crate::control::cluster::calvin::scheduler::driver::core) fn next_request_id( + &self, + ) -> RequestId { + self.shared.next_request_id() + } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs index 9a7fa84f0..e35f67a60 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/routing.rs @@ -10,10 +10,11 @@ #![deny(clippy::wildcard_enum_match_arm)] use nodedb_physical::physical_plan::{ - ArrayOp, ColumnarOp, CrdtOp, DocumentOp, GraphOp, KvOp, PhysicalPlan, TimeseriesOp, VectorOp, + ArrayOp, ColumnarOp, CrdtOp, DocumentOp, GraphOp, KvOp, MetaOp, PhysicalPlan, TimeseriesOp, + VectorOp, }; -use crate::types::{DatabaseId, VShardId}; +use crate::types::{DatabaseId, RecordHomes, VShardId}; use nodedb_types::{CollectionKey, QualifiedCollection}; /// Where a `PhysicalPlan` routes for Calvin cross-shard scheduling purposes. @@ -33,7 +34,7 @@ pub(crate) enum PlanRouting { /// Calvin-scheduled write txn is itself a bug. NotAWrite, /// A write whose vshard cannot be determined from the plan alone. Named - /// so the caller's error states WHY, not just that routing failed. + /// so the caller's error states WHY, not only that routing failed. Unroutable(&'static str), } @@ -46,10 +47,23 @@ pub(crate) fn homes_versioned_read( database_id: DatabaseId, vshard_id: u32, ) -> bool { - reads.iter().any(|entry| { - CollectionKey::from_qualified_str(database_id, &entry.collection) - .is_ok_and(|key| key.vshard().as_u32() == vshard_id) - }) + reads + .iter() + .any(|entry| versioned_read_homes_on(entry, database_id, vshard_id)) +} + +/// Whether `vshard_id` validates `entry`: its `home_vshard` when set, else +/// its collection's vShard. +pub(crate) fn versioned_read_homes_on( + entry: &nodedb_types::calvin::VersionedReadEntry, + database_id: DatabaseId, + vshard_id: u32, +) -> bool { + match entry.home_vshard { + Some(home) => home == vshard_id, + None => CollectionKey::from_qualified_str(database_id, &entry.collection) + .is_ok_and(|key| key.vshard().as_u32() == vshard_id), + } } /// Route a collection-homed write to the vShard of its canonical key. The @@ -92,10 +106,15 @@ pub(crate) fn plan_vshard_in_database(plan: &PhysicalPlan, database_id: Database PlanRouting::ControlPlaneOnly } // Reads / query operators / metadata ops: `is_write_plan` already - // excludes every variant of these four families upstream. + // excludes every variant of these four families upstream, except a + // RESTORE batch. PhysicalPlan::Text(_) => PlanRouting::NotAWrite, PhysicalPlan::Spatial(_) => PlanRouting::NotAWrite, PhysicalPlan::Query(_) => PlanRouting::NotAWrite, + // A RESTORE batch runs on the vShard it names. + PhysicalPlan::Meta(MetaOp::RestoreRedo(batch)) => { + PlanRouting::Vshards(vec![VShardId::new(batch.vshard)]) + } PhysicalPlan::Meta(_) => PlanRouting::NotAWrite, } } @@ -131,7 +150,7 @@ fn document_routing(op: &DocumentOp, database_id: DatabaseId) -> PlanRouting { DocumentOp::Merge { .. } | DocumentOp::UpdateFromJoin { .. } => PlanRouting::Unroutable( "cross-collection write: source/target co-location is not enforced", ), - // Read-only: it reports what the wrapped write would do and mutates + // Read-only: it reports what the wrapped write will do and mutates // nothing. DocumentOp::ResolveWrite(_) | DocumentOp::PointGet { .. } @@ -176,13 +195,13 @@ fn kv_routing(op: &KvOp, database_id: DatabaseId) -> PlanRouting { KvOp::TransferItem { .. } => PlanRouting::Unroutable( "cross-collection write: source/target co-location is not enforced", ), - // A resolved write carries per-mutation collections and may span two + // A resolved write carries per-mutation collections and can span two // (a resolved `TransferItem`), so the plan alone does not name one // home — same gap as `TransferItem` above. KvOp::ResolvedWrite { .. } => PlanRouting::Unroutable( "resolved KV write: mutations may span collections with no co-location guarantee", ), - // Read-only: it reports what a write would do and mutates nothing. + // Read-only: it reports what a write will do and mutates nothing. KvOp::ResolveWrite(_) | KvOp::Get { .. } | KvOp::Scan { .. } @@ -224,7 +243,7 @@ fn vector_routing(op: &VectorOp, database_id: DatabaseId) -> PlanRouting { VectorOp::ResolvedDirectWrite { .. } => PlanRouting::Unroutable( "resolved governed vector write: proposed directly by the write-resolve orchestrator", ), - // Read-only: it reports what the wrapped write would do and mutates + // Read-only: it reports what the wrapped write will do and mutates // nothing. VectorOp::ResolveDirectWrite(_) | VectorOp::Search { .. } @@ -245,22 +264,15 @@ fn graph_routing(op: &GraphOp) -> PlanRouting { // Edge plans are key-homed (dual-homed across endpoints), not // collection-homed: route to from_key(src) ∪ from_key(dst). GraphOp::EdgePut { src_id, dst_id, .. } | GraphOp::EdgeDelete { src_id, dst_id, .. } => { - let src_vshard = VShardId::from_key(src_id.as_bytes()); - let dst_vshard = VShardId::from_key(dst_id.as_bytes()); - if src_vshard.as_u32() == dst_vshard.as_u32() { - PlanRouting::Vshards(vec![src_vshard]) - } else { - PlanRouting::Vshards(vec![src_vshard, dst_vshard]) - } + PlanRouting::Vshards(RecordHomes::edge(src_id, dst_id).iter().collect()) } // A batch is the union of its edges' homes, under the same key-homing // rule as the single-edge plans above. GraphOp::EdgePutBatch { edges } | GraphOp::EdgeDeleteBatch { edges } => { let mut vshards: Vec = Vec::new(); for edge in edges { - for endpoint in [edge.src_id.as_bytes(), edge.dst_id.as_bytes()] { - let vshard = VShardId::from_key(endpoint); - if !vshards.iter().any(|v| v.as_u32() == vshard.as_u32()) { + for vshard in RecordHomes::edge(&edge.src_id, &edge.dst_id).iter() { + if !vshards.contains(&vshard) { vshards.push(vshard); } } @@ -276,9 +288,18 @@ fn graph_routing(op: &GraphOp) -> PlanRouting { } // Node-label writes are key-homed on `node_id`, the same mechanism the // edge plans use for their endpoints. - GraphOp::SetNodeLabels { node_id, .. } | GraphOp::RemoveNodeLabels { node_id, .. } => { + // A node delete's guard runs on the node's key home, which holds + // every edge incident on the node. + GraphOp::SetNodeLabels { node_id, .. } + | GraphOp::RemoveNodeLabels { node_id, .. } + | GraphOp::NodeEdgeGuard { node_id, .. } => { PlanRouting::Vshards(vec![VShardId::from_key(node_id.as_bytes())]) } + // A TRUNCATE's edge share and a CRDT delete's presence guard run on + // the vShard they name. + GraphOp::TruncateEdges { vshard, .. } | GraphOp::NodePresenceGuard { vshard, .. } => { + PlanRouting::Vshards(vec![VShardId::new(*vshard)]) + } // Read-only: it decides the wrapped delete's policy and mutates nothing. GraphOp::ResolveEdgeDelete(_) | GraphOp::Hop { .. } @@ -295,7 +316,8 @@ fn graph_routing(op: &GraphOp) -> PlanRouting { | GraphOp::WccSuperstep(_) | GraphOp::TemporalNeighbors { .. } | GraphOp::TemporalAlgorithm { .. } - | GraphOp::Stats { .. } => PlanRouting::NotAWrite, + | GraphOp::Stats { .. } + | GraphOp::NodePresenceRead { .. } => PlanRouting::NotAWrite, } } @@ -304,7 +326,7 @@ fn timeseries_routing(op: &TimeseriesOp, database_id: DatabaseId) -> PlanRouting TimeseriesOp::Ingest { collection, .. } | TimeseriesOp::Truncate { collection, .. } => { collection_routing(database_id, collection) } - // Read-only: it reports the lines the wrapped ingest would store and + // Read-only: it reports the lines the wrapped ingest will store and // mutates nothing. TimeseriesOp::ResolveIngest(_) | TimeseriesOp::Scan { .. } => PlanRouting::NotAWrite, } @@ -351,19 +373,20 @@ fn crdt_routing(op: &CrdtOp, database_id: DatabaseId) -> PlanRouting { fn array_routing(op: &ArrayOp) -> PlanRouting { match op { - // Array writes are tile-partitioned; tile->vshard needs catalog - // tile_extents not present on the plan. `Flush` is a write per - // `is_write_plan` (whole-memtable, not per-cell) but is likewise - // keyed only by `ArrayId`, with no collection/tile vshard on the op. - ArrayOp::Put { .. } | ArrayOp::Delete { .. } | ArrayOp::Flush { .. } => { - PlanRouting::Unroutable( - "array writes are tile-partitioned; tile->vshard needs catalog tile_extents not present on the plan", - ) + // A cell write names the vShard its cells' tiles live on. + ArrayOp::Put { vshard_id, .. } | ArrayOp::Delete { vshard_id, .. } => { + PlanRouting::Vshards(vec![VShardId::new(*vshard_id)]) } + // A flush writes every tile of the array on a node, and a + // transaction never stages one. + ArrayOp::Flush { .. } => PlanRouting::Unroutable( + "an array flush writes every tile of the array on a node, and runs outside a \ + transaction", + ), ArrayOp::OpenArray { .. } | ArrayOp::Compact { .. } | ArrayOp::DropArray { .. } - | ArrayOp::RestoreArrayDrop { .. } + | ArrayOp::RekeyArray { .. } | ArrayOp::PurgeArrayDrop { .. } | ArrayOp::Slice { .. } | ArrayOp::Project { .. } @@ -687,14 +710,28 @@ mod tests { } #[test] - fn array_put_is_unroutable() { - let plan = PhysicalPlan::Array(ArrayOp::Put { + fn array_cell_writes_route_to_their_tile_vshard() { + let put = PhysicalPlan::Array(ArrayOp::Put { array_id: ArrayId::new(TenantId::new(1), "genome"), cells_msgpack: Vec::new(), wal_lsn: 0, provenance: None, + vshard_id: 77, }); - assert!(matches!(plan_vshard(&plan), PlanRouting::Unroutable(_))); + assert_eq!(vshards_of(&put), vec![77]); + let delete = PhysicalPlan::Array(ArrayOp::Delete { + array_id: ArrayId::new(TenantId::new(1), "genome"), + coords_msgpack: Vec::new(), + wal_lsn: 0, + provenance: None, + vshard_id: 12, + }); + assert_eq!(vshards_of(&delete), vec![12]); + let flush = PhysicalPlan::Array(ArrayOp::Flush { + array_id: ArrayId::new(TenantId::new(1), "genome"), + wal_lsn: 0, + }); + assert!(matches!(plan_vshard(&flush), PlanRouting::Unroutable(_))); } #[test] diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs index b9c66e849..812a7e563 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/scheduler.rs @@ -173,6 +173,18 @@ pub struct Scheduler { pub(in crate::control::cluster::calvin::scheduler::driver::core) intake: IntakeGate, /// First halt cause, once set. See [`super::halt`]. pub(in crate::control::cluster::calvin::scheduler::driver::core) halt: HaltLatch, + /// The sequenced txn waiting for this node's metadata apply. See + /// [`super::metadata_hold`]. + pub(in crate::control::cluster::calvin::scheduler::driver::core) metadata_hold: + Option, + /// The multi-part transactions this vShard participates in and has not + /// staged yet. See [`super::parts`]. + pub(in crate::control::cluster::calvin::scheduler::driver::core) parts: + super::parts::PartsState, + /// Orders flushes against the data group's snapshots. See + /// [`super::install_gate`]. + pub(in crate::control::cluster::calvin::scheduler::driver::core) install_gate: + super::install_gate::InstallGate, } /// Parameters for [`Scheduler::new`]. @@ -247,6 +259,7 @@ impl Scheduler { // A backup's cut waits on every scheduler this node runs. shared.calvin.cuts.register(vshard_id); + let install_gate = super::install_gate::InstallGate::new(&shared, vshard_id); let capacity_freed = shared .dispatcher @@ -283,6 +296,9 @@ impl Scheduler { capacity_freed, intake: IntakeGate::default(), halt: HaltLatch::default(), + metadata_hold: None, + parts: Default::default(), + install_gate, } } @@ -298,7 +314,7 @@ impl Scheduler { /// `fully_applied_epoch()` is conservatively seeded to the same sentinel by /// recovery (the watermark only advances once the sequencer's re-fan-out /// supplies per-epoch expected-position counts) — it does NOT mean "nothing - /// left to apply". Naively comparing `u64::MAX >= rebuild_target_epoch` would + /// left to apply". Comparing `u64::MAX >= rebuild_target_epoch` will /// therefore report caught-up before a single epoch was actually /// re-applied. So: sentinel `fully_applied_epoch` is caught-up ONLY when /// there is genuinely no rebuild target; otherwise it must NOT be treated as @@ -322,7 +338,7 @@ impl Scheduler { /// `BEGIN` reads `CalvinLocalState::last_applied_epoch` to anchor a /// session's cross-shard snapshot version, so it MUST reflect the /// FULLY-applied epoch — never an epoch that has only some of its positions - /// committed, which would let a session anchor on a torn epoch. `fetch_max` + /// committed, which lets a session anchor on a torn epoch. `fetch_max` /// keeps it monotonic across all per-vShard schedulers writing the counter. pub(in crate::control::cluster::calvin::scheduler::driver::core) fn publish_watermark( &mut self, @@ -338,26 +354,6 @@ impl Scheduler { self.report_passed_cuts(); } - /// Spawn a bridge task that awaits a single executor response and forwards - /// it to the scheduler's fan-in completion channel. - /// - /// The bridge task is cancel-safe: it holds only a cloned sender and the - /// per-request receiver. Dropping the scheduler's `completion_rx` causes - /// the bridge's `send` to fail silently, which is fine on shutdown. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn spawn_response_bridge( - &self, - txn_id: TxnId, - request_id: RequestId, - mut response_rx: crate::control::ResponseReceiver, - ) { - let tx = self.completion_tx.clone(); - tokio::spawn(async move { - let result = response_rx.recv().await; - // Ignore send error: scheduler has shut down. - let _ = tx.send((txn_id, request_id, result)).await; - }); - } - /// Run the scheduler event loop until shutdown is signaled. pub async fn run(mut self, mut shutdown: ShutdownReceiver) { info!( @@ -369,7 +365,7 @@ impl Scheduler { // Low-frequency liveness timer so the top-of-loop stall/barrier sweeps // run even on an otherwise-idle vShard. Without it, a dropped verdict - // push plus zero further events for this vShard would leave a parked txn + // push plus zero further events for this vShard will leave a parked txn // never re-probing the durable verdict. A fraction of the stall-warn // window re-probes well within it. let mut stall_tick = tokio::time::interval(self.config.verdict_stall_warn() / 4); @@ -394,8 +390,12 @@ impl Scheduler { self.check_dependent_barrier_timeouts(); self.check_awaiting_verdict_stalls(); + self.resume_metadata_hold(); - let intake_open = self.refresh_intake_gate(); + // Open only once the inputs a closed gate held are processed. A + // gate closed for the backlog still takes awaited parts and + // releases (see `super::parts_lane`). + let intake_open = self.pass_intake_lane(); if intake_open && catch_up_resume { catch_up_resume = false; stall_tick.reset_immediately(); @@ -409,38 +409,54 @@ impl Scheduler { break; } + // A closed channel below means this node retired the vShard or + // started a new scheduler for it. A closed receiver yields at + // once on every poll, so the loop must leave rather than spin. maybe_completion = self.completion_rx.recv() => { - if let Some((txn_id, request_id, resp_opt)) = maybe_completion { - // Awaited in the arm: the loop takes no other input - // until this completion, its durability wait included, - // is fully handled. - self.handle_completion(txn_id, request_id, resp_opt).await; - } + let Some((txn_id, request_id, resp_opt)) = maybe_completion else { + self.log_superseded("completion"); + break; + }; + // Awaited in the arm: the loop takes no other input + // until this completion, its durability wait included, + // is fully handled. + self.handle_completion(txn_id, request_id, resp_opt).await; } maybe_verdict = self.verdict_rx.recv() => { - if let Some(signal) = maybe_verdict { - // A durable global verdict landed: resume the matching - // parked txn into its flush (commit) or drop (abort). - self.handle_verdict_signal(signal); - } + let Some(signal) = maybe_verdict else { + self.log_superseded("verdict"); + break; + }; + // A durable global verdict landed: resume the matching + // parked txn into its flush (commit) or drop (abort). + self.handle_verdict_signal(signal); } maybe_event = self.read_result_rx.recv() => { - if let Some(event) = maybe_event { - self.handle_read_result(event); - } + let Some(event) = maybe_event else { + self.log_superseded("read result"); + break; + }; + self.handle_read_result(event); } maybe_promoted = self.promotion_rx.recv() => { - if let Some(promoted) = maybe_promoted { - // A fast-path write-admission guard released an uncontended - // key that one of this scheduler's txns had queued behind; - // `release` already promoted it to holder. Run the normal - // promotion -> dispatch path so it stops being a stalled - // holder in `blocked` and actually executes. - self.dispatch_promoted(promoted); - } + let Some(promoted) = maybe_promoted else { + self.log_superseded("promotion"); + break; + }; + // A fast-path write-admission guard released an uncontended + // key that one of this scheduler's txns had queued behind; + // `release` already promoted it to holder. Run the normal + // promotion -> dispatch path so it stops being a stalled + // holder in `blocked` and actually executes. + self.dispatch_promoted(promoted); + } + + _ = tokio::time::sleep(super::metadata_hold::HOLD_POLL), + if self.metadata_hold.is_some() => { + // The next loop pass re-checks the held txn's floor. } _ = &mut capacity_notified, if self.resends_deferred() => { @@ -448,6 +464,12 @@ impl Scheduler { // requests in FIFO order. } + _ = self.install_gate.released(), if self.install_gate.is_waiting() => { + // A snapshot released a group's gate: the waiting flush + // tries again. + self.pump_flush_turn(); + } + maybe_txn = self.receiver.recv(), if intake_open => { match maybe_txn { Some(input) => self.process_scheduler_input(input), @@ -482,11 +504,13 @@ impl Scheduler { self.hold_all_redo_records(); } - /// Allocate a fresh request ID for a dispatch. - pub(in crate::control::cluster::calvin::scheduler::driver::core) fn next_request_id( - &self, - ) -> RequestId { - self.shared.next_request_id() + /// Log that a scheduler channel closed and the loop exits. + fn log_superseded(&self, channel: &str) { + info!( + vshard_id = self.vshard_id, + channel, + "calvin scheduler: a channel closed; the vShard left this node or a new scheduler took it" + ); } } @@ -496,12 +520,26 @@ mod tests { use super::*; use std::collections::BTreeSet; - use crate::control::cluster::calvin::scheduler::driver::core::test_support::build_test_scheduler; + use crate::control::cluster::calvin::scheduler::driver::core::test_support::{ + build_test_scheduler, spawn_scheduler_loop, + }; + + /// Retiring a vShard drops its verdict sender. The scheduler then leaves + /// its loop. A closed receiver is always ready, so a loop that ignored the + /// close would spin and starve the runtime. + #[tokio::test] + async fn a_closed_verdict_channel_ends_the_loop() { + let (scheduler, _dir) = build_test_scheduler(0); + let registry = Arc::clone(&scheduler.registry); + let running = spawn_scheduler_loop(scheduler); + registry.unregister_verdict_signal_sender(0); + assert!(running.exits_unprompted().await); + } /// A freshly-recovered scheduler (`fully_applied_epoch` still the /// `NOT_YET_APPLIED_EPOCH` sentinel) with a REAL, non-zero rebuild target must - /// NOT report caught-up. Naively comparing `u64::MAX >= rebuild_target_epoch` - /// (the bug) would say "caught up" before a single epoch was re-applied. + /// NOT report caught-up. Comparing `u64::MAX >= rebuild_target_epoch` + /// will say "caught up" before a single epoch was re-applied. #[tokio::test] async fn is_caught_up_false_when_fully_applied_is_sentinel_and_target_is_real() { let (mut scheduler, _dir) = build_test_scheduler(0); @@ -541,7 +579,7 @@ mod tests { /// seeds `max_applied_epoch` (hence `rebuild_target_epoch`) to /// `NOT_YET_APPLIED_EPOCH` too (see `recovery.rs`'s /// `greenfield_returns_sentinel_and_empty_tail` test) — this is distinct from - /// a real target of epoch 0 (which would report `max_applied_epoch == 0`). + /// a real target of epoch 0 (which will report `max_applied_epoch == 0`). /// With nothing to rebuild, the scheduler is trivially caught up even though /// `fully_applied_epoch` is still the sentinel. #[tokio::test] diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs index d5549ea9e..a1715242b 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/sequencer_proposer/raft.rs @@ -62,7 +62,7 @@ pub struct RaftSequencerProposer { multi_raft: Arc>, /// Source of the cluster transport and topology for forwards. Weak: the /// node's `SharedState` holds this proposer, and a strong handle back - /// would keep the state, and every file it holds open, alive after + /// will keep the state, and every file it holds open, alive after /// shutdown. shared: Weak, /// One permit per forward RPC in flight. @@ -268,6 +268,8 @@ mod tests { entries: Vec::new(), leader_commit: 0, group_id: SEQUENCER_GROUP_ID, + round: 1, + replicated_floor: 0, }) .expect("heartbeat"); assert_eq!(guard.group_leader(SEQUENCER_GROUP_ID), REMOTE_LEADER); diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/staged_vote.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/staged_vote.rs index fc9f7b70f..b0c7f92db 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/staged_vote.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/staged_vote.rs @@ -5,7 +5,7 @@ use nodedb_cluster::calvin::AbortReason; -use crate::bridge::envelope::{Response, Status}; +use crate::bridge::envelope::{ErrorCode, Response, Status}; /// A participant's local verdict on its staged slice. #[derive(Clone, Copy, Debug, PartialEq, Eq)] @@ -16,6 +16,12 @@ pub(super) enum StagedVote { SerializationConflict, /// The participant never staged, so no read-set was ever validated. ParticipantError, + /// A collection the transaction names no longer holds its planned + /// incarnation in this replica's catalog. + CollectionSuperseded, + /// The stage refused with `OllpRetryRequired`: the state the coordinator + /// predicted moved. The coordinator reads again and retries. + PredictionDrift, } impl StagedVote { @@ -25,16 +31,21 @@ impl StagedVote { Self::Commit => None, Self::SerializationConflict => Some(AbortReason::SerializationConflict), Self::ParticipantError => Some(AbortReason::ParticipantError), + Self::CollectionSuperseded => Some(AbortReason::CollectionSuperseded), + Self::PredictionDrift => Some(AbortReason::PredictionDrift), } } } /// Derive a local staged vote without ever treating an executor error as a -/// commit, and keep the two abort causes apart. A `None` read-set stays +/// commit, and keep the abort causes apart. A `None` read-set stays /// affirmative only for a successful dependent-read or active staged response, /// which has no versioned read-set. pub(super) fn staged_commit_vote(response: &Response) -> StagedVote { if response.status != Status::Ok { + if response.error_code.as_deref() == Some(&ErrorCode::OllpRetryRequired) { + return StagedVote::PredictionDrift; + } return StagedVote::ParticipantError; } if response.read_set_valid == Some(false) { @@ -67,7 +78,7 @@ mod tests { #[test] fn executor_error_votes_participant_error_whatever_the_read_set_field_says() { // The participant never staged, so no read-set was ever validated — - // reporting a serialization conflict here would be a lie. + // reporting a serialization conflict here will be a lie. assert_eq!( staged_commit_vote(&staged_response(Status::Error, None)), StagedVote::ParticipantError @@ -82,6 +93,17 @@ mod tests { ); } + #[test] + fn an_ollp_retry_stage_error_votes_prediction_drift() { + let mut response = staged_response(Status::Error, None); + response.error_code = Some(Box::new(ErrorCode::OllpRetryRequired)); + assert_eq!(staged_commit_vote(&response), StagedVote::PredictionDrift); + assert_eq!( + StagedVote::PredictionDrift.abort_reason(), + Some(AbortReason::PredictionDrift) + ); + } + #[test] fn stale_read_set_on_a_successful_response_votes_serialization_conflict() { assert_eq!( @@ -113,5 +135,9 @@ mod tests { StagedVote::ParticipantError.abort_reason(), Some(AbortReason::ParticipantError) ); + assert_eq!( + StagedVote::CollectionSuperseded.abort_reason(), + Some(AbortReason::CollectionSuperseded) + ); } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs index 0a91ddf3f..a8ca94fc2 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/core/test_support.rs @@ -76,7 +76,7 @@ pub(super) fn build_test_scheduler(vshard_id: u32) -> (Scheduler, tempfile::Temp sequencer_state_machine, // A freshly-built scheduler has applied nothing, so its watermark is the // not-yet-applied sentinel (matching `read_applied_recovery` for a clean - // node). Hardcoding `0` here would instead claim epoch 0 is fully applied, + // node). Hardcoding `0` here will instead claim epoch 0 is fully applied, // making the exactly-once gate (`AppliedGate::is_applied`) short-circuit // every epoch-0 replay before it reaches the lock table — silently // defeating the end-to-end drain tests below. @@ -198,6 +198,8 @@ pub(super) fn make_validate_only_txn(epoch: u64, position: u32) -> SequencedTxn collection: "test_coll".to_string(), key: ReadKeyIdent::Point(KeyRepr::Surrogate(1)), read_lsn: Lsn::ZERO, + home_vshard: None, + served_by: 0, }]); let tx_class = TxClass::new_single_vshard( ReadWriteSet::new(vec![]), @@ -273,7 +275,7 @@ fn filler_request(request_id: RequestId, tenant_id: TenantId) -> Request { plan: PhysicalPlan::Document(DocumentOp::PointGet { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "filler"), document_id: "d".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: None, pk_bytes: Vec::new(), rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, @@ -292,6 +294,7 @@ fn filler_request(request_id: RequestId, tenant_id: TenantId) -> Request { txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: Admission::Exempt(ExemptReason::Read), } } @@ -384,6 +387,14 @@ impl RunningScheduler { &self.input_tx } + /// Whether the loop exits by itself, with no shutdown signal, within + /// [`DATA_PLANE_WAIT`]. + pub(super) async fn exits_unprompted(self) -> bool { + tokio::time::timeout(DATA_PLANE_WAIT, self.handle) + .await + .is_ok_and(|joined| joined.is_ok()) + } + /// Signal shutdown and wait for the loop to exit. pub(super) async fn stop(self) { self.shutdown.signal(); @@ -433,6 +444,10 @@ pub(super) fn staged_pending(txn: SequencedTxn, txn_id: TxnId) -> PendingTxn { redo_records: None, flush_scope: crate::control::cluster::calvin::scheduler::driver::types::FlushScope::default( ), + superseded: false, + gates: Vec::new(), + ungated: false, + install_permit: None, } } diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/helpers.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/helpers.rs index 9cac56088..fb5816af2 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/helpers.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/helpers.rs @@ -50,7 +50,7 @@ pub(super) fn decode_lock_key(wire: &LockKeyWire) -> LockKey { /// /// All keys in both the read set and write set are acquired as exclusive /// (write) locks. -pub(super) fn expand_rw_set(txn: &SequencedTxn) -> BTreeSet { +pub(crate) fn expand_rw_set(txn: &SequencedTxn) -> BTreeSet { let mut keys = BTreeSet::new(); let add_key_set = |keys: &mut BTreeSet, ks: &EngineKeySet| match ks { EngineKeySet::Document { @@ -93,6 +93,14 @@ pub(super) fn expand_rw_set(txn: &SequencedTxn) -> BTreeSet { }); } } + // Each participating vShard's lock table holds the whole array on + // that vShard: the collection key every writer of the array takes. + EngineKeySet::Array { collection, .. } => { + keys.insert(LockKey::Surrogate { + collection: Arc::from(collection.as_str()), + surrogate: crate::control::planner::calvin::tx_class::write_keys::COLLECTION_KEY, + }); + } }; for ks in &txn.tx_class.read_set.0 { diff --git a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs index f786560f9..f3cda4a1a 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/driver/types.rs @@ -58,8 +58,8 @@ pub(super) struct PendingTxn { /// `Some(instant)` only while parked: if the deadline passes with the /// durable global verdict still unknown, the scheduler emits a stall /// warning and re-arms this deadline — it NEVER releases locks and NEVER - /// aborts (a unilateral abort while a peer may have already flushed a commit - /// would tear the transaction). `None` in every other state. + /// aborts (a unilateral abort while a peer can have already flushed a commit + /// will tear the transaction). `None` in every other state. /// /// `Instant::now()` is used for this deadline (observability / liveness /// only; never influences WAL bytes). @@ -78,6 +78,21 @@ pub(super) struct PendingTxn { /// What a committed flush names beside its redo record, derived once /// from this vShard's slice when it stages. pub flush_scope: FlushScope, + /// A collection the transaction names no longer held its planned + /// incarnation when this replica staged it. The vote is an abort. + pub superseded: bool, + /// The gates of the collections the transaction names, held shared from + /// the incarnation check until the txn leaves `pending`. A purge of one + /// waits, so a staged slice never flushes into a recreated collection. + pub gates: Vec, + /// A halt released `gates` while the txn had no write in flight. Its + /// flush checks the incarnations again and takes the gates back first. + pub ungated: bool, + /// The shared hold of the vShard's data-group apply gate, taken when the + /// flush dispatches and dropped once the txn left `pending` and its + /// position is marked applied. A data-group snapshot capture or install + /// holds the gate exclusive, so it never sees a flush half applied. + pub install_permit: Option, } /// What a vShard's committed flush carries: the collections its slice @@ -99,6 +114,17 @@ pub(super) struct FlushScope { /// Flushes sent for this txn. A refused install resends the flush until /// the count reaches its bound. pub sends: u32, + /// Every plan of this vShard's slice is one a trigger body buffered, and + /// a client plan of the transaction homes on another vShard. The slice + /// then commits under `Trigger`: a single overlay holding the whole + /// transaction tags each of these rows `Trigger` beside the client's. + pub body_only: bool, + /// The slice reads a whole collection at resolve: a TRUNCATE of rows. A + /// TRUNCATE's edge share reads none. A resolve otherwise runs as soon as + /// its verdict arrives, + /// beside lower txns that have not flushed. A TRUNCATE's resolve then + /// misses their rows, so it waits for its flush turn. + pub resolve_at_turn: bool, } impl FlushScope { @@ -116,10 +142,32 @@ impl FlushScope { // The resolve fills these once it answers. redo: Vec::new(), sends: 0, + body_only: false, + resolve_at_turn: plans.iter().any(reads_whole_collection_at_resolve), } } } +/// Whether the resolve of `plan` reads a whole collection's stored state. +/// +/// A TRUNCATE's edge share is not among them. Its resolve records a cut at +/// the transaction's ordinal and reads no stored edge, and the cut hides by +/// ordinal whatever this vShard flushes before or after it. The rows' +/// truncate on the collection's vShard still reads every row at resolve. +fn reads_whole_collection_at_resolve(plan: &nodedb_physical::physical_plan::PhysicalPlan) -> bool { + use nodedb_physical::physical_plan::{ + ColumnarOp, DocumentOp, KvOp, PhysicalPlan, TimeseriesOp, VectorOp, + }; + matches!( + plan, + PhysicalPlan::Document(DocumentOp::Truncate { .. }) + | PhysicalPlan::Kv(KvOp::Truncate { .. }) + | PhysicalPlan::Columnar(ColumnarOp::Truncate { .. }) + | PhysicalPlan::Timeseries(TimeseriesOp::Truncate { .. }) + | PhysicalPlan::Vector(VectorOp::DirectTruncate { .. }) + ) +} + /// Commit-resolution state of a staged static Calvin transaction. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub(in crate::control::cluster::calvin::scheduler::driver) enum CommitState { @@ -142,6 +190,15 @@ pub(in crate::control::cluster::calvin::scheduler::driver) enum CommitState { /// post-images into a replayable `RedoRecord`; awaiting that response /// before the redo is WAL-appended and the flush dispatched. AwaitingRedoResolve, + /// The txn committed, and its slice reads a whole collection at resolve + /// (a TRUNCATE). Its resolve waits for its turn: every lower txn of this + /// vShard finished, so the resolve reads every write sequenced before it + /// and none after. See [`FlushScope::resolve_at_turn`]. + AwaitingResolveTurn, + /// The txn committed and its redo record is appended. Its flush waits for + /// its turn: every lower txn of this vShard finished. See + /// [`super::core::flush_turn`]. + AwaitingFlushTurn { redo_lsn: Option }, /// A flush (`committed = true`) or drop (`committed = false`) has been /// dispatched, or parked for re-send at dispatcher capacity; awaiting its /// response before the commit tail runs. diff --git a/nodedb/src/control/cluster/calvin/scheduler/metrics.rs b/nodedb/src/control/cluster/calvin/scheduler/metrics.rs index 9019eb688..3b5d4e62d 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/metrics.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/metrics.rs @@ -41,7 +41,7 @@ pub struct SchedulerMetrics { /// Times a staged txn parked in `AwaitingVerdict` passed its stall deadline /// with the durable global verdict still unknown. This is NOT an abort: the /// scheduler keeps waiting and holding locks (a unilateral abort while a - /// peer may already have flushed a commit would tear the transaction). A + /// peer can already have flushed a commit will tear the transaction). A /// non-zero, growing value flags a stuck sequencer / partitioned verdict /// path that needs operator attention, never a correctness action here. pub verdict_stall_count: AtomicU64, @@ -70,7 +70,7 @@ pub struct SchedulerMetrics { pub intake_backlog: AtomicU64, /// Intake gate closures by reason. Indexes are the constants in /// [`intake_closure_reason`]. - pub intake_gate_closed_counts: [AtomicU64; 3], + pub intake_gate_closed_counts: [AtomicU64; 4], /// Apply halt state: 1 once the scheduler halted, 0 while it applies. pub apply_halted: AtomicU64, /// Reason of the halt. An index into [`apply_halt_reason`], read only @@ -106,8 +106,14 @@ pub mod intake_closure_reason { pub const DEFERRED_DISPATCH: usize = 0; pub const BACKLOG_FULL: usize = 1; pub const APPLY_HALTED: usize = 2; + pub const METADATA_CATCH_UP: usize = 3; - pub const LABELS: &[&str] = &["deferred_dispatch", "backlog_full", "apply_halted"]; + pub const LABELS: &[&str] = &[ + "deferred_dispatch", + "backlog_full", + "apply_halted", + "metadata_catch_up", + ]; } /// Reason codes for `nodedb_calvin_apply_halted`. @@ -120,6 +126,7 @@ pub mod apply_halt_reason { pub const LOCAL_STAGE_FAILED: usize = 5; pub const IDENTITY_BIND_FAILED: usize = 6; pub const WAL_APPEND_FAILED: usize = 7; + pub const METADATA_GROUP_GONE: usize = 8; pub const LABELS: &[&str] = &[ "draining", @@ -130,6 +137,7 @@ pub mod apply_halt_reason { "local_stage_failed", "identity_bind_failed", "wal_append_failed", + "metadata_group_gone", ]; } diff --git a/nodedb/src/control/cluster/calvin/scheduler/recovery.rs b/nodedb/src/control/cluster/calvin/scheduler/recovery.rs index 4d61ae75e..cac03b1e4 100644 --- a/nodedb/src/control/cluster/calvin/scheduler/recovery.rs +++ b/nodedb/src/control/cluster/calvin/scheduler/recovery.rs @@ -14,7 +14,7 @@ //! Each `CalvinApplied` marker is per `(epoch, position, vShard)` — one per //! independent transaction position — so the scan preserves `position` rather //! than collapsing an epoch to a single "applied" bit. Collapsing to the max -//! epoch would mark a whole epoch applied on the strength of its first committed +//! epoch will mark a whole epoch applied on the strength of its first committed //! position and, on restart, skip every other position of that epoch: a lost / //! torn transaction. //! @@ -61,12 +61,33 @@ pub struct AppliedRecovery { /// Records that fail to decode are logged and skipped — a corrupt record does /// not abort the scan. pub fn read_applied_recovery(wal: &WalManager, vshard_id: u32) -> crate::Result { + scan_applied(wal, vshard_id, None) +} + +/// [`read_applied_recovery`] for a vShard of data group `group_id`: a +/// snapshot install of the group drops every marker before it. The install +/// replaced the vShard's storage, and the applied state it brought replaced +/// this node's in the catalog. +fn scan_applied( + wal: &WalManager, + vshard_id: u32, + group_id: Option, +) -> crate::Result { let records = wal.replay()?; let mut applied_tail = BTreeSet::new(); let mut max_applied_epoch = NOT_YET_APPLIED_EPOCH; for record in &records { match record_type_of(record) { + Some(RecordType::SnapshotInstalled) => { + let installed = <[u8; 8]>::try_from(record.payload.as_slice()) + .ok() + .map(u64::from_le_bytes); + if group_id.is_some() && installed == group_id { + applied_tail.clear(); + max_applied_epoch = NOT_YET_APPLIED_EPOCH; + } + } Some(RecordType::CalvinApplied) => { match CalvinAppliedPayload::from_bytes(&record.payload) { Ok(p) if p.vshard_id == vshard_id => { @@ -129,15 +150,19 @@ pub fn read_applied_recovery(wal: &WalManager, vshard_id: u32) -> crate::Result< /// /// A checkpoint deletes WAL segments that hold applied markers, while the /// sequencer log keeps delivering their entries after a restart. Without the -/// saved state the scheduler would take an applied transaction for a new +/// saved state the scheduler will take an applied transaction for a new /// one: its local stage refuses the rows it already wrote, the scheduler /// halts, and the transaction's completion ack never settles. +/// +/// `group_id` is the data group of the vShard, when known: the markers +/// before the group's last snapshot install no longer hold. pub fn recover_applied( wal: &WalManager, catalog: &crate::control::security::catalog::SystemCatalog, vshard_id: u32, + group_id: Option, ) -> crate::Result { - let from_wal = read_applied_recovery(wal, vshard_id)?; + let from_wal = scan_applied(wal, vshard_id, group_id)?; let Some(saved) = catalog.load_calvin_applied(vshard_id)? else { return Ok(from_wal); }; @@ -307,6 +332,10 @@ mod tests { collections: Vec::new(), sum_targets: Vec::new(), }), + cross_shard_applied: None, + row_sources: Vec::new(), + publishes: Vec::new(), + row_changes: Vec::new(), }; wal.appender(crate::wal::manager::NO_APPLY_KEY) .with_event_source(crate::event::EventSource::User) @@ -327,6 +356,10 @@ mod tests { payload: vec![9, 9, 9], }], calvin_stamp: None, + cross_shard_applied: None, + row_sources: Vec::new(), + publishes: Vec::new(), + row_changes: Vec::new(), }; wal.appender(crate::wal::manager::NO_APPLY_KEY) .with_event_source(crate::event::EventSource::User) @@ -378,14 +411,41 @@ mod tests { ]) .unwrap(); - let rec = recover_applied(&wal, &catalog, 1).unwrap(); + let rec = recover_applied(&wal, &catalog, 1, None).unwrap(); assert_eq!(rec.fully_applied_epoch, 2); assert!(rec.applied_tail.contains(&(4, 1))); assert!(rec.applied_tail.contains(&(6, 0))); assert_eq!(rec.max_applied_epoch, 6); // A vShard with nothing saved keeps the plain WAL scan. - let other = recover_applied(&wal, &catalog, 9).unwrap(); + let other = recover_applied(&wal, &catalog, 9, None).unwrap(); assert_eq!(other, read_applied_recovery(&wal, 9).unwrap()); } + + /// A snapshot install of the vShard's group drops every marker before + /// it: the install replaced the storage those positions wrote. + #[test] + fn a_group_snapshot_install_drops_the_markers_before_it() { + use crate::types::VShardId; + let dir = TempDir::new().unwrap(); + let wal = open_wal(&dir); + let appender = || wal.appender(crate::wal::manager::NO_APPLY_KEY); + appender() + .append_calvin_applied(VShardId::new(1), 3, 0) + .unwrap(); + appender().append_snapshot_installed(8).unwrap(); + appender() + .append_calvin_applied(VShardId::new(1), 5, 2) + .unwrap(); + appender().append_snapshot_installed(9).unwrap(); + wal.sync().unwrap(); + + let in_group = scan_applied(&wal, 1, Some(8)).unwrap(); + assert_eq!(in_group.applied_tail, [(5, 2)].into_iter().collect()); + assert_eq!(in_group.max_applied_epoch, 5); + + let other_group = scan_applied(&wal, 1, Some(4)).unwrap(); + assert_eq!(other_group, read_applied_recovery(&wal, 1).unwrap()); + assert!(other_group.applied_tail.contains(&(3, 0))); + } } diff --git a/nodedb/src/control/cluster/calvin_snapshot/capture.rs b/nodedb/src/control/cluster/calvin_snapshot/capture.rs new file mode 100644 index 000000000..ff95938c9 --- /dev/null +++ b/nodedb/src/control/cluster/calvin_snapshot/capture.rs @@ -0,0 +1,127 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The Calvin cut a data-group snapshot build takes on the leader. +//! +//! The build first places a cut marker in the sequencer log and waits until +//! the scheduler of every group vShard here passed it (see +//! [`await_calvin_cut`]). Every input sequenced below the marker then +//! finished here. The build then fences the group: the apply gate it holds +//! exclusive also stops every Calvin install of the group's vShards, since +//! a scheduler installs only under a shared hold of that gate. The applied +//! positions it reads under the fence (see [`capture_calvin_cut`]) are +//! exactly the ones the captured storage holds. + +use std::collections::{BTreeSet, HashSet}; +use std::time::Duration; + +use nodedb_cluster::calvin::SequencerEntry; + +use crate::Error; +use crate::control::security::catalog::calvin_base::CalvinBase; +use crate::control::state::SharedState; +use crate::types::{GroupCalvinCut, VShardCalvinState}; + +/// How long the build waits for the group's schedulers to pass its marker. +/// A failed build is retried on the next heartbeat. +const CUT_WAIT: Duration = Duration::from_secs(5); + +/// How long one marker proposal waits before it is proposed again. A leader +/// change can drop a proposed marker. +const MARKER_RETRY: Duration = Duration::from_secs(1); + +/// Place a cut marker and wait until the scheduler of every group vShard +/// here passed it. Returns the marker's sequencer log index. +/// +/// Fails when this node holds no whole Calvin state of a group vShard: its +/// scheduler did not start from a base that reaches the sequencer log, so +/// its storage can lack Calvin transactions. Another replica sends the +/// snapshot, or this one once its own snapshot installed. +pub async fn await_calvin_cut( + shared: &SharedState, + group_id: u64, + group_vshards: &HashSet, +) -> Result { + if let Some(vshard_id) = group_vshards + .iter() + .copied() + .find(|vshard_id| !CalvinBase::is_kept(shared.calvin.bases.base(*vshard_id))) + { + return Err(Error::Internal { + detail: format!( + "snapshot build: group {group_id}: this node holds no whole Calvin state of \ + vShard {vshard_id}; the build retries on the next heartbeat" + ), + }); + } + let proposer = shared + .calvin + .sequencer_proposer + .get() + .ok_or_else(|| Error::Internal { + detail: format!( + "snapshot build: group {group_id}: no sequencer proposer is set on this node, \ + so the build cannot place its Calvin cut marker; the build retries on the \ + next heartbeat" + ), + })?; + let hlc = shared.hlc_clock.now().wall_ns; + let marker = zerompk::to_msgpack_vec(&SequencerEntry::CutMarker { + hlc, + restore_point: 0, + }) + .map_err(|error| Error::Internal { + detail: format!("snapshot build: group {group_id}: encode the Calvin cut marker: {error}"), + })?; + let vshards: BTreeSet = group_vshards.iter().copied().collect(); + let deadline = tokio::time::Instant::now() + CUT_WAIT; + let mut last_refusal = None; + loop { + if let Err(error) = proposer.propose(marker.clone()) { + last_refusal = Some(error.to_string()); + } + let attempt = deadline.min(tokio::time::Instant::now() + MARKER_RETRY); + if let Some(index) = shared + .calvin + .cuts + .await_vshards_passed(hlc, &vshards, attempt) + .await + { + return Ok(index); + } + if tokio::time::Instant::now() >= deadline { + return Err(Error::Internal { + detail: format!( + "snapshot build: group {group_id}: its Calvin schedulers did not pass the \ + cut marker in time (last marker refusal: {}); the build retries on the \ + next heartbeat", + last_refusal.as_deref().unwrap_or("none") + ), + }); + } + } +} + +/// The group's Calvin cut through `through`: the applied positions of every +/// group vShard here. The caller holds the group's apply gate exclusive, so +/// no Calvin install of these vShards runs. +pub fn capture_calvin_cut( + shared: &SharedState, + group_vshards: &HashSet, + through: u64, +) -> GroupCalvinCut { + let mirrors = shared.authorization_fence.calvin_mirrors(); + let mut ids: Vec = group_vshards.iter().copied().collect(); + ids.sort_unstable(); + let vshards = ids + .into_iter() + .filter_map(|vshard_id| { + let (fully_applied_epoch, tail) = mirrors.get(vshard_id)?.snapshot(); + Some(VShardCalvinState { + vshard_id, + fully_applied_epoch, + tail: tail.into_iter().collect(), + }) + }) + .collect(); + GroupCalvinCut { through, vshards } +} diff --git a/nodedb/src/control/cluster/calvin_snapshot/install.rs b/nodedb/src/control/cluster/calvin_snapshot/install.rs new file mode 100644 index 000000000..b528d7cbd --- /dev/null +++ b/nodedb/src/control/cluster/calvin_snapshot/install.rs @@ -0,0 +1,73 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Install the Calvin cut a data-group snapshot carries. +//! +//! The snapshot replaced the storage of the group's vShards with the +//! leader's, which holds exactly the Calvin positions the cut names. So +//! each vShard's applied state here becomes the cut's: in the applied +//! mirror, which a checkpoint saves from, and in the catalog, which a +//! scheduler starts from. Its base becomes the cut's sequencer index, which +//! moves its generation: a scheduler started before the install installs +//! nothing more, and the next scheduler reconcile starts it again from the +//! installed state. +//! +//! A WAL `SnapshotInstalled` record of the group, written before this runs, +//! makes boot recovery drop the vShards' applied markers from before it. + +use std::collections::{BTreeSet, HashSet}; + +use crate::control::cluster::snapshot_install::SnapshotInstallError; +use crate::control::security::catalog::calvin_applied::StoredCalvinApplied; +use crate::control::state::SharedState; +use crate::types::GroupCalvinCut; + +/// Install `cut` as the Calvin state of every vShard of `group_id` in +/// `group_vshards`. A vShard the cut does not name had no scheduler on the +/// builder: it holds no Calvin position. +pub fn install_calvin_cut( + shared: &SharedState, + group_id: u64, + group_vshards: &HashSet, + cut: GroupCalvinCut, +) -> Result<(), SnapshotInstallError> { + let mut ids: Vec = group_vshards.iter().copied().collect(); + ids.sort_unstable(); + let states: Vec = ids + .iter() + .map(|&vshard_id| { + let named = cut + .vshards + .iter() + .find(|state| state.vshard_id == vshard_id); + StoredCalvinApplied { + vshard_id, + fully_applied_epoch: named.map_or( + crate::control::cluster::calvin::scheduler::NOT_YET_APPLIED_EPOCH, + |state| state.fully_applied_epoch, + ), + tail: named + .map(|state| state.tail.iter().copied().collect()) + .unwrap_or_else(BTreeSet::new), + } + }) + .collect(); + + let mirrors = shared.authorization_fence.calvin_mirrors(); + for state in &states { + mirrors.register(state.vshard_id, state.fully_applied_epoch, &state.tail); + } + let catalog = shared.credentials.catalog(); + let settle_error = |source| SnapshotInstallError::Settle { + group_id, + step: crate::control::cluster::snapshot_install::SettleStep::CalvinState, + source, + }; + catalog + .replace_calvin_applied(&states) + .map_err(settle_error)?; + shared + .calvin + .bases + .record_snapshot(catalog, &ids, cut.through) + .map_err(settle_error) +} diff --git a/nodedb/src/control/cluster/calvin_snapshot/mod.rs b/nodedb/src/control/cluster/calvin_snapshot/mod.rs new file mode 100644 index 000000000..1f7b75a92 --- /dev/null +++ b/nodedb/src/control/cluster/calvin_snapshot/mod.rs @@ -0,0 +1,11 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod capture; +pub mod install; +pub mod requirement; + +pub use capture::{await_calvin_cut, capture_calvin_cut}; +pub use install::install_calvin_cut; +pub use requirement::{ + forget_left, install_snapshot_requirement, may_start, note_kept, rebased, retain_mounted, +}; diff --git a/nodedb/src/control/cluster/calvin_snapshot/requirement.rs b/nodedb/src/control/cluster/calvin_snapshot/requirement.rs new file mode 100644 index 000000000..8a47eb1b5 --- /dev/null +++ b/nodedb/src/control/cluster/calvin_snapshot/requirement.rs @@ -0,0 +1,213 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Which data-group replicas need a snapshot for their Calvin state, and +//! which vShard schedulers can start. +//! +//! A replica's state of a vShard is its data group's log plus every Calvin +//! input the sequencer sequenced for the vShard. Calvin writes barely grow a +//! data group's log, so the group can never compact, and a new replica will +//! catch up by log replay alone. Its scheduler then catches up from the +//! first index the sequencer log still holds, and every input below that +//! index is lost on the replica. So: +//! +//! - A data group mounted here whose vShards have no base that reaches the +//! sequencer log refuses log entries, and its leader sends a snapshot +//! instead (see [`install_snapshot_requirement`]). The snapshot carries +//! the Calvin cut its state holds. +//! - A vShard's scheduler starts only from a base that reaches the +//! sequencer log, and then keeps the base (see [`may_start`]). +//! - The sequencer log keeps the range a waiting vShard will replay (see +//! [`crate::control::state::CalvinBases::replay_floor`]). + +use std::collections::BTreeSet; +use std::sync::{Arc, Mutex, RwLock}; + +use nodedb_cluster::multi_raft::MultiRaft; + +use crate::control::security::catalog::calvin_base::CalvinBase; +use crate::control::state::SharedState; + +/// Load this node's Calvin bases and install the data-group snapshot +/// requirement on `multi_raft`, before its data groups take entries. +pub fn install_snapshot_requirement( + shared: &Arc, + routing: Arc>, + multi_raft: &mut MultiRaft, +) -> crate::Result<()> { + let catalog = shared.credentials.catalog(); + shared.calvin.bases.load( + catalog.load_calvin_bases()?, + catalog.load_calvin_sequencer_install()?, + ); + let weak = Arc::downgrade(shared); + multi_raft.set_snapshot_requirement(Arc::new(move |group_id, sequencer_first| { + let Some(shared) = weak.upgrade() else { + return false; + }; + let vshards = routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .vshards_for_group(group_id); + !shared.calvin.bases.group_reaches(&vshards, sequencer_first) + })); + Ok(()) +} + +/// Whether the scheduler of `vshard_id` can start, with the sequencer log +/// holding entries from `sequencer_start`. A vShard whose base does not +/// reach the log needs a snapshot: its group's replica here refuses entries +/// until one installs. +/// +/// `None` is a sequencer log with no known start: it holds no entry and no +/// snapshot boundary. Its leader can still send a snapshot that skips the +/// inputs a scheduler will wait for, so no scheduler starts yet. Nor does +/// one start while a sequencer snapshot installs here. +pub fn may_start( + shared: &SharedState, + multi_raft: &Mutex, + vshard_id: u32, + group_id: Option, + sequencer_start: Option, +) -> bool { + let Some(sequencer_first) = sequencer_start else { + return false; + }; + if shared + .calvin + .bases + .sequencer_install_pending(sequencer_first) + { + return false; + } + let base = shared.calvin.bases.base(vshard_id); + if CalvinBase::reaches(base, sequencer_first) { + return true; + } + if let Some(group_id) = group_id { + let mut multi_raft = multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + multi_raft.set_snapshot_required(group_id, true); + // Without a scheduler here the vShard casts no vote, so a leader here + // blocks every Calvin transaction on it. A replica whose state is + // whole takes the leadership. When none qualifies, the transactions + // end at their deadline with an error. + let sole_replica = multi_raft + .group_membership(group_id) + .is_some_and(|m| m.voters.len() + m.learners.len() <= 1); + if sole_replica && shared.calvin.bases.first_stopped_report(vshard_id) { + tracing::error!( + vshard_id, + group_id, + sequencer_first, + ?base, + "calvin: vShard {vshard_id} lost its Calvin base on group {group_id}, where \ + this node is the only replica; Calvin stays stopped for the vShard" + ); + } + if let Some(target) = multi_raft.hand_off_leadership(group_id) { + tracing::info!( + vshard_id, + group_id, + target, + "calvin: the vShard's scheduler cannot start here; the data-group \ + leadership moves to a replica that runs one" + ); + } + } + tracing::debug!( + vshard_id, + ?group_id, + sequencer_first, + ?base, + "calvin: the vShard's Calvin state does not reach the sequencer log; its scheduler \ + waits for a data-group snapshot" + ); + false +} + +/// Record that the schedulers of `vshards`, each `(vshard_id, from)`, +/// started from bases that reach the sequencer log and caught up from +/// index `from`. They keep them whole from here. +pub fn note_kept(shared: &SharedState, vshards: &[(u32, u64)]) { + if let Err(error) = shared + .calvin + .bases + .record_kept(shared.credentials.catalog(), vshards) + { + tracing::error!( + ?vshards, + %error, + "calvin: the vShards' kept bases did not persist; a restart takes a snapshot for \ + them" + ); + } +} + +/// The served vShards whose base a snapshot install replaced since their +/// scheduler started, or a sequencer snapshot install ended. Their +/// scheduler stops and starts again from the state a snapshot installs. +pub fn rebased(shared: &SharedState, served: &[u32]) -> Vec { + served + .iter() + .copied() + .filter(|vshard_id| !CalvinBase::is_kept(shared.calvin.bases.base(*vshard_id))) + .collect() +} + +/// Forget the Calvin state of `vshards`: this node left their groups. A +/// later return catches up from nothing, so neither their bases nor their +/// applied positions can survive. +pub fn forget_left(shared: &SharedState, vshards: &[u32]) { + if vshards.is_empty() { + return; + } + let catalog = shared.credentials.catalog(); + let mirrors = shared.authorization_fence.calvin_mirrors(); + let resets: Vec = + vshards + .iter() + .map(|&vshard_id| { + crate::control::security::catalog::calvin_applied::StoredCalvinApplied { + vshard_id, + fully_applied_epoch: + crate::control::cluster::calvin::scheduler::NOT_YET_APPLIED_EPOCH, + tail: BTreeSet::new(), + } + }) + .collect(); + for &vshard_id in vshards { + mirrors.remove(vshard_id); + } + let forgotten = catalog + .replace_calvin_applied(&resets) + .and_then(|()| shared.calvin.bases.forget(catalog, vshards)); + if let Err(error) = forgotten { + tracing::error!( + ?vshards, + %error, + "calvin: the left vShards' Calvin state was not forgotten" + ); + } +} + +/// Keep the sequencer log range only for vShards of the data groups this +/// node mounts. +pub fn retain_mounted( + shared: &SharedState, + multi_raft: &Mutex, + routing: &RwLock, +) { + let groups = multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .group_ids(); + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let mounted: BTreeSet = groups + .into_iter() + .filter(|group_id| { + *group_id != nodedb_cluster::METADATA_GROUP_ID + && *group_id != nodedb_cluster::calvin::SEQUENCER_GROUP_ID + }) + .flat_map(|group_id| routing.vshards_for_group(group_id)) + .collect(); + shared.calvin.bases.retain_waiting(&mounted); +} diff --git a/nodedb/src/control/cluster/core_stall.rs b/nodedb/src/control/cluster/core_stall.rs index 910d1a520..2880b893e 100644 --- a/nodedb/src/control/cluster/core_stall.rs +++ b/nodedb/src/control/cluster/core_stall.rs @@ -30,7 +30,7 @@ //! Unlike the sibling markers //! [`MetadataApplyWedge`](super::metadata_applier::MetadataApplyWedge) and //! [`SequencerHaltMarker`](super::SequencerHaltMarker), which latch -//! first-writer-wins because their conditions never clear, a stall can +//! first-writer-wins and clear only through their own writer, a stall can //! recover on its own. [`CoreStallMarker`] is therefore replaced on every //! sampling window and reports the current set of stalled cores. @@ -190,7 +190,7 @@ mod tests { // // This test does NOT assert the detector can tell the two causes // apart — it asserts the opposite: the same (previous, current) - // pair, which could equally have come from either cause, always + // pair, which can equally have come from either cause, always // produces the same "stalled" verdict. Do not read a passing test // here as proof of stall detection; it is proof of the ambiguity. let previous = vec![42u64]; @@ -229,7 +229,7 @@ mod tests { #[test] fn marker_replaces_rather_than_accumulates_across_samples() { // Unlike `MetadataApplyWedge` / `SequencerHaltMarker` (first writer - // wins, never clears), a stall can recover, so each new sampling + // wins), a stall can recover, so each new sampling // window's `set` must replace — not merge with — the previous // report. let marker = CoreStallMarker::default(); diff --git a/nodedb/src/control/cluster/data_plane_error_wire.rs b/nodedb/src/control/cluster/data_plane_error_wire.rs index 958957427..d39f47279 100644 --- a/nodedb/src/control/cluster/data_plane_error_wire.rs +++ b/nodedb/src/control/cluster/data_plane_error_wire.rs @@ -21,17 +21,26 @@ use crate::bridge::envelope::{CounterFault, ErrorCode, SyncHold}; /// the coordinator rebuilds `Error::DataPlane(code)` and renders the SQLSTATE /// single-node execution renders. Every other error keeps its own numeric /// classification from `NodeDbError::from(err).code()` — never a hardcoded -/// plan-decode code, which would misname what failed. +/// plan-decode code, which will misname what failed. pub(crate) fn execution_error_to_typed(err: crate::Error) -> TypedClusterError { match err { crate::Error::DataPlane(code) => TypedClusterError::DataPlane { code: code.into() }, // A statement that ran out of time keeps the wire's own deadline // variant, which the coordinator rebuilds as `Error::DeadlineExceeded`. - // Folding it into `Internal` would report a client's own timeout as an + // Folding it into `Internal` will report a client's own timeout as an // internal failure once it crossed a node boundary. crate::Error::DeadlineExceeded { .. } => { TypedClusterError::DeadlineExceeded { elapsed_ms: 0 } } + // A redirect crosses as the wire's own redirect, with the leader and + // the term this node knows it at, so the coordinator moves its + // routing hint and retries against that leader. + not_leader @ crate::Error::NotLeader { .. } => TypedClusterError::from(not_leader), + // A Calvin abort keeps its verdict and a schema change stays + // retryable, so a routed submit answers as a local one does. + typed @ (crate::Error::CalvinSerializationConflict + | crate::Error::CalvinParticipantError + | crate::Error::RetryableSchemaChanged { .. }) => TypedClusterError::from(typed), // A Control-Plane constraint refusal crosses verbatim, same as a // Data-Plane verdict, so the coordinator answers 23502 vs 23505 // instead of flattening both into one numeric class. @@ -57,8 +66,6 @@ pub(crate) fn execution_error_to_typed(err: crate::Error) -> TypedClusterError { | crate::Error::RejectedAuthz { .. } | crate::Error::OffsetRegression { .. } | crate::Error::ConflictRetry { .. } - | crate::Error::CalvinSerializationConflict - | crate::Error::CalvinParticipantError | crate::Error::RejectedPrevalidation { .. } | crate::Error::RetryableRefusal { .. } | crate::Error::AppendOnlyViolation { .. } @@ -87,10 +94,7 @@ pub(crate) fn execution_error_to_typed(err: crate::Error) -> TypedClusterError { | crate::Error::NotInTransactionBlock { .. } | crate::Error::CrdtAdmissionTimeout { .. } | crate::Error::NoLeader { .. } - | crate::Error::NotLeader { .. } - | crate::Error::FanOutExceeded { .. } | crate::Error::CrossCollectionNotColocated { .. } - | crate::Error::SourceFrozen { .. } | crate::Error::CloneWriteRequiresMaterialize { .. } | crate::Error::BadRequest { .. } | crate::Error::BackupTenantMismatch { .. } @@ -107,12 +111,15 @@ pub(crate) fn execution_error_to_typed(err: crate::Error) -> TypedClusterError { | crate::Error::DivisionByZero | crate::Error::DataException { .. } | crate::Error::InvalidLimitValue { .. } - | crate::Error::RetryableSchemaChanged { .. } | crate::Error::RetryableLeaderChange { .. } + | crate::Error::CommittedResultUnavailable { .. } + | crate::Error::ProposalOutcomeUnknown { .. } | crate::Error::GroupQuorumUnavailable { .. } | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::BackupCaptureMoved { .. } | crate::Error::MetadataLeaderUnavailable | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::LinearizableReadRefused { .. } | crate::Error::ExecutionLimitExceeded { .. } | crate::Error::LimitExceeded { .. } | crate::Error::Wal(_) @@ -130,12 +137,15 @@ pub(crate) fn execution_error_to_typed(err: crate::Error) -> TypedClusterError { | crate::Error::Encryption { .. } | crate::Error::Bridge { .. } | crate::Error::VersionCompat { .. } + | crate::Error::RestoreTargetNotEmpty { .. } + | crate::Error::RestoreVerificationFailed { .. } | crate::Error::Internal { .. } | crate::Error::Shaping(_) | crate::Error::RemoteTyped { .. } | crate::Error::Ddl(_) | crate::Error::DescriptorVersionAnomaly { .. } | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CollectionUnstamped { .. } | crate::Error::CatalogIntegrityViolation { .. } | crate::Error::Promql(_) | crate::Error::DependentObjectsExist { .. } @@ -213,7 +223,6 @@ impl From for DataPlaneErrorCode { ErrorCode::CrdtFrontierMismatch { expected, actual } => { Self::CrdtFrontierMismatch { expected, actual } } - ErrorCode::FanOutExceeded => Self::FanOutExceeded, ErrorCode::ResourcesExhausted => Self::ResourcesExhausted, ErrorCode::RejectedDanglingEdge { missing_node } => { Self::RejectedDanglingEdge { missing_node } @@ -340,7 +349,6 @@ impl From for ErrorCode { DataPlaneErrorCode::CrdtFrontierMismatch { expected, actual } => { Self::CrdtFrontierMismatch { expected, actual } } - DataPlaneErrorCode::FanOutExceeded => Self::FanOutExceeded, DataPlaneErrorCode::ResourcesExhausted => Self::ResourcesExhausted, DataPlaneErrorCode::RejectedDanglingEdge { missing_node } => { Self::RejectedDanglingEdge { missing_node } diff --git a/nodedb/src/control/cluster/handle.rs b/nodedb/src/control/cluster/handle.rs index 677531d49..bcbdd8a9c 100644 --- a/nodedb/src/control/cluster/handle.rs +++ b/nodedb/src/control/cluster/handle.rs @@ -16,6 +16,9 @@ use nodedb_cluster::GroupAppliedWatchers; /// rationale. pub struct PendingSubsystems { pub config: nodedb_cluster::ClusterConfig, + /// SWIM socket bound before `start_cluster`, so its address is already + /// in this node's topology entry. + pub swim_transport: Arc, } /// Everything the main server needs to wire the cluster into the rest of @@ -66,4 +69,7 @@ pub struct ClusterHandle { /// calls [`nodedb_cluster::start_cluster_subsystems`] with the /// loop's shared `multi_raft` handle. pub pending_subsystems: Mutex>, + /// State of every vShard migration this node's rebalancer runs. Shared + /// with `SharedState` for `SHOW MIGRATIONS` and the status route. + pub migration_tracker: Arc, } diff --git a/nodedb/src/control/cluster/init.rs b/nodedb/src/control/cluster/init.rs index 059eb608f..859bb9c8b 100644 --- a/nodedb/src/control/cluster/init.rs +++ b/nodedb/src/control/cluster/init.rs @@ -16,7 +16,21 @@ use crate::control::cluster::handle::ClusterHandle; /// Node id for the synthesized single-node Calvin deployment. A standalone /// server is a cluster of one, so the id is fixed and non-zero. -const SINGLE_NODE_CALVIN_NODE_ID: u64 = 1; +pub const SINGLE_NODE_CALVIN_NODE_ID: u64 = 1; + +/// The node id a server booted from `config` runs as. +/// +/// It is `[cluster] node_id` when the section is present. A server without +/// `[cluster]` runs the synthesized one-node cluster as +/// [`SINGLE_NODE_CALVIN_NODE_ID`]. The WAL archive and the PITR base +/// snapshots key on this id, so an offline restore reads the node life the +/// server wrote under it. +pub fn configured_node_id(config: &crate::ServerConfig) -> u64 { + match &config.cluster { + Some(cluster) => cluster.node_id, + None => SINGLE_NODE_CALVIN_NODE_ID, + } +} /// Raft group count for the synthesized single-node Calvin deployment. This /// node is the sole member of every group, so the value only affects how the @@ -84,6 +98,35 @@ pub async fn init_cluster_with_transport( })?, ); + // Raise this boot's epoch durably, on a blocking thread, before either + // transport sends. Peers keep their replay windows for this node across + // a restart, so every boot sends in a sequence range above every earlier + // boot's. No clock goes into it. + let boot_epoch = { + let catalog = Arc::clone(&catalog); + tokio::task::spawn_blocking(move || catalog.advance_boot_epoch()) + .await + .map_err(|e| crate::Error::Config { + detail: format!("cluster boot epoch task: {e}"), + })? + .map_err(|e| crate::Error::Config { + detail: format!("cluster boot epoch: {e}"), + })? + }; + transport.enter_boot_epoch(boot_epoch); + + // SWIM binds before startup so bootstrap, join, and restart advertise the + // bound address in this node's topology entry. A bind error fails startup. + let swim_transport = Arc::new( + nodedb_cluster::bind_swim_listener(config.swim_listen, config.listen, transport.mac_key()) + .await + .map_err(|e| crate::Error::Config { + detail: format!("cluster SWIM listener: {e}"), + })?, + ); + swim_transport.enter_boot_epoch(boot_epoch); + let swim_addr = nodedb_cluster::swim::Transport::local_addr(swim_transport.as_ref()); + // 3. Bootstrap, join, or restart. let cluster_config = nodedb_cluster::ClusterConfig { node_id: config.node_id, @@ -97,7 +140,7 @@ pub async fn init_cluster_with_transport( max_attempts: config.join_retry_max_attempts, max_backoff_secs: config.join_retry_max_backoff_secs, }, - swim_udp_addr: None, + swim_udp_addr: Some(swim_addr), election_timeout_min: std::time::Duration::from_millis( transport_tuning.effective_election_timeout_min_ms(), ), @@ -107,6 +150,7 @@ pub async fn init_cluster_with_transport( install_snapshot_chunk_bytes: 4 * 1024 * 1024, orphan_partial_max_age_secs: 300, log_compaction_threshold: config.log_compaction_threshold, + wire_build_id: nodedb_types::wire_version::WIRE_BUILD_ID.to_owned(), }; let lifecycle = nodedb_cluster::ClusterLifecycleTracker::new(); @@ -163,11 +207,13 @@ pub async fn init_cluster_with_transport( running_cluster: Mutex::new(None), pending_subsystems: Mutex::new(Some(crate::control::cluster::handle::PendingSubsystems { config: cluster_config, + swim_transport, })), + migration_tracker: Arc::new(nodedb_cluster::MigrationTracker::new()), }) } -/// Initialize a flag-gated single-node Calvin deployment on a standalone server. +/// Initialize the single-node Calvin deployment of a server with no `[cluster]`. /// /// Synthesizes a one-node cluster configuration — this node as its own sole /// seed, replication factor 1 — and drives the SAME cluster startup a real @@ -175,14 +221,12 @@ pub async fn init_cluster_with_transport( /// to an ephemeral loopback port and never dials a peer; a single-member Raft /// group is the deterministic bootstrapper and self-elects, committing locally. /// The single node therefore hosts every vShard, so the sequencer group and -/// per-vShard schedulers all come up and `calvin_available` becomes true. +/// per-vShard schedulers all come up. /// /// The caller must still call [`super::start_raft::start_raft`] after /// `SharedState` is constructed, exactly as for [`init_cluster`]. /// -/// Only reached when `server.single_node_calvin` is set and `[cluster]` is -/// absent; when the flag is off (the default) the standalone boot path never -/// calls this and no Calvin stack is started. +/// Boot calls this whenever `[cluster]` is absent. pub async fn init_single_node_calvin( data_dir: &std::path::Path, transport_tuning: &ClusterTransportTuning, @@ -229,6 +273,9 @@ pub async fn init_single_node_calvin( log_compaction_threshold: None, join_retry_max_attempts: 8, join_retry_max_backoff_secs: 32, + // The QUIC port is OS-assigned, so `port + 1` can be taken. No peer + // ever probes this node, so an OS-assigned loopback port serves. + swim_listen: Some(listen_placeholder), }; init_cluster_with_transport(&settings, transport, data_dir, transport_tuning).await diff --git a/nodedb/src/control/cluster/leased_read.rs b/nodedb/src/control/cluster/leased_read.rs new file mode 100644 index 000000000..0d7150fd3 --- /dev/null +++ b/nodedb/src/control/cluster/leased_read.rs @@ -0,0 +1,190 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Serving a read on a group's leader under its leader lease. +//! +//! A read answered by a node that lost leadership can miss writes a newer +//! leader committed. A node answers a leased read of a group only while it +//! holds the group's leader lease, and only once it has applied the group +//! through the lease read index. The lease lapses before any other node can +//! win an election, so no newer leader has committed anything the read +//! misses. A group whose lease this node does not hold is refused, with the +//! leader and term its routing table names. + +use std::time::Instant; + +use crate::control::state::SharedState; + +use super::linearizable_read::wait_applied_through; + +/// A group this node does not serve a leased read of. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct LeaseRefusal { + pub group_id: u64, + /// The leader this node's routing table names, `0` when it names none or + /// names this node. + pub leader_node: u64, + /// The term `leader_node` is known at, `0` when unknown. + pub leader_term: u64, +} + +/// Make a read of `groups` safe to serve on this node under its leader +/// leases. +/// +/// Returns the groups whose lease this node does not hold. Every other group +/// is applied here through its lease read index before this returns. A node +/// with no routing table serves every group: without a cluster there is one +/// copy and nothing to prove. +pub async fn confirm_leased_read( + state: &SharedState, + groups: &[u64], + deadline: Instant, +) -> crate::Result> { + let Some(routing) = state.cluster_routing.as_ref() else { + return Ok(Vec::new()); + }; + let Some(gate) = state.raft_read_gate.get() else { + // A routing table without a gate: `start_raft` has not published it. + return Ok(groups + .iter() + .map(|&group_id| LeaseRefusal { + group_id, + leader_node: 0, + leader_term: 0, + }) + .collect()); + }; + // The gate locks `MultiRaft`, so every lease is read before the routing + // guard is taken. Holding the guard across the gate inverts the lock order. + let leases: Vec<(u64, Option)> = groups + .iter() + .map(|&group_id| (group_id, gate.lease_read_index(group_id))) + .collect(); + let mut refusals = Vec::new(); + let mut leased: Vec<(u64, u64)> = Vec::new(); + { + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + for (group_id, lease) in leases { + match lease { + Some(read_index) => leased.push((group_id, read_index)), + None => { + let (leader, term) = routing + .group_info(group_id) + .map(|info| (info.leader, info.leader_term)) + .unwrap_or((0, 0)); + refusals.push(LeaseRefusal { + group_id, + leader_node: if leader == state.node_id { 0 } else { leader }, + leader_term: term, + }); + } + } + } + } + let waits = leased + .into_iter() + .map(|(group_id, read_index)| wait_applied_through(state, group_id, read_index, deadline)); + futures::future::try_join_all(waits).await?; + Ok(refusals) +} + +#[cfg(test)] +mod tests { + use std::sync::atomic::{AtomicBool, Ordering}; + use std::sync::{Arc, RwLock}; + use std::time::Duration; + + use async_trait::async_trait; + use nodedb_cluster::RoutingTable; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::control::cluster::read_index::{RaftReadGate, ReadIndexRefusal}; + use crate::wal::WalManager; + + /// A gate that holds no lease. Each lease question probes the routing lock. + /// + /// The production gate locks `MultiRaft`, which re-reads routing under that lock. + /// `try_write` fails while any routing read guard is held, so a failed probe + /// proves the caller asked the gate under a routing guard. + struct ProbingGate { + routing: Arc>, + held_guard: AtomicBool, + } + + impl ProbingGate { + fn probe(&self) { + if self.routing.try_write().is_err() { + self.held_guard.store(true, Ordering::SeqCst); + } + } + } + + #[async_trait] + impl RaftReadGate for ProbingGate { + async fn confirm_leader( + &self, + _group_id: u64, + _timeout: Duration, + ) -> Result { + self.probe(); + Err(ReadIndexRefusal::NotLeader) + } + + fn within_staleness_bound(&self, _group_id: u64, _max_staleness: Duration) -> bool { + self.probe(); + false + } + + fn holds_leader_lease(&self, _group_id: u64) -> bool { + self.probe(); + false + } + + fn leader_lease_term(&self, _group_id: u64) -> Option { + self.probe(); + None + } + + fn lease_read_index(&self, _group_id: u64) -> Option { + self.probe(); + None + } + } + + /// A leased read asks the gate for every lease before it takes the routing guard. + /// + /// Holding the guard across the gate inverts the `MultiRaft` then routing order. + /// A queued routing writer then deadlocks the node. + #[tokio::test] + async fn leases_are_read_with_no_routing_guard_held() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = + Arc::new(WalManager::open_for_testing(&dir.path().join("leased.wal")).expect("wal")); + let (dispatcher, _sides) = Dispatcher::new(1, 64); + let mut state = SharedState::new(dispatcher, wal).expect("shared state"); + let routing = Arc::new(RwLock::new(RoutingTable::uniform(2, &[1, 2], 2))); + Arc::get_mut(&mut state) + .expect("sole owner of fresh state") + .cluster_routing = Some(Arc::clone(&routing)); + let gate = Arc::new(ProbingGate { + routing: Arc::clone(&routing), + held_guard: AtomicBool::new(false), + }); + assert!( + state + .raft_read_gate + .set(Arc::clone(&gate) as Arc) + .is_ok() + ); + let groups = routing.read().expect("routing lock").group_ids(); + + let refusals = confirm_leased_read(&state, &groups, Instant::now()) + .await + .expect("a refused lease is not an error"); + assert!( + !gate.held_guard.load(Ordering::SeqCst), + "a lease was read under a routing guard" + ); + assert_eq!(refusals.len(), groups.len(), "no lease is held here"); + } +} diff --git a/nodedb/src/control/cluster/linearizable_read.rs b/nodedb/src/control/cluster/linearizable_read.rs new file mode 100644 index 000000000..369026938 --- /dev/null +++ b/nodedb/src/control/cluster/linearizable_read.rs @@ -0,0 +1,193 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Making a linearizable read safe on the node that serves it. +//! +//! A read observes every write committed before it began only when the node +//! that runs it has applied its group through a read index taken after the +//! read began. The read index comes from the group's leader: this node when it +//! leads (under its lease or a quorum round), the leader over the transport +//! otherwise. Every path that serves a strong read calls +//! [`confirm_linearizable_read`] on the serving node before it reads. + +use std::time::{Duration, Instant}; + +use nodedb_cluster::WaitOutcome; + +use crate::control::cluster::read_index::ReadIndexRefusal; +use crate::control::state::SharedState; + +/// Budget for one linearizable read to get each group's read index and apply +/// through it. Several production election timeouts (150-300ms), so an +/// ordinary round trip fits and a partition is refused rather than hung on. +pub const LINEARIZABLE_READ_TIMEOUT: Duration = Duration::from_millis(750); + +/// The Raft groups that hold `vshard_ids`, deduplicated. Empty on a node with +/// no routing table: without a cluster there is one copy and nothing to prove. +pub fn groups_of_vshards( + state: &SharedState, + vshard_ids: impl IntoIterator, +) -> crate::Result> { + let Some(routing) = state.cluster_routing.as_ref() else { + return Ok(Vec::new()); + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let mut groups: Vec = Vec::new(); + for vshard_id in vshard_ids { + let group_id = routing + .group_for_vshard(vshard_id) + .map_err(|_| crate::Error::NoLeader { + vshard_id: crate::types::VShardId::new(vshard_id), + })?; + if !groups.contains(&group_id) { + groups.push(group_id); + } + } + Ok(groups) +} + +/// Every data group this node replicates, as a voter or a learner. A read that +/// fans across all local cores observes each of them. The metadata group holds +/// no user rows, so it is left out. +pub fn groups_hosted_here(state: &SharedState) -> Vec { + let Some(routing) = state.cluster_routing.as_ref() else { + return Vec::new(); + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + routing + .group_ids() + .into_iter() + .filter(|&group_id| group_id != nodedb_cluster::METADATA_GROUP_ID) + .filter(|&group_id| hosts_group(state, &routing, group_id)) + .collect() +} + +/// The groups that hold `vshard_ids` and that this node replicates. +/// +/// A read on this node's cores observes only groups this node replicates. A +/// group held elsewhere has no rows here and never applies here, so it is +/// left out. +pub fn hosted_groups_of_vshards( + state: &SharedState, + vshard_ids: impl IntoIterator, +) -> crate::Result> { + let groups = groups_of_vshards(state, vshard_ids)?; + let Some(routing) = state.cluster_routing.as_ref() else { + return Ok(groups); + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + Ok(groups + .into_iter() + .filter(|&group_id| hosts_group(state, &routing, group_id)) + .collect()) +} + +fn hosts_group(state: &SharedState, routing: &nodedb_cluster::RoutingTable, group_id: u64) -> bool { + routing.group_info(group_id).is_some_and(|info| { + info.members.contains(&state.node_id) || info.learners.contains(&state.node_id) + }) +} + +/// The deadline a linearizable read in the running statement gets: never past +/// what is left of the statement's budget. +pub fn statement_read_deadline(state: &SharedState) -> Instant { + let remaining_ms = crate::control::server::shared::session::statement_deadline_ms( + state.tuning.network.default_deadline_secs, + ); + linearizable_read_deadline(Duration::from_millis(remaining_ms)) +} + +/// The deadline a linearizable read gets, never past `statement_budget`. +pub fn linearizable_read_deadline(statement_budget: Duration) -> Instant { + Instant::now() + LINEARIZABLE_READ_TIMEOUT.min(statement_budget) +} + +/// Make a linearizable read of `groups` safe to serve on this node. +/// +/// Each group gets a read index taken after this call, then this node waits +/// until it has applied the group through that index. The groups run +/// concurrently under one `deadline`. +pub async fn confirm_linearizable_read( + state: &SharedState, + groups: &[u64], + deadline: Instant, +) -> crate::Result<()> { + if groups.is_empty() { + return Ok(()); + } + let Some(gate) = state.raft_read_gate.get() else { + // A routing table without a gate: `start_raft` has not published it. + return Err(refused(groups[0], "raft is not serving on this node yet")); + }; + let reads = groups.iter().map(|&group_id| async move { + let remaining = deadline.saturating_duration_since(Instant::now()); + let read_index = + gate.read_index(group_id, remaining) + .await + .map_err(|refusal| match refusal { + ReadIndexRefusal::NotLeader => { + refused(group_id, "no leader confirmed a read index") + } + ReadIndexRefusal::Timeout { waited_ms } => refused( + group_id, + &format!("no quorum confirmed a read index within {waited_ms}ms"), + ), + })?; + wait_applied_through(state, group_id, read_index, deadline).await + }); + futures::future::try_join_all(reads).await.map(|_| ()) +} + +/// Wait until this node has applied `group_id` through `read_index`, or +/// refuse at `deadline`. +pub(crate) async fn wait_applied_through( + state: &SharedState, + group_id: u64, + read_index: u64, + deadline: Instant, +) -> crate::Result<()> { + let watcher = state.applied_index_watcher(group_id); + if watcher.current() >= read_index { + return Ok(()); + } + let remaining = deadline.saturating_duration_since(Instant::now()); + // The watcher parks its caller on a condition variable, so the wait runs + // on the blocking pool. + let outcome = tokio::task::spawn_blocking(move || watcher.wait_for(read_index, remaining)) + .await + .map_err(|e| refused(group_id, &format!("the apply wait did not finish: {e}")))?; + match outcome { + WaitOutcome::Reached => Ok(()), + WaitOutcome::TimedOut => Err(refused( + group_id, + &format!("this node did not apply through read index {read_index} in time"), + )), + WaitOutcome::GroupGone => Err(refused(group_id, "the group left this node")), + } +} + +fn refused(group_id: u64, detail: &str) -> crate::Error { + crate::Error::LinearizableReadRefused { + group_id, + detail: detail.to_owned(), + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use super::{LINEARIZABLE_READ_TIMEOUT, linearizable_read_deadline}; + + #[test] + fn the_deadline_never_exceeds_the_statement_budget() { + let short = Duration::from_millis(10); + let before = std::time::Instant::now(); + let deadline = linearizable_read_deadline(short); + assert!(deadline <= std::time::Instant::now() + short); + assert!(deadline >= before + short); + + let long = LINEARIZABLE_READ_TIMEOUT * 10; + let deadline = linearizable_read_deadline(long); + assert!(deadline <= std::time::Instant::now() + LINEARIZABLE_READ_TIMEOUT); + } +} diff --git a/nodedb/src/control/cluster/metadata_applier/audit.rs b/nodedb/src/control/cluster/metadata_applier/audit.rs index e03dd3de0..bffc2e766 100644 --- a/nodedb/src/control/cluster/metadata_applier/audit.rs +++ b/nodedb/src/control/cluster/metadata_applier/audit.rs @@ -5,18 +5,24 @@ use crate::control::catalog_entry; use crate::control::state::SharedState; +use super::audit_describe::describe_entry; + /// Apply a `MetadataEntry::CaTrustChange` on the host side: write or /// delete `tls/ca.d/.crt`, emit an [`AuditEvent::CertRotation`] /// record, and log for the operator. Hot-reload of the rustls config /// picks up the new trust set on the next connection; the overlap /// window guarantees existing connections keep working through the /// rotation. +/// +/// A failed write or remove returns `Err`, so the entry is re-delivered. +/// Both are idempotent: a rewrite replaces the same file, and removing an +/// absent file succeeds. pub(super) fn apply_ca_trust_change( shared: &SharedState, add: Option<&[u8]>, remove: Option<&[u8; 32]>, raft_index: u64, -) { +) -> crate::Result<()> { use crate::control::cluster::tls::{TLS_SUBDIR, remove_trusted_ca, write_trusted_ca}; use crate::control::security::audit::{AuditAuth, AuditEvent}; @@ -24,33 +30,21 @@ pub(super) fn apply_ca_trust_change( let mut added_fp: Option<[u8; 32]> = None; if let Some(der) = add { - match write_trusted_ca(&tls_dir, der) { - Ok(fp) => { - added_fp = Some(fp); - tracing::info!( - fingerprint = %nodedb_cluster::ca_fingerprint_hex(&fp), - raft_index, - "cluster CA trust: added overlap anchor" - ); - } - Err(e) => { - tracing::warn!(error = %e, raft_index, "ca trust add failed"); - } - } + let fp = write_trusted_ca(&tls_dir, der)?; + added_fp = Some(fp); + tracing::info!( + fingerprint = %nodedb_cluster::ca_fingerprint_hex(&fp), + raft_index, + "cluster CA trust: added overlap anchor" + ); } if let Some(fp) = remove { - match remove_trusted_ca(&tls_dir, fp) { - Ok(()) => { - tracing::info!( - fingerprint = %nodedb_cluster::ca_fingerprint_hex(fp), - raft_index, - "cluster CA trust: removed overlap anchor" - ); - } - Err(e) => { - tracing::warn!(error = %e, raft_index, "ca trust remove failed"); - } - } + remove_trusted_ca(&tls_dir, fp)?; + tracing::info!( + fingerprint = %nodedb_cluster::ca_fingerprint_hex(fp), + raft_index, + "cluster CA trust: removed overlap anchor" + ); } let detail = sonic_rs::to_string(&sonic_rs::json!({ @@ -69,6 +63,7 @@ pub(super) fn apply_ca_trust_change( &AuditAuth::default(), ); } + Ok(()) } /// Emit a [`AuditEvent::DdlChange`] record describing one applied @@ -83,6 +78,16 @@ pub(super) fn emit_ddl_audit( use crate::control::security::audit::{AuditAuth, AuditEvent, DdlAuditDetail}; use crate::control::security::catalog::StoredCollection; + // A consumer offset commit and a backup schedule mark are progress, not + // a schema change. + if matches!( + stamped, + catalog_entry::CatalogEntry::CommitConsumerOffsets(_) + | catalog_entry::CatalogEntry::PutBackupScheduleMark(_) + ) { + return; + } + let (descriptor_name, version_after, hlc) = describe_entry(stamped); let version_before = version_after.saturating_sub(1); @@ -104,7 +109,7 @@ pub(super) fn emit_ddl_audit( // `tenant_id` on the audit entry: the authoritative tenant for // most descriptor types is available on the `Stored*` value, but - // extracting it per-variant would bloat this helper. Leave it + // extracting it per-variant will bloat this helper. Leave it // `None` at this layer — consumers that care route by // `descriptor_kind` + `descriptor_name`. let _ = std::any::type_name::(); @@ -154,312 +159,3 @@ pub(super) fn emit_ddl_audit( Err(_) => emit(), } } - -/// Return `(descriptor_name, version_after, hlc_string)` for a -/// stamped `CatalogEntry`. Delete* variants return `version_after = 0` -/// since the object is being removed. A soft delete keeps its row, so it -/// reports the version and HLC it stamped. -pub(super) fn describe_entry(e: &catalog_entry::CatalogEntry) -> (String, u64, String) { - use catalog_entry::CatalogEntry as E; - match e { - E::PutCollection(c) => ( - c.name.clone(), - c.descriptor_version, - format!("{:?}", c.modification_hlc), - ), - E::PutCollectionIfAbsent(c) => ( - c.name.clone(), - c.descriptor_version, - format!("{:?}", c.modification_hlc), - ), - E::DeactivateCollection { - name, - descriptor_version, - modification_hlc, - .. - } => ( - name.clone(), - *descriptor_version, - format!("{modification_hlc:?}"), - ), - E::PurgeCollection { name, .. } => (name.clone(), 0, String::new()), - E::RecordWalTombstone { collection, .. } => (collection.clone(), 0, String::new()), - E::PutSequence(s) => ( - s.name.clone(), - s.descriptor_version, - format!("{:?}", s.modification_hlc), - ), - E::DeleteSequence { name, .. } => (name.clone(), 0, String::new()), - E::PutSequenceState(s) => (s.name.clone(), 0, String::new()), - E::PutTrigger(t) => ( - t.name.clone(), - t.descriptor_version, - format!("{:?}", t.modification_hlc), - ), - E::DeleteTrigger { name, .. } => (name.clone(), 0, String::new()), - E::PutFunction(f) => ( - f.name.clone(), - f.descriptor_version, - format!("{:?}", f.modification_hlc), - ), - E::DeleteFunction { name, .. } => (name.clone(), 0, String::new()), - E::PutProcedure(p) => ( - p.name.clone(), - p.descriptor_version, - format!("{:?}", p.modification_hlc), - ), - E::DeleteProcedure { name, .. } => (name.clone(), 0, String::new()), - E::PutSchedule(s) => (s.name.clone(), 0, String::new()), - E::DeleteSchedule { name, .. } => (name.clone(), 0, String::new()), - E::PutChangeStream(cs) => (cs.name.clone(), 0, String::new()), - E::DeleteChangeStream { name, .. } => (name.clone(), 0, String::new()), - E::PutUser(u) => (u.username.clone(), 0, String::new()), - E::DropUser { username, .. } => (username.clone(), 0, String::new()), - E::PutRole(r) => (r.name.clone(), 0, String::new()), - E::DeleteRole { name, .. } => (name.clone(), 0, String::new()), - E::PutApiKey(k) => (k.key_id.clone(), 0, String::new()), - E::RevokeApiKey { key_id, .. } => (key_id.clone(), 0, String::new()), - E::PutAuthUser(u) => (u.id.clone(), 0, String::new()), - E::PutMaterializedView(m) => (m.name.clone(), 0, String::new()), - E::DeleteMaterializedView { name, .. } => (name.clone(), 0, String::new()), - E::PutStreamingMaterializedView(m) => (m.name.clone(), 0, String::new()), - E::DeleteStreamingMaterializedView { name, .. } => (name.clone(), 0, String::new()), - E::PutContinuousAggregate(c) => ( - c.name.clone(), - c.descriptor_version, - format!("{:?}", c.modification_hlc), - ), - E::DeleteContinuousAggregate { name, .. } => (name.clone(), 0, String::new()), - E::PutTenant(t) => (t.name.clone(), 0, String::new()), - E::PutTenantWithAdmin { tenant, admin } => (tenant.name.clone(), 0, admin.username.clone()), - E::DeleteTenant { tenant_id, .. } => (tenant_id.to_string(), 0, String::new()), - E::PutRlsPolicy(p) => (p.name.clone(), 0, String::new()), - E::DeleteRlsPolicy { name, .. } => (name.clone(), 0, String::new()), - E::PutRedactionPolicy(p) => (p.name.clone(), 0, String::new()), - E::DeleteRedactionPolicy { for_role, .. } => (for_role.clone(), 0, String::new()), - E::PutPermission(p) => ( - format!("{}@{}:{}", p.grantee, p.target, p.permission), - 0, - String::new(), - ), - E::DeletePermission { - target, - grantee, - permission, - } => (format!("{grantee}@{target}:{permission}"), 0, String::new()), - E::PutScopeGrant(g) => ( - format!("{}:{}@{}", g.grantee_type, g.grantee_id, g.scope_name), - 0, - String::new(), - ), - E::DeleteScopeGrant { - scope_name, - grantee_type, - grantee_id, - } => ( - format!("{grantee_type}:{grantee_id}@{scope_name}"), - 0, - String::new(), - ), - E::PutDatabaseQuota { db_id, .. } => (format!("quota:db:{db_id}"), 0, String::new()), - E::DeleteDatabaseQuota { db_id } => (format!("quota:db:{db_id}"), 0, String::new()), - E::PutTenantQuota { - db_id, tenant_id, .. - } => ( - format!("quota:db:{db_id}:tenant:{tenant_id}"), - 0, - String::new(), - ), - E::DeleteTenantQuota { db_id, tenant_id } => ( - format!("quota:db:{db_id}:tenant:{tenant_id}"), - 0, - String::new(), - ), - E::PutScopeQuota(q) => (format!("quota:scope:{}", q.scope_name), 0, String::new()), - E::DeleteScopeQuota { scope_name } => { - (format!("quota:scope:{scope_name}"), 0, String::new()) - } - E::PutRetentionPolicy(p) => ( - format!("retention:{}:{}:{}", p.database_id, p.tenant_id, p.name), - 0, - String::new(), - ), - E::DeleteRetentionPolicy { - database_id, - tenant_id, - name, - .. - } => ( - format!("retention:{database_id}:{tenant_id}:{name}"), - 0, - String::new(), - ), - E::PutAlertRule(a) => ( - format!("alert:{}:{}:{}", a.database_id, a.tenant_id, a.name), - 0, - String::new(), - ), - E::DeleteAlertRule { - database_id, - tenant_id, - name, - } => ( - format!("alert:{database_id}:{tenant_id}:{name}"), - 0, - String::new(), - ), - E::CreateTopicIfAbsent(t) => ( - format!("topic:{}:{}:{}", t.database_id, t.tenant_id, t.name), - 0, - String::new(), - ), - E::DeleteTopicWithConsumerGroups { - database_id, - tenant_id, - name, - } => ( - format!("topic:{database_id}:{tenant_id}:{name}"), - 0, - String::new(), - ), - E::PutConsumerGroupIfAbsent(g) => ( - format!( - "consumer_group:{}:{}:{}:{}", - g.database_id, g.tenant_id, g.stream_name, g.name - ), - 0, - String::new(), - ), - E::DeleteConsumerGroup { - database_id, - tenant_id, - stream_name, - name, - } => ( - format!("consumer_group:{database_id}:{tenant_id}:{stream_name}:{name}"), - 0, - String::new(), - ), - E::MigrateConsumerGroupStream { def, legacy_stream } => ( - format!( - "consumer_group:{}:{}:{legacy_stream}:{}", - def.database_id, def.tenant_id, def.name - ), - 0, - String::new(), - ), - E::PutCheckpoint(c) => ( - format!( - "checkpoint:{}:{}:{}:{}:{}", - c.database_id, c.tenant_id, c.collection, c.doc_id, c.checkpoint_name - ), - 0, - String::new(), - ), - E::DeleteCheckpoint { - database_id, - tenant_id, - collection, - doc_id, - checkpoint_name, - } => ( - format!("checkpoint:{database_id}:{tenant_id}:{collection}:{doc_id}:{checkpoint_name}"), - 0, - String::new(), - ), - E::CompactHistory { - database_id, - tenant_id, - collection, - doc_id, - before_timestamp, - .. - } => ( - format!( - "checkpoint:{database_id}:{tenant_id}:{collection}:{doc_id}:<{before_timestamp}" - ), - 0, - String::new(), - ), - E::PutVectorModel(m) => ( - format!( - "vector_model:{}:{}:{}:{}", - m.database_id, m.tenant_id, m.collection, m.column - ), - 0, - String::new(), - ), - E::DeleteVectorModel { - database_id, - tenant_id, - collection, - column, - } => ( - format!("vector_model:{database_id}:{tenant_id}:{collection}:{column}"), - 0, - String::new(), - ), - E::PutColumnStats(rows) => ( - rows.first().map_or_else(String::new, |r| { - format!( - "column_stats:{}:{}:{}", - r.database_id, r.tenant_id, r.collection - ) - }), - 0, - String::new(), - ), - E::PutVectorIndexParams(p) => ( - format!( - "vector_index_params:{}:{}:{}:{}", - p.database_id, p.tenant_id, p.collection, p.field_name - ), - 0, - String::new(), - ), - E::DeleteVectorIndexParams { - database_id, - tenant_id, - collection, - field_name, - } => ( - format!("vector_index_params:{database_id}:{tenant_id}:{collection}:{field_name}"), - 0, - String::new(), - ), - E::PutOwner(o) => (o.object_name.clone(), 0, String::new()), - E::DeleteOwner { object_name, .. } => (object_name.clone(), 0, String::new()), - E::PutSynonymGroup(g) => (g.name.clone(), 0, String::new()), - E::DeleteSynonymGroup { name, .. } => (name.clone(), 0, String::new()), - E::PutCustomType(t) => (t.name.clone(), 0, String::new()), - E::DeleteCustomType { name, .. } => (name.clone(), 0, String::new()), - E::PutDatabase(d) => (d.name.clone(), 0, String::new()), - E::DeleteDatabase { db_id } => (db_id.to_string(), 0, String::new()), - E::PutDatabaseGrant { - db_id, - user_id, - privilege, - } => ( - format!("db:{db_id}:user:{user_id}:{privilege}"), - 0, - String::new(), - ), - E::DeleteDatabaseGrant { - db_id, - user_id, - privilege, - } => ( - format!("db:{db_id}:user:{user_id}:{privilege}"), - 0, - String::new(), - ), - E::CloneDatabase { - target_descriptor, .. - } => (target_descriptor.name.clone(), 0, String::new()), - E::MoveTenantCutover { tenant_id, .. } => (format!("tenant:{tenant_id}"), 0, String::new()), - E::PutIndexRecord(r) => (r.name.clone(), 0, String::new()), - E::DeleteIndexRecord { name, .. } => (name.clone(), 0, String::new()), - E::PutOidcProvider(p) => (p.provider_name.clone(), 0, String::new()), - E::DeleteOidcProvider { name } => (name.clone(), 0, String::new()), - } -} diff --git a/nodedb/src/control/cluster/metadata_applier/audit_describe.rs b/nodedb/src/control/cluster/metadata_applier/audit_describe.rs new file mode 100644 index 000000000..f2967c69f --- /dev/null +++ b/nodedb/src/control/cluster/metadata_applier/audit_describe.rs @@ -0,0 +1,376 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Audit description of a stamped `CatalogEntry`. + +use crate::control::catalog_entry; + +/// Return `(descriptor_name, version_after, hlc_string)` for a +/// stamped `CatalogEntry`. Delete* variants return `version_after = 0` +/// since the object is being removed. A soft delete keeps its row, so it +/// reports the version and HLC it stamped. +pub(super) fn describe_entry(e: &catalog_entry::CatalogEntry) -> (String, u64, String) { + use catalog_entry::CatalogEntry as E; + match e { + E::PutCollection(c) => ( + c.name.clone(), + c.descriptor_version, + format!("{:?}", c.modification_hlc), + ), + E::PutCollectionIfAbsent(c) => ( + c.name.clone(), + c.descriptor_version, + format!("{:?}", c.modification_hlc), + ), + E::DeactivateCollection { + name, + descriptor_version, + modification_hlc, + .. + } => ( + name.clone(), + *descriptor_version, + format!("{modification_hlc:?}"), + ), + E::PurgeCollection { name, .. } => (name.clone(), 0, String::new()), + E::RecordWalTombstone { collection, .. } => (collection.clone(), 0, String::new()), + E::PutSequence(s) => ( + s.name.clone(), + s.descriptor_version, + format!("{:?}", s.modification_hlc), + ), + E::DeleteSequence { name, .. } => (name.clone(), 0, String::new()), + E::PutSequenceState(s) => (s.name.clone(), 0, String::new()), + E::PutTrigger(t) => ( + t.name.clone(), + t.descriptor_version, + format!("{:?}", t.modification_hlc), + ), + E::DeleteTrigger { name, .. } => (name.clone(), 0, String::new()), + E::PutFunction(f) => ( + f.name.clone(), + f.descriptor_version, + format!("{:?}", f.modification_hlc), + ), + E::DeleteFunction { name, .. } => (name.clone(), 0, String::new()), + E::PutProcedure(p) => ( + p.name.clone(), + p.descriptor_version, + format!("{:?}", p.modification_hlc), + ), + E::DeleteProcedure { name, .. } => (name.clone(), 0, String::new()), + E::PutSchedule(s) => (s.name.clone(), 0, String::new()), + E::DeleteSchedule { name, .. } => (name.clone(), 0, String::new()), + E::PutChangeStream(cs) => (cs.name.clone(), 0, String::new()), + E::DeleteChangeStream { name, .. } => (name.clone(), 0, String::new()), + E::PutUser(u) => (u.username.clone(), 0, String::new()), + E::DropUser { username, .. } => (username.clone(), 0, String::new()), + E::PutRole(r) => (r.name.clone(), 0, String::new()), + E::DeleteRole { name, .. } => (name.clone(), 0, String::new()), + E::PutApiKey(k) => (k.key_id.clone(), 0, String::new()), + E::RevokeApiKey { key_id, .. } => (key_id.clone(), 0, String::new()), + E::PutAuthUser(u) => (u.id.clone(), 0, String::new()), + E::PutMaterializedView(m) => (m.name.clone(), 0, String::new()), + E::DeleteMaterializedView { name, .. } => (name.clone(), 0, String::new()), + E::PutStreamingMaterializedView(m) => (m.name.clone(), 0, String::new()), + E::DeleteStreamingMaterializedView { name, .. } => (name.clone(), 0, String::new()), + E::PutContinuousAggregate(c) => ( + c.name.clone(), + c.descriptor_version, + format!("{:?}", c.modification_hlc), + ), + E::DeleteContinuousAggregate { name, .. } => (name.clone(), 0, String::new()), + E::PutTenant(t) => (t.name.clone(), 0, String::new()), + E::PutTenantWithAdmin { tenant, admin } => (tenant.name.clone(), 0, admin.username.clone()), + E::DeleteTenant { tenant_id, .. } => (tenant_id.to_string(), 0, String::new()), + E::PutRlsPolicy(p) => (p.name.clone(), 0, String::new()), + E::DeleteRlsPolicy { name, .. } => (name.clone(), 0, String::new()), + E::PutRedactionPolicy(p) => (p.name.clone(), 0, String::new()), + E::DeleteRedactionPolicy { for_role, .. } => (for_role.clone(), 0, String::new()), + E::PutPermission(p) => ( + format!("{}@{}:{}", p.grantee, p.target, p.permission), + 0, + String::new(), + ), + E::DeletePermission { + target, + grantee, + permission, + } => (format!("{grantee}@{target}:{permission}"), 0, String::new()), + E::PutScopeGrant(g) => ( + format!("{}:{}@{}", g.grantee_type, g.grantee_id, g.scope_name), + 0, + String::new(), + ), + E::DeleteScopeGrant { + scope_name, + grantee_type, + grantee_id, + } => ( + format!("{grantee_type}:{grantee_id}@{scope_name}"), + 0, + String::new(), + ), + E::PutDatabaseQuota { db_id, .. } => (format!("quota:db:{db_id}"), 0, String::new()), + E::DeleteDatabaseQuota { db_id } => (format!("quota:db:{db_id}"), 0, String::new()), + E::PutTenantQuota { + db_id, tenant_id, .. + } => ( + format!("quota:db:{db_id}:tenant:{tenant_id}"), + 0, + String::new(), + ), + E::DeleteTenantQuota { db_id, tenant_id } => ( + format!("quota:db:{db_id}:tenant:{tenant_id}"), + 0, + String::new(), + ), + E::PutScopeQuota(q) => (format!("quota:scope:{}", q.scope_name), 0, String::new()), + E::DeleteScopeQuota { scope_name } => { + (format!("quota:scope:{scope_name}"), 0, String::new()) + } + E::PutRetentionPolicy(p) => ( + format!("retention:{}:{}:{}", p.database_id, p.tenant_id, p.name), + 0, + String::new(), + ), + E::DeleteRetentionPolicy { + database_id, + tenant_id, + name, + .. + } => ( + format!("retention:{database_id}:{tenant_id}:{name}"), + 0, + String::new(), + ), + E::PutAlertRule(a) => ( + format!("alert:{}:{}:{}", a.database_id, a.tenant_id, a.name), + 0, + String::new(), + ), + E::DeleteAlertRule { + database_id, + tenant_id, + name, + } => ( + format!("alert:{database_id}:{tenant_id}:{name}"), + 0, + String::new(), + ), + E::CreateTopicIfAbsent(t) => ( + format!("topic:{}:{}:{}", t.database_id, t.tenant_id, t.name), + 0, + String::new(), + ), + E::DeleteTopicWithConsumerGroups { + database_id, + tenant_id, + name, + .. + } => ( + format!("topic:{database_id}:{tenant_id}:{name}"), + 0, + String::new(), + ), + E::PutConsumerGroupIfAbsent(g) => ( + format!( + "consumer_group:{}:{}:{}:{}", + g.database_id, g.tenant_id, g.stream_name, g.name + ), + 0, + String::new(), + ), + E::DeleteConsumerGroup { + database_id, + tenant_id, + stream_name, + name, + .. + } => ( + format!("consumer_group:{database_id}:{tenant_id}:{stream_name}:{name}"), + 0, + String::new(), + ), + E::MigrateConsumerGroupStream { def, legacy_stream } => ( + format!( + "consumer_group:{}:{}:{legacy_stream}:{}", + def.database_id, def.tenant_id, def.name + ), + 0, + String::new(), + ), + E::PutBackupScheduleMark(m) => ( + format!("backup_schedule:{}:{:016x}", m.job, m.incarnation), + 0, + String::new(), + ), + E::CommitConsumerOffsets(c) => ( + format!( + "consumer_group:{}:{}:{}:{}", + c.database_id, c.tenant_id, c.stream_name, c.group_name + ), + 0, + format!("{:?}", c.group_hlc), + ), + E::PutCheckpoint(c) => ( + format!( + "checkpoint:{}:{}:{}:{}:{}", + c.database_id, c.tenant_id, c.collection, c.doc_id, c.checkpoint_name + ), + 0, + String::new(), + ), + E::DeleteCheckpoint { + database_id, + tenant_id, + collection, + doc_id, + checkpoint_name, + } => ( + format!("checkpoint:{database_id}:{tenant_id}:{collection}:{doc_id}:{checkpoint_name}"), + 0, + String::new(), + ), + E::CompactHistory { + database_id, + tenant_id, + collection, + doc_id, + before_timestamp, + .. + } => ( + format!( + "checkpoint:{database_id}:{tenant_id}:{collection}:{doc_id}:<{before_timestamp}" + ), + 0, + String::new(), + ), + E::PutVectorModel(m) => ( + format!( + "vector_model:{}:{}:{}:{}", + m.database_id, m.tenant_id, m.collection, m.column + ), + 0, + String::new(), + ), + E::DeleteVectorModel { + database_id, + tenant_id, + collection, + column, + } => ( + format!("vector_model:{database_id}:{tenant_id}:{collection}:{column}"), + 0, + String::new(), + ), + E::PutCloneCopyup { + database_id, + tenant_id, + collection, + source_surrogate, + .. + } + | E::PutCloneTombstone { + database_id, + tenant_id, + collection, + source_surrogate, + } => ( + format!("clone_cow:{database_id}:{tenant_id}:{collection}:{source_surrogate}"), + 0, + String::new(), + ), + E::PutKvCloneTombstone { + database_id, + tenant_id, + collection, + kv_key, + } => ( + format!("clone_kv_tombstone:{database_id}:{tenant_id}:{collection}:{kv_key}"), + 0, + String::new(), + ), + E::PutCloneSourceDrain(row) => ( + format!( + "clone_source_drain:{}:{}:{}", + row.clone_database, row.tenant_id, row.clone_collection + ), + 0, + String::new(), + ), + E::DeleteCloneSourceDrain { + clone_database, + tenant_id, + clone_collection, + } => ( + format!("clone_source_drain:{clone_database}:{tenant_id}:{clone_collection}"), + 0, + String::new(), + ), + E::PutColumnStats(rows) => ( + rows.first().map_or_else(String::new, |r| { + format!( + "column_stats:{}:{}:{}", + r.database_id, r.tenant_id, r.collection + ) + }), + 0, + String::new(), + ), + E::PutVectorIndexParams(p) => ( + format!( + "vector_index_params:{}:{}:{}:{}", + p.database_id, p.tenant_id, p.collection, p.field_name + ), + 0, + String::new(), + ), + E::DeleteVectorIndexParams { + database_id, + tenant_id, + collection, + field_name, + .. + } => ( + format!("vector_index_params:{database_id}:{tenant_id}:{collection}:{field_name}"), + 0, + String::new(), + ), + E::PutOwner(o) => (o.object_name.clone(), 0, String::new()), + E::DeleteOwner { object_name, .. } => (object_name.clone(), 0, String::new()), + E::PutSynonymGroup(g) => (g.name.clone(), 0, String::new()), + E::DeleteSynonymGroup { name, .. } => (name.clone(), 0, String::new()), + E::PutArray(a) => (a.name.clone(), 0, String::new()), + E::DeleteArray { name, .. } => (name.clone(), 0, String::new()), + E::PutCustomType(t) => (t.name.clone(), 0, String::new()), + E::DeleteCustomType { name, .. } => (name.clone(), 0, String::new()), + E::PutDatabase(d) => (d.name.clone(), 0, String::new()), + E::DeleteDatabase { db_id } => (db_id.to_string(), 0, String::new()), + E::PutDatabaseGrant { + db_id, + user_id, + privilege, + } => ( + format!("db:{db_id}:user:{user_id}:{privilege}"), + 0, + String::new(), + ), + E::DeleteDatabaseGrant { + db_id, + user_id, + privilege, + } => ( + format!("db:{db_id}:user:{user_id}:{privilege}"), + 0, + String::new(), + ), + E::CloneDatabase { + target_descriptor, .. + } => (target_descriptor.name.clone(), 0, String::new()), + E::MoveTenantCutover { tenant_id, .. } => (format!("tenant:{tenant_id}"), 0, String::new()), + E::PutIndexRecord(r) => (r.name.clone(), 0, String::new()), + E::DeleteIndexRecord { name, .. } => (name.clone(), 0, String::new()), + E::PutOidcProvider(p) => (p.provider_name.clone(), 0, String::new()), + E::DeleteOidcProvider { name } => (name.clone(), 0, String::new()), + } +} diff --git a/nodedb/src/control/cluster/metadata_applier/boot_seed.rs b/nodedb/src/control/cluster/metadata_applier/boot_seed.rs new file mode 100644 index 000000000..9df6fa735 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_applier/boot_seed.rs @@ -0,0 +1,271 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Boot seeding of the metadata group's durable host state. +//! +//! The applier persists descriptor leases, drains, the cluster version, the +//! DDL preparation owner, and pending DDL records as it applies them. Boot +//! loads those rows into the in-memory state before any entry applies. +//! State that no row holds starts fresh: +//! - `last_applied_hlc` is the highest expiry over the seeded leases and drains. +//! - `metadata_ddl_applied_token` is 0. +//! - `topology_log`, `routing_log`, and `catalog_entries_applied` are empty. +//! +//! A read error fails boot: running with a partial view of leases or drains +//! admits lease acquires that the cluster has fenced. + +use std::sync::RwLock; + +use nodedb_cluster::MetadataCache; + +use crate::control::security::catalog::SystemCatalog; +use crate::control::state::SharedState; + +/// Replace the drains, pending DDL records, and DDL preparation owner in +/// `shared` with the persisted rows. Runs in every deployment mode: a single +/// node writes drain rows too. +pub fn seed_host_tables(shared: &SharedState) -> crate::Result<()> { + let catalog = shared.credentials.catalog(); + let drains = catalog.load_descriptor_drains()?; + let pending = catalog.load_pending_ddl()?; + let owner = catalog.load_ddl_owner()?; + shared.lease_drain.clear(); + shared.pending_ddl.clear(); + for drain in drains { + shared.lease_drain.install_start( + drain.descriptor_id, + drain.owner, + drain.up_to_version, + drain.expires_at, + drain.proposer_node_id, + ); + } + for record in pending { + shared + .pending_ddl + .insert(record.token, record.objects, record.proposed_at); + } + *shared + .metadata_ddl_owner + .lock() + .unwrap_or_else(|p| p.into_inner()) = + owner.map( + |(token, node_id)| crate::control::metadata_proposer::DdlPrepareOwner { + token, + node_id, + acquired_at: std::time::Instant::now(), + }, + ); + Ok(()) +} + +/// Replace the leases and cluster version in the metadata group's `cache` +/// with the persisted rows, and raise `last_applied_hlc` to the highest +/// persisted lease or drain expiry. +pub fn seed_metadata_cache( + cache: &RwLock, + catalog: &SystemCatalog, +) -> crate::Result<()> { + let leases = catalog.load_descriptor_leases()?; + let drains = catalog.load_descriptor_drains()?; + let cluster_version = catalog.load_cluster_version()?; + + let mut cache = cache.write().unwrap_or_else(|p| p.into_inner()); + cache.leases.clear(); + let mut hlc_floor = cache.last_applied_hlc; + for lease in leases { + if lease.expires_at > hlc_floor { + hlc_floor = lease.expires_at; + } + cache + .leases + .insert((lease.descriptor_id.clone(), lease.node_id), lease); + } + for drain in &drains { + if drain.expires_at > hlc_floor { + hlc_floor = drain.expires_at; + } + } + cache.last_applied_hlc = hlc_floor; + cache.cluster_version = cluster_version.unwrap_or(0); + Ok(()) +} + +#[cfg(test)] +mod tests { + use nodedb_cluster::{ + DescriptorId, DescriptorKind, DescriptorLease, DrainOwner, MetadataApplier, MetadataEntry, + PendingDdlObject, encode_entry, + }; + use nodedb_types::Hlc; + + use crate::control::catalog_entry; + use crate::control::catalog_entry::CatalogEntry; + use crate::control::lease::DrainEntry; + use crate::control::security::catalog::StoredCollection; + + use super::super::MetadataCommitApplier; + use super::super::test_fixture::applier_with_shared_at; + use super::{seed_host_tables, seed_metadata_cache}; + + async fn apply(applier: &MetadataCommitApplier, index: u64, entry: &MetadataEntry) { + assert_eq!( + applier + .apply(&[(index, encode_entry(entry).unwrap())]) + .await, + index, + "entry {index} must apply" + ); + } + + fn orders() -> DescriptorId { + DescriptorId::new(0, 7, DescriptorKind::Collection, "orders") + } + + fn pending_object() -> PendingDdlObject { + let stored = StoredCollection::new(7, "pending_orders", "tester"); + PendingDdlObject::Create { + entry: Box::new(MetadataEntry::CatalogDdl { + payload: catalog_entry::encode(&CatalogEntry::PutCollection(Box::new(stored))) + .unwrap(), + }), + } + } + + /// Every durable host table round-trips: apply in one session, reopen the + /// catalog, seed a fresh session, and read the same state back. + #[tokio::test(flavor = "multi_thread")] + async fn applied_host_state_is_seeded_after_reopen() { + let dir = tempfile::tempdir().unwrap(); + let lease = DescriptorLease { + descriptor_id: orders(), + version: 3, + node_id: 2, + expires_at: Hlc::new(70, 0), + }; + let moving = DrainOwner::MoveTenant { + tenant_id: 7, + source_db_id: 0, + }; + // Built once: `StoredCollection::new` stamps `created_at` from the + // clock, so two builds differ when they straddle a second. + let pending = pending_object(); + { + let (applier, _state) = applier_with_shared_at(dir.path(), "first.wal"); + apply( + &applier, + 1, + &MetadataEntry::DescriptorLeaseGrant(lease.clone()), + ) + .await; + apply( + &applier, + 2, + &MetadataEntry::ClusterVersionBump { from: 0, to: 3 }, + ) + .await; + apply( + &applier, + 3, + &MetadataEntry::DdlPrepareAcquire { + token: 42, + node_id: 2, + }, + ) + .await; + // A pending propose reserves only under the lease owner's token. + apply( + &applier, + 4, + &MetadataEntry::DdlPendingPropose { + token: 42, + objects: vec![pending.clone()], + proposed_at: Hlc::new(60, 0), + }, + ) + .await; + for (index, owner, up_to) in [(5, DrainOwner::Ddl, 4), (6, moving.clone(), 6)] { + apply( + &applier, + index, + &MetadataEntry::DescriptorDrainStart { + descriptor_id: orders(), + up_to_version: up_to, + expires_at: Hlc::new(90, 0), + proposer_node_id: 2, + owner, + }, + ) + .await; + } + apply( + &applier, + 7, + &MetadataEntry::DescriptorDrainEnd { + descriptor_id: orders(), + owner: moving, + }, + ) + .await; + } + + let (applier, state) = applier_with_shared_at(dir.path(), "second.wal"); + seed_host_tables(&state).unwrap(); + seed_metadata_cache(&state.metadata_cache, state.credentials.catalog()).unwrap(); + { + let cache = state.metadata_cache.read().unwrap(); + assert_eq!(cache.leases.get(&(orders(), 2)), Some(&lease)); + assert_eq!(cache.cluster_version, 3); + assert_eq!( + cache.last_applied_hlc, + Hlc::new(90, 0), + "the floor is the highest seeded lease or drain expiry" + ); + assert!(cache.topology_log.is_empty() && cache.routing_log.is_empty()); + } + assert_eq!( + state.lease_drain.snapshot(), + vec![( + orders(), + DrainOwner::Ddl, + DrainEntry { + up_to_version: 4, + expires_at: Hlc::new(90, 0), + proposer_node_id: 2, + } + )], + "an ended owner's drain is not seeded" + ); + let record = state.pending_ddl.get(42).expect("pending record seeded"); + assert_eq!(record.objects, vec![pending]); + assert_eq!(record.proposed_at, Hlc::new(60, 0)); + assert_eq!( + state + .metadata_ddl_owner + .lock() + .unwrap() + .map(|owner| (owner.token, owner.node_id)), + Some((42, 2)) + ); + assert_eq!( + state + .metadata_ddl_applied_token + .load(std::sync::atomic::Ordering::Acquire), + 0 + ); + + // Releases remove the rows the grants wrote. + apply( + &applier, + 1, + &MetadataEntry::DescriptorLeaseRelease { + node_id: 2, + descriptor_ids: vec![orders()], + }, + ) + .await; + apply(&applier, 2, &MetadataEntry::DdlPrepareRelease { token: 42 }).await; + let catalog = state.credentials.catalog(); + assert!(catalog.load_descriptor_leases().unwrap().is_empty()); + assert_eq!(catalog.load_ddl_owner().unwrap(), None); + } +} diff --git a/nodedb/src/control/cluster/metadata_applier/catalog_ddl.rs b/nodedb/src/control/cluster/metadata_applier/catalog_ddl.rs index 56d491cdd..be9bfdc24 100644 --- a/nodedb/src/control/cluster/metadata_applier/catalog_ddl.rs +++ b/nodedb/src/control/cluster/metadata_applier/catalog_ddl.rs @@ -2,8 +2,8 @@ //! `CatalogDdl` / `CatalogDdlAudited` host-side effects: decode the //! opaque payload as a `CatalogEntry`, write through to `SystemCatalog` -//! redb, run synchronous post-apply side effects, emit the DDL audit -//! record, and spawn async post-apply side effects. +//! redb, run synchronous post-apply side effects, run the post-apply +//! dispatch, and emit the DDL audit record. use tracing::{debug, warn}; @@ -12,10 +12,13 @@ use nodedb_cluster::MetadataEntry; use crate::control::catalog_entry; use super::audit::emit_ddl_audit; +use crate::control::state::SharedState; + use super::types::MetadataCommitApplier; impl MetadataCommitApplier { - /// Release the descriptor drain a `Put*` DDL installed. + /// Release the descriptor drain a `Put*` DDL installed, and no drain + /// another owner holds on the same descriptor. /// /// A drain is proposed *before* the DDL and is meant to end when that DDL /// concludes. Concluding includes the outcomes that write nothing — an @@ -24,24 +27,25 @@ impl MetadataCommitApplier { /// catalog changed, so every path that finishes handling the entry must /// clear it. /// - /// Missing one of those paths does not fail loudly: the drain simply + /// Missing one of those paths does not fail loudly: the drain /// survives, and `is_draining` then rejects every plan for that /// descriptor as a retryable schema change with no error explaining /// why. The drain has no self-healing wall-clock expiry (see /// `lease::drain`) — the only backstop is the proposer's own wait /// loop timing out and proposing `DescriptorDrainEnd` explicitly, so /// every code path here must clear the drain it opened. - fn clear_implicit_drain(&self, stamped: &catalog_entry::CatalogEntry) { - if let Some(weak) = self.shared.get() - && let Some(shared) = weak.upgrade() - && let Some(drained_id) = - crate::control::lease::descriptor_id_for_implicit_clear(stamped) - { - shared.lease_drain.install_end(&drained_id); - } + /// + /// The drain rows go first, so a crash after them never leaves a drain + /// that boot will seed back. + fn clear_implicit_drain( + &self, + shared: &SharedState, + stamped: &catalog_entry::CatalogEntry, + ) -> Result<(), crate::Error> { + crate::control::lease::clear_implicit_drains(shared, stamped) } - pub(super) fn apply_catalog_ddl( + pub(super) async fn apply_catalog_ddl( &self, entry: &MetadataEntry, raft_index: u64, @@ -75,6 +79,23 @@ impl MetadataCommitApplier { return Ok(()); } }; + let shared = self.shared_state()?; + + // Holds back one node's apply of backup schedule marks with a + // transient error, so a test can make that node's catalog lag the + // metadata group. Raft re-delivers the entry once it is cleared. + #[cfg(feature = "failpoints")] + if matches!( + stamped, + catalog_entry::CatalogEntry::PutBackupScheduleMark(_) + ) { + nodedb_types::fail_point_err!(&backup_mark_fail_point(shared.node_id), |detail| { + crate::Error::Storage { + engine: "catalog".into(), + detail, + } + }); + } // Descriptor versions (and the constraint_version / // modification_hlc that travel with them) are frozen at PROPOSE @@ -88,8 +109,9 @@ impl MetadataCommitApplier { // node's local prior. Historical entries encountered during a full-log // replay are acknowledged without overwriting newer state or repeating // post-apply side effects. Forward gaps and same-version divergent - // payloads remain loud typed errors. A version of `0` (compat mode / - // unit tests) is applied without version fencing. + // payloads remain loud typed errors. A version of `0` (unit-test + // fixtures that bypass the proposer) is applied without version + // fencing. if matches!( catalog_entry::descriptor_validate::validate(&stamped, catalog)?, catalog_entry::descriptor_validate::ValidationOutcome::AlreadyApplied @@ -102,7 +124,7 @@ impl MetadataCommitApplier { // changed nothing — release it, or every read of the descriptor // stays rejected indefinitely (no wall-clock expiry backstops // this path; see `lease::drain`). - self.clear_implicit_drain(&stamped); + self.clear_implicit_drain(&shared, &stamped)?; return Ok(()); } @@ -113,7 +135,7 @@ impl MetadataCommitApplier { // descriptor that already exists) still concludes its DDL. So // does a refused entry: every node refuses it at this position, // and the proposer reports the refusal to its client. - self.clear_implicit_drain(&stamped); + self.clear_implicit_drain(&shared, &stamped)?; return Ok(()); } // Implicit drain clear: if the entry is a `Put*` for one @@ -122,10 +144,8 @@ impl MetadataCommitApplier { // entry from every node's host tracker. Happens before // post_apply so a subsequent `acquire_lease` fired from // post_apply doesn't see a stale drain. - if let Some(weak) = self.shared.get() - && let Some(shared) = weak.upgrade() { - self.clear_implicit_drain(&stamped); + self.clear_implicit_drain(&shared, &stamped)?; // Run synchronous post-apply side effects INLINE so every // in-memory cache update (install_replicated_user, // install_replicated_owner, etc.) is visible before the @@ -133,21 +153,30 @@ impl MetadataCommitApplier { // moving past `last` is guaranteed to see the sync side // effects of every entry up to `last`. // - // `PutCollection` Register dispatch runs synchronously - // (block_in_place) inside spawn_post_apply_async_side_effects - // and IS part of the applied-index contract: the watcher - // only bumps after doc_configs is populated on every core, - // so subsequent scans always find the schema. + // The awaited post-apply lane is part of the same contract: + // `PutCollection`'s Register dispatch completes on every core + // before the entry counts as applied, so a later scan always + // finds the schema. catalog_entry::post_apply::apply_post_apply_side_effects_sync(&stamped, &shared); - // Emit a DdlChange audit record on every replica. - // Executed BEFORE spawning async post-apply side effects - // so the audit entry lands synchronously with the rest of - // the commit. - emit_ddl_audit(&shared, raft_index, &stamped, audit.as_ref()); + // A reclaim that failed with no durable retry returns `Err`. The + // batch stops here and the re-delivered entry retries the reclaim. + catalog_entry::post_apply::run_post_apply_async_side_effects( + stamped.clone(), + std::sync::Arc::clone(&shared), + ) + .await?; - catalog_entry::post_apply::spawn_post_apply_async_side_effects(stamped, shared); + // Emit a DdlChange audit record on every replica, once the entry + // fully applied, so a re-delivered entry is audited once. + emit_ddl_audit(&shared, raft_index, &stamped, audit.as_ref()); } Ok(()) } } + +/// The fail point that holds back node `node_id`'s apply of backup schedule +/// marks. +pub fn backup_mark_fail_point(node_id: u64) -> String { + format!("backup_schedule::mark_apply::node{node_id}") +} diff --git a/nodedb/src/control/cluster/metadata_applier/database_id.rs b/nodedb/src/control/cluster/metadata_applier/database_id.rs new file mode 100644 index 000000000..acf388b19 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_applier/database_id.rs @@ -0,0 +1,120 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `DatabaseIdReserve` host-side effect. + +use tracing::debug; + +use super::types::MetadataCommitApplier; + +impl MetadataCommitApplier { + /// Carve the next database id at `raft_index` and persist the new hwm + /// with the index before the watermark advances. + /// + /// Every node runs this in log order against the same hwm, so every + /// node issues the same id. A persist error returns `Err`: the + /// watermark stays on this entry and Raft re-delivers it. + pub(super) fn apply_database_id_reserve( + &self, + node_id: u64, + request_id: u64, + raft_index: u64, + ) -> Result<(), crate::Error> { + let shared = self.shared_state()?; + let registry = &shared.database_registry; + let Some(id) = registry.reserve_at_index(raft_index, self.credentials.catalog())? else { + debug!(raft_index, "database_id_reserve: already folded into hwm"); + return Ok(()); + }; + if node_id == shared.node_id { + registry.complete_request(request_id, id); + } + debug!( + node_id, + request_id, + raft_index, + database_id = id.as_u64(), + "database id reserved via raft" + ); + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_cluster::{MetadataApplier, MetadataEntry, encode_entry}; + use nodedb_types::DatabaseId; + + use crate::control::database::USER_DB_START; + use crate::control::state::SharedState; + + use super::super::test_fixture::applier_with_shared_at; + use super::MetadataCommitApplier; + + fn applier_with_shared() -> (MetadataCommitApplier, Arc, tempfile::TempDir) { + let tmp = tempfile::tempdir().expect("tmpdir"); + let (applier, state) = applier_with_shared_at(tmp.path(), "test.wal"); + (applier, state, tmp) + } + + fn reserve(node_id: u64, request_id: u64) -> Vec { + encode_entry(&MetadataEntry::DatabaseIdReserve { + node_id, + request_id, + }) + .expect("encode") + } + + /// The requesting node receives the carved id, the hwm and cursor are + /// durable, and a re-delivered entry issues nothing. + #[tokio::test(flavor = "multi_thread")] + async fn reserve_issues_persists_and_ignores_redelivery() { + let (applier, state, _tmp) = applier_with_shared(); + let registry = &state.database_registry; + let catalog = state.credentials.catalog(); + + let request = registry.begin_request(); + assert_eq!( + applier.apply(&[(4, reserve(state.node_id, request))]).await, + 4 + ); + assert_eq!( + registry.finish_request(request), + Some(DatabaseId::new(USER_DB_START)) + ); + assert_eq!(catalog.get_database_hwm().unwrap(), USER_DB_START); + assert_eq!(catalog.get_database_reserve_index().unwrap(), 4); + + let again = registry.begin_request(); + assert_eq!( + applier.apply(&[(4, reserve(state.node_id, again))]).await, + 4 + ); + assert_eq!(registry.finish_request(again), None); + assert_eq!(catalog.get_database_hwm().unwrap(), USER_DB_START); + } + + /// Another node's reservation advances this node's hwm, so the next + /// local request never receives the id the other node owns. + #[tokio::test(flavor = "multi_thread")] + async fn peer_reservation_advances_hwm_without_completing_local_request() { + let (applier, state, _tmp) = applier_with_shared(); + let registry = &state.database_registry; + let peer = state.node_id + 1; + + let local = registry.begin_request(); + assert_eq!(applier.apply(&[(1, reserve(peer, local))]).await, 1); + assert_eq!(registry.finish_request(local), None); + + let local = registry.begin_request(); + assert_eq!( + applier.apply(&[(2, reserve(state.node_id, local))]).await, + 2 + ); + assert_eq!( + registry.finish_request(local), + Some(DatabaseId::new(USER_DB_START + 1)) + ); + } +} diff --git a/nodedb/src/control/cluster/metadata_applier/dispatch.rs b/nodedb/src/control/cluster/metadata_applier/dispatch.rs index 38d84776b..34dabf8c1 100644 --- a/nodedb/src/control/cluster/metadata_applier/dispatch.rs +++ b/nodedb/src/control/cluster/metadata_applier/dispatch.rs @@ -5,10 +5,17 @@ use tracing::{debug, error, warn}; -use nodedb_cluster::{MetadataApplier, MetadataEntry, RoutingChange, TopologyChange, decode_entry}; +use nodedb_cluster::{ + CommittedMetadata, MetadataApplier, MetadataEntry, MetadataPayload, RoutingChange, + TopologyChange, +}; use super::types::{CatalogChangeEvent, MetadataCommitApplier}; +/// The future of one entry's host-side effects. +pub(super) type HostEffects<'a> = + std::pin::Pin> + Send + 'a>>; + impl MetadataCommitApplier { /// Apply a single decoded `MetadataEntry`'s host-side effects. /// @@ -28,69 +35,76 @@ impl MetadataCommitApplier { /// failure clears on retry; a persistent one leaves the watermark loudly /// stuck (proposer waiters time out) rather than silently diverging from the /// quorum with a false-success ACK. - pub(super) fn apply_host_side_effects( + /// + /// The future is boxed: a prepared DDL and a batch apply their inner + /// entries through this function again. + pub(super) fn apply_host_side_effects<'a>( + &'a self, + entry: &'a MetadataEntry, + raft_index: u64, + ) -> HostEffects<'a> { + Box::pin(async move { + let result = self.apply_host_side_effects_inner(entry, raft_index).await; + // A drain this entry ended wakes the statements waiting it out, + // now that every effect of the entry is visible to their re-plan. + if let Ok(shared) = self.shared_state() { + shared.lease_drain.settle(); + } + result + }) + } + + async fn apply_host_side_effects_inner( &self, entry: &MetadataEntry, raft_index: u64, ) -> Result<(), crate::Error> { - // A prepared DDL is conditionally applied under the replicated owner - // token. A superseded proposal is a deterministic no-op: rejecting a - // committed stale token would wedge the Raft apply watermark forever. - if let MetadataEntry::DdlPrepared { token, entry } = entry { - let Some(shared) = self.shared.get().and_then(std::sync::Weak::upgrade) else { - return Ok(()); - }; - let owns_lease = shared - .metadata_ddl_owner - .lock() - .unwrap_or_else(|poisoned| poisoned.into_inner()) - .is_some_and(|(current, _)| current == *token); - if !owns_lease { - debug!(token, raft_index, "skipping superseded prepared DDL"); - return Ok(()); - } - self.apply_host_side_effects(entry.as_ref(), raft_index)?; - shared - .metadata_ddl_applied_token - .store(*token, std::sync::atomic::Ordering::Release); - return Ok(()); - } - - // Atomic batches unpack one level: the sub-entries are - // applied individually so each gets its own audit record - // stamped with the same raft_index (they committed at the - // same log position). - if let MetadataEntry::Batch { entries } = entry { - for sub in entries { - self.apply_host_side_effects(sub, raft_index)?; - } - return Ok(()); - } - // Handle non-CatalogDdl variants that still have host-side // effects. Drain start/end land on `shared.lease_drain` on // every node so the next `force_refresh_lease` check sees // the replicated drain state. match entry { + MetadataEntry::DdlPrepared { token, entry } => { + return self.apply_prepared_ddl(*token, entry, raft_index).await; + } + MetadataEntry::Batch { entries } => { + return self.apply_batch(entries, raft_index).await; + } MetadataEntry::DescriptorDrainStart { descriptor_id, up_to_version, expires_at, - } => return self.apply_drain_start(descriptor_id, *up_to_version, *expires_at), - MetadataEntry::DescriptorDrainEnd { descriptor_id } => { - return self.apply_drain_end(descriptor_id); + proposer_node_id, + owner, + } => { + return self.apply_drain_start( + descriptor_id, + owner, + *up_to_version, + *expires_at, + *proposer_node_id, + ); } - MetadataEntry::DdlPrepareAcquire { token } => { - if let Some(shared) = self.shared.get().and_then(std::sync::Weak::upgrade) { - let mut owner = shared - .metadata_ddl_owner - .lock() - .unwrap_or_else(|poisoned| poisoned.into_inner()); - if owner.is_none() || owner.is_some_and(|(current, _)| current == *token) { - *owner = Some((*token, std::time::Instant::now())); - } - } - return Ok(()); + MetadataEntry::DescriptorDrainEnd { + descriptor_id, + owner, + } => { + return self.apply_drain_end(descriptor_id, owner); + } + MetadataEntry::DescriptorLeaseRelease { + node_id, + descriptor_ids, + } => { + return self.apply_lease_release(*node_id, descriptor_ids); + } + MetadataEntry::DdlPrepareAcquire { token, node_id } => { + return self.apply_ddl_prepare_acquire(*token, *node_id); + } + MetadataEntry::DescriptorLeaseGrant(lease) => { + return self.apply_lease_grant(lease); + } + MetadataEntry::ClusterVersionBump { to, .. } => { + return self.apply_cluster_version(*to); } MetadataEntry::DdlPendingPropose { token, @@ -100,22 +114,13 @@ impl MetadataCommitApplier { return self.apply_ddl_pending_propose(*token, objects, *proposed_at); } MetadataEntry::DdlPendingFinalize { token } => { - return self.apply_ddl_pending_finalize(*token, raft_index); + return self.apply_ddl_pending_finalize(*token, raft_index).await; } MetadataEntry::DdlPendingCancel { token } => { return self.apply_ddl_pending_cancel(*token); } MetadataEntry::DdlPrepareRelease { token } => { - if let Some(shared) = self.shared.get().and_then(std::sync::Weak::upgrade) { - let mut owner = shared - .metadata_ddl_owner - .lock() - .unwrap_or_else(|poisoned| poisoned.into_inner()); - if owner.is_some_and(|(current, _)| current == *token) { - *owner = None; - } - } - return Ok(()); + return self.apply_ddl_prepare_release(*token); } MetadataEntry::CaTrustChange { add_ca_cert, @@ -130,6 +135,15 @@ impl MetadataCommitApplier { MetadataEntry::SurrogateAlloc { hwm } => { return self.apply_surrogate_alloc(*hwm, raft_index); } + MetadataEntry::DatabaseIdReserve { + node_id, + request_id, + } => { + return self.apply_database_id_reserve(*node_id, *request_id, raft_index); + } + MetadataEntry::RestorePoint { hlc, created_at_ms } => { + return self.apply_restore_point(*hlc, *created_at_ms, raft_index); + } MetadataEntry::SurrogateReserve { node_id, request_id, @@ -190,75 +204,19 @@ impl MetadataCommitApplier { transition, ts_ms, } => { - nodedb_cluster::apply_token_transition_to_mirror( - &self.token_state, - *token_hash, - transition, - *ts_ms, - ); - let state = self - .token_state - .lock() - .unwrap_or_else(|poisoned| poisoned.into_inner()) - .get(token_hash) - .cloned(); - if let Some(state) = state { - self.credentials.catalog().put_join_token_state(&state)?; - } - return Ok(()); + return self.apply_join_token_transition(token_hash, transition, *ts_ms); } MetadataEntry::EnrollmentPreauthorization { spki, expires_at_ms, } => { - let now_ms = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .map(|duration| duration.as_millis() as u64) - .unwrap_or(u64::MAX); - if *expires_at_ms <= now_ms { - return Ok(()); - } - self.credentials - .catalog() - .put_enrollment_preauthorization(spki, *expires_at_ms)?; - let ttl = std::time::Duration::from_millis(expires_at_ms - now_ms); - let transport = self.transport.get().ok_or_else(|| crate::Error::Internal { - detail: "metadata enrollment apply has no cluster transport".into(), - })?; - if !transport.preauthorize_peer_identity(*spki, ttl) { - // Admission remains fail-closed, but replicated metadata - // application must never wedge on a bounded runtime cache. - // The issuer reserves capacity before proposing, so this is - // only a defensive path for stale/corrupt excess entries. - tracing::error!( - ?spki, - "metadata enrollment preauthorization capacity exhausted; entry persisted but not admitted" - ); - } - return Ok(()); + return self.apply_enrollment_preauthorization(spki, *expires_at_ms); } MetadataEntry::EnrollmentPreauthorizationRevoke { spki, expires_at_ms, } => { - let now_ms = std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .map(|duration| duration.as_millis() as u64) - .unwrap_or(u64::MAX); - if *expires_at_ms <= now_ms { - return Ok(()); - } - self.credentials - .catalog() - .remove_enrollment_preauthorization(spki)?; - let transport = self.transport.get().ok_or_else(|| crate::Error::Internal { - detail: "metadata enrollment revoke has no cluster transport".into(), - })?; - transport.revoke_peer_preauthorization( - spki, - std::time::Duration::from_millis(expires_at_ms - now_ms), - ); - return Ok(()); + return self.apply_enrollment_revoke(spki, *expires_at_ms); } MetadataEntry::RoutingChange(RoutingChange::SetPlacement { group_id, @@ -266,35 +224,66 @@ impl MetadataCommitApplier { }) => { return self.apply_set_placement(*group_id, placement, raft_index); } + MetadataEntry::RoutingChange(RoutingChange::ReassignVShard { + vshard_id, + new_group_id, + new_leaseholder_node_id, + }) => { + return self.apply_reassign_vshard( + *vshard_id, + *new_group_id, + *new_leaseholder_node_id, + raft_index, + ); + } MetadataEntry::TopologyChange(TopologyChange::Leave { node_id }) => { - // Lease GC: a node that left the topology can never release - // its own leases. Spawn (do NOT propose-and-wait inline — - // apply runs on the raft loop task; blocking here would - // deadlock the applied-index watcher). - if let Some(shared) = self.shared.get().and_then(std::sync::Weak::upgrade) { - let shared = std::sync::Arc::clone(&shared); - let left_node_id = *node_id; - tokio::spawn(async move { - if !shared.is_singleton_worker() { - return; - } - if let Err(e) = - crate::control::lease::gc::gc_leases_for_node(&shared, left_node_id) - { - tracing::warn!( - node_id = left_node_id, - error = %e, - "lease GC after Leave failed; periodic sweep will retry" - ); - } - }); - } - return Ok(()); + return self.apply_node_leave(*node_id, raft_index); } _ => {} } - self.apply_catalog_ddl(entry, raft_index) + self.apply_catalog_ddl(entry, raft_index).await + } + + /// Apply a prepared DDL under the replicated owner token. A superseded + /// proposal is a deterministic no-op: rejecting a committed stale token + /// will wedge the Raft apply watermark forever. + async fn apply_prepared_ddl( + &self, + token: u64, + entry: &MetadataEntry, + raft_index: u64, + ) -> Result<(), crate::Error> { + let shared = self.shared_state()?; + if !crate::control::metadata_proposer::ddl_owner::owns_ddl_lease(&shared, token) { + debug!(token, raft_index, "skipping superseded prepared DDL"); + return Ok(()); + } + self.apply_host_side_effects(entry, raft_index).await?; + shared + .metadata_ddl_applied_token + .store(token, std::sync::atomic::Ordering::Release); + Ok(()) + } + + /// Apply an atomic batch one level deep. Each sub-entry applies on its + /// own, so each gets its own audit record stamped with the same + /// `raft_index` (they committed at the same log position). + async fn apply_batch( + &self, + entries: &[MetadataEntry], + raft_index: u64, + ) -> Result<(), crate::Error> { + let shared = self.shared.get().and_then(std::sync::Weak::upgrade); + for sub in entries { + self.apply_host_side_effects(sub, raft_index).await?; + if let Some(shared) = &shared { + shared + .metadata_apply_progress + .fetch_add(1, std::sync::atomic::Ordering::Release); + } + } + Ok(()) } /// Publish a permanent apply failure on the node-wide readiness marker. @@ -325,47 +314,90 @@ impl MetadataCommitApplier { } } +/// The fail point that holds back node `node_id`'s whole metadata apply. +pub fn metadata_apply_hold_point(node_id: u64) -> String { + format!("metadata_apply::hold::node{node_id}") +} + +#[async_trait::async_trait] impl MetadataApplier for MetadataCommitApplier { - fn apply(&self, entries: &[(u64, Vec)]) -> u64 { + async fn apply_decoded(&self, entries: &[CommittedMetadata<'_>]) -> u64 { + let shared = self.shared.get().and_then(std::sync::Weak::upgrade); + // Holds back this node's whole metadata apply, so a test can make + // its catalog lag the metadata group. Nothing applies and Raft + // re-delivers the batch once the hold is released. + #[cfg(feature = "failpoints")] + if let Some(shared) = &shared + && crate::control::fail_gate::holds_metadata_apply(shared.node_id) + { + return 0; + } + // Before any entry mutates a registry or the catalog, declare the + // batch's end. A reader that sees an effect of this batch then stamps + // a floor at or above the entry that made it visible. + if let (Some(batch_end), Some(shared)) = (entries.last(), &shared) { + shared + .applied_index_watcher(nodedb_cluster::METADATA_GROUP_ID) + .begin_batch(batch_end.index); + } // `last` is the highest index whose state is GUARANTEED visible. We // only advance it past an entry that fully applied — a durable apply // failure stops the batch here so Raft re-delivers the entry and the // apply is retried (never a silent divergence with a false-success ACK). let mut last = 0u64; - for (index, data) in entries { - if data.is_empty() { - // Raft no-op: nothing to apply, but advance the cache watermark - // in lockstep with the Raft applied index the tick loop reports - // from our return value. Skipping this leaves `cache.applied_index` - // behind the watcher and the startup applied-index sanity check - // fails the boot with a spurious gap (every group's first - // committed entry on a fresh start is an election no-op). - self.cache - .write() - .unwrap_or_else(|p| p.into_inner()) - .advance_applied_index(*index); - last = *index; - continue; - } - let entry = match decode_entry(data) { - Ok(e) => e, - Err(e) => { + for committed in entries { + let index = &committed.index; + let data = committed.data; + let entry = match &committed.payload { + MetadataPayload::Empty => { + // Raft no-op: nothing to apply, but advance the cache + // watermark in lockstep with the Raft applied index the + // tick loop reports from our return value. Skipping this + // leaves `cache.applied_index` behind the watcher and the + // startup applied-index sanity check fails the boot with a + // spurious gap (every group's first committed entry on a + // fresh start is an election no-op). + self.cache + .write() + .unwrap_or_else(|p| p.into_inner()) + .advance_applied_index(*index); + last = *index; + continue; + } + MetadataPayload::Undecodable(e) => { // Undecodable committed entry: deterministic poison, won't // decode on retry — skip (advance) rather than wedge. warn!(index = *index, error = %e, "metadata decode failed"); last = *index; continue; } + MetadataPayload::Decoded(entry) => entry, }; + if let Err(e) = self.observe_stamp(data) { + error!( + index = *index, + last_applied = last, + error = %e, + "metadata apply: recording the entry stamp failed; not advancing \ + watermark — Raft will re-deliver and retry" + ); + break; + } // 1. Cluster-owned cache state (topology, routing, // leases, catalog_entries_applied counter). { let mut guard = self.cache.write().unwrap_or_else(|p| p.into_inner()); - guard.apply(*index, &entry); + guard.apply(*index, entry); } - // 2. Host side effects (redb writeback + async post-apply). A + // 2. Host side effects (redb writeback + awaited post-apply). A // durable failure halts the watermark at the last good index. - if let Err(e) = self.apply_host_side_effects(&entry, *index) { + // Every WAL record the effects append carries the entry's stamp. + let applied = crate::wal::manager::effect_stamp::with_effect_stamp( + nodedb_cluster::entry_stamp(data), + self.apply_host_side_effects(entry, *index), + ) + .await; + if let Err(e) = applied { // Both classes stop the batch — skipping a committed metadata // entry is silent divergence from the quorum and is strictly // worse than halting. What differs is whether waiting for a @@ -373,14 +405,14 @@ impl MetadataApplier for MetadataCommitApplier { let class = super::wedge::classify(&e); // A deterministic failure here re-fails on every re-delivery and // wedges this node's applier forever while /healthz stays green, - // so it is filed as a structured report — not just a log line — + // so it is filed as a structured report — not only a log line — // at the one site that detects it. - crate::diag::metadata_apply_wedged(&e, &entry, *index, last, class.is_permanent()); + crate::diag::metadata_apply_wedged(&e, entry, *index, last, class.is_permanent()); if class.is_permanent() { // Retrying cannot help, so the node must stop advertising // readiness rather than serve queries that will all die on // an unrelated-looking descriptor-lease timeout. - self.record_permanent_wedge(&e, &entry, *index, last); + self.record_permanent_wedge(&e, entry, *index, last); error!( index = *index, last_applied = last, @@ -416,6 +448,12 @@ impl MetadataApplier for MetadataCommitApplier { } last } + + /// Host effects land in redb, or in state boot rebuilds from redb, before + /// `apply` returns an index. + fn durable_effects(&self) -> bool { + true + } } #[cfg(test)] @@ -454,7 +492,7 @@ mod tests { } fn put_collection_entry(name: &str) -> MetadataEntry { - let stored = StoredCollection::new(7, name, "tester"); + let stored = StoredCollection::stamped_for_test(7, name, "tester"); let catalog_entry = CatalogEntry::PutCollection(Box::new(stored)); MetadataEntry::CatalogDdl { payload: catalog_entry::encode(&catalog_entry).unwrap(), @@ -467,23 +505,121 @@ mod tests { } } - /// An applier wired to a real `SharedState` (weak handle installed), the - /// only shape under which `DdlPendingPropose` / `DdlPendingFinalize` / - /// `DdlPendingCancel` do anything — they are no-ops without it, matching - /// every other `self.shared`-gated apply path in this module. + /// Install `token` as the DDL preparation owner, as an applied + /// `DdlPrepareAcquire` does. A pending propose and finalize apply only + /// under the owner's token. + fn own_ddl_lease(state: &SharedState, token: u64) { + *state.metadata_ddl_owner.lock().unwrap() = + Some(crate::control::metadata_proposer::DdlPrepareOwner { + token, + node_id: state.node_id, + acquired_at: std::time::Instant::now(), + }); + } + + /// A reclaimed owner's late propose and finalize apply nothing on any + /// replica, and a prepared entry under its token is skipped too. + #[tokio::test(flavor = "multi_thread")] + async fn a_reclaimed_owners_late_entries_apply_nothing() { + let (applier, state, _core, _tmp) = make_applier_with_acknowledging_core(); + let dead = 21; + let acquire = + |token: u64, node_id: u64| MetadataEntry::DdlPrepareAcquire { token, node_id }; + let entries = [ + acquire(dead, 2), + // The metadata leader reclaims the dead owner's lease. + MetadataEntry::DdlPrepareRelease { token: dead }, + acquire(22, 3), + MetadataEntry::DdlPendingPropose { + token: dead, + objects: vec![pending_create_object("late_pending")], + proposed_at: Hlc::default(), + }, + MetadataEntry::DdlPendingFinalize { token: dead }, + MetadataEntry::DdlPrepared { + token: dead, + entry: Box::new(put_collection_entry("late_prepared")), + }, + ]; + let batch: Vec<_> = entries + .iter() + .zip(1u64..) + .map(|(entry, index)| (index, encode_entry(entry).unwrap())) + .collect(); + assert_eq!(applier.apply(&batch).await, 6, "no late entry wedges apply"); + + assert!( + !state.pending_ddl.contains(dead), + "the late propose reserved nothing" + ); + let catalog = state.credentials.catalog(); + for name in ["late_pending", "late_prepared"] { + assert!( + catalog + .get_collection(DatabaseId::DEFAULT, 7, name) + .unwrap() + .is_none(), + "{name} must not apply under a reclaimed token" + ); + } + assert_ne!( + state + .metadata_ddl_applied_token + .load(std::sync::atomic::Ordering::Acquire), + dead, + "the dead owner's proposer must see its entries superseded" + ); + assert_eq!(catalog.load_ddl_owner().unwrap(), Some((22, 3))); + } + + /// An applier wired to a real `SharedState` (weak handle installed). Every + /// entry with host effects needs it: without it the apply returns a + /// transient error and the watermark stays put. fn make_applier_with_shared() -> (MetadataCommitApplier, Arc, tempfile::TempDir) { + let (applier, state, _data_sides, tmp) = applier_over_shared_state(); + // `_data_sides` drops here, so no Data Plane request is ever answered. + (applier, state, tmp) + } + + /// [`make_applier_with_shared`] with one Data Plane core that + /// acknowledges every request. A create's post-apply clears the name's + /// storage on every core before it registers, so an applied create needs + /// the answer. Runs inside a multi-thread tokio runtime: the core is a + /// spawned task, and the applier blocks its own worker for the post-apply. + fn make_applier_with_acknowledging_core() -> ( + MetadataCommitApplier, + Arc, + tokio::task::JoinHandle<()>, + tempfile::TempDir, + ) { + let (applier, state, mut data_sides, tmp) = applier_over_shared_state(); + let side = data_sides.pop().expect("one data side"); + let core = tokio::spawn(crate::control::state::test_core::acknowledge_every_request( + Arc::clone(&state), + side, + )); + (applier, state, core, tmp) + } + + /// An applier over a one-core `SharedState`, and that core's data side. + fn applier_over_shared_state() -> ( + MetadataCommitApplier, + Arc, + Vec, + tempfile::TempDir, + ) { let tmp = tempfile::tempdir().expect("tmpdir"); let wal = Arc::new(WalManager::open_for_testing(&tmp.path().join("test.wal")).expect("open wal")); let credentials = Arc::new(CredentialStore::open(&tmp.path().join("system.redb")).expect("open catalog")); - let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let (dispatcher, data_sides) = Dispatcher::new(1, 64); let mut state = SharedState::new_with_credentials(dispatcher, wal, credentials, false) .expect("construct shared state"); - // `_data_sides` is dropped with this fixture, so the schema-register - // barrier can never be answered. Keep its deadline short: the test - // covers finalize semantics, not the production deadline. Sole - // reference here, so `get_mut` always succeeds. + // A fixture whose data side drops never answers a Data Plane request. + // Keep the request deadline short: the tests cover apply semantics, + // not the production deadline. Sole reference here, so `get_mut` + // always succeeds. Arc::get_mut(&mut state) .expect("sole reference to the fixture's SharedState") .tuning @@ -498,19 +634,23 @@ mod tests { token_state, ); applier.install_shared(Arc::downgrade(&state)); - (applier, state, tmp) + (applier, state, data_sides, tmp) } #[tokio::test(flavor = "multi_thread")] async fn propose_then_finalize_applies_and_clears_the_record() { - let (applier, state, _tmp) = make_applier_with_shared(); + let (applier, state, _core, _tmp) = make_applier_with_acknowledging_core(); let token = 1; + own_ddl_lease(&state, token); let propose = MetadataEntry::DdlPendingPropose { token, objects: vec![pending_create_object("pending_orders")], proposed_at: Hlc::default(), }; - assert_eq!(applier.apply(&[(1, encode_entry(&propose).unwrap())]), 1); + assert_eq!( + applier.apply(&[(1, encode_entry(&propose).unwrap())]).await, + 1 + ); assert!( state.pending_ddl.contains(token), "propose reserves the record" @@ -526,7 +666,12 @@ mod tests { ); let finalize = MetadataEntry::DdlPendingFinalize { token }; - assert_eq!(applier.apply(&[(2, encode_entry(&finalize).unwrap())]), 2); + assert_eq!( + applier + .apply(&[(2, encode_entry(&finalize).unwrap())]) + .await, + 2 + ); assert!( !state.pending_ddl.contains(token), "finalize clears the record" @@ -543,7 +688,12 @@ mod tests { // Double-apply (Raft re-delivery): no record left, so this must be a // silent no-op rather than an error or a repeat write. - assert_eq!(applier.apply(&[(3, encode_entry(&finalize).unwrap())]), 3); + assert_eq!( + applier + .apply(&[(3, encode_entry(&finalize).unwrap())]) + .await, + 3 + ); assert!(!state.pending_ddl.contains(token)); } @@ -551,20 +701,40 @@ mod tests { async fn propose_then_cancel_clears_without_touching_the_catalog() { let (applier, state, _tmp) = make_applier_with_shared(); let token = 2; + own_ddl_lease(&state, token); let propose = MetadataEntry::DdlPendingPropose { token, objects: vec![pending_create_object("pending_widgets")], proposed_at: Hlc::default(), }; - assert_eq!(applier.apply(&[(1, encode_entry(&propose).unwrap())]), 1); + assert_eq!( + applier.apply(&[(1, encode_entry(&propose).unwrap())]).await, + 1 + ); assert!(state.pending_ddl.contains(token)); let cancel = MetadataEntry::DdlPendingCancel { token }; - assert_eq!(applier.apply(&[(2, encode_entry(&cancel).unwrap())]), 2); + assert_eq!( + applier.apply(&[(2, encode_entry(&cancel).unwrap())]).await, + 2 + ); assert!( !state.pending_ddl.contains(token), "cancel clears the record" ); + // The fixture's Data Plane never answers, so the spawned teardown + // cannot succeed and remove the row: it was queued before the spawn. + let queued = state + .credentials + .catalog() + .load_pending_reclaim_queue() + .unwrap(); + assert!( + queued + .iter() + .any(|entry| entry.tenant_id == 7 && entry.name == "pending_widgets"), + "cancel queues a durable reclaim for the teardown: {queued:?}" + ); assert!( state .credentials @@ -576,45 +746,108 @@ mod tests { ); // Double-apply (Raft re-delivery): no record left, must stay a no-op. - assert_eq!(applier.apply(&[(3, encode_entry(&cancel).unwrap())]), 3); + assert_eq!( + applier.apply(&[(3, encode_entry(&cancel).unwrap())]).await, + 3 + ); assert!(!state.pending_ddl.contains(token)); } + /// A reclaim that fails before any durable retry is queued stops the + /// batch at its entry. The re-delivered entry runs the reclaim again. + #[cfg(feature = "failpoints")] + #[tokio::test(flavor = "multi_thread")] + async fn a_reclaim_with_no_retry_queued_stops_the_batch_and_redelivery_retries_it() { + let (applier, state, _tmp) = make_applier_with_shared(); + let purge = MetadataEntry::CatalogDdl { + payload: catalog_entry::encode(&CatalogEntry::PurgeCollection { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 7, + name: "orders".to_string(), + target_descriptor_version: 0, + target_hlc: Hlc::ZERO, + }) + .unwrap(), + }; + let batch = [(1, encode_entry(&purge).unwrap()), (2, Vec::new())]; + + { + let _fail = nodedb_types::fail_point::FailGuard::fail( + "collection_reclaim::before_tombstone", + "injected tombstone write error", + ); + assert_eq!( + applier.apply(&batch).await, + 0, + "the failed reclaim stops the batch before its entry" + ); + } + assert!( + state + .credentials + .catalog() + .load_pending_reclaim_queue() + .unwrap() + .is_empty(), + "the injected error precedes every durable reclaim step" + ); + + // Re-delivery: the reclaim runs again. The fixture's Data Plane never + // answers, so it ends with a queued durable retry, which counts as done. + assert_eq!(applier.apply(&batch).await, 2); + let queued = state + .credentials + .catalog() + .load_pending_reclaim_queue() + .unwrap(); + assert!( + queued + .iter() + .any(|entry| entry.tenant_id == 7 && entry.name == "orders"), + "the re-delivered purge ran its reclaim: {queued:?}" + ); + } + #[tokio::test(flavor = "multi_thread")] async fn finalize_and_cancel_for_an_unknown_token_are_noops() { let (applier, state, _tmp) = make_applier_with_shared(); let unknown = 999; assert_eq!( - applier.apply(&[( - 1, - encode_entry(&MetadataEntry::DdlPendingFinalize { token: unknown }).unwrap() - )]), + applier + .apply(&[( + 1, + encode_entry(&MetadataEntry::DdlPendingFinalize { token: unknown }).unwrap() + )]) + .await, 1, "finalize with no matching propose must not wedge the watermark" ); assert_eq!( - applier.apply(&[( - 2, - encode_entry(&MetadataEntry::DdlPendingCancel { token: unknown }).unwrap() - )]), + applier + .apply(&[( + 2, + encode_entry(&MetadataEntry::DdlPendingCancel { token: unknown }).unwrap() + )]) + .await, 2, "cancel with no matching propose must not wedge the watermark" ); assert!(!state.pending_ddl.contains(unknown)); } - #[test] - fn apply_put_collection_writes_through_to_redb() { - let (applier, cache, credentials, _tmp) = make_applier(); + #[tokio::test(flavor = "multi_thread")] + async fn apply_put_collection_writes_through_to_redb() { + let (applier, state, _core, _tmp) = make_applier_with_acknowledging_core(); let bytes = encode_entry(&put_collection_entry("orders")).unwrap(); - assert_eq!(applier.apply(&[(11, bytes)]), 11); + assert_eq!(applier.apply(&[(11, bytes)]).await, 11); - let cache_guard = cache.read().unwrap(); + let cache_guard = state.metadata_cache.read().unwrap(); assert_eq!(cache_guard.applied_index, 11); assert_eq!(cache_guard.catalog_entries_applied, 1); drop(cache_guard); - let loaded = credentials + let loaded = state + .credentials .catalog() .get_collection(DatabaseId::DEFAULT, 7, "orders") .unwrap() @@ -623,12 +856,18 @@ mod tests { assert_eq!(loaded.owner, "tester"); } - #[test] - fn apply_deactivate_preserves_record() { - let (applier, _cache, credentials, _tmp) = make_applier(); + #[tokio::test(flavor = "multi_thread")] + async fn apply_deactivate_preserves_record() { + let (applier, state, _core, _tmp) = make_applier_with_acknowledging_core(); // Seed. - applier.apply(&[(1, encode_entry(&put_collection_entry("archived")).unwrap())]); + assert_eq!( + applier + .apply(&[(1, encode_entry(&put_collection_entry("archived")).unwrap())]) + .await, + 1, + "the seeding create applies" + ); let drop_entry = MetadataEntry::CatalogDdl { payload: catalog_entry::encode(&CatalogEntry::DeactivateCollection { @@ -640,9 +879,12 @@ mod tests { }) .unwrap(), }; - applier.apply(&[(2, encode_entry(&drop_entry).unwrap())]); + applier + .apply(&[(2, encode_entry(&drop_entry).unwrap())]) + .await; - let loaded = credentials + let loaded = state + .credentials .catalog() .get_collection(DatabaseId::DEFAULT, 7, "archived") .unwrap() @@ -650,8 +892,8 @@ mod tests { assert!(!loaded.is_active); } - #[test] - fn join_token_transition_updates_and_persists_shared_mirror() { + #[tokio::test] + async fn join_token_transition_updates_and_persists_shared_mirror() { let (applier, _cache, credentials, _tmp) = make_applier(); let hash = [0x44; 32]; let entries = [ @@ -683,7 +925,9 @@ mod tests { for (offset, entry) in entries.iter().enumerate() { let index = offset as u64 + 1; assert_eq!( - applier.apply(&[(index, encode_entry(entry).expect("encode"))]), + applier + .apply(&[(index, encode_entry(entry).expect("encode"))]) + .await, index ); } @@ -700,21 +944,37 @@ mod tests { )); } - #[test] - fn apply_empty_batch_is_noop() { + /// Without `SharedState` a catalog DDL cannot land its host effects, so + /// the watermark stays below the entry and Raft re-delivers it. + #[tokio::test] + async fn catalog_ddl_without_shared_state_is_not_applied() { + let (applier, _cache, credentials, _tmp) = make_applier(); + let bytes = encode_entry(&put_collection_entry("orders")).unwrap(); + assert_eq!(applier.apply(&[(4, bytes)]).await, 0); + assert!( + credentials + .catalog() + .get_collection(DatabaseId::DEFAULT, 7, "orders") + .unwrap() + .is_none() + ); + } + + #[tokio::test] + async fn apply_empty_batch_is_noop() { let (applier, _cache, _credentials, _tmp) = make_applier(); - assert_eq!(applier.apply(&[]), 0); + assert_eq!(applier.apply(&[]).await, 0); } - #[test] - fn apply_noop_entry_advances_cache_watermark() { + #[tokio::test] + async fn apply_noop_entry_advances_cache_watermark() { let (applier, cache, _credentials, _tmp) = make_applier(); // A committed Raft no-op (empty payload) at index 1 — the shape of every // group's first entry on a fresh single-node start. It mutates nothing, but // the cache watermark must advance in lockstep with the Raft applied index // the tick loop takes from the return value; otherwise the startup // applied-index sanity check reads a spurious gap and fails the boot. - assert_eq!(applier.apply(&[(1, Vec::new())]), 1); + assert_eq!(applier.apply(&[(1, Vec::new())]).await, 1); assert_eq!(cache.read().unwrap().applied_index, 1); assert_eq!( cache.read().unwrap().catalog_entries_applied, @@ -738,12 +998,16 @@ mod tests { async fn a_proposers_hlc_advances_the_local_clock() { let (applier, state, _tmp) = make_applier_with_shared(); let ahead = Hlc::new(now_ns() + 500_000_000, 0); + own_ddl_lease(&state, 9); let propose = MetadataEntry::DdlPendingPropose { token: 9, objects: vec![pending_create_object("hlc_fold")], proposed_at: ahead, }; - assert_eq!(applier.apply(&[(1, encode_entry(&propose).unwrap())]), 1); + assert_eq!( + applier.apply(&[(1, encode_entry(&propose).unwrap())]).await, + 1 + ); assert!( state.hlc_clock.peek() >= ahead, "the local clock must absorb a proposer's observation" @@ -756,12 +1020,16 @@ mod tests { async fn a_far_future_proposer_hlc_is_refused_but_the_entry_still_applies() { let (applier, state, _tmp) = make_applier_with_shared(); let far = Hlc::new(now_ns() + nodedb_types::MAX_CLOCK_SKEW_NS * 10, 0); + own_ddl_lease(&state, 11); let propose = MetadataEntry::DdlPendingPropose { token: 11, objects: vec![pending_create_object("hlc_skew")], proposed_at: far, }; - assert_eq!(applier.apply(&[(1, encode_entry(&propose).unwrap())]), 1); + assert_eq!( + applier.apply(&[(1, encode_entry(&propose).unwrap())]).await, + 1 + ); assert!( state.hlc_clock.peek() < far, "a skewed proposer must never move this node's clock" @@ -772,4 +1040,46 @@ mod tests { protection, wedging the state machine is not" ); } + + /// A topic the batch in progress creates is visible before the applied + /// watermark reaches its entry. The floor a reader stamps once it sees + /// the topic is at or above the entry that created it. + #[tokio::test(flavor = "multi_thread")] + async fn an_object_visible_mid_batch_yields_a_floor_at_its_entry() { + let (applier, state, _tmp) = make_applier_with_shared(); + let topic = crate::event::topic::TopicDef { + database_id: DatabaseId::DEFAULT, + tenant_id: 1, + name: "floor_feed".into(), + retention: crate::event::cdc::stream_def::RetentionConfig::default(), + owner: "tester".into(), + created_at: 0, + last_sequence: 0, + last_lsn: 0, + last_epoch: 0, + modification_hlc: Hlc::ZERO, + }; + let create = MetadataEntry::CatalogDdl { + payload: catalog_entry::encode(&CatalogEntry::CreateTopicIfAbsent(Box::new(topic))) + .unwrap(), + }; + let watcher = state.applied_index_watcher(nodedb_cluster::METADATA_GROUP_ID); + // The entry creating the topic sits inside a batch that ends later. + let batch = [(7, encode_entry(&create).unwrap()), (8, Vec::new())]; + assert_eq!(applier.apply(&batch).await, 8); + + // The Raft loop bumps the applied watermark only after the batch + // returns, so the topic is visible with `applied` still below it. + assert!( + state + .ep_topic_registry + .get(DatabaseId::DEFAULT, 1, "floor_feed") + .is_some() + ); + assert!(watcher.current() < 7); + assert!( + watcher.floor() >= 7, + "a reader that sees the topic stamps a floor at or above its creation" + ); + } } diff --git a/nodedb/src/control/cluster/metadata_applier/host_state.rs b/nodedb/src/control/cluster/metadata_applier/host_state.rs new file mode 100644 index 000000000..c47f7d6f6 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_applier/host_state.rs @@ -0,0 +1,65 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Durable host state of the metadata group: descriptor leases, the cluster +//! version, and the owner of the DDL preparation lease. +//! +//! Each apply writes its `SystemCatalog` row before it returns, and boot +//! seeds the in-memory state from the rows. No state here depends on +//! replaying the metadata log. + +use nodedb_cluster::DescriptorLease; + +use super::types::MetadataCommitApplier; + +impl MetadataCommitApplier { + /// `DescriptorLeaseGrant`: the cache holds the lease; persist it. + pub(super) fn apply_lease_grant(&self, lease: &DescriptorLease) -> Result<(), crate::Error> { + self.credentials.catalog().put_descriptor_lease(lease) + } + + /// `ClusterVersionBump`: the cache holds the new version; persist it. + pub(super) fn apply_cluster_version(&self, to: u16) -> Result<(), crate::Error> { + self.credentials.catalog().put_cluster_version(to) + } + + /// `DdlPrepareAcquire`: `token`, proposed by `node_id`, takes the + /// preparation lease when it is free. A re-delivered acquire of the + /// current owner keeps its first apply time. + pub(super) fn apply_ddl_prepare_acquire( + &self, + token: u64, + node_id: u64, + ) -> Result<(), crate::Error> { + let shared = self.shared_state()?; + let mut owner = shared + .metadata_ddl_owner + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()); + if owner.is_none() { + self.credentials + .catalog() + .put_ddl_owner(Some((token, node_id)))?; + *owner = Some(crate::control::metadata_proposer::DdlPrepareOwner { + token, + node_id, + acquired_at: std::time::Instant::now(), + }); + } + Ok(()) + } + + /// `DdlPrepareRelease`: `token` gives the preparation lease up, if it + /// holds it. + pub(super) fn apply_ddl_prepare_release(&self, token: u64) -> Result<(), crate::Error> { + let shared = self.shared_state()?; + let mut owner = shared + .metadata_ddl_owner + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()); + if owner.is_some_and(|current| current.token == token) { + self.credentials.catalog().put_ddl_owner(None)?; + *owner = None; + } + Ok(()) + } +} diff --git a/nodedb/src/control/cluster/metadata_applier/lease_events.rs b/nodedb/src/control/cluster/metadata_applier/lease_events.rs index d33af6feb..7568957b8 100644 --- a/nodedb/src/control/cluster/metadata_applier/lease_events.rs +++ b/nodedb/src/control/cluster/metadata_applier/lease_events.rs @@ -1,10 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Descriptor-drain and CA-trust-change host-side effects. +//! Descriptor-drain, lease-release, and CA-trust-change host-side effects. use tracing::debug; -use nodedb_cluster::DescriptorId; +use nodedb_cluster::{DescriptorId, DrainOwner}; use nodedb_types::Hlc; use super::audit::apply_ca_trust_change; @@ -14,40 +14,60 @@ impl MetadataCommitApplier { pub(super) fn apply_drain_start( &self, descriptor_id: &DescriptorId, + owner: &DrainOwner, up_to_version: u64, expires_at: Hlc, + proposer_node_id: u64, ) -> Result<(), crate::Error> { - if let Some(weak) = self.shared.get() - && let Some(shared) = weak.upgrade() - { - // This exact gate also covers plan admission's drain check, - // refcount increment, and first-holder acquire. - let _admission_gate = shared - .lease_admission_gate - .lock() - .unwrap_or_else(|poison| poison.into_inner()); - shared - .lease_drain - .install_start(descriptor_id.clone(), up_to_version, expires_at); - debug!( - descriptor = ?descriptor_id, - up_to_version, - "drain_start applied to host tracker" - ); - } + let shared = self.shared_state()?; + crate::control::lease::apply_drain_start( + &shared, + descriptor_id, + owner, + up_to_version, + expires_at, + proposer_node_id, + )?; + debug!( + descriptor = ?descriptor_id, + up_to_version, + "drain_start applied to host tracker" + ); Ok(()) } - pub(super) fn apply_drain_end(&self, descriptor_id: &DescriptorId) -> Result<(), crate::Error> { - if let Some(weak) = self.shared.get() - && let Some(shared) = weak.upgrade() - { - shared.lease_drain.install_end(descriptor_id); - debug!( - descriptor = ?descriptor_id, - "drain_end applied to host tracker" - ); - } + /// A release of this node's lease ends the statements still running + /// under it: other nodes now treat the lease as gone. + pub(super) fn apply_lease_release( + &self, + node_id: u64, + descriptor_ids: &[DescriptorId], + ) -> Result<(), crate::Error> { + let shared = self.shared_state()?; + self.credentials + .catalog() + .remove_descriptor_leases(node_id, descriptor_ids)?; + crate::control::lease::revoke_on_release(&shared, node_id, descriptor_ids); + // A drain waiting on one of these leases counts again now. + shared.lease_drain.wake_drain_waiters(); + Ok(()) + } + + pub(super) fn apply_drain_end( + &self, + descriptor_id: &DescriptorId, + owner: &DrainOwner, + ) -> Result<(), crate::Error> { + let shared = self.shared_state()?; + crate::control::lease::apply_drain_ends( + &shared, + &[(descriptor_id.clone(), owner.clone())], + )?; + debug!( + descriptor = ?descriptor_id, + ?owner, + "drain_end applied to host tracker" + ); Ok(()) } @@ -57,11 +77,7 @@ impl MetadataCommitApplier { remove_ca_fingerprint: Option<&[u8; 32]>, raft_index: u64, ) -> Result<(), crate::Error> { - if let Some(weak) = self.shared.get() - && let Some(shared) = weak.upgrade() - { - apply_ca_trust_change(&shared, add_ca_cert, remove_ca_fingerprint, raft_index); - } - Ok(()) + let shared = self.shared_state()?; + apply_ca_trust_change(&shared, add_ca_cert, remove_ca_fingerprint, raft_index) } } diff --git a/nodedb/src/control/cluster/metadata_applier/membership_effects.rs b/nodedb/src/control/cluster/metadata_applier/membership_effects.rs new file mode 100644 index 000000000..19087da11 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_applier/membership_effects.rs @@ -0,0 +1,129 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Host-side effects of committed membership entries: join tokens, +//! enrollment preauthorizations, and a node's leave. + +use nodedb_cluster::JoinTokenTransitionKind; + +use super::types::MetadataCommitApplier; + +/// Unix-epoch milliseconds. A clock before the epoch reads as the far +/// future, so no preauthorization is applied against it. +fn unix_now_ms() -> u64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|duration| duration.as_millis() as u64) + .unwrap_or(u64::MAX) +} + +impl MetadataCommitApplier { + /// Mirror a join-token transition and persist the token's new state. + pub(super) fn apply_join_token_transition( + &self, + token_hash: &[u8; 32], + transition: &JoinTokenTransitionKind, + ts_ms: u64, + ) -> Result<(), crate::Error> { + nodedb_cluster::apply_token_transition_to_mirror( + &self.token_state, + *token_hash, + transition, + ts_ms, + ); + let state = self + .token_state + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .get(token_hash) + .cloned(); + if let Some(state) = state { + self.credentials.catalog().put_join_token_state(&state)?; + } + Ok(()) + } + + /// Persist an enrollment preauthorization and admit its identity until it + /// expires. An expired entry applies nothing. + pub(super) fn apply_enrollment_preauthorization( + &self, + spki: &[u8; 32], + expires_at_ms: u64, + ) -> Result<(), crate::Error> { + let now_ms = unix_now_ms(); + if expires_at_ms <= now_ms { + return Ok(()); + } + self.credentials + .catalog() + .put_enrollment_preauthorization(spki, expires_at_ms)?; + let ttl = std::time::Duration::from_millis(expires_at_ms - now_ms); + let transport = self.transport.get().ok_or_else(|| crate::Error::Internal { + detail: "metadata enrollment apply has no cluster transport".into(), + })?; + if !transport.preauthorize_peer_identity(*spki, ttl) { + // Admission remains fail-closed, but replicated metadata + // application must never wedge on a bounded runtime cache. + // The issuer reserves capacity before proposing, so this is + // only a defensive path for stale/corrupt excess entries. + tracing::error!( + ?spki, + "metadata enrollment preauthorization capacity exhausted; entry persisted but not admitted" + ); + } + Ok(()) + } + + /// Remove an enrollment preauthorization and revoke its admission. An + /// expired entry applies nothing. + pub(super) fn apply_enrollment_revoke( + &self, + spki: &[u8; 32], + expires_at_ms: u64, + ) -> Result<(), crate::Error> { + let now_ms = unix_now_ms(); + if expires_at_ms <= now_ms { + return Ok(()); + } + self.credentials + .catalog() + .remove_enrollment_preauthorization(spki)?; + let transport = self.transport.get().ok_or_else(|| crate::Error::Internal { + detail: "metadata enrollment revoke has no cluster transport".into(), + })?; + transport.revoke_peer_preauthorization( + spki, + std::time::Duration::from_millis(expires_at_ms - now_ms), + ); + Ok(()) + } + + /// Queue the cleanup a node that left owes. + /// + /// A node that left can never release its own leases or end its own + /// drains. The owed cleanup is durable before the entry counts as + /// applied. The spawned drive, the boot drain, and the retry worker + /// carry it out. The proposals run off the raft loop task: waiting on + /// them here will deadlock the applied-index watcher. + pub(super) fn apply_node_leave( + &self, + node_id: u64, + raft_index: u64, + ) -> Result<(), crate::Error> { + let shared = self.shared_state()?; + self.credentials + .catalog() + .enqueue_pending_leave_cleanup(node_id, raft_index)?; + tokio::spawn(async move { + if let Err(error) = + crate::control::lease::leave_cleanup::drive_leave_cleanup(&shared, node_id).await + { + tracing::warn!( + node_id, + %error, + "leave cleanup did not finish; the retry worker re-drives it" + ); + } + }); + Ok(()) + } +} diff --git a/nodedb/src/control/cluster/metadata_applier/mod.rs b/nodedb/src/control/cluster/metadata_applier/mod.rs index c626df56a..e5c28f67e 100644 --- a/nodedb/src/control/cluster/metadata_applier/mod.rs +++ b/nodedb/src/control/cluster/metadata_applier/mod.rs @@ -12,36 +12,57 @@ //! The applier broadcasts `CatalogChangeEvent` (for future //! prepared-statement / catalog cache invalidation). The per-group //! apply watermark is maintained by the Raft tick loop directly via -//! [`nodedb_cluster::GroupAppliedWatchers`] — the applier no longer -//! owns its own watcher because that primitive is now keyed by -//! `group_id` and shared across every Raft group on the node. +//! [`nodedb_cluster::GroupAppliedWatchers`] — the applier owns +//! no watcher because that primitive is keyed by `group_id` and shared +//! across every Raft group on the node. //! //! Split by concern: //! - [`types`]: the `MetadataCommitApplier` struct, construction, and //! `CatalogChangeEvent`. //! - [`lease_events`]: descriptor-drain and CA-trust-change effects. +//! - [`membership_effects`]: join tokens, enrollment preauthorizations, +//! and a node's leave. //! - [`surrogate`]: cross-engine surrogate HWM + HiLo batch reservation. +//! - [`database_id`]: replicated database-id reservation. +//! - [`restore_point`]: cluster restore points. //! - [`sync_and_routing`]: Lite sync-producer register/fence + live //! routing-table `SetPlacement`. //! - [`catalog_ddl`]: `CatalogDdl` / `CatalogDdlAudited` decode + apply. //! - [`pending_ddl`]: `DdlPendingPropose` / `DdlPendingFinalize` / //! `DdlPendingCancel` apply. +//! - [`host_state`]: durable leases, cluster version, and DDL preparation +//! owner. +//! - [`boot_seed`]: loads the persisted host state before the Raft loop +//! ticks. //! - [`dispatch`]: the recursive `apply_host_side_effects` entry point //! and `impl MetadataApplier for MetadataCommitApplier`. //! - [`audit`]: audit and CA-trust helpers (kept as its own file; used //! by [`catalog_ddl`] and [`lease_events`]). +//! - [`audit_describe`]: the name, version, and HLC an audit record reports +//! for a catalog entry. //! - [`wedge`]: transient-vs-permanent classification of an apply failure //! and the readiness marker a permanent one leaves behind. mod audit; +mod audit_describe; +mod boot_seed; mod catalog_ddl; +mod database_id; mod dispatch; +mod host_state; mod lease_events; +mod membership_effects; mod pending_ddl; +mod restore_point; mod surrogate; mod sync_and_routing; +#[cfg(test)] +mod test_fixture; mod types; mod wedge; +pub use boot_seed::{seed_host_tables, seed_metadata_cache}; +pub use catalog_ddl::backup_mark_fail_point; +pub use dispatch::metadata_apply_hold_point; pub use types::{CATALOG_CHANNEL_CAPACITY, CatalogChangeEvent, MetadataCommitApplier}; pub use wedge::{ApplyFailureClass, MetadataApplyWedge, WedgeReport, classify}; diff --git a/nodedb/src/control/cluster/metadata_applier/pending_ddl.rs b/nodedb/src/control/cluster/metadata_applier/pending_ddl.rs index 3c25fea9c..c9a415c0d 100644 --- a/nodedb/src/control/cluster/metadata_applier/pending_ddl.rs +++ b/nodedb/src/control/cluster/metadata_applier/pending_ddl.rs @@ -4,8 +4,9 @@ //! `DdlPendingCancel`. //! //! Applies the entries `ddl_flush::begin_commit` / `finalize_pending` propose -//! at COMMIT, and the ones `metadata_proposer::acquire_ddl_prepare_lease` -//! proposes to reclaim a dead owner's stranded record. Finalize and cancel +//! at COMMIT, and the cancel `metadata_proposer::ddl_reclaim` proposes for a +//! reclaimed owner's stranded record. Propose and finalize apply only while +//! their token owns the DDL preparation lease. Finalize and cancel //! are idempotent: applying either twice, or applying either for a token //! with no pending record, is a no-op. Raft replay relies on exactly that //! shape. @@ -16,6 +17,7 @@ use nodedb_cluster::{MetadataEntry, PendingDdlObject}; use nodedb_types::Hlc; use crate::control::catalog_entry; +use crate::control::security::catalog::StoredPendingReclaim; use super::types::MetadataCommitApplier; @@ -23,18 +25,27 @@ impl MetadataCommitApplier { /// `DdlPendingPropose`: insert the pending record. Re-delivery of the /// same propose overwrites with an identical record, so no ordering /// hazard exists. + /// + /// Applies only while `token` owns the preparation lease. An owner whose + /// lease the metadata leader reclaimed reserves nothing: its propose is a + /// deterministic no-op on every replica, and the proposer sees no record. pub(super) fn apply_ddl_pending_propose( &self, token: u64, objects: &[PendingDdlObject], proposed_at: Hlc, ) -> Result<(), crate::Error> { - let Some(shared) = self.shared.get().and_then(std::sync::Weak::upgrade) else { + let shared = self.shared_state()?; + if !crate::control::metadata_proposer::ddl_owner::owns_ddl_lease(&shared, token) { + debug!( + token, + "pending DDL propose: token does not own the lease, no-op" + ); return Ok(()); - }; + } // `proposed_at` is the only remote HLC observation the metadata group // carries — every other `Hlc` on a `MetadataEntry` is a future - // deadline, and folding one would jump this node's clock forward. + // deadline, and folding one will jump this node's clock forward. // // The entry is already committed, so a refused fold must not stop the // apply: refusing to move the clock IS the protection. Applying still @@ -48,6 +59,13 @@ impl MetadataCommitApplier { "refusing to fold a proposer's HLC: {skew}" ); } + self.credentials.catalog().put_pending_ddl( + &crate::control::security::catalog::StoredPendingDdl { + token, + objects: objects.to_vec(), + proposed_at, + }, + )?; shared .pending_ddl .insert(token, objects.to_vec(), proposed_at); @@ -58,22 +76,36 @@ impl MetadataCommitApplier { /// effects, then drop the pending record. The record is peeked rather /// than removed up front, so a mid-replay failure leaves it in place /// for the next re-delivery instead of silently skipping the rest. - pub(super) fn apply_ddl_pending_finalize( + /// + /// Applies only while `token` owns the preparation lease. A finalize + /// that applies records `token` in `metadata_ddl_applied_token`, which + /// is how its proposer learns the objects landed. + pub(super) async fn apply_ddl_pending_finalize( &self, token: u64, raft_index: u64, ) -> Result<(), crate::Error> { - let Some(shared) = self.shared.get().and_then(std::sync::Weak::upgrade) else { - return Ok(()); - }; + let shared = self.shared_state()?; let Some(record) = shared.pending_ddl.get(token) else { debug!(token, "pending DDL finalize: no pending record, no-op"); return Ok(()); }; + if !crate::control::metadata_proposer::ddl_owner::owns_ddl_lease(&shared, token) { + debug!( + token, + "pending DDL finalize: token does not own the lease, no-op" + ); + return Ok(()); + } for object in &record.objects { - self.apply_host_side_effects(object_entry(object), raft_index)?; + self.apply_host_side_effects(object_entry(object), raft_index) + .await?; } + self.credentials.catalog().remove_pending_ddl(token)?; shared.pending_ddl.take(token); + shared + .metadata_ddl_applied_token + .store(token, std::sync::atomic::Ordering::Release); Ok(()) } @@ -82,39 +114,66 @@ impl MetadataCommitApplier { /// record. A collection's engine is registered eagerly at CREATE /// statement time, independent of buffering, so an abandoned create /// still needs the same `UnregisterCollection` teardown a real purge - /// uses. The dispatch is spawned rather than awaited inline — apply - /// runs on the raft loop task, and blocking here would deadlock the - /// applied-index watcher (same reasoning as the `TopologyChange::Leave` - /// lease-GC spawn in `dispatch.rs`). + /// uses. A name with any committed row keeps its engine. + /// + /// Each teardown is queued as a durable `_system.pending_reclaim` row + /// before the record is dropped, so a crash before the teardown finishes + /// leaves it to the boot drain. The dispatch is spawned rather than + /// awaited inline — apply runs on the raft loop task, and blocking here + /// will deadlock the applied-index watcher (same reasoning as the + /// `TopologyChange::Leave` lease-GC spawn in `dispatch.rs`). pub(super) fn apply_ddl_pending_cancel(&self, token: u64) -> Result<(), crate::Error> { - let Some(shared) = self.shared.get().and_then(std::sync::Weak::upgrade) else { - return Ok(()); - }; - let Some(record) = shared.pending_ddl.take(token) else { + let shared = self.shared_state()?; + let Some(record) = shared.pending_ddl.get(token) else { debug!(token, "pending DDL cancel: no pending record, no-op"); return Ok(()); }; + // Decide every teardown before dropping the record, so a failed catalog + // read leaves the record for the re-delivered cancel. + let catalog = self.credentials.catalog(); + let mut teardown = Vec::new(); for object in &record.objects { let PendingDdlObject::Create { entry } = object else { continue; }; - let Some((database_id, tenant_id, name)) = created_collection_target(entry.as_ref()) - else { + let Some(target) = created_collection_target(entry.as_ref()) else { continue; }; + if cancel_owns_engine(&target, catalog)? { + teardown.push(target); + } else { + debug!( + collection = %target.name, + tenant = target.tenant_id, + "pending DDL cancel: a later incarnation holds the name, teardown skipped" + ); + } + } + let queued = queue_teardowns( + catalog, + teardown, + shared.wal.next_lsn().as_u64(), + crate::control::lease::wall_now_ns(), + )?; + // A same-name CREATE waits on the pending-reclaim path's hold until + // the teardown finishes. A re-delivered cancel finds it already held. + for entry in &queued { + shared.quiesce.ensure_reclaim_hold(&entry.owner()); + } + catalog.remove_pending_ddl(token)?; + shared.pending_ddl.take(token); + for entry in queued { let shared = std::sync::Arc::clone(&shared); tokio::spawn(async move { - let purge_lsn = shared.wal.next_lsn().as_u64(); - if let Err(error) = crate::control::server::shared::ddl::neutral::collection::purge::dispatch_unregister_collection( - &shared, database_id, tenant_id, &name, purge_lsn, - ) - .await + if let Err(error) = + crate::event::collection_gc::pending_reclaim::retry_one(&shared, &entry).await { tracing::warn!( - collection = %name, - tenant = tenant_id, + collection = %entry.name, + tenant = entry.tenant_id, error = %error, - "pending DDL cancel: Data Plane teardown failed" + "pending DDL cancel: Data Plane teardown failed; the pending-reclaim \ + worker retries it" ); } }); @@ -123,6 +182,64 @@ impl MetadataCommitApplier { } } +/// A collection a pending create registered. +struct CreatedCollection { + database_id: u64, + tenant_id: u64, + name: String, + /// The create's own clock, the incarnation the teardown reclaims. + hlc: Hlc, +} + +/// Record one durable pending-reclaim row per teardown. The boot drain and the +/// pending-reclaim worker finish any teardown these rows still name. +fn queue_teardowns( + catalog: &crate::control::security::catalog::SystemCatalog, + teardown: Vec, + purge_lsn: u64, + enqueued_at_ns: u64, +) -> crate::Result> { + let mut queued = Vec::with_capacity(teardown.len()); + for CreatedCollection { + database_id, + tenant_id, + name, + hlc, + } in teardown + { + let entry = StoredPendingReclaim { + database_id, + tenant_id, + name, + purge_lsn, + enqueued_at_ns, + last_error: String::new(), + attempts: 0, + target_hlc: Some(hlc), + cancelled_create: true, + }; + catalog.enqueue_pending_reclaim(&entry)?; + queued.push(entry); + } + Ok(queued) +} + +/// Whether the engine registered under `target`'s name still belongs to the +/// cancelled create. A cancelled create never commits its row, so any +/// committed row under the name belongs to a finalized create or a later +/// incarnation, and its engine stays. +fn cancel_owns_engine( + target: &CreatedCollection, + catalog: &crate::control::security::catalog::SystemCatalog, +) -> crate::Result { + let row = catalog.get_committed_collection( + crate::types::DatabaseId::new(target.database_id), + target.tenant_id, + &target.name, + )?; + Ok(row.is_none()) +} + /// The `MetadataEntry` wrapped by a pending object, regardless of shape. fn object_entry(object: &PendingDdlObject) -> &MetadataEntry { match object { @@ -134,7 +251,7 @@ fn object_entry(object: &PendingDdlObject) -> &MetadataEntry { /// `(database_id, tenant_id, name)` when `entry` is a collection create — /// the only shape that registers a Data Plane engine eagerly at DDL time. -fn created_collection_target(entry: &MetadataEntry) -> Option<(u64, u64, String)> { +fn created_collection_target(entry: &MetadataEntry) -> Option { let payload = match entry { MetadataEntry::CatalogDdl { payload } | MetadataEntry::CatalogDdlAudited { payload, .. } => payload, @@ -142,9 +259,93 @@ fn created_collection_target(entry: &MetadataEntry) -> Option<(u64, u64, String) }; match catalog_entry::decode(payload).ok()? { catalog_entry::CatalogEntry::PutCollection(stored) - | catalog_entry::CatalogEntry::PutCollectionIfAbsent(stored) => { - Some((stored.database_id.as_u64(), stored.tenant_id, stored.name)) - } + | catalog_entry::CatalogEntry::PutCollectionIfAbsent(stored) => Some(CreatedCollection { + database_id: stored.database_id.as_u64(), + tenant_id: stored.tenant_id, + hlc: stored.modification_hlc, + name: stored.name, + }), _ => None, } } + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_types::DatabaseId; + + use super::*; + use crate::control::security::catalog::StoredCollection; + use crate::control::security::credential::CredentialStore; + + fn target() -> CreatedCollection { + CreatedCollection { + database_id: DatabaseId::DEFAULT.as_u64(), + tenant_id: 1, + name: "orders".to_string(), + hlc: Hlc::new(10, 0), + } + } + + fn open() -> (Arc, tempfile::TempDir) { + let tmp = tempfile::tempdir().expect("tmpdir"); + let store = Arc::new(CredentialStore::open(&tmp.path().join("system.redb")).expect("open")); + (store, tmp) + } + + fn seed(store: &CredentialStore, hlc: Hlc) { + let mut row = StoredCollection::stamped_for_test(1, "orders", "tester"); + row.modification_hlc = hlc; + store + .catalog() + .put_collection(DatabaseId::DEFAULT, &row) + .expect("seed collection"); + } + + /// A cancel replayed after the same name was created for real must leave + /// the later collection's engine registered. + #[test] + fn replayed_cancel_spares_a_later_same_name_collection() { + let (store, _tmp) = open(); + seed(&store, Hlc::new(30, 0)); + assert!(!cancel_owns_engine(&target(), store.catalog()).expect("read")); + } + + #[test] + fn cancel_tears_down_when_no_row_holds_the_name() { + let (store, _tmp) = open(); + assert!(cancel_owns_engine(&target(), store.catalog()).expect("read")); + } + + /// The cancel records the teardown durably before it spawns the dispatch, + /// so the boot drain finishes a teardown a crash interrupted. + #[test] + fn cancel_queues_a_durable_reclaim_for_each_teardown() { + let (store, _tmp) = open(); + let queued = + queue_teardowns(store.catalog(), vec![target()], 42, 7).expect("queue teardown"); + assert_eq!(queued.len(), 1); + + let rows = store + .catalog() + .load_pending_reclaim_queue() + .expect("load queue"); + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].database_id, DatabaseId::DEFAULT.as_u64()); + assert_eq!(rows[0].tenant_id, 1); + assert_eq!(rows[0].name, "orders"); + assert_eq!(rows[0].purge_lsn, 42); + assert_eq!(rows[0].target_hlc, Some(Hlc::new(10, 0))); + assert!(rows[0].cancelled_create); + } + + /// A committed row at the create's own clock means the create was + /// finalized. Its engine is live. + #[test] + fn cancel_spares_a_finalized_create() { + let (store, _tmp) = open(); + seed(&store, Hlc::new(10, 0)); + assert!(!cancel_owns_engine(&target(), store.catalog()).expect("read")); + } +} diff --git a/nodedb/src/control/cluster/metadata_applier/restore_point.rs b/nodedb/src/control/cluster/metadata_applier/restore_point.rs new file mode 100644 index 000000000..e0f8693a5 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_applier/restore_point.rs @@ -0,0 +1,69 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `RestorePoint` host-side effect, and the clock step every stamped entry +//! takes. + +use nodedb_cluster::METADATA_GROUP_ID; +use nodedb_types::Hlc; +use nodedb_wal::record::RestorePointPayload; + +use super::types::MetadataCommitApplier; +use crate::control::pitr::restore_point::{record_group_point, spawn_node_cut}; +use crate::control::security::catalog::restore_points::StoredRestorePoint; + +impl MetadataCommitApplier { + /// Move this node's clock past the leader's stamp of the entry `data`, + /// and record the stamp as the applied high-water. + /// + /// A write that follows the entry's effect on this node then carries a + /// later HLC, so a restore that drops the entry drops the write too. A + /// leader's next stamp lands above it, and the high-water carries that + /// across a restart and a snapshot install. + pub(super) fn observe_stamp(&self, data: &[u8]) -> crate::Result<()> { + let Some(stamp) = nodedb_cluster::entry_stamp(data) else { + return Ok(()); + }; + self.credentials.catalog().raise_metadata_stamp_hwm(stamp)?; + if let Ok(shared) = self.shared_state() { + shared.hlc_clock.update(Hlc::new(stamp, 0)); + } + Ok(()) + } + + /// Record the restore point created at `raft_index`, record the metadata + /// group's place at it in this node's WAL, then cut every group this node + /// hosts at `hlc`. + /// + /// The catalog row is the durable effect: a persist error returns `Err`, + /// so the watermark stays on this entry and Raft re-delivers it. + pub(super) fn apply_restore_point( + &self, + hlc: u64, + created_at_ms: u64, + raft_index: u64, + ) -> Result<(), crate::Error> { + let shared = self.shared_state()?; + self.credentials + .catalog() + .put_restore_point(&StoredRestorePoint { + id: raft_index, + hlc, + created_at_ms, + })?; + record_group_point( + &shared, + RestorePointPayload { + id: raft_index, + hlc, + group_id: METADATA_GROUP_ID, + applied_index: raft_index, + term: 0, + next_epoch: 0, + epoch_system_ms: 0, + vshards: Vec::new(), + }, + ); + spawn_node_cut(shared, raft_index, hlc); + Ok(()) + } +} diff --git a/nodedb/src/control/cluster/metadata_applier/surrogate.rs b/nodedb/src/control/cluster/metadata_applier/surrogate.rs index ac499c81f..6f90fb276 100644 --- a/nodedb/src/control/cluster/metadata_applier/surrogate.rs +++ b/nodedb/src/control/cluster/metadata_applier/surrogate.rs @@ -14,16 +14,14 @@ impl MetadataCommitApplier { /// node. `restore_hwm` is idempotent and monotonic: calling /// it with a value at or below the current HWM is a no-op, /// so duplicate or reordered delivery cannot push the - /// counter backwards. Also persist the hwm to the catalog so - /// the local node survives a restart without re-reading the - /// full log. + /// counter backwards. The hwm is persisted before the entry counts as + /// applied, so boot never depends on re-reading the log. pub(super) fn apply_surrogate_alloc( &self, hwm: u32, raft_index: u64, ) -> Result<(), crate::Error> { - if let Some(weak) = self.shared.get() - && let Some(shared) = weak.upgrade() + let shared = self.shared_state()?; { let reg = shared .surrogate_assigner @@ -31,9 +29,13 @@ impl MetadataCommitApplier { .read() .unwrap_or_else(|p| p.into_inner()); let restored = reg.restore_hwm(hwm); + // The raised watermark, never below what this node already + // issued: persisting the entry's `hwm` can lower the catalog + // copy and reissue surrogates after a restart. + let current = reg.current_hwm(); drop(reg); // The in-memory HWM advance is correctness-critical: if it - // fails this replica could re-issue a surrogate the cluster + // fails this replica can re-issue a surrogate the cluster // already allocated. Do not advance past this entry — retry. if let Err(e) = restored { warn!(hwm, error = %e, "surrogate_alloc apply: restore_hwm failed — halting watermark for retry"); @@ -41,18 +43,9 @@ impl MetadataCommitApplier { detail: format!("surrogate_alloc apply: restore_hwm failed: {e}"), }); } - // Best-effort catalog persist: a failure means the - // next restart will re-derive the HWM from the log - // (the log is the source of truth), which is correct — - // just slightly slower. Tolerate and continue. - let catalog = self.credentials.catalog(); - if let Err(e) = catalog.put_surrogate_hwm(hwm) { - warn!( - hwm, - error = %e, - "surrogate_alloc apply: failed to persist hwm to catalog (tolerable; log is authoritative)" - ); - } + // A failed persist returns `Err`: the entry is re-delivered, and + // `restore_hwm` is a no-op on the retry, which persists again. + self.credentials.catalog().put_surrogate_hwm(current)?; debug!(hwm, raft_index, "surrogate hwm advanced via raft"); } Ok(()) @@ -86,7 +79,7 @@ impl MetadataCommitApplier { /// 2. The reserved batch must NOT be installed during replay. /// A node that crashed mid-batch already consumed part of /// its pre-crash `[start, end)`; re-installing it on replay - /// would hand those surrogates out AGAIN. So `G` advances + /// will hand those surrogates out AGAIN. So `G` advances /// (deterministic, every node) but the batch install is /// gated on a LIVE pending waiter, which only exists during /// a genuine in-process reservation (`pending_reservations` @@ -105,12 +98,11 @@ impl MetadataCommitApplier { batch_size: u32, raft_index: u64, ) -> Result<(), crate::Error> { - if let Some(weak) = self.shared.get() - && let Some(shared) = weak.upgrade() + let shared = self.shared_state()?; { // Read guard is sufficient: `reserve_at_index` mutates via // interior atomics (counter + last_reserve_index). Taking a - // write guard here would risk deadlocking the allocation + // write guard here will risk deadlocking the allocation // path, which holds no registry lock across the propose+wait // but does re-take it to retry. let reg = shared @@ -121,7 +113,7 @@ impl MetadataCommitApplier { // Advancing the global watermark is correctness-critical and // must be deterministic across nodes incl. replay: an // exhaustion error must NOT advance the apply watermark past - // this entry, or replicas would diverge on `G`. Surface it + // this entry, or replicas will diverge on `G`. Surface it // so Raft re-delivers. This apply path only runs when // `start_raft` is active, which only happens when // `config.cluster.is_some()` — the same condition that puts @@ -166,14 +158,35 @@ impl MetadataCommitApplier { }); } }; + let current_hwm = reg.current_hwm(); drop(reg); + let catalog = self.credentials.catalog(); let Some((start, end)) = reserved else { - // Already applied (full-log replay / duplicate - // delivery): do NOT advance `G`, do NOT persist, do NOT - // install a batch. Advancing the apply watermark past a - // replayed entry is correct — its effect is already in - // the seeded state. + // Already applied in memory (replay, duplicate delivery, or + // a re-delivery after a failed persist): do NOT advance `G`. + // When the persisted cursor is behind this entry, the earlier + // persist failed, so this delivery writes it. The applier + // stops at a failed entry, so the in-memory `G` is exactly + // this entry's carve. + if catalog.get_surrogate_reserve_index()? < raft_index { + catalog.put_surrogate_reserve_state(current_hwm, raft_index)?; + } + // The carve kept from the failed first application goes to + // the allocator still waiting on it. After a restart no carve + // is kept and no allocator waits. + let kept = self + .unpersisted_carve + .lock() + .unwrap_or_else(|p| p.into_inner()) + .take_if(|(index, _, _)| *index == raft_index); + if let Some((_, start, end)) = kept + && node_id == shared.node_id + { + shared + .surrogate_assigner + .complete_reservation(request_id, start, end); + } debug!( node_id, request_id, @@ -185,25 +198,25 @@ impl MetadataCommitApplier { // First application: persist `(hwm = end - 1, cursor = // raft_index)` ATOMICALLY so a restart can skip this - // reservation on replay (no double-count) and seed an - // already-equal `G` on every node. Best-effort (warn on - // fail) like the `SurrogateAlloc` arm: if the persist fails, - // the log is still authoritative and the next restart - // re-derives `G` by replaying from the last durable cursor — - // correct, just slightly slower. The hwm and cursor are - // written together in one redb txn, so a crash can never - // leave them inconsistent. - let catalog = self.credentials.catalog(); + // reservation (no double-count) and seed an already-equal `G` + // on every node. A failed persist returns `Err`: the entry is + // re-delivered, and the branch above persists it and hands the + // kept carve to the waiting allocator. if let Err(e) = catalog.put_surrogate_reserve_state(end - 1, raft_index) { + *self + .unpersisted_carve + .lock() + .unwrap_or_else(|p| p.into_inner()) = Some((raft_index, start, end)); warn!( node_id, request_id, hwm = end - 1, raft_index, error = %e, - "surrogate_reserve apply: failed to persist reserve state to catalog \ - (tolerable; log is authoritative)" + "surrogate_reserve apply: failed to persist reserve state; the entry is \ + re-delivered" ); + return Err(e); } if node_id == shared.node_id { @@ -227,3 +240,57 @@ impl MetadataCommitApplier { Ok(()) } } + +#[cfg(test)] +mod tests { + use nodedb_cluster::{MetadataApplier, MetadataEntry, encode_entry}; + + use super::super::test_fixture::{applier_with_shared_at, cluster_applier_with_shared_at}; + + /// A failed hwm persist stops the entry. The re-delivered entry persists + /// and applies. + #[tokio::test(flavor = "multi_thread")] + async fn surrogate_hwm_persist_error_stops_the_entry() { + let dir = tempfile::tempdir().unwrap(); + let (applier, state) = applier_with_shared_at(dir.path(), "test.wal"); + let entry = encode_entry(&MetadataEntry::SurrogateAlloc { hwm: 500 }).unwrap(); + let catalog = state.credentials.catalog(); + + catalog.fail_next_surrogate_write_for_test(); + assert_eq!(applier.apply(&[(3, entry.clone())]).await, 0); + + assert_eq!(applier.apply(&[(3, entry)]).await, 3); + assert!(catalog.get_surrogate_hwm().unwrap() >= 500); + } + + /// A reserve whose persist failed keeps its carve. The re-delivered entry + /// persists it and hands that exact batch to the allocator waiting on it. + #[tokio::test(flavor = "multi_thread")] + async fn reserve_redelivery_hands_the_kept_carve_to_the_waiter() { + let dir = tempfile::tempdir().unwrap(); + let (applier, state) = cluster_applier_with_shared_at(dir.path(), "test.wal"); + let catalog = state.credentials.catalog(); + let mut waiter = state.surrogate_assigner.await_reservation_for_test(77); + let entry = encode_entry(&MetadataEntry::SurrogateReserve { + node_id: state.node_id, + request_id: 77, + batch_size: 16, + }) + .unwrap(); + + catalog.fail_next_surrogate_write_for_test(); + assert_eq!(applier.apply(&[(5, entry.clone())]).await, 0); + assert!( + waiter.try_recv().is_err(), + "no batch before the carve is durable" + ); + + assert_eq!(applier.apply(&[(5, entry)]).await, 5); + let (start, end) = waiter + .try_recv() + .expect("the kept carve reaches the waiter"); + assert_eq!(end - start, 16); + assert_eq!(catalog.get_surrogate_reserve_index().unwrap(), 5); + assert_eq!(catalog.get_surrogate_hwm().unwrap(), end - 1); + } +} diff --git a/nodedb/src/control/cluster/metadata_applier/sync_and_routing.rs b/nodedb/src/control/cluster/metadata_applier/sync_and_routing.rs index b412aa6c1..0e0140ddb 100644 --- a/nodedb/src/control/cluster/metadata_applier/sync_and_routing.rs +++ b/nodedb/src/control/cluster/metadata_applier/sync_and_routing.rs @@ -6,6 +6,19 @@ use tracing::{debug, warn}; use super::types::MetadataCommitApplier; +use crate::control::state::SharedState; +use crate::control::sync_producer::registry::SyncProducerRegistry; + +/// The durable producer registry. Production boot always opens it, so a +/// missing one is transient: the entry is re-delivered. +fn producer_registry(shared: &SharedState) -> Result<&SyncProducerRegistry, crate::Error> { + shared + .producer_registry + .as_deref() + .ok_or_else(|| crate::Error::Internal { + detail: "sync producer registry is not open; the entry is re-delivered".into(), + }) +} pub(super) struct SyncPeerBindApply<'a> { pub(super) database_id: u64, @@ -39,30 +52,25 @@ impl MetadataCommitApplier { epoch, created_ms, } = registration; - if let Some(weak) = self.shared.get() - && let Some(shared) = weak.upgrade() + let shared = self.shared_state()?; + let registry = producer_registry(&shared)?; + // The registration row is durable replicated state. A write + // failure must not advance the watermark — Raft re-delivers + // and `apply_register` is idempotent, so the retry is safe. + if let Err(e) = + registry.apply_register(lite_id, producer_id, tenant_id, user_id, epoch, created_ms) { - let Some(registry) = shared.producer_registry.as_deref() else { - return Ok(()); - }; - // The registration row is durable replicated state. A write - // failure must not advance the watermark — Raft re-delivers - // and `apply_register` is idempotent, so the retry is safe. - if let Err(e) = - registry.apply_register(lite_id, producer_id, tenant_id, user_id, epoch, created_ms) - { - warn!( - lite_id = %lite_id, - producer_id, - error = %e, - "sync_producer_register apply failed — halting watermark for retry" - ); - return Err(crate::Error::Internal { - detail: format!("sync_producer_register apply failed: {e}"), - }); - } - debug!(lite_id = %lite_id, producer_id, raft_index, "sync producer registered via raft"); + warn!( + lite_id = %lite_id, + producer_id, + error = %e, + "sync_producer_register apply failed — halting watermark for retry" + ); + return Err(crate::Error::Internal { + detail: format!("sync_producer_register apply failed: {e}"), + }); } + debug!(lite_id = %lite_id, producer_id, raft_index, "sync producer registered via raft"); Ok(()) } @@ -72,27 +80,22 @@ impl MetadataCommitApplier { new_epoch: u64, raft_index: u64, ) -> Result<(), crate::Error> { - if let Some(weak) = self.shared.get() - && let Some(shared) = weak.upgrade() - { - let Some(registry) = shared.producer_registry.as_deref() else { - return Ok(()); - }; - // Durable epoch advance; `apply_fence` is idempotent - // (max-wins) so re-delivery on failure is safe. - if let Err(e) = registry.apply_fence(lite_id, new_epoch) { - warn!( - lite_id = %lite_id, - new_epoch, - error = %e, - "sync_producer_fence apply failed — halting watermark for retry" - ); - return Err(crate::Error::Internal { - detail: format!("sync_producer_fence apply failed: {e}"), - }); - } - debug!(lite_id = %lite_id, new_epoch, raft_index, "sync producer fenced via raft"); + let shared = self.shared_state()?; + let registry = producer_registry(&shared)?; + // Durable epoch advance; `apply_fence` is idempotent + // (max-wins) so re-delivery on failure is safe. + if let Err(e) = registry.apply_fence(lite_id, new_epoch) { + warn!( + lite_id = %lite_id, + new_epoch, + error = %e, + "sync_producer_fence apply failed — halting watermark for retry" + ); + return Err(crate::Error::Internal { + detail: format!("sync_producer_fence apply failed: {e}"), + }); } + debug!(lite_id = %lite_id, new_epoch, raft_index, "sync producer fenced via raft"); Ok(()) } @@ -109,40 +112,35 @@ impl MetadataCommitApplier { producer_id, bound_ms, } = binding; - if let Some(weak) = self.shared.get() - && let Some(shared) = weak.upgrade() - { - let Some(registry) = shared.producer_registry.as_deref() else { - return Ok(()); - }; - let key = crate::control::security::catalog::sync_producer::PeerBindingKey::new( - database_id, - tenant_id, - collection, - peer_id, - ); - // Lowest-producer-id-wins, so re-delivery and reordering both - // converge; a write failure must not advance the watermark. - if let Err(e) = registry.apply_bind_peer(&key, producer_id, bound_ms) { - warn!( - collection = %collection, - peer_id, - producer_id, - error = %e, - "sync_peer_bind apply failed — halting watermark for retry" - ); - return Err(crate::Error::Internal { - detail: format!("sync_peer_bind apply failed: {e}"), - }); - } - debug!( + let shared = self.shared_state()?; + let registry = producer_registry(&shared)?; + let key = crate::control::security::catalog::sync_producer::PeerBindingKey::new( + database_id, + tenant_id, + collection, + peer_id, + ); + // Lowest-producer-id-wins, so re-delivery and reordering both + // converge; a write failure must not advance the watermark. + if let Err(e) = registry.apply_bind_peer(&key, producer_id, bound_ms) { + warn!( collection = %collection, peer_id, producer_id, - raft_index, - "loro peer id bound to producer via raft" + error = %e, + "sync_peer_bind apply failed — halting watermark for retry" ); + return Err(crate::Error::Internal { + detail: format!("sync_peer_bind apply failed: {e}"), + }); } + debug!( + collection = %collection, + peer_id, + producer_id, + raft_index, + "loro peer id bound to producer via raft" + ); Ok(()) } @@ -154,18 +152,20 @@ impl MetadataCommitApplier { /// same `RwLock` the reconciler and the /// learner-promotion gate read. Without this write the placement /// never leaves the metadata log and N>RF voter-cap convergence - /// is inert. The other `RoutingChange` variants are intentionally - /// not handled here to avoid double-applying the conf-change path. + /// is inert. `ReassignVShard` has no conf-change equivalent either, and + /// is written through by `apply_reassign_vshard`. The other + /// `RoutingChange` variants are not handled here, so the conf-change path + /// applies them once. pub(super) fn apply_set_placement( &self, group_id: u64, placement: &[u64], raft_index: u64, ) -> Result<(), crate::Error> { - if let Some(weak) = self.shared.get() - && let Some(shared) = weak.upgrade() - && let Some(routing) = shared.cluster_routing.as_ref() - { + let shared = self.shared_state()?; + // A node without a live routing table (single-node origin) has no + // placement to update: the placement lives in the metadata cache. + if let Some(routing) = shared.cluster_routing.as_ref() { routing .write() .unwrap_or_else(|p| p.into_inner()) @@ -177,4 +177,29 @@ impl MetadataCommitApplier { } Ok(()) } + /// Write a committed vShard reassignment through to the live routing + /// table. `raft_index` becomes the vShard's epoch in the new group, so + /// change-data-capture positions of the vShard keep rising across the + /// move on every node. + pub(super) fn apply_reassign_vshard( + &self, + vshard_id: u32, + new_group_id: u64, + new_leaseholder_node_id: u64, + raft_index: u64, + ) -> Result<(), crate::Error> { + let shared = self.shared_state()?; + if let Some(routing) = shared.cluster_routing.as_ref() { + let mut routing = routing.write().unwrap_or_else(|p| p.into_inner()); + routing.reassign_vshard(vshard_id, new_group_id, raft_index); + // The entry names a planned leaseholder with no term. It fills + // only a hint that holds no term. + routing.set_leader(new_group_id, new_leaseholder_node_id); + debug!( + vshard_id, + new_group_id, raft_index, "vShard reassignment applied to live routing table" + ); + } + Ok(()) + } } diff --git a/nodedb/src/control/cluster/metadata_applier/test_fixture.rs b/nodedb/src/control/cluster/metadata_applier/test_fixture.rs new file mode 100644 index 000000000..af5c8bfe3 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_applier/test_fixture.rs @@ -0,0 +1,55 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Test fixture: an applier wired to a real `SharedState` over an on-disk +//! catalog. + +use std::path::Path; +use std::sync::{Arc, Mutex}; + +use tokio::sync::broadcast; + +use crate::bridge::dispatch::Dispatcher; +use crate::control::security::credential::CredentialStore; +use crate::control::state::SharedState; +use crate::wal::WalManager; + +use super::MetadataCommitApplier; + +/// One node session over the catalog `dir/system.redb`. `wal` names the WAL +/// file, so a second session over the same catalog can use a fresh WAL. +pub(super) fn applier_with_shared_at( + dir: &Path, + wal: &str, +) -> (MetadataCommitApplier, Arc) { + applier_over(dir, wal, false) +} + +/// [`applier_with_shared_at`] with the surrogate registry in cluster mode. +pub(super) fn cluster_applier_with_shared_at( + dir: &Path, + wal: &str, +) -> (MetadataCommitApplier, Arc) { + applier_over(dir, wal, true) +} + +fn applier_over( + dir: &Path, + wal: &str, + is_cluster: bool, +) -> (MetadataCommitApplier, Arc) { + let wal = Arc::new(WalManager::open_for_testing(&dir.join(wal)).unwrap()); + let credentials = Arc::new(CredentialStore::open(&dir.join("system.redb")).unwrap()); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let state = + SharedState::new_with_credentials(dispatcher, wal, credentials, is_cluster).unwrap(); + let (tx, _rx) = broadcast::channel(16); + let token_state = Arc::new(Mutex::new(std::collections::HashMap::new())); + let applier = MetadataCommitApplier::new( + state.metadata_cache.clone(), + tx, + state.credentials.clone(), + token_state, + ); + applier.install_shared(Arc::downgrade(&state)); + (applier, state) +} diff --git a/nodedb/src/control/cluster/metadata_applier/types.rs b/nodedb/src/control/cluster/metadata_applier/types.rs index 0c856901b..cf2d352c1 100644 --- a/nodedb/src/control/cluster/metadata_applier/types.rs +++ b/nodedb/src/control/cluster/metadata_applier/types.rs @@ -3,7 +3,7 @@ //! `MetadataCommitApplier` struct definition, construction, and the //! `CatalogChangeEvent` it broadcasts. -use std::sync::{Arc, OnceLock, RwLock, Weak}; +use std::sync::{Arc, Mutex, OnceLock, RwLock, Weak}; use tokio::sync::broadcast; @@ -36,6 +36,10 @@ pub struct MetadataCommitApplier { /// break the Arc cycle (SharedState → raft loop → applier → /// SharedState). `None` in unit tests. pub(super) shared: OnceLock>, + /// `(raft_index, start, end)` of a surrogate carve whose persist failed. + /// The re-delivered entry persists it and hands this exact batch to the + /// allocator still waiting on it. + pub(super) unpersisted_carve: Mutex>, } impl MetadataCommitApplier { @@ -52,6 +56,7 @@ impl MetadataCommitApplier { token_state, transport: OnceLock::new(), shared: OnceLock::new(), + unpersisted_carve: Mutex::new(None), } } @@ -66,4 +71,20 @@ impl MetadataCommitApplier { pub fn install_shared(&self, shared: Weak) { let _ = self.shared.set(shared); } + + /// The installed `SharedState`. + /// + /// Missing means the applier runs before `start_raft` installed it, or + /// after shutdown dropped it. The entry's host effects cannot land, so + /// this is a transient error: the entry is re-delivered. + pub(super) fn shared_state(&self) -> Result, crate::Error> { + self.shared + .get() + .and_then(Weak::upgrade) + .ok_or_else(|| crate::Error::Internal { + detail: "metadata applier has no SharedState (not installed yet, or the node \ + shut down); the entry is re-delivered" + .into(), + }) + } } diff --git a/nodedb/src/control/cluster/metadata_applier/wedge.rs b/nodedb/src/control/cluster/metadata_applier/wedge.rs index f4e8c1c57..a457fa543 100644 --- a/nodedb/src/control/cluster/metadata_applier/wedge.rs +++ b/nodedb/src/control/cluster/metadata_applier/wedge.rs @@ -3,7 +3,7 @@ //! Classification of a failed metadata host-side apply, and the durable //! marker a permanent failure leaves behind. //! -//! The apply loop must never advance its watermark past an entry it could not +//! The apply loop must never advance its watermark past an entry it cannot //! apply — skipping a committed metadata entry is silent divergence from the //! quorum. So both a transient and a permanent failure stop the batch. What //! they must NOT share is the *story told to operators*: @@ -17,12 +17,12 @@ //! only symptom operators ever see is an unrelated-looking lease timeout on //! every subsequent query. -use std::sync::OnceLock; +use std::sync::Mutex; /// Whether a failed host-side apply can plausibly succeed on re-delivery. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum ApplyFailureClass { - /// May clear on its own; halt-and-retry is sufficient. + /// Can clear on its own; halt-and-retry is sufficient. Transient, /// Deterministic in the entry and local state — re-delivery re-fails. Permanent, @@ -39,7 +39,7 @@ impl ApplyFailureClass { /// Deliberately an allowlist of the variants that are *provably* a pure /// function of the entry plus local persisted state. Everything else is /// treated as transient, because the cost of the two mistakes is asymmetric: -/// calling a transient failure permanent takes a node that would have healed +/// calling a transient failure permanent takes a node that heals /// itself out of rotation, while calling a permanent failure transient only /// costs the loud health signal — the watermark halts either way. pub fn classify(error: &crate::Error) -> ApplyFailureClass { @@ -52,12 +52,15 @@ pub fn classify(error: &crate::Error) -> ApplyFailureClass { // applier code that ran them; re-delivery replays the same writes // and finds the same orphan every time. crate::Error::CatalogIntegrityViolation { .. } => ApplyFailureClass::Permanent, + // The committed entry carries the row it writes, so an unstamped row + // is unstamped on every re-delivery. + crate::Error::CollectionUnstamped { .. } => ApplyFailureClass::Permanent, // The bytes being encoded/decoded are fixed by the committed entry, so // a codec rejection is reproducible. crate::Error::Serialization { .. } | crate::Error::Codec { .. } => { ApplyFailureClass::Permanent } - // A committed entry that the host rejects as malformed will be just as + // A committed entry that the host rejects as malformed will be as // malformed next time. crate::Error::BadRequest { .. } | crate::Error::TypeMismatch { .. } => { ApplyFailureClass::Permanent @@ -99,9 +102,7 @@ pub fn classify(error: &crate::Error) -> ApplyFailureClass { | crate::Error::CrdtAdmissionTimeout { .. } | crate::Error::NoLeader { .. } | crate::Error::NotLeader { .. } - | crate::Error::FanOutExceeded { .. } | crate::Error::CrossCollectionNotColocated { .. } - | crate::Error::SourceFrozen { .. } | crate::Error::CloneWriteRequiresMaterialize { .. } | crate::Error::BackupTenantMismatch { .. } | crate::Error::BackupKeyMismatch @@ -119,10 +120,14 @@ pub fn classify(error: &crate::Error) -> ApplyFailureClass { | crate::Error::InvalidLimitValue { .. } | crate::Error::RetryableSchemaChanged { .. } | crate::Error::RetryableLeaderChange { .. } + | crate::Error::CommittedResultUnavailable { .. } + | crate::Error::ProposalOutcomeUnknown { .. } | crate::Error::GroupQuorumUnavailable { .. } | crate::Error::GroupMarksUnavailable { .. } + | crate::Error::BackupCaptureMoved { .. } | crate::Error::MetadataLeaderUnavailable | crate::Error::AuthorizationStateBehind { .. } + | crate::Error::LinearizableReadRefused { .. } | crate::Error::ExecutionLimitExceeded { .. } | crate::Error::LimitExceeded { .. } | crate::Error::Wal(_) @@ -139,6 +144,8 @@ pub fn classify(error: &crate::Error) -> ApplyFailureClass { | crate::Error::Encryption { .. } | crate::Error::Bridge { .. } | crate::Error::VersionCompat { .. } + | crate::Error::RestoreTargetNotEmpty { .. } + | crate::Error::RestoreVerificationFailed { .. } | crate::Error::Internal { .. } | crate::Error::Shaping(_) | crate::Error::Ddl(_) @@ -171,9 +178,9 @@ pub fn classify(error: &crate::Error) -> ApplyFailureClass { } /// What the applier recorded when it stopped on a permanent failure. -#[derive(Debug, Clone)] +#[derive(Debug, Clone, PartialEq, Eq)] pub struct WedgeReport { - /// Raft index of the entry that could not be applied. + /// Raft index of the entry that cannot be applied. pub raft_index: u64, /// Highest index whose state is guaranteed visible — one below the stall. pub last_applied_watermark: u64, @@ -183,34 +190,56 @@ pub struct WedgeReport { pub error: String, } -/// Node-wide marker set once when the metadata applier stops on a permanent -/// failure. Read by the readiness probe so a wedged node stops reporting -/// itself healthy. +/// Node-wide marker set when the metadata applier stops on a permanent +/// failure, or when a cut barrier's floor write keeps failing. Read by the +/// readiness probe so a wedged node stops reporting itself healthy. /// -/// First writer wins: the applier retries the same entry on every re-delivery -/// and would otherwise overwrite the original cause with an identical copy on -/// every tick. There is no clear path — the applier only resumes if the entry -/// applies, and if it applies the process has already made progress past the -/// point this marker describes, so operator intervention is required either -/// way. +/// First writer wins: a stalled apply retries the same step and will +/// otherwise overwrite the original cause with an identical copy on every +/// attempt. The metadata applier never clears its report: its entry cannot +/// apply on re-delivery, so operator intervention is required. A floor write +/// that later succeeds clears its own report through [`Self::clear`]. A clear +/// never removes a report another writer recorded. #[derive(Debug, Default)] pub struct MetadataApplyWedge { - report: OnceLock, + report: Mutex>, } impl MetadataApplyWedge { - /// Record the first permanent failure. Later calls are ignored. - pub fn record(&self, report: WedgeReport) { - let _ = self.report.set(report); + /// Record `report` unless a report is already held. Returns whether this + /// call recorded it. + pub fn record(&self, report: WedgeReport) -> bool { + let mut held = self.report.lock().unwrap_or_else(|p| p.into_inner()); + if held.is_some() { + return false; + } + *held = Some(report); + true + } + + /// Remove the held report if it equals `report`. Returns whether it did. + pub fn clear(&self, report: &WedgeReport) -> bool { + let mut held = self.report.lock().unwrap_or_else(|p| p.into_inner()); + if held.as_ref() != Some(report) { + return false; + } + *held = None; + true } - /// The recorded failure, if this node's metadata applier is wedged. - pub fn report(&self) -> Option<&WedgeReport> { - self.report.get() + /// The recorded failure, if this node is wedged. + pub fn report(&self) -> Option { + self.report + .lock() + .unwrap_or_else(|p| p.into_inner()) + .clone() } pub fn is_wedged(&self) -> bool { - self.report.get().is_some() + self.report + .lock() + .unwrap_or_else(|p| p.into_inner()) + .is_some() } } @@ -263,8 +292,31 @@ mod tests { }); assert!(wedge.is_wedged()); assert_eq!( - wedge.report().map(|report| report.error.as_str()), - Some("first") + wedge.report().map(|report| report.error), + Some("first".to_string()) ); } + + #[test] + fn a_clear_removes_only_its_own_report() { + let wedge = MetadataApplyWedge::default(); + let applier = WedgeReport { + raft_index: 3, + last_applied_watermark: 2, + entry_kind: "DdlPrepared".into(), + error: "applier".into(), + }; + let floor = WedgeReport { + raft_index: 9, + last_applied_watermark: 8, + entry_kind: "CutBarrier".into(), + error: "floor".into(), + }; + assert!(wedge.record(applier.clone())); + assert!(!wedge.record(floor.clone())); + assert!(!wedge.clear(&floor)); + assert_eq!(wedge.report(), Some(applier.clone())); + assert!(wedge.clear(&applier)); + assert!(!wedge.is_wedged()); + } } diff --git a/nodedb/src/control/cluster/metadata_image/capture.rs b/nodedb/src/control/cluster/metadata_image/capture.rs new file mode 100644 index 000000000..3d77799a1 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_image/capture.rs @@ -0,0 +1,129 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Capture of a group 0 [`MetadataImage`]. +//! +//! [`MetadataImageCapture::from_live`] runs on the Raft tick thread between +//! metadata apply batches. It opens read transactions and copies the small +//! in-memory cluster state, so the capture holds exactly the applied +//! entries; the rows are read later, off the tick. [`capture_metadata_image`] +//! reads catalogs that no Raft loop writes, for an offline capture. + +use std::path::Path; + +use nodedb_cluster::{ClusterCatalog, MetadataSnapshotCapture}; + +use crate::control::cluster::ca_trust::trusted_ca_ders; +use crate::control::cluster::tls::TLS_SUBDIR; +use crate::control::security::catalog::SystemCatalog; +use crate::control::security::catalog::replicated_image::ReplicatedCatalogRead; +use crate::control::state::SharedState; +use crate::event::cdc::OffsetStore; +use crate::event::cdc::consumer_group::state::OffsetImageRead; + +use super::format::{MetadataImage, encode_metadata_image, encode_routing}; + +/// Group 0 state held at one applied index, not yet read out. +pub struct MetadataImageCapture { + applied_index: u64, + applied_term: u64, + cluster_epoch: u64, + routing: Vec, + catalog: ReplicatedCatalogRead, + offsets: OffsetImageRead, + trusted_cas: Vec>, +} + +impl MetadataImageCapture { + /// Capture this node's live group 0 state at `applied_index`. + /// + /// Call only between metadata apply batches: the read transactions see + /// every commit made so far, and the cluster epoch and routing are read + /// from memory at the same point. + pub fn from_live( + shared: &SharedState, + applied_index: u64, + applied_term: u64, + ) -> crate::Result { + let routing = shared + .cluster_routing + .as_ref() + .ok_or_else(|| crate::Error::Internal { + detail: "group 0 capture: this node has no routing table".into(), + })?; + let routing = encode_routing(&routing.read().unwrap_or_else(|p| p.into_inner()))?; + let cluster_epoch = shared + .cluster_epoch + .get() + .map(|epoch| epoch.applied()) + .ok_or_else(|| crate::Error::Internal { + detail: "group 0 capture: this node has no cluster epoch state".into(), + })?; + Ok(Self { + applied_index, + applied_term, + cluster_epoch, + routing, + catalog: shared.credentials.catalog().begin_replicated_read()?, + offsets: shared.offset_store.begin_image_read()?, + trusted_cas: trusted_ca_ders(&shared.data_dir.join(TLS_SUBDIR))?, + }) + } + + /// Read the captured state into an image. + pub fn into_image(self) -> crate::Result { + Ok(MetadataImage { + applied_index: self.applied_index, + applied_term: self.applied_term, + cluster_epoch: self.cluster_epoch, + routing: self.routing, + tables: self.catalog.dump()?, + consumer_offsets: self.offsets.dump()?, + trusted_cas: self.trusted_cas, + }) + } +} + +impl MetadataSnapshotCapture for MetadataImageCapture { + fn serialize( + self: Box, + ) -> std::result::Result, Box> { + let image = self.into_image()?; + Ok(encode_metadata_image(&image)?) + } +} + +/// Capture a group 0 image from catalogs no Raft loop is writing. +/// +/// `applied_index` and `applied_term` name the group 0 entry the catalogs +/// hold state through. The cluster epoch and routing come from `cluster`. +pub fn capture_metadata_image( + system: &SystemCatalog, + cluster: &ClusterCatalog, + offsets: &OffsetStore, + tls_dir: &Path, + applied_index: u64, + applied_term: u64, +) -> crate::Result { + let cluster_err = |e: nodedb_cluster::ClusterError| crate::Error::Internal { + detail: format!("group 0 capture: read cluster catalog: {e}"), + }; + let routing = cluster + .load_routing() + .map_err(cluster_err)? + .ok_or_else(|| crate::Error::Internal { + detail: "group 0 capture: the cluster catalog holds no routing table".into(), + })?; + let cluster_epoch = cluster + .load_cluster_epoch() + .map_err(cluster_err)? + .unwrap_or(0); + Ok(MetadataImage { + applied_index, + applied_term, + cluster_epoch, + routing: encode_routing(&routing)?, + tables: system.begin_replicated_read()?.dump()?, + consumer_offsets: offsets.begin_image_read()?.dump()?, + trusted_cas: trusted_ca_ders(tls_dir)?, + }) +} diff --git a/nodedb/src/control/cluster/metadata_image/format.rs b/nodedb/src/control/cluster/metadata_image/format.rs new file mode 100644 index 000000000..8d7cd024b --- /dev/null +++ b/nodedb/src/control/cluster/metadata_image/format.rs @@ -0,0 +1,95 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The metadata Raft group 0 snapshot image and its wire codec. +//! +//! One image is the replicated state of group 0 at one applied index: +//! - every `_system.*` table group 0 replicates, as raw rows (the list and +//! the excluded tables are in +//! [`crate::control::security::catalog::replicated_image`]); +//! - the committed consumer offsets (`event_plane/consumer_offsets.redb`); +//! - the overlap CA trust set (`tls/ca.d/*.crt`), as DER; +//! - the cluster catalog keys group 0 drives: the applied cluster epoch and +//! the routing table. +//! +//! [`encode_metadata_image`] and [`decode_metadata_image`] are the only +//! codec. The Raft snapshot path and point-in-time restore both use them. + +use crate::control::security::catalog::replicated_image::RawRows; + +/// Group 0 state at `applied_index`. +#[derive(Debug, Clone, PartialEq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +#[msgpack(map)] +pub struct MetadataImage { + /// The last group 0 log index the image includes. + pub applied_index: u64, + /// The term of the entry at `applied_index`. + pub applied_term: u64, + /// The cluster epoch applied through `applied_index`. + pub cluster_epoch: u64, + /// The routing table, encoded as `ClusterCatalog::save_routing` stores it. + pub routing: Vec, + /// `(label, rows)` for every replicated `_system.*` table. + pub tables: Vec<(String, RawRows)>, + /// `(key, encoded offset)` for every committed consumer offset. + pub consumer_offsets: Vec<(String, Vec)>, + /// DER of every overlap CA in the trust set. + pub trusted_cas: Vec>, +} + +impl MetadataImage { + /// The routing table the image carries. + pub fn routing_table(&self) -> crate::Result { + zerompk::from_msgpack(&self.routing).map_err(|e| crate::Error::Internal { + detail: format!("metadata image: decode routing table: {e}"), + }) + } +} + +/// Encode `image` for the wire or for storage. +pub fn encode_metadata_image(image: &MetadataImage) -> crate::Result> { + zerompk::to_msgpack_vec(image).map_err(|e| crate::Error::Internal { + detail: format!( + "encode metadata image at index {}: {e}", + image.applied_index + ), + }) +} + +/// Decode an image [`encode_metadata_image`] produced. +pub fn decode_metadata_image(bytes: &[u8]) -> crate::Result { + zerompk::from_msgpack(bytes).map_err(|e| crate::Error::Internal { + detail: format!("decode metadata image: {e}"), + }) +} + +/// Encode `routing` the way the image carries it. +pub fn encode_routing(routing: &nodedb_cluster::RoutingTable) -> crate::Result> { + zerompk::to_msgpack_vec(routing).map_err(|e| crate::Error::Internal { + detail: format!("metadata image: encode routing table: {e}"), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn an_image_round_trips() { + let routing = nodedb_cluster::RoutingTable::uniform(2, &[1, 2], 2); + let image = MetadataImage { + applied_index: 42, + applied_term: 3, + cluster_epoch: 5, + routing: encode_routing(&routing).unwrap(), + tables: vec![("users".into(), vec![(vec![1], vec![2, 3])])], + consumer_offsets: vec![("v2:0:1".into(), vec![9; 16])], + trusted_cas: vec![vec![7; 8]], + }; + let decoded = decode_metadata_image(&encode_metadata_image(&image).unwrap()).unwrap(); + assert_eq!(decoded, image); + assert_eq!( + decoded.routing_table().unwrap().num_groups(), + routing.num_groups() + ); + } +} diff --git a/nodedb/src/control/cluster/metadata_image/install.rs b/nodedb/src/control/cluster/metadata_image/install.rs new file mode 100644 index 000000000..8f42cf8f8 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_image/install.rs @@ -0,0 +1,252 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Install a group 0 [`MetadataImage`]. +//! +//! [`write_metadata_image`] is the durable half both installs share: the +//! replicated `_system.*` tables in one write transaction, the committed +//! consumer offsets, the CA trust set, and the routing and cluster epoch +//! keys of the cluster catalog. A crash part-way leaves the staged snapshot +//! install in place, and boot runs the install again from the start. +//! +//! [`install_metadata_image_offline`] writes a plain data directory no +//! server has open. [`install_metadata_image`] installs into a running node: +//! after the durable write it rebuilds every registry the catalog feeds and +//! reconciles the Data Plane with the new catalog. + +use std::path::Path; +use std::sync::Arc; + +use nodedb_cluster::{ClusterCatalog, METADATA_GROUP_ID, RoutingTable}; + +use crate::control::cluster::ca_trust::replace_trusted_cas; +use crate::control::cluster::tls::TLS_SUBDIR; +use crate::control::security::catalog::SystemCatalog; +use crate::control::state::SharedState; +use crate::data::executor::snapshot::layout::{CLUSTER_CATALOG_FILE, SYSTEM_CATALOG_FILE}; +use crate::event::cdc::OffsetStore; + +use super::format::{MetadataImage, decode_metadata_image}; +use super::inventory::Inventory; +use super::reconcile::reconcile_data_plane; +use super::reload::{RaftOwnedState, reload_registries}; + +fn cluster_err(what: &str) -> impl Fn(nodedb_cluster::ClusterError) -> crate::Error + '_ { + move |e| crate::Error::Internal { + detail: format!("metadata image: {what}: {e}"), + } +} + +/// How an install sets the cluster epoch the cluster catalog holds. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum EpochPolicy { + /// Keep the higher of the stored epoch and the image's. A live install + /// uses this: a node never stamps an older generation than it applied. + KeepHigher, + /// Store the image's epoch, even when it is lower than the stored one. + /// For a restore that rewinds the whole cluster to the image. + UseImage, +} + +/// Write `image` durably: the replicated tables, the consumer offsets, the +/// CA trust set, `routing`, and the cluster epoch under `epoch`. +pub fn write_metadata_image( + system: &SystemCatalog, + cluster: &ClusterCatalog, + offsets: &OffsetStore, + tls_dir: &Path, + image: &MetadataImage, + routing: &RoutingTable, + epoch: EpochPolicy, +) -> crate::Result<()> { + system.replace_replicated_tables(&image.tables)?; + offsets.replace_all_offsets(&image.consumer_offsets)?; + replace_trusted_cas(tls_dir, &image.trusted_cas)?; + cluster + .save_routing(routing) + .map_err(cluster_err("save routing"))?; + let persisted = cluster + .load_cluster_epoch() + .map_err(cluster_err("load cluster epoch"))? + .unwrap_or(0); + let write = match epoch { + EpochPolicy::KeepHigher => image.cluster_epoch > persisted, + EpochPolicy::UseImage => image.cluster_epoch != persisted, + }; + if write { + cluster + .save_cluster_epoch(image.cluster_epoch) + .map_err(cluster_err("save cluster epoch"))?; + } + Ok(()) +} + +/// Install `image` into the catalogs of the node data directory `data_dir`. +/// +/// No server can have the directory open. The Raft logs are not touched. +/// `epoch` sets the cluster epoch rule: a point-in-time restore that rewinds +/// every node passes [`EpochPolicy::UseImage`]; a restore that must never +/// lower the generation passes [`EpochPolicy::KeepHigher`]. +pub fn install_metadata_image_offline( + data_dir: &Path, + image: &MetadataImage, + epoch: EpochPolicy, +) -> crate::Result<()> { + let system = SystemCatalog::open(&data_dir.join(SYSTEM_CATALOG_FILE))?; + let cluster = ClusterCatalog::open(&data_dir.join(CLUSTER_CATALOG_FILE)) + .map_err(cluster_err("open cluster catalog"))?; + let offsets = OffsetStore::open(data_dir)?; + write_metadata_image( + &system, + &cluster, + &offsets, + &data_dir.join(TLS_SUBDIR), + image, + &image.routing_table()?, + epoch, + ) +} + +/// `image_routing` with this node's own view of every data group it is a +/// member or learner of. Those groups apply their own conf changes on this +/// node, so their membership here is at least as new as the image's. +pub fn merge_routing( + local: &RoutingTable, + image_routing: RoutingTable, + node_id: u64, +) -> RoutingTable { + let mut merged = image_routing; + for (group_id, info) in local.group_members() { + if *group_id == METADATA_GROUP_ID { + continue; + } + if info.members.contains(&node_id) || info.learners.contains(&node_id) { + merged.set_group_members(*group_id, info.members.clone()); + merged.set_group_learners(*group_id, info.learners.clone()); + } + } + merged +} + +/// Install an encoded image into this running node. +/// +/// The Raft snapshot applier calls this under the group 0 install gate, so +/// no metadata entry applies until it returns. Only then does the caller +/// adopt the snapshot index. +pub async fn install_metadata_image( + shared: &Arc, + cluster: &ClusterCatalog, + raft: &RaftOwnedState, + bytes: &[u8], +) -> crate::Result<()> { + let image = decode_metadata_image(bytes)?; + let catalog = shared.credentials.catalog(); + let before = Inventory::read(catalog)?; + + let live_routing = shared + .cluster_routing + .as_ref() + .ok_or_else(|| crate::Error::Internal { + detail: "metadata snapshot install: this node has no routing table".into(), + })?; + let routing = { + let local = live_routing.read().unwrap_or_else(|p| p.into_inner()); + merge_routing(&local, image.routing_table()?, shared.node_id) + }; + write_metadata_image( + catalog, + cluster, + &shared.offset_store, + &shared.data_dir.join(TLS_SUBDIR), + &image, + &routing, + EpochPolicy::KeepHigher, + )?; + *live_routing.write().unwrap_or_else(|p| p.into_inner()) = routing; + // The image's entries never apply here, so their stamps reach the clock + // only through the high-water the image carries. + crate::control::cluster::metadata_stamp::fold_metadata_stamp_hwm(shared)?; + if let Some(epoch) = shared.cluster_epoch.get() { + epoch.advance_applied(image.cluster_epoch); + } + + let after = Inventory::read(catalog)?; + reload_registries(shared, raft, &before, &after).await?; + reconcile_data_plane(shared, &before, &after).await?; + tracing::info!( + applied_index = image.applied_index, + tables = image.tables.len(), + "installed metadata snapshot image" + ); + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::security::catalog::StoredCollection; + use nodedb_types::DatabaseId; + + /// A data group this node belongs to keeps its local membership; the + /// metadata group and foreign groups take the image's. + #[test] + fn merge_keeps_local_membership_of_hosted_data_groups() { + let mut local = RoutingTable::uniform(2, &[1, 2, 3], 2); + local.set_group_members(1, vec![1, 3]); + local.set_group_members(0, vec![1]); + let mut image = RoutingTable::uniform(2, &[1, 2, 3], 2); + image.set_group_members(1, vec![2, 3]); + image.set_group_members(0, vec![1, 2, 3]); + let merged = merge_routing(&local, image, 1); + assert_eq!(merged.group_info(1).unwrap().members, vec![1, 3]); + assert_eq!(merged.group_info(0).unwrap().members, vec![1, 2, 3]); + } + + /// An image captured from one data directory installs offline into + /// another: the replicated tables, offsets, routing, and epoch arrive. + #[test] + fn offline_capture_then_install_copies_the_image() { + let source = tempfile::tempdir().unwrap(); + let target = tempfile::tempdir().unwrap(); + let routing = RoutingTable::uniform(2, &[1, 2], 2); + let image = { + let system = SystemCatalog::open(&source.path().join(SYSTEM_CATALOG_FILE)).unwrap(); + system + .put_collection( + DatabaseId::DEFAULT, + &StoredCollection::stamped_for_test(1, "orders", "admin"), + ) + .unwrap(); + let cluster = ClusterCatalog::open(&source.path().join(CLUSTER_CATALOG_FILE)).unwrap(); + cluster.save_routing(&routing).unwrap(); + cluster.save_cluster_epoch(4).unwrap(); + let offsets = OffsetStore::open(source.path()).unwrap(); + super::super::capture::capture_metadata_image( + &system, + &cluster, + &offsets, + &source.path().join(TLS_SUBDIR), + 17, + 2, + ) + .unwrap() + }; + + install_metadata_image_offline(target.path(), &image, EpochPolicy::UseImage).unwrap(); + + let system = SystemCatalog::open(&target.path().join(SYSTEM_CATALOG_FILE)).unwrap(); + assert!( + system + .get_collection(DatabaseId::DEFAULT, 1, "orders") + .unwrap() + .is_some() + ); + let cluster = ClusterCatalog::open(&target.path().join(CLUSTER_CATALOG_FILE)).unwrap(); + assert_eq!(cluster.load_cluster_epoch().unwrap(), Some(4)); + assert_eq!( + cluster.load_routing().unwrap().unwrap().num_groups(), + routing.num_groups() + ); + assert_eq!(image.applied_index, 17); + assert_eq!(image.applied_term, 2); + } +} diff --git a/nodedb/src/control/cluster/metadata_image/inventory.rs b/nodedb/src/control/cluster/metadata_image/inventory.rs new file mode 100644 index 000000000..c01f4cd56 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_image/inventory.rs @@ -0,0 +1,161 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The catalog objects a metadata image install reconciles, read before and +//! after the replicated tables are replaced. + +use std::collections::{HashMap, HashSet}; + +use nodedb_types::{DatabaseId, StoredVectorIndexParams}; + +use crate::control::array_catalog::ArrayCatalogEntry; +use crate::control::security::catalog::{ + SequenceState, StoredCollection, StoredContinuousAggregate, StoredSequence, SystemCatalog, +}; +use crate::types::TenantId; + +/// `(database, tenant, name)` of a catalog object. +pub(super) type ObjectKey = (u64, u64, String); + +/// Catalog objects whose change the Data Plane or a live registry must see. +pub(super) struct Inventory { + pub(super) collections: HashMap, + pub(super) arrays: HashMap, + /// `(database, tenant, collection, field)` → encoded params. + pub(super) vector_params: HashMap<(u64, u64, String, String), Vec>, + /// Aggregate key → encoded definition. + pub(super) continuous_aggregates: HashMap>, + /// Collection key → the version its CRDT history was last compacted to. + pub(super) compaction_points: HashMap, + /// Sequence key → encoded definition and counter state. + pub(super) sequences: HashMap, Option>)>, + pub(super) synonym_groups: HashSet, + pub(super) topics: HashSet, + pub(super) change_streams: HashSet, + pub(super) tenants: HashSet, + pub(super) database_quotas: HashSet, + pub(super) tenant_quotas: HashSet<(DatabaseId, TenantId)>, + /// SPKI → expiry (epoch-ms) of every live enrollment pre-authorization. + pub(super) preauthorizations: HashMap<[u8; 32], u64>, +} + +fn encode(value: &T, what: &str) -> crate::Result> { + zerompk::to_msgpack_vec(value).map_err(|e| crate::Error::Internal { + detail: format!("metadata image inventory: encode {what}: {e}"), + }) +} + +/// Wall clock in epoch-ms, for pre-authorization expiry. +pub(super) fn now_ms() -> u64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| d.as_millis() as u64) + .unwrap_or(u64::MAX) +} + +impl Inventory { + /// Read the inventory from `catalog`. + pub(super) fn read(catalog: &SystemCatalog) -> crate::Result { + let collections = catalog + .load_all_collections_across_databases()? + .into_iter() + .map(|c| ((c.database_id.as_u64(), c.tenant_id, c.name.clone()), c)) + .collect(); + let arrays = catalog + .load_all_arrays()? + .into_iter() + .map(|a| { + ( + ( + a.array_id.database_id.as_u64(), + a.array_id.tenant_id.as_u64(), + a.name.clone(), + ), + a, + ) + }) + .collect(); + let mut vector_params = HashMap::new(); + for p in catalog.list_all_vector_index_params()? { + let bytes = encode::(&p, "vector index params")?; + vector_params.insert( + (p.database_id, p.tenant_id, p.collection, p.field_name), + bytes, + ); + } + let mut continuous_aggregates = HashMap::new(); + for a in catalog.load_all_continuous_aggregates()? { + let bytes = encode::(&a, "continuous aggregate")?; + continuous_aggregates.insert((a.database_id, a.tenant_id, a.name), bytes); + } + let compaction_points = catalog + .load_compaction_points()? + .into_iter() + .map(|p| { + ( + (p.database_id, p.tenant_id, p.collection), + p.target_version_json, + ) + }) + .collect(); + let mut sequences = HashMap::new(); + for def in catalog.load_all_sequences()? { + let state = catalog + .get_sequence_state(def.database_id, def.tenant_id, &def.name)? + .map(|state| encode::(&state, "sequence state")) + .transpose()?; + let bytes = encode::(&def, "sequence")?; + sequences.insert((def.database_id, def.tenant_id, def.name), (bytes, state)); + } + let synonym_groups = catalog + .load_all_synonym_groups()? + .into_iter() + .map(|g| (g.database_id, g.tenant_id, g.name)) + .collect(); + let topics = catalog + .load_all_ep_topics()? + .into_iter() + .map(|t| (t.database_id.as_u64(), t.tenant_id, t.name)) + .collect(); + let change_streams = catalog + .load_all_change_streams()? + .into_iter() + .map(|s| (s.database_id.as_u64(), s.tenant_id, s.name)) + .collect(); + let tenants = catalog + .load_all_tenants()? + .into_iter() + .map(|t| t.tenant_id) + .collect(); + let database_quotas = catalog + .list_database_quotas_lossy()? + .0 + .into_iter() + .map(|(db, _)| db) + .collect(); + let tenant_quotas = catalog + .list_all_tenant_quotas_lossy()? + .0 + .into_iter() + .map(|(db, tenant, _)| (db, tenant)) + .collect(); + let preauthorizations = catalog + .list_enrollment_preauthorizations(now_ms())? + .into_iter() + .collect(); + Ok(Self { + collections, + arrays, + vector_params, + continuous_aggregates, + compaction_points, + sequences, + synonym_groups, + topics, + change_streams, + tenants, + database_quotas, + tenant_quotas, + preauthorizations, + }) + } +} diff --git a/nodedb/src/control/cluster/metadata_image/mod.rs b/nodedb/src/control/cluster/metadata_image/mod.rs new file mode 100644 index 000000000..f85ce7b87 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_image/mod.rs @@ -0,0 +1,28 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The metadata Raft group 0 snapshot image: its format, capture, and +//! install. +//! +//! - [`format`]: the image and its codec. +//! - [`capture`]: capture from a live node between apply batches, or from +//! catalogs on disk. +//! - [`install`]: the durable write, the offline install into a data +//! directory, and the live install. +//! - [`reload`]: rebuild of the registries the catalog feeds. +//! - [`reconcile`]: the Data Plane effects of the entries an install skips. +//! - [`inventory`]: the catalog objects the reload and reconcile compare. + +pub mod capture; +pub mod format; +pub mod install; +mod inventory; +mod reconcile; +pub mod reload; + +pub use capture::{MetadataImageCapture, capture_metadata_image}; +pub use format::{MetadataImage, decode_metadata_image, encode_metadata_image}; +pub use install::{ + EpochPolicy, install_metadata_image, install_metadata_image_offline, merge_routing, + write_metadata_image, +}; +pub use reload::RaftOwnedState; diff --git a/nodedb/src/control/cluster/metadata_image/reconcile.rs b/nodedb/src/control/cluster/metadata_image/reconcile.rs new file mode 100644 index 000000000..d1c6794f7 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_image/reconcile.rs @@ -0,0 +1,437 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Bring this node's Data Plane in line with a metadata image. +//! +//! An install skips every entry the image covers, and with them the Data +//! Plane effects their post-apply made on a node that applied them. This pass +//! reproduces those effects from the difference between the catalog before +//! and after the install: +//! - `PurgeCollection`, and the source side of `MoveTenantCutover`: a +//! collection the image no longer holds has its storage reclaimed. +//! - `PutCollection` / `PutCollectionIfAbsent`: a new incarnation's storage +//! is cleared first, then every active collection is registered. +//! - `PutArray` / `DeleteArray`: a dropped array is dropped on every core, a +//! moved one is rekeyed, a new or changed one is opened, and the array +//! mirror is rebuilt from the catalog. +//! - `PutVectorIndexParams` / `DeleteVectorIndexParams`: changed parameters +//! are installed, removed ones dropped. +//! - `PutContinuousAggregate` / `DeleteContinuousAggregate`: changed +//! aggregates are registered, removed ones unregistered. +//! - `PutSynonymGroup` / `DeleteSynonymGroup`: every group is installed, +//! removed ones deleted. +//! - `DeleteMaterializedView`: a view target is a collection, reclaimed by +//! the collection rule. +//! - `DeleteTopicWithConsumerGroups`: a removed topic's node-local messages +//! and CDC buffer go. +//! - `DeleteChangeStream`: a removed stream's CDC buffer goes. +//! - `CompactHistory`: a collection whose compaction point moved owes a +//! compaction to the image's point. The owed row is written, then every +//! core compacts. A failed compaction stays owed for the retry worker. + +use std::collections::HashMap; +use std::sync::Arc; + +use crate::control::array_catalog::ArrayCatalogEntry; +use crate::control::catalog_entry::post_apply::{ + ContinuousAggregateRegisterFailure, array, clear_before_recreate, compact_async, + drop_array_on_every_core, install_synonym_group, install_vector_index_params, + open_array_on_every_core, reclaim_collection_storage, + register_continuous_aggregate_on_every_core, remove_synonym_group, remove_vector_index_params, + unregister_continuous_aggregate, +}; +use crate::control::security::catalog::StoredPendingHistoryCompaction; +use crate::control::state::SharedState; + +use super::inventory::{Inventory, ObjectKey}; + +/// Reproduce on this node's Data Plane every effect the skipped entries +/// make. +pub(super) async fn reconcile_data_plane( + shared: &Arc, + before: &Inventory, + after: &Inventory, +) -> crate::Result<()> { + reconcile_collections(shared, before, after).await?; + reconcile_compactions(shared, before, after).await?; + reconcile_arrays(shared, before, after).await?; + reconcile_vector_params(shared, before, after).await?; + reconcile_continuous_aggregates(shared, before, after).await?; + reconcile_synonym_groups(shared, before, after).await?; + reconcile_event_buffers(shared, before, after)?; + invalidate_plans(shared, before, after); + Ok(()) +} + +/// Owe and run a compaction for every collection whose compaction point the +/// image moved. Each point comes from a committed `CompactHistory` the +/// install skipped. +async fn reconcile_compactions( + shared: &Arc, + before: &Inventory, + after: &Inventory, +) -> crate::Result<()> { + let owed = owed_compactions(&before.compaction_points, &after.compaction_points); + let catalog = shared.credentials.catalog(); + for row in &owed { + catalog.enqueue_pending_history_compaction(row)?; + } + for row in owed { + if let Err(error) = compact_async( + row.database_id, + row.tenant_id, + &row.collection, + &row.target_version_json, + shared, + ) + .await + { + tracing::warn!( + collection = %row.collection, + tenant = row.tenant_id, + error = %error, + "metadata snapshot install: history compaction still owed; the retry worker owns it" + ); + } + } + Ok(()) +} + +/// The owed compaction of every collection whose point in `after` differs +/// from the one in `before`. +fn owed_compactions( + before: &HashMap, + after: &HashMap, +) -> Vec { + after + .iter() + .filter(|(key, target)| before.get(*key) != Some(*target)) + .map( + |((database_id, tenant_id, collection), target)| StoredPendingHistoryCompaction { + database_id: *database_id, + tenant_id: *tenant_id, + collection: collection.clone(), + target_version_json: target.clone(), + last_error: String::new(), + attempts: 0, + }, + ) + .collect() +} + +/// Drop this node's CDC buffer of every change stream the image removed, +/// and the messages and buffer of every removed topic. +fn reconcile_event_buffers( + shared: &SharedState, + before: &Inventory, + after: &Inventory, +) -> crate::Result<()> { + for (db, tenant, name) in before.change_streams.difference(&after.change_streams) { + shared + .cdc_router + .remove_buffer(nodedb_types::DatabaseId::new(*db), *tenant, name); + } + for (db, tenant, name) in before.topics.difference(&after.topics) { + shared + .credentials + .catalog() + .delete_ep_topic_with_consumer_groups_unchecked( + nodedb_types::DatabaseId::new(*db), + *tenant, + name, + )?; + crate::control::catalog_entry::post_apply::topic::delete_with_consumer_groups( + *db, *tenant, name, shared, + ); + } + Ok(()) +} + +async fn reconcile_collections( + shared: &Arc, + before: &Inventory, + after: &Inventory, +) -> crate::Result<()> { + for (db, tenant, name) in before.collections.keys() { + if after + .collections + .contains_key(&(*db, *tenant, name.clone())) + { + continue; + } + let purge_lsn = shared.wal.next_lsn().as_u64(); + match reclaim_collection_storage(shared, *db, *tenant, name, purge_lsn, false).await { + Ok(()) => {} + Err(failure) if failure.retry_queued => tracing::warn!( + collection = %name, + tenant = *tenant, + error = %failure.error, + "metadata snapshot install: reclaim failed; the pending-reclaim worker owns the retry" + ), + Err(failure) => return Err(failure.error), + } + } + for (key, stored) in &after.collections { + if !stored.is_active { + continue; + } + let same_incarnation = before + .collections + .get(key) + .is_some_and(|old| old.created_at == stored.created_at); + if !same_incarnation { + clear_before_recreate(shared, key.0, key.1, &key.2).await?; + } + } + crate::bootstrap::schema_rehydrate::rehydrate_schema_registry(shared) + .await + .map_err(|e| crate::Error::Internal { + detail: format!("metadata snapshot install: register collections: {e}"), + }) +} + +/// The target database of a move of the array `entry` that the catalog no +/// longer holds at `key`. +/// +/// A move keeps the array's incarnation, the stamp of the `PutArray` that +/// created it, and changes only its database. So the target is the one array +/// the install added with the same tenant, name, and incarnation in another +/// database. A same-named array with another incarnation is an unrelated +/// array, never a move. An unstamped incarnation names no array, so it never +/// matches. +fn move_target( + key: &ObjectKey, + entry: &ArrayCatalogEntry, + before: &HashMap, + after: &HashMap, +) -> Option { + if entry.incarnation == nodedb_types::Hlc::ZERO { + return None; + } + let mut targets = after.iter().filter(|((db, tenant, name), added)| { + added.incarnation == entry.incarnation + && *tenant == key.1 + && *name == key.2 + && *db != key.0 + && !before.contains_key(&(*db, *tenant, name.clone())) + }); + let target = targets.next()?; + targets.next().is_none().then_some(target.0.0) +} + +async fn reconcile_arrays( + shared: &Arc, + before: &Inventory, + after: &Inventory, +) -> crate::Result<()> { + let mut moved_in: Vec = Vec::new(); + for (key, entry) in &before.arrays { + if after.arrays.contains_key(key) { + continue; + } + let target = move_target(key, entry, &before.arrays, &after.arrays); + if let Some(to) = target { + moved_in.push((to, key.1, key.2.clone())); + } + drop_array_on_every_core(shared, key.0, key.1, &key.2, target).await?; + } + + // The Data Plane opens an array from the mirror, so the mirror is the + // catalog's before any open. + let mirror = crate::control::array_catalog::persist::load_all(shared.credentials.catalog())?; + *shared + .array_catalog + .write() + .unwrap_or_else(|p| p.into_inner()) = mirror; + for entry in after.arrays.values() { + array::put_sync(entry, shared); + } + + for (key, entry) in &after.arrays { + if moved_in.contains(key) { + continue; + } + let unchanged = before + .arrays + .get(key) + .is_some_and(|old| old.schema_hash == entry.schema_hash); + if !unchanged { + open_array_on_every_core(shared, entry).await?; + } + } + Ok(()) +} + +async fn reconcile_vector_params( + shared: &Arc, + before: &Inventory, + after: &Inventory, +) -> crate::Result<()> { + for (key, _) in before + .vector_params + .iter() + .filter(|(k, _)| !after.vector_params.contains_key(*k)) + { + remove_vector_index_params( + key.0, + key.1, + key.2.clone(), + key.3.clone(), + Arc::clone(shared), + ) + .await; + } + for (key, bytes) in &after.vector_params { + if before.vector_params.get(key) == Some(bytes) { + continue; + } + let params = zerompk::from_msgpack(bytes).map_err(|e| crate::Error::Internal { + detail: format!("metadata snapshot install: decode vector index params: {e}"), + })?; + install_vector_index_params(params, Arc::clone(shared)).await; + } + Ok(()) +} + +async fn reconcile_continuous_aggregates( + shared: &Arc, + before: &Inventory, + after: &Inventory, +) -> crate::Result<()> { + for (db, tenant, name) in before.continuous_aggregates.keys() { + if !after + .continuous_aggregates + .contains_key(&(*db, *tenant, name.clone())) + { + unregister_continuous_aggregate(*db, *tenant, name.clone(), Arc::clone(shared)).await; + } + } + for (key, bytes) in &after.continuous_aggregates { + if before.continuous_aggregates.get(key) == Some(bytes) { + continue; + } + let stored: crate::control::security::catalog::StoredContinuousAggregate = + zerompk::from_msgpack(bytes).map_err(|e| crate::Error::Internal { + detail: format!("metadata snapshot install: decode continuous aggregate: {e}"), + })?; + register_continuous_aggregate_on_every_core( + shared, + stored.tenant_id, + &stored.name, + &stored.def_bytes, + ) + .await + .map_err(|failure: ContinuousAggregateRegisterFailure| failure.error)?; + } + Ok(()) +} + +async fn reconcile_synonym_groups( + shared: &Arc, + before: &Inventory, + after: &Inventory, +) -> crate::Result<()> { + for (db, tenant, name) in before.synonym_groups.difference(&after.synonym_groups) { + remove_synonym_group(*db, *tenant, name.clone(), shared).await; + } + for group in shared.credentials.catalog().load_all_synonym_groups()? { + install_synonym_group(group, shared).await; + } + Ok(()) +} + +/// Drop every cached plan that names a collection whose descriptor changed. +fn invalidate_plans(shared: &SharedState, before: &Inventory, after: &Inventory) { + let Some(invalidator) = shared.gateway_invalidator.get() else { + return; + }; + for (key, old) in &before.collections { + let version = after + .collections + .get(key) + .map_or(u64::MAX, |new| new.descriptor_version); + if version != old.descriptor_version { + invalidator.invalidate(&key.2, version); + } + } + for (key, new) in &after.collections { + if !before.collections.contains_key(key) { + invalidator.invalidate(&key.2, new.descriptor_version); + } + } +} + +#[cfg(test)] +mod tests { + use nodedb_array::types::ArrayId; + use nodedb_types::{DatabaseId, Hlc, TenantId}; + + use super::*; + + const TENANT: u64 = 3; + + fn array(db: u64, name: &str, incarnation: Hlc) -> (ObjectKey, ArrayCatalogEntry) { + let entry = ArrayCatalogEntry { + array_id: ArrayId::in_database(TenantId::new(TENANT), DatabaseId::new(db), name), + name: name.to_string(), + schema_msgpack: Vec::new(), + schema_hash: 7, + created_at_ms: 0, + prefix_bits: 8, + audit_retain_ms: None, + minimum_audit_retain_ms: None, + modification_hlc: incarnation, + incarnation, + }; + ((db, TENANT, name.to_string()), entry) + } + + #[test] + fn a_moved_compaction_point_is_owed_and_a_held_one_is_not() { + let key = |name: &str| (2, TENANT, name.to_string()); + let before = HashMap::from([ + (key("held"), "{\"1\":4}".to_string()), + (key("behind"), "{\"1\":4}".to_string()), + ]); + let after = HashMap::from([ + (key("held"), "{\"1\":4}".to_string()), + (key("behind"), "{\"1\":9}".to_string()), + (key("new"), "{\"1\":2}".to_string()), + ]); + let mut owed: Vec<(String, String)> = owed_compactions(&before, &after) + .into_iter() + .map(|row| (row.collection, row.target_version_json)) + .collect(); + owed.sort(); + assert_eq!( + owed, + vec![ + ("behind".to_string(), "{\"1\":9}".to_string()), + ("new".to_string(), "{\"1\":2}".to_string()), + ] + ); + } + + #[test] + fn a_move_keeps_the_incarnation() { + let (key, entry) = array(1, "grid", Hlc::new(10, 0)); + let before = HashMap::from([(key.clone(), entry.clone())]); + let after = HashMap::from([array(2, "grid", Hlc::new(10, 0))]); + assert_eq!(move_target(&key, &entry, &before, &after), Some(2)); + } + + #[test] + fn a_same_named_array_in_another_database_is_not_a_move() { + let (key, entry) = array(1, "grid", Hlc::new(10, 0)); + let before = HashMap::from([(key.clone(), entry.clone())]); + let after = HashMap::from([array(2, "grid", Hlc::new(11, 0))]); + assert_eq!(move_target(&key, &entry, &before, &after), None); + } + + #[test] + fn an_unstamped_array_is_never_a_move() { + let (key, entry) = array(1, "grid", Hlc::ZERO); + let before = HashMap::from([(key.clone(), entry.clone())]); + let after = HashMap::from([array(2, "grid", Hlc::ZERO)]); + assert_eq!(move_target(&key, &entry, &before, &after), None); + } +} diff --git a/nodedb/src/control/cluster/metadata_image/reload.rs b/nodedb/src/control/cluster/metadata_image/reload.rs new file mode 100644 index 000000000..a5a435969 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_image/reload.rs @@ -0,0 +1,192 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Rebuild every in-memory registry the replicated tables feed, after a +//! metadata image replaced them. +//! +//! Each registry is cleared and loaded from the catalog through the loader +//! boot uses, so the node holds exactly what a restart on the new catalog +//! loads. + +use std::sync::Arc; +use std::time::Duration; + +use crate::control::catalog_entry::post_apply::{quota, tenant}; +use crate::control::cluster::metadata_applier::{seed_host_tables, seed_metadata_cache}; +use crate::control::cluster::recovery_check::registry_verify::{ + alert, api_keys, change_stream, consumer_group, credential, materialized_view, permissions, + redaction_policy, retention_policy, rls_policy, roles, schedule, triggers, +}; +use crate::control::state::SharedState; +use crate::control::surrogate::{SurrogateRegistry, SurrogateRegistryMode}; + +use super::inventory::{Inventory, now_ms}; + +/// Registries owned by the Raft wiring rather than by `SharedState`. +pub struct RaftOwnedState { + /// The join-token mirror the metadata applier and the bootstrap listener + /// share. + pub token_state: nodedb_cluster::SharedTokenStateMirror, + /// The transport whose pre-authorized enrollment identities mirror the + /// catalog rows. + pub transport: Option>, +} + +/// Rebuild every catalog-derived registry. `before` is the inventory read +/// before the replace: objects it holds and the catalog no longer does are +/// removed from the registries that keep them outside the catalog. +pub(super) async fn reload_registries( + shared: &Arc, + raft: &RaftOwnedState, + before: &Inventory, + after: &Inventory, +) -> crate::Result<()> { + let catalog = shared.credentials.catalog(); + + // Auth and tenancy. + credential::repair_credentials(&shared.credentials, catalog)?; + api_keys::repair_api_keys(&shared.api_keys, catalog)?; + roles::repair_roles(&shared.roles, catalog)?; + permissions::repair_permissions(&shared.permissions, catalog)?; + rls_policy::repair_rls_policies(&shared.rls, catalog)?; + redaction_policy::repair_redaction_policies(&shared.redaction, catalog)?; + shared.auth_users.reload_from_catalog(catalog)?; + shared.scope_grants.reload_from_catalog(catalog)?; + shared.quota_manager.load_from(catalog)?; + reload_tenants(shared, before, after); + + // DDL objects. + triggers::repair_triggers(&shared.trigger_registry, catalog)?; + shared + .sequence_registry + .reload_from_catalog(catalog, |database_id, tenant_id, name| { + let key = (database_id, tenant_id, name.to_string()); + before + .sequences + .get(&key) + .is_some_and(|was| after.sequences.get(&key) == Some(was)) + })?; + shared.synonym_registry.reload_from_catalog(catalog)?; + shared.custom_type_registry.reload_from_catalog(catalog)?; + shared.block_cache.clear(); + + // Event Plane definitions. + schedule::repair_schedules(&shared.schedule_registry, catalog)?; + alert::repair_alerts(&shared.alert_registry, catalog)?; + materialized_view::repair_mvs(&shared.mv_registry, catalog)?; + change_stream::repair_change_streams(&shared.stream_registry, catalog)?; + consumer_group::repair_consumer_groups(&shared.group_registry, catalog)?; + retention_policy::repair_retention_policies(&shared.retention_policy_registry, catalog)?; + shared.ep_topic_registry.load_from_catalog(catalog)?; + + // Caches keyed by database or collection. + shared.audit_dml_cache.load_from_catalog(catalog)?; + shared.collection_to_database.load_from_catalog(catalog)?; + shared.idle_timeout_cache.load_from_catalog(catalog)?; + + // Replicated watermarks. + shared.database_registry.restore_persisted( + catalog.get_database_hwm()?, + catalog.get_database_reserve_index()?, + ); + reload_surrogate_registry(shared)?; + if let Some(registry) = shared.producer_registry.as_deref() { + registry.reload_from_catalog()?; + } + + // Quota enforcement. + for db in before.database_quotas.difference(&after.database_quotas) { + quota::delete_database(*db, shared); + } + for (db, tenant_id) in before.tenant_quotas.difference(&after.tenant_quotas) { + quota::delete_tenant(*db, *tenant_id, shared); + } + crate::bootstrap::quota_replay::replay_quotas(shared); + + // Metadata-group host state. + seed_host_tables(shared)?; + seed_metadata_cache(&shared.metadata_cache, catalog)?; + + // Join tokens and enrollment identities. + let tokens = catalog + .list_join_token_states()? + .into_iter() + .map(|state| (state.token_hash, state)) + .collect(); + *raft.token_state.lock().unwrap_or_else(|p| p.into_inner()) = tokens; + if let Some(transport) = raft.transport.as_ref() { + reload_preauthorizations(transport, before, after); + } + + // Authorization state last: it reads everything above. + crate::control::security::permission_tree::reload::reload_all(shared, None).await?; + shared.authorization_fence.note_snapshot_installed(); + Ok(()) +} + +/// Seed the default quota for every tenant in the catalog and drop the +/// quota of every tenant it no longer holds. +fn reload_tenants(shared: &Arc, before: &Inventory, after: &Inventory) { + for tenant_id in before.tenants.difference(&after.tenants) { + tenant::delete(*tenant_id, Arc::clone(shared)); + } + let mut tenants = shared.tenants.lock().unwrap_or_else(|p| p.into_inner()); + for tenant_id in &after.tenants { + let tid = crate::types::TenantId::new(*tenant_id); + if !tenants.has_quota(tid) { + tenants.set_quota( + tid, + crate::control::security::tenant::TenantQuota::default(), + ); + } + } +} + +/// Reset the surrogate registry to the catalog's watermark and reserve +/// cursor, never below what this node already issued. +fn reload_surrogate_registry(shared: &SharedState) -> crate::Result<()> { + let catalog = shared.credentials.catalog(); + let persisted = catalog + .get_surrogate_hwm()? + .max(catalog.max_bound_surrogate()?.as_u32()); + let reserve_index = catalog.get_surrogate_reserve_index()?; + let mut registry = shared + .surrogate_registry + .write() + .unwrap_or_else(|p| p.into_inner()); + let floor = persisted.max(registry.current_hwm()); + let cluster = matches!(registry.mode(), SurrogateRegistryMode::Cluster(_)); + *registry = if cluster { + SurrogateRegistry::from_persisted_cluster(floor, reserve_index) + } else { + SurrogateRegistry::from_persisted_hwm(floor) + }; + Ok(()) +} + +/// Revoke the pre-authorizations the catalog dropped and install the ones +/// it holds. +fn reload_preauthorizations( + transport: &nodedb_cluster::NexarTransport, + before: &Inventory, + after: &Inventory, +) { + let now = now_ms(); + for (spki, expires_at_ms) in &before.preauthorizations { + if !after.preauthorizations.contains_key(spki) { + transport.revoke_peer_preauthorization( + spki, + Duration::from_millis(expires_at_ms.saturating_sub(now)), + ); + } + } + for (spki, expires_at_ms) in &after.preauthorizations { + let ttl = Duration::from_millis(expires_at_ms.saturating_sub(now)); + if !transport.preauthorize_peer_identity(*spki, ttl) { + tracing::error!( + ?spki, + "enrollment preauthorization capacity exhausted during snapshot install; \ + identity not admitted" + ); + } + } +} diff --git a/nodedb/src/control/cluster/metadata_stamp.rs b/nodedb/src/control/cluster/metadata_stamp.rs new file mode 100644 index 000000000..5fded3fe0 --- /dev/null +++ b/nodedb/src/control/cluster/metadata_stamp.rs @@ -0,0 +1,23 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The floor of the stamps this node takes as metadata leader. +//! +//! A leader stamps each metadata entry as it appends it, above its HLC and +//! above every stamp its log holds (see +//! `MultiRaft::propose_stamped_metadata`). Entries compacted out of the log +//! no longer show there. The applier records the highest stamp it applied in +//! the catalog, and this folds it into the HLC wherever the in-memory clock +//! can have fallen behind it: at boot, and after a snapshot install replaced +//! the catalog. + +use nodedb_types::Hlc; + +use crate::control::state::SharedState; + +/// Move the node HLC past the highest metadata stamp the catalog records. +pub fn fold_metadata_stamp_hwm(shared: &SharedState) -> crate::Result<()> { + if let Some(hwm) = shared.credentials.catalog().load_metadata_stamp_hwm()? { + shared.hlc_clock.update(Hlc::new(hwm, 0)); + } + Ok(()) +} diff --git a/nodedb/src/control/cluster/mod.rs b/nodedb/src/control/cluster/mod.rs index 547f495a5..95d86d319 100644 --- a/nodedb/src/control/cluster/mod.rs +++ b/nodedb/src/control/cluster/mod.rs @@ -8,26 +8,39 @@ pub mod array_cluster_exec; pub mod array_cluster_helpers; pub mod array_executor; -pub mod boot_restore; pub mod bootstrap_listener; +pub mod ca_trust; pub mod calvin; +pub mod calvin_snapshot; pub mod core_stall; pub mod data_plane_error_wire; pub mod decommission_bridge; pub mod handle; pub mod init; +pub mod leased_read; +pub mod linearizable_read; pub mod metadata_applier; +pub mod metadata_image; +pub mod metadata_stamp; pub mod pem_io; pub mod read_index; pub mod recovery_check; +pub mod sequencer_compaction; pub mod sequencer_halt; +pub mod sequencer_snapshot; pub mod snapshot_applier; pub mod snapshot_builder; +pub mod snapshot_builder_filter; +pub mod snapshot_cut; pub mod snapshot_hook; +pub mod snapshot_install; pub mod spsc_applier; pub mod start_raft; pub mod start_raft_helpers; +#[cfg(test)] +pub(crate) mod test_one_node; pub mod tls; +pub mod vshard_envelope_handler; pub mod warm_peers; pub use array_cluster_exec::ClusterArrayExecutor; @@ -35,7 +48,10 @@ pub use array_executor::DataPlaneArrayExecutor; pub use core_stall::CoreStallMarker; pub use decommission_bridge::spawn_decommission_shutdown_bridge; pub use handle::ClusterHandle; -pub use init::{init_cluster, init_cluster_with_transport, init_single_node_calvin}; +pub use init::{ + SINGLE_NODE_CALVIN_NODE_ID, configured_node_id, init_cluster, init_cluster_with_transport, + init_single_node_calvin, +}; pub use metadata_applier::MetadataCommitApplier; pub use read_index::{MultiRaftReadGate, RaftReadGate, ReadIndexRefusal}; pub use recovery_check::{VerifyReport, verify_and_repair}; diff --git a/nodedb/src/control/cluster/read_index.rs b/nodedb/src/control/cluster/read_index.rs index 28a93ed20..d2dcd06ec 100644 --- a/nodedb/src/control/cluster/read_index.rs +++ b/nodedb/src/control/cluster/read_index.rs @@ -5,8 +5,7 @@ //! The Control Plane decides what a read needs — a confirmed leader, or a //! replica within a staleness bound — but only the Raft coordinator, over in //! `nodedb-cluster`, can answer either. This trait is the seam between them: -//! `start_raft` installs an implementation, and single-node deployments -//! leave it unset because there is no quorum and no replica to weigh. +//! `start_raft` installs an implementation on every node. //! //! A refusal carries no leader hint. The caller already read the routing //! table to decide the read belonged here, so it builds the redirect from @@ -33,7 +32,7 @@ pub enum ReadIndexRefusal { #[async_trait] pub trait RaftReadGate: Send + Sync { /// Confirm leadership of `group_id` against a quorum, returning the index - /// the read may be served at. + /// the read can be served at. async fn confirm_leader( &self, group_id: u64, @@ -44,6 +43,28 @@ pub trait RaftReadGate: Send + Sync { /// the leader. Local state only — no quorum round, so this does not block. fn within_staleness_bound(&self, group_id: u64, max_staleness: Duration) -> bool; + /// Whether this node leads `group_id` under a leader lease that is valid + /// now. Local state only, so this does not block. + /// + /// The lease lapses before any other node can win an election for the + /// group, so at most one node answers `true` for a group at a time. + fn holds_leader_lease(&self, group_id: u64) -> bool; + + /// The Raft term of this node's valid leader lease on `group_id`, `None` + /// when it holds none. Local state only, so this does not block. + /// + /// Each later leader of the group leads at a higher term, so the term + /// fences work a previous leaseholder started. + fn leader_lease_term(&self, group_id: u64) -> Option; + + /// The index a read of `group_id` can be served at under this node's + /// valid leader lease: the commit index, `None` when it holds no lease. + /// Local state only, so this does not block. + /// + /// The read is linearizable once this node has applied the group through + /// the index. + fn lease_read_index(&self, group_id: u64) -> Option; + /// A read index for `group_id` on any node: confirmed here when this node /// leads the group, asked of the leader otherwise. /// @@ -65,7 +86,7 @@ type GateRaftLoop = nodedb_cluster::RaftLoop< /// Production implementation, backed by the Raft loop's coordinator. /// /// Holds the loop weakly: the loop keeps `SharedState` alive, and the gate -/// lives on `SharedState`, so a strong reference would pin both. +/// lives on `SharedState`, so a strong reference will pin both. pub struct MultiRaftReadGate { multi_raft: Arc>, raft_loop: Weak, @@ -122,6 +143,7 @@ fn refusal_of(error: ClusterError) -> ReadIndexRefusal { | ClusterError::SpatialGather(_) | ClusterError::Bm25Gather(_) | ClusterError::TsGather(_) + | ClusterError::ShufflePush(_) | ClusterError::RemoteUntyped { .. } | ClusterError::ShardExecution { .. } => ReadIndexRefusal::NotLeader, } @@ -156,4 +178,26 @@ impl RaftReadGate for MultiRaftReadGate { .unwrap_or_else(|p| p.into_inner()) .within_staleness_bound(group_id, max_staleness) } + + fn holds_leader_lease(&self, group_id: u64) -> bool { + self.multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .lease_read_index(group_id) + .is_some() + } + + fn leader_lease_term(&self, group_id: u64) -> Option { + self.multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .lease_term(group_id) + } + + fn lease_read_index(&self, group_id: u64) -> Option { + self.multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .lease_read_index(group_id) + } } diff --git a/nodedb/src/control/cluster/recovery_check/applied_index.rs b/nodedb/src/control/cluster/recovery_check/applied_index.rs index d01516fc7..7391ce823 100644 --- a/nodedb/src/control/cluster/recovery_check/applied_index.rs +++ b/nodedb/src/control/cluster/recovery_check/applied_index.rs @@ -8,7 +8,7 @@ //! behind between `raft_ready_rx` firing (which only waits for //! the first entry) and the recovery check running. Serving //! client traffic against that state is a correctness bug — -//! the next DDL would race an unapplied prior entry. +//! the next DDL will race an unapplied prior entry. //! //! Implementation note: `MetadataCache.applied_index` is the //! local applier's watermark. The "expected committed index" @@ -39,20 +39,7 @@ impl AppliedIndexGate { /// Read both the `MetadataCache.applied_index` and the /// `AppliedIndexWatcher::current` and report any gap. -/// -/// Single-node mode (no cluster handle) returns a gate with -/// zero gap and zero indexes — there is nothing to replay. pub fn check_applied_index(shared: &SharedState) -> AppliedIndexGate { - // If we're in single-node mode, neither source exists in a - // meaningful sense. Return a trivially-ok gate. - if shared.cluster_topology.is_none() { - return AppliedIndexGate { - cache_applied: 0, - watcher_current: 0, - gap: 0, - }; - } - let cache_applied = { let cache = match shared.metadata_cache.read() { Ok(c) => c, diff --git a/nodedb/src/control/cluster/recovery_check/integrity.rs b/nodedb/src/control/cluster/recovery_check/integrity.rs index 204b06695..36f0fe836 100644 --- a/nodedb/src/control/cluster/recovery_check/integrity.rs +++ b/nodedb/src/control/cluster/recovery_check/integrity.rs @@ -30,7 +30,7 @@ //! //! Three ownership/grant classes are healed before the abort gate, //! because they are reachable from ordinary DDL rather than from -//! storage corruption, and each would otherwise leave an existing +//! storage corruption, and each will otherwise leave an existing //! data directory permanently unbootable with no repair path. //! `verify_and_repair` runs `repair_integrity::heal_orphan_rows` //! to a fixpoint over this module's output: @@ -49,8 +49,13 @@ use std::collections::HashSet; -use crate::control::security::catalog::SystemCatalog; use crate::control::security::catalog::auth_types::object_type; +use crate::control::security::catalog::{ + StoredMaterializedView, StoredOwner, StoredPermission, StoredRlsPolicy, StoredTrigger, + SystemCatalog, +}; +use crate::event::cdc::stream_def::ChangeStreamDef; +use crate::event::scheduler::ScheduleDef; use super::divergence::{Divergence, DivergenceKind}; @@ -150,7 +155,7 @@ pub fn verify_redb_integrity(catalog: &SystemCatalog) -> Vec { // Active AND soft-deleted collections both require an // owner row. `DeactivateCollection` preserves the // primary record for undrop and must preserve the - // owner alongside it; splitting them would break + // owner alongside it; splitting them will break // undrop ownership restoration. collections .iter() @@ -221,7 +226,33 @@ pub fn verify_redb_integrity(catalog: &SystemCatalog) -> Vec { .collect(), ), ]; - for (kind, rows) in &parent_replicated { + check_parent_owners(&parent_replicated, &owner_keys, &mut violations); + check_owner_users(&owners, &user_names, &mut violations); + check_permission_grantees(&permissions, &user_names, &role_names, &mut violations); + check_trigger_collections(&triggers, &collection_keys, &mut violations); + check_rls_collections(&rls, &legacy_collection_keys, &mut violations); + check_materialized_view_sources(&materialized_views, &collection_keys, &mut violations); + check_change_stream_collections(&change_streams, &legacy_collection_keys, &mut violations); + check_schedule_targets(&schedules, &collection_keys, &mut violations); + + violations +} + +/// Database-scoped collection keys: `(database_id, tenant_id, name)`. +type CollectionKeys = HashSet<(u64, u64, String)>; +/// Collection keys of the object families with no database scope: +/// `(tenant_id, name)`. +type LegacyCollectionKeys = HashSet<(u64, String)>; +/// Owner row keys: `(object_type, database_id, tenant_id, name)`. +type OwnerKeys = HashSet<(String, u64, u64, String)>; + +/// Check 1: every parent-replicated DDL object has an owner. +fn check_parent_owners( + parent_replicated: &[ParentOwnerRows], + owner_keys: &OwnerKeys, + violations: &mut Vec, +) { + for (kind, rows) in parent_replicated { for (database_id, tenant, name) in rows { let key = ((*kind).to_string(), *database_id, *tenant, name.clone()); if !owner_keys.contains(&key) { @@ -233,9 +264,15 @@ pub fn verify_redb_integrity(catalog: &SystemCatalog) -> Vec { } } } +} - // ── Check 2: every owner.owner_username resolves to a user. ── - for o in &owners { +/// Check 2: every owner.owner_username resolves to a user. +fn check_owner_users( + owners: &[StoredOwner], + user_names: &HashSet, + violations: &mut Vec, +) { + for o in owners { if !user_names.contains(&o.owner_username) { violations.push(Divergence::new(DivergenceKind::DanglingReference { from_kind: "owner", @@ -248,9 +285,16 @@ pub fn verify_redb_integrity(catalog: &SystemCatalog) -> Vec { })); } } +} - // ── Check 3: every permission.grantee resolves. ── - for p in &permissions { +/// Check 3: every permission.grantee resolves. +fn check_permission_grantees( + permissions: &[StoredPermission], + user_names: &HashSet, + role_names: &HashSet, + violations: &mut Vec, +) { + for p in permissions { // `grantee` is either `"user:"` or `""`. if let Some(username) = p.grantee.strip_prefix("user:") { if !user_names.contains(username) { @@ -277,9 +321,15 @@ pub fn verify_redb_integrity(catalog: &SystemCatalog) -> Vec { } } } +} - // ── Check 4: every trigger.collection exists. ── - for t in &triggers { +/// Check 4: every trigger.collection exists. +fn check_trigger_collections( + triggers: &[StoredTrigger], + collection_keys: &CollectionKeys, + violations: &mut Vec, +) { + for t in triggers { let database_id = t.database_id.as_u64(); let key = (database_id, t.tenant_id, t.collection.clone()); if !collection_keys.contains(&key) { @@ -291,9 +341,15 @@ pub fn verify_redb_integrity(catalog: &SystemCatalog) -> Vec { })); } } +} - // ── Check 5: every rls_policy.collection exists. ── - for p in &rls { +/// Check 5: every rls_policy.collection exists. +fn check_rls_collections( + rls: &[StoredRlsPolicy], + legacy_collection_keys: &LegacyCollectionKeys, + violations: &mut Vec, +) { + for p in rls { let key = (p.tenant_id, p.collection.clone()); if !legacy_collection_keys.contains(&key) { violations.push(Divergence::new(DivergenceKind::DanglingReference { @@ -304,16 +360,21 @@ pub fn verify_redb_integrity(catalog: &SystemCatalog) -> Vec { })); } } +} - // ── Check 6: every materialized_view.source exists as a - // collection. ── - // - // An MV whose source was purged (or never existed on this node) - // will silently refresh against nothing. Surface as a dangling - // reference so operators know to drop the stale MV or restore - // the source. Cascade-delete of MVs on `PurgeCollection` is the - // preventive path; this check is the detective path. - for mv in &materialized_views { +/// Check 6: every materialized_view.source exists as a collection. +/// +/// An MV whose source was purged (or never existed on this node) +/// will silently refresh against nothing. Surface as a dangling +/// reference so operators know to drop the stale MV or restore +/// the source. Cascade-delete of MVs on `PurgeCollection` is the +/// preventive path; this check is the detective path. +fn check_materialized_view_sources( + materialized_views: &[StoredMaterializedView], + collection_keys: &CollectionKeys, + violations: &mut Vec, +) { + for mv in materialized_views { let key = (mv.database_id, mv.tenant_id, mv.source.clone()); if !collection_keys.contains(&key) { violations.push(Divergence::new(DivergenceKind::DanglingReference { @@ -324,11 +385,17 @@ pub fn verify_redb_integrity(catalog: &SystemCatalog) -> Vec { })); } } +} - // ── Check 7: every change_stream.collection exists as a - // collection, unless it's the wildcard `*` which - // matches any collection for the tenant. ── - for cs in &change_streams { +/// Check 7: every change_stream.collection exists as a collection, +/// unless it's the wildcard `*` which matches any collection for the +/// tenant. +fn check_change_stream_collections( + change_streams: &[ChangeStreamDef], + legacy_collection_keys: &LegacyCollectionKeys, + violations: &mut Vec, +) { + for cs in change_streams { if cs.collection == "*" { continue; } @@ -342,12 +409,17 @@ pub fn verify_redb_integrity(catalog: &SystemCatalog) -> Vec { })); } } +} - // ── Check 8: every schedule.target_collection (when Some) exists - // as a collection. `None` means the schedule is - // cross-collection or opaque (runs on `_system` - // coordinator) and is exempt. ── - for sch in &schedules { +/// Check 8: every schedule.target_collection (when Some) exists as a +/// collection. `None` means the schedule is cross-collection or opaque +/// (runs on `_system` coordinator) and is exempt. +fn check_schedule_targets( + schedules: &[ScheduleDef], + collection_keys: &CollectionKeys, + violations: &mut Vec, +) { + for sch in schedules { let Some(target) = &sch.target_collection else { continue; }; @@ -361,10 +433,6 @@ pub fn verify_redb_integrity(catalog: &SystemCatalog) -> Vec { })); } } - - let _ = (functions, procedures, sequences); - - violations } /// Built-in role names that exist outside the `StoredRole` @@ -395,7 +463,7 @@ mod tests { .expect("open integrity catalog"); let catalog = store.catalog(); - let collection = StoredCollection::new(1, "orders", "owner"); + let collection = StoredCollection::stamped_for_test(1, "orders", "owner"); catalog .put_collection(DatabaseId::DEFAULT, &collection) .expect("store default database collection"); diff --git a/nodedb/src/control/cluster/sequencer_compaction.rs b/nodedb/src/control/cluster/sequencer_compaction.rs new file mode 100644 index 000000000..a5131f3d2 --- /dev/null +++ b/nodedb/src/control/cluster/sequencer_compaction.rs @@ -0,0 +1,116 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Compaction of this node's Calvin sequencer log. +//! +//! The sequencer state machine lives in memory and is rebuilt from the log. +//! A compaction discards entries, so the state they built must survive a +//! restart some other way. Once the log retained the group's compaction +//! threshold of entries past the last capture, this node: +//! +//! 1. captures the state machine at its applied index, on the Raft tick +//! thread, right after the apply; +//! 2. writes the capture durably as the kept sequencer snapshot (see +//! [`super::sequencer_snapshot`]), off the async threads; +//! 3. records the applied index as the group's durable applied floor, so a +//! restart restores the state machine from the file and the log delivers +//! only the entries after it; +//! 4. compacts the log under the compactor's floors: every index a Calvin +//! scheduler here still replays stays (see the `raft_compactor` wiring). +//! +//! One run is in flight at a time. A run that fails leaves the log as it is; +//! a later apply starts the next one. + +use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; +use std::sync::{Arc, Weak}; + +use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; +use tracing::warn; + +use crate::control::cluster::sequencer_snapshot::SequencerSnapshotStore; +use crate::control::state::SharedState; + +/// Captures, persists and compacts the sequencer log on this node. +pub struct SequencerCompaction { + store: Arc, + shared: Weak, + /// The group's compaction threshold: entries applied past the last + /// capture before the next run. + threshold: u64, + /// The applied index of the last capture written durably. + persisted: AtomicU64, + /// Whether a run is in flight. + running: AtomicBool, +} + +impl SequencerCompaction { + /// A compaction for the sequencer log of `shared`'s node, or `None` when + /// the group has no compaction threshold: its log is never compacted. + pub fn new( + store: Arc, + shared: &Arc, + threshold: Option, + ) -> Option> { + let threshold = threshold?; + Some(Arc::new(Self { + store, + shared: Arc::downgrade(shared), + threshold: threshold.max(1), + persisted: AtomicU64::new(0), + running: AtomicBool::new(false), + })) + } + + /// Start a run when the state machine applied through `applied` and the + /// threshold passed since the last capture. The caller released the + /// state machine's lock. Never blocks: the capture is in memory and the + /// disk work runs on a blocking thread. + pub fn after_apply(self: &Arc, applied: u64) { + let persisted = self.persisted.load(Ordering::Acquire); + if applied < persisted.saturating_add(self.threshold) { + return; + } + if self.running.swap(true, Ordering::AcqRel) { + return; + } + let bytes = match self.store.capture(applied) { + Ok(bytes) => bytes, + Err(error) => { + warn!(applied, %error, "sequencer compaction: the capture failed"); + self.running.store(false, Ordering::Release); + return; + } + }; + let this = Arc::clone(self); + tokio::spawn(async move { + let blocking = Arc::clone(&this); + let run = + tokio::task::spawn_blocking(move || blocking.persist_and_compact(applied, &bytes)) + .await; + match run { + Ok(Ok(())) => {} + Ok(Err(error)) => warn!(applied, %error, "sequencer compaction did not complete"), + Err(error) => warn!(applied, %error, "sequencer compaction task failed"), + } + this.running.store(false, Ordering::Release); + }); + } + + /// Steps 2 to 4 of the module docs. Blocks on disk. + fn persist_and_compact(&self, applied: u64, bytes: &[u8]) -> crate::Result<()> { + self.store.persist(bytes)?; + let shared = self + .shared + .upgrade() + .ok_or_else(|| crate::Error::Internal { + detail: "sequencer compaction: the node shut down".into(), + })?; + if let Some(sink) = shared.raft_applied_index_sink.get() { + sink(SEQUENCER_GROUP_ID, applied)?; + } + self.persisted.store(applied, Ordering::Release); + if let Some(compactor) = shared.raft_compactor.get() { + compactor(SEQUENCER_GROUP_ID, applied)?; + } + Ok(()) + } +} diff --git a/nodedb/src/control/cluster/sequencer_snapshot.rs b/nodedb/src/control/cluster/sequencer_snapshot.rs new file mode 100644 index 000000000..96c7bdec6 --- /dev/null +++ b/nodedb/src/control/cluster/sequencer_snapshot.rs @@ -0,0 +1,193 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The Calvin sequencer group's Raft snapshot on this node. +//! +//! The sequencer state machine lives in memory and is rebuilt by applying +//! the sequencer log. A follower that installs a snapshot never applies the +//! entries it covers, so the snapshot carries the state they built (see +//! `nodedb_cluster::calvin::SequencerSnapshot`). +//! +//! - Send path: the leader captures its state machine at the group's applied +//! index, on the Raft tick thread. +//! - Receive path: the follower writes the payload durably, then restores +//! its state machine from it. The install is durable before Raft advances +//! the group's log boundary, as every install is. +//! - Own compaction: the node writes its state machine's capture before it +//! compacts its own log (see [`super::sequencer_compaction`]). +//! - Boot: a node whose sequencer log holds every entry after the kept +//! snapshot restores the state machine from the file before the log +//! applies. + +use std::path::PathBuf; +use std::sync::{Arc, Mutex}; + +use nodedb_cluster::calvin::{SEQUENCER_GROUP_ID, SequencerSnapshot, SequencerStateMachine}; + +/// The file an installed sequencer snapshot is kept in, under the data +/// directory's `calvin` directory. +const SNAPSHOT_FILE: &str = "sequencer.snapshot"; + +/// Captures, installs, and reloads the sequencer group's snapshot. +pub struct SequencerSnapshotStore { + state_machine: Arc>, + dir: PathBuf, +} + +impl SequencerSnapshotStore { + /// A store for `state_machine`, keeping its file under `data_dir`. + pub fn new( + state_machine: Arc>, + data_dir: &std::path::Path, + ) -> Self { + Self { + state_machine, + dir: data_dir.join("calvin"), + } + } + + /// The state machine's snapshot at `applied_index`, encoded. + pub fn capture(&self, applied_index: u64) -> crate::Result> { + let snapshot = self + .state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .capture_snapshot(applied_index); + snapshot.encode().map_err(codec_error) + } + + /// Install a received snapshot payload: record the install in `bases`, + /// write the payload durably, then restore the state machine from it. + /// Blocks on disk: call it off the async threads. + /// + /// The install skips entries this node's schedulers never received. The + /// record is durable first, so every kept base it skips is whole no + /// more, also after a crash. + pub fn install( + &self, + bytes: &[u8], + bases: &crate::control::state::CalvinBases, + catalog: &crate::control::security::catalog::SystemCatalog, + ) -> crate::Result<()> { + let snapshot = SequencerSnapshot::decode(bytes).map_err(codec_error)?; + bases.record_sequencer_install(catalog, snapshot.applied_index())?; + self.persist(bytes)?; + self.state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .restore_snapshot(snapshot); + Ok(()) + } + + /// Write an encoded capture durably as this node's sequencer snapshot. + /// Blocks on disk: call it off the async threads. + pub fn persist(&self, bytes: &[u8]) -> crate::Result<()> { + std::fs::create_dir_all(&self.dir).map_err(|e| storage_error(&self.dir, "create", e))?; + nodedb_wal::segment::atomic_write_fsync(&self.dir, SNAPSHOT_FILE, bytes).map_err(|e| { + crate::Error::Storage { + engine: "calvin".into(), + detail: format!( + "write the sequencer snapshot under {}: {e}", + self.dir.display() + ), + } + }) + } + + /// Restore the state machine at boot from the kept snapshot when the + /// sequencer log still holds every entry after it. Returns whether it + /// restored. + /// + /// The group's log boundary is the index of the last snapshot the log + /// adopted: an installed one, or this node's own compaction. Every own + /// compaction writes the file at or above its boundary first, so the + /// file holds the state whenever its index is at or above the boundary. + /// The log then delivers only the entries after the file's index again: + /// the state machine skips every entry at or below the index it + /// restored. + pub fn restore_at_boot( + &self, + multi_raft: &nodedb_cluster::multi_raft::MultiRaft, + ) -> crate::Result { + let path = self.dir.join(SNAPSHOT_FILE); + let bytes = match std::fs::read(&path) { + Ok(bytes) => bytes, + Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(false), + Err(e) => return Err(storage_error(&path, "read", e)), + }; + let snapshot = SequencerSnapshot::decode(&bytes).map_err(codec_error)?; + let Ok((_, boundary, _)) = multi_raft.snapshot_metadata(SEQUENCER_GROUP_ID) else { + return Ok(false); + }; + if snapshot.applied_index() < boundary { + return Ok(false); + } + self.state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()) + .restore_snapshot(snapshot); + Ok(true) + } +} + +fn codec_error(e: nodedb_cluster::ClusterError) -> crate::Error { + crate::Error::Internal { + detail: format!("sequencer snapshot: {e}"), + } +} + +fn storage_error(path: &std::path::Path, op: &str, e: std::io::Error) -> crate::Error { + crate::Error::Storage { + engine: "calvin".into(), + detail: format!("{op} {}: {e}", path.display()), + } +} + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + + use nodedb_cluster::calvin::CalvinCompletionRegistry; + + use super::*; + + fn state_machine() -> Arc> { + Arc::new(Mutex::new(SequencerStateMachine::new( + HashMap::new(), + CalvinCompletionRegistry::new_detached(), + ))) + } + + /// An installed payload is kept durably and restores the state machine: + /// its applied index becomes the state machine's. The kept bases it + /// skips are whole no more. + #[test] + fn an_installed_snapshot_is_kept_and_restored() { + let dir = tempfile::tempdir().expect("tempdir"); + let leader = SequencerSnapshotStore::new(state_machine(), dir.path()); + let bytes = leader.capture(9).expect("capture"); + + let catalog = + crate::control::security::catalog::SystemCatalog::open(&dir.path().join("system.redb")) + .expect("catalog"); + let bases = crate::control::state::CalvinBases::default(); + bases + .record_kept(&catalog, &[(4, 3), (5, 10)]) + .expect("kept"); + + let follower_sm = state_machine(); + let follower = SequencerSnapshotStore::new(Arc::clone(&follower_sm), dir.path()); + follower.install(&bytes, &bases, &catalog).expect("install"); + assert_eq!(bases.base(4), None, "inputs 3..=9 were skipped"); + assert!(bases.base(5).is_some()); + assert_eq!(catalog.load_calvin_sequencer_install().expect("load"), 9); + assert_eq!( + follower_sm + .lock() + .unwrap_or_else(|p| p.into_inner()) + .current_committed_index(), + Some(9) + ); + let kept = std::fs::read(dir.path().join("calvin").join(SNAPSHOT_FILE)).expect("kept"); + assert_eq!(kept, bytes); + } +} diff --git a/nodedb/src/control/cluster/snapshot_applier.rs b/nodedb/src/control/cluster/snapshot_applier.rs index 15784306d..e75512b64 100644 --- a/nodedb/src/control/cluster/snapshot_applier.rs +++ b/nodedb/src/control/cluster/snapshot_applier.rs @@ -6,195 +6,525 @@ //! `nodedb-cluster` defines the [`nodedb_cluster::SnapshotApplier`] trait but //! cannot depend on `nodedb` (circular), so the host crate supplies this //! implementation. The install-snapshot finalize path calls it on the FOLLOWER -//! after the atomic `.partial`→`.snap` rename and before advancing Raft. +//! after staging the snapshot and before advancing Raft. //! -//! The apply reuses the existing Data-Plane restore handler -//! (`MetaOp::RestoreTenantSnapshot`) with `replace_mode: true`, so a Raft -//! install OVERWRITES keys present in the snapshot (a Raft install must replace -//! local state, unlike user RESTORE which fail-closes on collisions). The -//! per-group snapshot bytes are the same vshard-filtered `TenantDataSnapshot` -//! the leader's `DataPlaneSnapshotBuilder` ships; the handler installs by -//! payload keys regardless of the `tenant_id` plan field, so a multi-tenant -//! snapshot applies correctly. +//! The per-group snapshot bytes are the vshard-filtered `TenantDataSnapshot` +//! the leader's `DataPlaneSnapshotBuilder` ships. The install: //! -//! Scope: this makes a FRESH/new-replica follower fully correct, plus OVERWRITE -//! of keys PRESENT in the snapshot. The `collections_to_clear` field of -//! `RestoreTenantSnapshot` carries the pre-resolved collection list so the Data -//! Plane handler performs an exact clear-then-install for lagging followers — -//! keys deleted before the snapshot index and dropped collections do not linger. +//! 0. waits until this node's metadata group applied the snapshot's +//! `metadata_floor`, the same hold a live replicated write takes. A new +//! collection incarnation clears its storage when it applies, so the clear +//! runs before the snapshot's rows of that incarnation land, never after. +//! The wait is bounded: a group 0 snapshot this node still needs can queue +//! behind this install, so a lagging node refuses the install and the +//! leader resends it; +//! 1. splits the snapshot into one share per local core, routed the way live +//! writes route ([`crate::control::cluster::snapshot_install::split`]); +//! 2. appends and fsyncs the WAL install barrier, so restart replay never +//! applies a pre-install record over the installed state +//! ([`crate::control::cluster::snapshot_install::barrier`]); +//! 3. sends every core its share and its clear list as a +//! `MetaOp::RestoreTenantSnapshot` with `replace_mode: true`, so a Raft +//! install replaces local state instead of failing against it. Each core +//! checkpoints its memory-only engines before it answers, so an +//! acknowledged install survives a restart without the WAL; +//! 4. settles only after every core acknowledged: rebinds the PK→surrogate +//! identities and takes the group's tenant write marks. +//! +//! Any error returns a [`SnapshotInstallError`]. The caller then leaves the +//! Raft boundary and durable floor unmoved. A re-install clears every core +//! first, so it converges from whatever share a failed attempt left. +//! +//! Metadata group 0 ships a [`crate::control::cluster::metadata_image`] +//! image instead, installed by +//! [`crate::control::cluster::metadata_image::install_metadata_image`]. use std::collections::HashSet; use std::sync::Arc; use std::time::Duration; -use nodedb_cluster::routing::vshard_for_collection; use nodedb_types::Surrogate; -use nodedb_types::id::{CollectionKey, DatabaseId, QualifiedCollection}; +use nodedb_types::id::{CollectionKey, DatabaseId}; use crate::bridge::envelope::PhysicalPlan; +use crate::control::cluster::snapshot_install::{ + CoreMap, CoreShares, SettleStep, SnapshotInstallError, append_install_barrier, + append_install_marker, clear_targets_per_core, group_collections, install_on_every_core, + split_by_core, +}; use crate::control::state::SharedState; -use crate::types::{TenantDataSnapshot, TenantId}; +use crate::types::{SurrogateBindEntry, TenantDataSnapshot, TenantId, VShardId}; use nodedb_physical::physical_plan::MetaOp; -/// The Raft group that owns cluster topology / metadata (group 0). Its state -/// machine is restored inline by `MultiRaft::handle_install_snapshot`, so the -/// applier is a no-op for it. -const METADATA_GROUP_ID: u64 = 0; +use crate::control::cluster::metadata_image::{RaftOwnedState, install_metadata_image}; +use nodedb_cluster::METADATA_GROUP_ID; /// Per-group snapshot apply dispatch timeout (mirrors the restore orchestrator's /// node timeout and the snapshot builder's per-tenant timeout). const SNAPSHOT_APPLY_TIMEOUT: Duration = Duration::from_secs(120); +/// How long an install waits for this node's metadata group to reach the +/// snapshot's floor before it refuses the install. +const METADATA_FLOOR_WAIT: Duration = Duration::from_secs(10); + /// Applies received per-group snapshots to the local Data Plane on the Raft /// snapshot RECEIVE path. pub struct DataPlaneSnapshotApplier { shared: Arc, + /// What a group 0 image install needs beyond `shared`. `None` on an + /// applier built for data groups only. + metadata: Option, + /// The Calvin sequencer group's snapshot store. `None` on an applier + /// built for data groups only. + sequencer: Option>, +} + +/// The state a group 0 image install writes that `SharedState` does not hold. +struct MetadataInstallParts { + /// The cluster catalog the image writes routing and epoch into. + cluster_catalog: Arc, + /// Registries the Raft wiring owns that the image reloads. + raft: RaftOwnedState, } impl DataPlaneSnapshotApplier { - /// Construct an applier bound to the node's shared state. + /// Construct an applier bound to the node's shared state. It installs + /// data-group snapshots only. pub fn new(shared: Arc) -> Self { - Self { shared } + Self { + shared, + metadata: None, + sequencer: None, + } } -} -#[async_trait::async_trait] -impl nodedb_cluster::SnapshotApplier for DataPlaneSnapshotApplier { - async fn apply_snapshot( + /// Let this applier install sequencer-group snapshots through `store`. + #[must_use] + pub fn with_sequencer( + mut self, + store: Arc, + ) -> Self { + self.sequencer = Some(store); + self + } + + /// Let this applier install group 0 images too. + pub fn with_metadata( + mut self, + cluster_catalog: Arc, + raft: RaftOwnedState, + ) -> Self { + self.metadata = Some(MetadataInstallParts { + cluster_catalog, + raft, + }); + self + } + + /// Install a non-empty data-group snapshot on every local core, then + /// settle its Control-Plane state. + pub async fn install( &self, group_id: u64, snapshot_bytes: &[u8], - ) -> std::result::Result<(), Box> { - // Metadata group is restored inline by the Raft state machine — nothing - // to apply to the Data Plane here. - if group_id == METADATA_GROUP_ID { - return Ok(()); - } - // Empty payload is the bootstrap stub — nothing to restore. - if snapshot_bytes.is_empty() { - return Ok(()); - } + ) -> Result<(), SnapshotInstallError> { + self.install_within(group_id, snapshot_bytes, METADATA_FLOOR_WAIT) + .await + } - // Resolve the target group's vshards from the LOCAL routing table, - // mirroring the builder's resolution exactly. Single-node (no routing) - // or an empty/ownerless group → empty set → no clear (correct: a fresh - // or single-node follower has no stale state to remove). - let group_vshards: HashSet = match self.shared.cluster_routing.as_ref() { - Some(routing) => { - let table = routing.read().map_err(|_| { - Box::new(crate::Error::Internal { - detail: "snapshot apply: cluster_routing RwLock poisoned".into(), - }) as Box - })?; - table.vshards_for_group(group_id).into_iter().collect() - } - None => HashSet::new(), + /// [`Self::install`], waiting at most `floor_wait` for the metadata floor. + async fn install_within( + &self, + group_id: u64, + snapshot_bytes: &[u8], + floor_wait: Duration, + ) -> Result<(), SnapshotInstallError> { + let snap: TenantDataSnapshot = + zerompk::from_msgpack(snapshot_bytes).map_err(|e| SnapshotInstallError::Decode { + group_id, + detail: e.to_string(), + })?; + // Before the catalog and routing reads below: both come from the + // metadata group, and the floor brings them to the builder's view. + self.await_metadata_floor(group_id, snap.metadata_floor, floor_wait) + .await?; + + let group_vshards = self.group_vshards(group_id)?; + let cores = { + let dispatcher = self + .shared + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()); + CoreMap::from_router(dispatcher.router()) }; + let CoreShares { + per_core, + surrogate_pk, + group_write_marks, + proposal_keys, + proposal_keys_complete_from, + cut_index, + event_lane, + calvin, + } = split_by_core(group_id, snap, &cores)?; + // Refused before any core changes: storage installed without its + // Calvin cut will hold Calvin transactions no applied state names. + let calvin = calvin.ok_or(SnapshotInstallError::NoCalvinCut { group_id })?; - // Build the clear list for an exact clear-then-install. A lagging - // follower's LOCAL catalog still lists collections that were dropped - // after its lag point, so clearing every in-group collection BEFORE - // installing the snapshot removes dropped collections (the snapshot - // omits them) and stale keys (the snapshot carries only survivors), - // while survivors are cleared-then-reinstalled. The Data Plane has no - // catalog access, so this resolution is Control-Plane (applier) only. - let mut clear_vshards: Vec = group_vshards.iter().copied().collect(); - clear_vshards.sort_unstable(); - // - // Every database's collections are listed. Each entry names its - // database and the collection as the Data Plane stores it there. - let mut collections_to_clear: Vec<(u64, u64, String)> = Vec::new(); - if !group_vshards.is_empty() { - let catalog = self.shared.credentials.catalog(); - let collections = catalog - .load_all_collections_across_databases() - .map_err(|e| Box::new(e) as Box)?; - for coll in collections.iter().filter(|c| { - c.is_active - && group_vshards.contains(&vshard_for_collection(CollectionKey::from_bare( - c.database_id, - &c.name, - ))) - }) { - collections_to_clear.push(( - coll.database_id.as_u64(), - coll.tenant_id, - QualifiedCollection::new(coll.database_id, &coll.name) - .as_str() - .to_string(), - )); - } - } + let collections = + group_collections(group_id, self.shared.credentials.catalog(), &group_vshards)?; + let clears = clear_targets_per_core(group_id, &collections, &cores)?; + // Every core replaces its array cells and graph edges of the group's + // vShards. A store with none of them stays as it is. + let mut vshards: Vec = group_vshards.iter().copied().collect(); + vshards.sort_unstable(); - // Reuse the existing local restore handler with replace_mode = true so a - // Raft install OVERWRITES present keys. The handler installs by the - // snapshot's own per-entry database and tenant, so the `tenant_id` plan - // field and the dispatch database are only routing keys — mirror the - // local RESTORE dispatch (DEFAULT db, "__system" collection). Tenant 0 - // is used as the representative routing tenant for the multi-tenant, - // multi-database payload. - let plan = PhysicalPlan::Meta(MetaOp::RestoreTenantSnapshot { - tenant_id: 0, - snapshot: snapshot_bytes.to_vec(), - replace_mode: true, - clear_vshards, - collections_to_clear, - }); + let plans = per_core + .into_iter() + .zip(clears) + .enumerate() + .map(|(core_id, (share, collections_to_clear))| { + let snapshot = + zerompk::to_msgpack_vec(&share).map_err(|e| SnapshotInstallError::Encode { + group_id, + core_id, + detail: e.to_string(), + })?; + Ok(PhysicalPlan::Meta(MetaOp::RestoreTenantSnapshot { + tenant_id: 0, + snapshot, + replace_mode: true, + collections_to_clear, + group_vshards: vshards.clone(), + })) + }) + .collect::, SnapshotInstallError>>()?; - crate::control::server::shared::ddl::sync_dispatch::dispatch_system( + append_install_barrier(&self.shared.wal, group_id, &collections)?; + install_on_every_core(&self.shared, group_id, plans, SNAPSHOT_APPLY_TIMEOUT).await?; + // The installed rows ride no WAL record: the install is a boundary a + // point-in-time restore starts after, from a base this forces. + append_install_marker(&self.shared.wal, group_id)?; + // After the install marker: boot recovery drops the vShards' applied + // markers from before it, and this replaces the rest. + crate::control::cluster::calvin_snapshot::install_calvin_cut( &self.shared, - crate::control::server::shared::ddl::sync_dispatch::SystemTask::new( - crate::control::server::shared::ddl::sync_dispatch::SystemReason::ClusterSnapshot, - TenantId::new(0), - nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "__system"), - plan, - ), - SNAPSHOT_APPLY_TIMEOUT, + group_id, + &group_vshards, + calvin, + )?; + crate::control::pitr::force_base_after_install(&self.shared); + self.settle( + group_id, + &surrogate_pk, + &group_write_marks, + &proposal_keys, + proposal_keys_complete_from, + )?; + // The snapshot's trigger actions and messages, so this node can own + // the group's firing and delivery without skipping an event. + crate::event::trigger::lane::snapshot::install( + &self.shared, + &group_vshards, + cut_index, + event_lane, ) - .await - .map_err(|e| Box::new(e) as Box)?; - - // Rebind the PK→surrogate identity map carried in the snapshot. The - // Data-Plane restore handler above installs the doc/kv/etc. blobs but - // has no catalog access, so it can NOT rebind identities; without this - // step PK point-lookups (`WHERE id=`) resolve to nothing on the - // newly caught-up follower even though full scans work. The catalog is - // Control-Plane state, available here. - let snap: TenantDataSnapshot = zerompk::from_msgpack(snapshot_bytes).map_err(|e| { - Box::new(crate::Error::Internal { - detail: format!("snapshot apply: decode group {group_id} snapshot: {e}"), - }) as Box + .map_err(|source| SnapshotInstallError::Settle { + group_id, + step: SettleStep::EventLane, + source, })?; - if !snap.surrogate_pk.is_empty() { - let catalog = self.shared.credentials.catalog(); - for e in &snap.surrogate_pk { - catalog - .put_surrogate( - CollectionKey::from_bare(DatabaseId::new(e.database_id), &e.collection), - TenantId::new(e.tenant_id), - &e.pk, - Surrogate::new(e.surrogate), - ) - .map_err(|err| Box::new(err) as Box)?; + // The snapshot holds every entry through its cut, which is at or + // above its Raft index. The entries between the two reach the apply + // loop from the log, and it concludes them without applying. Recorded + // while the install holds the group's apply gate, so no such entry + // starts before it. + if let Some(tracker) = self.shared.propose_tracker.get() { + tracker.cover_through(group_id, cut_index); + } + Ok(()) + } + + /// Wait until this node's metadata group applied `floor`, for at most + /// `limit`. `0` holds nothing. + async fn await_metadata_floor( + &self, + group_id: u64, + floor: u64, + limit: Duration, + ) -> Result<(), SnapshotInstallError> { + let watcher = self.shared.applied_index_watcher(METADATA_GROUP_ID); + if floor == 0 || watcher.current() >= floor { + return Ok(()); + } + let waiting = Arc::clone(&watcher); + let outcome = tokio::task::spawn_blocking(move || waiting.wait_for(floor, limit)).await; + match outcome { + Ok(nodedb_cluster::WaitOutcome::Reached) => Ok(()), + Ok(nodedb_cluster::WaitOutcome::TimedOut | nodedb_cluster::WaitOutcome::GroupGone) + | Err(_) => Err(SnapshotInstallError::MetadataBehind { + group_id, + floor, + applied: watcher.current(), + }), + } + } + + /// The group's vShards in the LOCAL routing table, resolved exactly as the + /// builder resolves them. No routing table or an ownerless group + /// yields an empty set, so nothing is cleared: such a node has no stale + /// state to remove. + fn group_vshards(&self, group_id: u64) -> Result, SnapshotInstallError> { + match self.shared.cluster_routing.as_ref() { + Some(routing) => { + let table = routing + .read() + .map_err(|_| SnapshotInstallError::RoutingPoisoned { group_id })?; + Ok(table.vshards_for_group(group_id).into_iter().collect()) } + None => Ok(HashSet::new()), + } + } + + /// Control-Plane steps that run once every core holds its share. + fn settle( + &self, + group_id: u64, + surrogate_pk: &[SurrogateBindEntry], + group_write_marks: &[(u64, u64, u8, String, u64)], + proposal_keys: &[(u64, u64)], + proposal_keys_complete_from: u64, + ) -> Result<(), SnapshotInstallError> { + // The Data Plane has no catalog access, so the PK→surrogate identities + // bind here. Without them PK point-lookups (`WHERE id=`) resolve to + // nothing on the caught-up follower even though full scans work. + let catalog = self.shared.credentials.catalog(); + for e in surrogate_pk { + catalog + .put_surrogate( + CollectionKey::from_bare(DatabaseId::new(e.database_id), &e.collection), + TenantId::new(e.tenant_id), + &e.pk, + Surrogate::new(e.surrogate), + ) + .map_err(|source| SnapshotInstallError::Settle { + group_id, + step: SettleStep::SurrogateRebind, + source, + })?; } // Take the group's tenant write marks, durably, before this node // reports the group applied through the snapshot. - if !snap.group_write_marks.is_empty() { + if !group_write_marks.is_empty() { self.shared .tenant_marks - .raise_group_entries(group_id, &snap.group_write_marks); + .raise_group_entries(group_id, group_write_marks); self.shared .tenant_marks - .persist(self.shared.credentials.catalog()) - .map_err(|err| Box::new(err) as Box)?; + .persist(catalog) + .map_err(|source| SnapshotInstallError::Settle { + group_id, + step: SettleStep::WriteMarks, + source, + })?; } + self.restore_proposal_keys(group_id, proposal_keys, proposal_keys_complete_from)?; + // The install emitted no per-row events, so the permission cache // reloads before this node reports coverage of the group again. self.shared.authorization_fence.note_snapshot_installed(); + Ok(()) + } +} +impl DataPlaneSnapshotApplier { + /// Make the covered entries' proposal keys this node's own. + /// + /// A `ProposalApplied` WAL marker per key, fsynced, lets the proposal + /// ledger recover them after a restart. The tracker then answers the + /// group's covered waiters by them at adopt, and hands them to the apply + /// loop's ledger, so a later copy of one of those proposals is skipped. + fn restore_proposal_keys( + &self, + group_id: u64, + proposal_keys: &[(u64, u64)], + complete_from: u64, + ) -> Result<(), SnapshotInstallError> { + let settle_error = |source| SnapshotInstallError::Settle { + group_id, + step: SettleStep::ProposalKeys, + source, + }; + for &(_, key) in proposal_keys { + self.shared + .wal + .appender(key) + .append_proposal_applied(TenantId::new(0), VShardId::new(0), DatabaseId::DEFAULT) + .map_err(settle_error)?; + } + if !proposal_keys.is_empty() { + self.shared.wal.sync().map_err(settle_error)?; + } + // Recorded even with no keys: `complete_from` tells adopt which + // covered indexes the missing keys say nothing about. + if let Some(tracker) = self.shared.propose_tracker.get() { + tracker.restore_committed_keys(group_id, proposal_keys, complete_from); + } Ok(()) } } + +#[async_trait::async_trait] +impl nodedb_cluster::SnapshotApplier for DataPlaneSnapshotApplier { + async fn apply_snapshot( + &self, + group_id: u64, + snapshot_bytes: &[u8], + ) -> std::result::Result<(), Box> { + if group_id == METADATA_GROUP_ID { + let parts = self.metadata.as_ref().ok_or_else(|| { + Box::new(crate::Error::Internal { + detail: "this snapshot applier installs no metadata images".into(), + }) as Box + })?; + return install_metadata_image( + &self.shared, + &parts.cluster_catalog, + &parts.raft, + snapshot_bytes, + ) + .await + .map_err(|e| Box::new(e) as Box); + } + // Empty payload is the bootstrap stub — nothing to restore. + if snapshot_bytes.is_empty() { + return Ok(()); + } + // The sequencer group's state is its state machine, not engine data. + if group_id == nodedb_cluster::calvin::SEQUENCER_GROUP_ID { + let store = self.sequencer.as_ref().ok_or_else(|| { + Box::new(crate::Error::Internal { + detail: "this snapshot applier installs no sequencer snapshots".into(), + }) as Box + })?; + // The install fsyncs its file: off the async workers. + let store = Arc::clone(store); + let shared = Arc::clone(&self.shared); + let bytes = snapshot_bytes.to_vec(); + return tokio::task::spawn_blocking(move || { + store.install(&bytes, &shared.calvin.bases, shared.credentials.catalog()) + }) + .await + .map_err(|e| Box::new(e) as Box)? + .map_err(|e| Box::new(e) as Box); + } + self.install(group_id, snapshot_bytes) + .await + .map_err(|e| Box::new(e) as Box) + } + + /// Answer every propose waiter the snapshot covers. Their entries + /// committed, but no apply on this node produces their results. + fn snapshot_adopted(&self, group_id: u64, last_included_index: u64) { + if group_id == METADATA_GROUP_ID { + // No entry at or below the index applies here any more, so the + // cache's applied index starts at it. + self.shared + .metadata_cache + .write() + .unwrap_or_else(|p| p.into_inner()) + .advance_applied_index(last_included_index); + } + if let Some(tracker) = self.shared.propose_tracker.get() { + tracker.resolve_covered(group_id, last_included_index); + } + // The covered entries' change events never route on this node. A + // data-group snapshot covers every entry through its cut, which can + // sit above the Raft index. + let covered = match self.shared.propose_tracker.get() { + Some(tracker) if group_id != METADATA_GROUP_ID => { + tracker.covered_through(group_id).max(last_included_index) + } + _ => last_included_index, + }; + if group_id != METADATA_GROUP_ID { + self.shared + .change_stream + .install_group_floor(group_id, covered); + } + crate::event::cdc::position::record_install_floor(&self.shared, group_id, covered); + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::wal::WalManager; + + /// A data-group snapshot that holds a new incarnation's rows reaches this + /// node before the metadata entry that created the incarnation. No core + /// receives the rows until the metadata group applied the snapshot's + /// floor, so the incarnation's storage clear, which runs inside that + /// apply, precedes them. + #[tokio::test(flavor = "multi_thread")] + async fn new_incarnation_rows_wait_for_the_metadata_floor() { + let directory = tempfile::tempdir().expect("temporary WAL directory"); + let wal = Arc::new( + WalManager::open_for_testing(&directory.path().join("floor.wal")).expect("test WAL"), + ); + let (dispatcher, mut sides) = Dispatcher::new(1, 64); + let shared = SharedState::new(dispatcher, wal).expect("shared state"); + let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); + let floor = watcher.current() + 3; + let snapshot = TenantDataSnapshot { + documents: vec![("1:1:recreated:doc1".to_string(), b"new".to_vec())], + metadata_floor: floor, + group_calvin: Some(crate::types::GroupCalvinCut::default()), + ..Default::default() + }; + let bytes = zerompk::to_msgpack_vec(&snapshot).expect("encode snapshot"); + let applier = Arc::new(DataPlaneSnapshotApplier::new(Arc::clone(&shared))); + + let refused = applier + .install_within(7, &bytes, Duration::from_millis(100)) + .await; + assert!(matches!( + refused, + Err(SnapshotInstallError::MetadataBehind { floor: f, .. }) if f == floor + )); + assert!( + sides[0].request_rx.try_pop().is_err(), + "no core receives the rows while the metadata group lags the floor" + ); + + let installing = { + let applier = Arc::clone(&applier); + let bytes = bytes.clone(); + tokio::spawn(async move { + applier + .install_within(7, &bytes, Duration::from_secs(30)) + .await + }) + }; + tokio::time::sleep(Duration::from_millis(200)).await; + assert!(sides[0].request_rx.try_pop().is_err()); + + watcher.bump(floor); + let deadline = std::time::Instant::now() + Duration::from_secs(10); + let delivered = loop { + if sides[0].request_rx.try_pop().is_ok() { + break true; + } + if std::time::Instant::now() > deadline { + break false; + } + tokio::time::sleep(Duration::from_millis(20)).await; + }; + installing.abort(); + assert!(delivered, "the rows reach the core once the floor applied"); + } +} diff --git a/nodedb/src/control/cluster/snapshot_builder.rs b/nodedb/src/control/cluster/snapshot_builder.rs index 754182730..debd6ddcd 100644 --- a/nodedb/src/control/cluster/snapshot_builder.rs +++ b/nodedb/src/control/cluster/snapshot_builder.rs @@ -10,17 +10,18 @@ //! //! The build reuses the existing Data-Plane snapshot builder //! (`MetaOp::CreateTenantSnapshot`) per tenant and per database the tenant has -//! collections in, then FILTERS every section down to the collections whose -//! vshard belongs to the target Raft group, and merges the slices into one +//! collections or arrays in, then FILTERS every section down to the records with a home +//! vShard in the target Raft group, and merges the slices into one //! `TenantDataSnapshot` for the wire. Every section entry names its database, //! so the follower installs each row in the database it came from. //! -//! The vshard-partitioned engines are filtered and shipped, including graph -//! `edges` (the edge key already embeds the collection, so it is routed through -//! the same vshard filter as every other section). CRDT is one Loro doc per -//! (tenant, collection); each `crdt_state` entry carries its single collection -//! and is shipped to the group that owns that collection's vshard — the same -//! per-collection vshard filter as every other section. +//! A row is homed on its collection's vShard. A graph edge is homed on both +//! endpoint vShards, `from_key(src)` and `from_key(dst)`, so a group ships +//! every edge with an endpoint home among its vShards and no other edge. CRDT +//! is one Loro doc per (tenant, collection); each `crdt_state` entry carries +//! its single collection and is shipped to the group that owns that +//! collection's vshard. An array cell is homed on the vShard its Hilbert +//! prefix routes to. use std::collections::{BTreeMap, BTreeSet, HashSet}; use std::sync::Arc; @@ -30,30 +31,54 @@ use nodedb_types::id::DatabaseId; use crate::Error; use crate::control::backup::snapshot_keys::{ - extract_db_scoped_collection, extract_db_tenant_scoped_collection, vshard_of_stored, + StoredRecord, extract_db_scoped_collection, extract_db_tenant_scoped_collection, + homes_of_stored, }; +use crate::control::cluster::calvin_snapshot::{await_calvin_cut, capture_calvin_cut}; +use crate::control::cluster::snapshot_builder_filter::{bind_in_group, push_group_edges}; +use crate::control::cluster::snapshot_cut::settled_cut; use crate::control::security::catalog::SystemCatalog; use crate::control::state::SharedState; -use crate::engine::graph::edge_store::parse_versioned_edge_key; +use crate::event::trigger::lane::snapshot as lane_snapshot; use crate::types::{SurrogateBindEntry, TenantDataSnapshot, TenantId}; /// Per-tenant snapshot dispatch timeout (mirrors the backup orchestrator). const TENANT_SNAPSHOT_TIMEOUT: Duration = Duration::from_secs(120); +/// How long the build waits for the group's apply gate. A Calvin install +/// holds it shared until the install finishes on its core. +const FENCE_WAIT: Duration = Duration::from_secs(5); + /// Builds per-group snapshot payloads from the local Data Plane for the Raft /// snapshot SEND path. pub struct DataPlaneSnapshotBuilder { shared: Arc, + /// The Calvin sequencer group's snapshot, or `None` on a builder for + /// data groups only. + sequencer: Option>, } impl DataPlaneSnapshotBuilder { /// Construct a builder bound to the node's shared state. pub fn new(shared: Arc) -> Self { - Self { shared } + Self { + shared, + sequencer: None, + } + } + + /// Capture the sequencer group's snapshot from `store`. + #[must_use] + pub fn with_sequencer( + mut self, + store: Arc, + ) -> Self { + self.sequencer = Some(store); + self } - /// Every tenant with an active collection, and the databases it has - /// active collections in. + /// Every tenant with an active collection or an array, and the databases + /// it has them in. fn tenant_databases(catalog: &SystemCatalog) -> Result>, Error> { let mut tenants: BTreeMap> = BTreeMap::new(); for coll in catalog @@ -66,16 +91,22 @@ impl DataPlaneSnapshotBuilder { .or_default() .insert(coll.database_id.as_u64()); } + for array in catalog.load_all_arrays()? { + tenants + .entry(array.array_id.tenant_id.as_u64()) + .or_default() + .insert(array.array_id.database_id.as_u64()); + } Ok(tenants) } - /// Capture PK→surrogate bindings for every active collection whose vshard - /// belongs to the target group, for each enumerated tenant, in every - /// database. + /// Capture the PK→surrogate binds with a home in the target group, for + /// every active collection of each enumerated tenant, in every database. /// - /// Routes each collection by its `(database, bare name)` key, the same - /// key every section's filter routes by, so only in-group collections' - /// identities ship — never more, never less than the data sections carry. + /// A bind homes on its collection's vShard and, in an edge-bearing + /// collection, on the key vShard of its key, where a live edge write binds + /// an endpoint. A group whose vShards hold an endpoint therefore ships + /// that endpoint's bind even when the collection homes elsewhere. fn capture_surrogates( catalog: &SystemCatalog, tenants: &BTreeMap>, @@ -88,12 +119,17 @@ impl DataPlaneSnapshotBuilder { .filter(|c| c.is_active && tenants.contains_key(&c.tenant_id)) { let key = nodedb_types::CollectionKey::from_bare(coll.database_id, &coll.name); - if !group_vshards.contains(&nodedb_cluster::routing::vshard_for_collection(key)) { + let holds_edges = coll.has_implicit_edges; + // Without edges every bind homes on the collection's vShard. + if !holds_edges && !group_vshards.contains(&key.vshard().as_u32()) { continue; } let bindings = catalog.scan_surrogates_for_collection(key, TenantId::new(coll.tenant_id))?; for (pk, surrogate) in bindings { + if !bind_in_group(key, &pk, holds_edges, group_vshards) { + continue; + } merged.surrogate_pk.push(SurrogateBindEntry { database_id: coll.database_id.as_u64(), tenant_id: coll.tenant_id, @@ -120,6 +156,7 @@ impl DataPlaneSnapshotBuilder { TenantId::new(tenant_id), database_id, TENANT_SNAPSHOT_TIMEOUT, + true, ) .await?; @@ -132,13 +169,14 @@ impl DataPlaneSnapshotBuilder { })?; // Every section names its collection as the Data Plane stores it in // `database_id`. - let in_group = - |stored: &str| group_vshards.contains(&vshard_of_stored(database_id, stored)); + let in_group = |collection: &str| { + homes_of_stored(database_id, StoredRecord::Row { collection }).intersects(group_vshards) + }; // db-tenant-scoped sections: key shape "{db}:{tid}:{collection}[:suffix]" let in_group_db_tenant_scoped = |key: &str| extract_db_tenant_scoped_collection(key, tenant_id).is_some_and(in_group); - // db-scoped sections: key shape "{db}:{tid}:{collection}" (coll may contain ':') + // db-scoped sections: key shape "{db}:{tid}:{collection}" (coll can contain ':') let in_group_db_scoped = |key: &str| extract_db_scoped_collection(key, tenant_id).is_some_and(in_group); @@ -199,36 +237,35 @@ impl DataPlaneSnapshotBuilder { } } - // Graph edges: the versioned edge key embeds the collection as its - // FIRST `\x00`-delimited component, and edge writes are homed at - // the vshard of the collection's `(database, bare name)` key — the SAME - // routing every other section's filter uses. The restore path parses - // the key and rebuilds CSR, so no key transformation is needed here. - // - // Unlike every other section, the edge key carries neither the - // database nor the tenant, so the merged snapshot (applied ONCE with no - // per-database or per-tenant dispatch) carries edges via - // `tenant_edges` — pushing to the plain `edges` field here would - // install them under the wrong database and tenant on apply. - for (key, value) in snap.edges { - match parse_versioned_edge_key(&key) { - Some((collection, ..)) => { - if in_group(collection) { - merged - .tenant_edges - .push((database_id.as_u64(), tenant_id, key, value)); - } - } - None => { - // All edge keys are the versioned format; an unparseable - // key has no determinable group, and restore would reject - // it via `put_edge_raw`. Do NOT silently drop it — surface - // it. Log only a short prefix, never the full key. - let key_prefix: String = key.chars().take(32).collect(); - tracing::warn!(key_prefix, "snapshot build: unparseable edge key, skipping"); - } - } - } + push_group_edges( + snap.edges, + database_id, + tenant_id, + group_vshards, + &mut merged.tenant_edges, + )?; + // A version a TRUNCATE hides is still history, so the follower gets + // it with the cuts and applied ordinals and resolves every read as + // the leader does. A cut covers every vShard of its collection. + push_group_edges( + snap.edge_hidden, + database_id, + tenant_id, + group_vshards, + &mut merged.tenant_edges, + )?; + push_group_edges( + snap.edge_applied, + database_id, + tenant_id, + group_vshards, + &mut merged.tenant_edge_applied, + )?; + merged.tenant_edge_cuts.extend( + snap.edge_cuts + .into_iter() + .map(|(collection, cut)| (database_id.as_u64(), tenant_id, collection, cut)), + ); // CRDT: one Loro doc per (tenant, collection). Each entry carries its // single collection; include it iff that collection's vshard belongs to @@ -248,6 +285,13 @@ impl DataPlaneSnapshotBuilder { } } + // Array cells: each blob holds the cells that route to one vShard. + merged.arrays.extend( + snap.arrays + .into_iter() + .filter(|blob| group_vshards.contains(&blob.vshard)), + ); + Ok(()) } } @@ -257,12 +301,15 @@ impl nodedb_cluster::SnapshotBuilder for DataPlaneSnapshotBuilder { async fn build_group_snapshot( &self, group_id: u64, - _last_included_index: u64, + last_included_index: u64, _last_included_term: u64, - ) -> Result, Box> { - // Resolve the group's vshards. Single-node (no routing) or an - // empty/ownerless group → nothing to ship; the sender falls back to the - // stub chunk. + ) -> Result> { + // No routing table or an empty group ships nothing: the sender falls + // back to the stub chunk, at the index asked for. + let nothing = nodedb_cluster::BuiltGroupSnapshot { + bytes: Vec::new(), + cut_index: last_included_index, + }; let group_vshards: HashSet = match self.shared.cluster_routing.as_ref() { Some(routing) => { let table = routing.read().map_err(|_| { @@ -272,12 +319,47 @@ impl nodedb_cluster::SnapshotBuilder for DataPlaneSnapshotBuilder { })?; table.vshards_for_group(group_id).into_iter().collect() } - None => return Ok(Vec::new()), + None => return Ok(nothing), }; if group_vshards.is_empty() { - return Ok(Vec::new()); + return Ok(nothing); } + // Every Calvin input of the group's vShards sequenced up to the + // marker this places finished here before the capture starts. + let calvin_through = await_calvin_cut(&self.shared, group_id, &group_vshards) + .await + .map_err(|e| Box::new(e) as Box)?; + + // Fence the group's apply for the whole capture, so every section, + // the write marks and the proposal keys hold the same entries: those + // at or below the cut. No entry of the group starts until the fence + // drops, and no Calvin install of its vShards runs: a scheduler + // installs under a shared hold of the same gate. Entries of other + // groups apply as usual. + let _fence = match self.shared.raft_apply_gates.get() { + Some(gates) => Some( + tokio::time::timeout(FENCE_WAIT, gates.install(group_id)) + .await + .map_err(|_| { + Box::new(Error::Internal { + detail: format!( + "snapshot build: group {group_id}: an install in flight held \ + the group's apply gate past {FENCE_WAIT:?}; the build \ + retries on the next heartbeat" + ), + }) as Box + })?, + ), + None => None, + }; + let cut_index = settled_cut(&self.shared, group_id, last_included_index) + .await + .map_err(|e| Box::new(e) as Box)?; + // Read under the fence: the storage captured below holds exactly the + // Calvin positions this names. + let calvin_cut = capture_calvin_cut(&self.shared, &group_vshards, calvin_through); + // Enumerate tenants and their databases from the system catalog — the // same source the backup orchestrator's catalog sections use. Every // active collection carries its `tenant_id` and `database_id`; each @@ -312,17 +394,80 @@ impl nodedb_cluster::SnapshotBuilder for DataPlaneSnapshotBuilder { .map_err(|e| Box::new(e) as Box)?; } - // The group's tenant write marks travel with its data: the follower - // that installs the snapshot never applies the entries it covers. + // Read after the capture. A row reached this node's storage only + // after its collection's incarnation applied here, so every row the + // snapshot holds belongs to an incarnation at or below this index. + merged.metadata_floor = self + .shared + .applied_index_watcher(nodedb_cluster::METADATA_GROUP_ID) + .floor(); + + // The group's write marks travel with its data: the follower never + // applies the entries the snapshot covers. So does its lane state. merged.group_write_marks = self.shared.tenant_marks.group_entries(group_id); - // Always return a well-formed serialized struct (even when empty) so the - // follower-apply unit receives a decodable payload rather than a stub. + // The keys of the entries the snapshot covers travel with it too: the + // follower learns from them which proposals committed, for its + // waiters and its duplicate check. They stop at the cut: an entry + // above it committed but is not in the captured state, so the + // follower must apply it. + // With no tracker no covered index is known. + match self.shared.propose_tracker.get() { + Some(tracker) => { + let carried = tracker.committed_keys_through(group_id, cut_index); + merged.group_proposal_keys = carried.keys; + merged.group_proposal_keys_complete_from = carried.complete_from; + } + None => { + merged.group_proposal_keys_complete_from = cut_index.saturating_add(1); + } + } + merged.group_cut_index = cut_index; + merged.group_calvin = Some(calvin_cut); + merged.group_event_lane = lane_snapshot::capture(&self.shared, &group_vshards) + .await + .map_err(|e| Box::new(e) as Box)?; + + // A well-formed struct, even when empty, so the follower can decode it. let out = zerompk::to_msgpack_vec(&merged).map_err(|e| { Box::new(Error::Internal { detail: format!("snapshot build: encode merged group {group_id} snapshot: {e}"), }) as Box })?; - Ok(out) + // Labelled with its cut: the follower resumes the log right after it. + Ok(nodedb_cluster::BuiltGroupSnapshot { + bytes: out, + cut_index, + }) + } + + fn capture_metadata( + &self, + applied_index: u64, + applied_term: u64, + ) -> Result< + Box, + Box, + > { + let capture = crate::control::cluster::metadata_image::MetadataImageCapture::from_live( + &self.shared, + applied_index, + applied_term, + )?; + Ok(Box::new(capture)) + } + + fn capture_sequencer( + &self, + applied_index: u64, + ) -> Result, Box> { + let store = self.sequencer.as_ref().ok_or_else(|| { + Box::new(Error::Internal { + detail: "this snapshot builder captures no sequencer snapshot".into(), + }) as Box + })?; + store + .capture(applied_index) + .map_err(|e| Box::new(e) as Box) } } diff --git a/nodedb/src/control/cluster/snapshot_builder_filter.rs b/nodedb/src/control/cluster/snapshot_builder_filter.rs new file mode 100644 index 000000000..03226352d --- /dev/null +++ b/nodedb/src/control/cluster/snapshot_builder_filter.rs @@ -0,0 +1,217 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The record filters a data-group snapshot build applies: which edges and +//! which PK→surrogate binds have a home in the target group. + +use std::collections::HashSet; + +use nodedb_types::id::DatabaseId; + +use crate::Error; +use crate::control::backup::snapshot_keys::{StoredRecord, homes_of_stored}; +use crate::types::{HomedRecord, RecordHomes}; + +/// The number of edge-key characters an unparseable-key error carries. +pub(crate) const EDGE_KEY_PREFIX_CHARS: usize = 32; + +/// Whether the bind of `pk` in `collection` has a home in `group_vshards`. +pub(crate) fn bind_in_group( + collection: nodedb_types::CollectionKey<'_>, + pk: &[u8], + holds_edges: bool, + group_vshards: &HashSet, +) -> bool { + RecordHomes::of(HomedRecord::Bind { + collection, + key: pk, + holds_edges, + }) + .intersects(group_vshards) +} + +/// Push every entry of `edges`, keyed by an edge version's key, with an +/// endpoint home in `group_vshards` onto `out`, tagged with its database and +/// tenant. +/// +/// The edge key carries neither the database nor the tenant, and the merged +/// snapshot applies once with no per-database dispatch, so edges travel in +/// `tenant_edges`, never in the plain `edges` field. +/// +/// An unparseable edge key has no home. Leaving it out makes the replica +/// diverge, so the build fails with a storage error naming the key prefix. +pub(crate) fn push_group_edges( + edges: Vec<(String, V)>, + database_id: DatabaseId, + tenant_id: u64, + group_vshards: &HashSet, + out: &mut Vec<(u64, u64, String, V)>, +) -> Result<(), Error> { + for (key, value) in edges { + let Some(record) = StoredRecord::from_edge_key(&key) else { + // The error carries a short prefix, never the full key. + let key_prefix: String = key.chars().take(EDGE_KEY_PREFIX_CHARS).collect(); + return Err(Error::Storage { + engine: "graph".into(), + detail: format!( + "snapshot build: unparseable edge key in database {} of tenant {tenant_id}, \ + key prefix {key_prefix:?}", + database_id.as_u64() + ), + }); + }; + if homes_of_stored(database_id, record).intersects(group_vshards) { + out.push((database_id.as_u64(), tenant_id, key, value)); + } + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use std::collections::{BTreeMap, BTreeSet}; + + use super::*; + use crate::engine::graph::edge_store::versioned_edge_key; + use crate::types::VShardId; + + const DB: DatabaseId = DatabaseId::new(1025); + const TID: u64 = 7; + + fn edge_key(src: &str, dst: &str) -> String { + let stored = nodedb_types::QualifiedCollection::new(DB, "follows"); + versioned_edge_key(stored.as_str(), src, "L", dst, 1).expect("edge key") + } + + /// A node key whose key vShard satisfies `pred`. + fn node_where(pred: impl Fn(u32) -> bool) -> String { + (0..100_000) + .map(|i| format!("n{i}")) + .find(|k| pred(VShardId::from_key(k.as_bytes()).as_u32())) + .expect("a matching node key") + } + + /// A group gets exactly the edges with an endpoint home among its + /// vShards: a dst-only edge is in, and an edge homed off the group is out + /// even when its collection homes on the group. + #[test] + fn a_group_gets_exactly_its_edges() { + let collection_home = nodedb_types::CollectionKey::from_bare(DB, "follows") + .vshard() + .as_u32(); + let group: HashSet = (0..VShardId::COUNT) + .filter(|v| v % 4 == collection_home % 4) + .collect(); + let inside = |v: u32| group.contains(&v); + + let a_in = node_where(inside); + let b_in = node_where(|v| inside(v) && v != VShardId::from_key(a_in.as_bytes()).as_u32()); + let x_out = node_where(|v| !inside(v)); + let y_out = + node_where(|v| !inside(v) && v != VShardId::from_key(x_out.as_bytes()).as_u32()); + + let both = edge_key(&a_in, &b_in); + let src_only = edge_key(&a_in, &x_out); + let dst_only = edge_key(&x_out, &a_in); + let neither = edge_key(&x_out, &y_out); + let edges: Vec<(String, Vec)> = [&both, &src_only, &dst_only, &neither] + .into_iter() + .map(|k| (k.clone(), vec![1])) + .collect(); + + let mut out = Vec::new(); + push_group_edges(edges, DB, TID, &group, &mut out).expect("parseable edges"); + let got: BTreeSet = out.iter().map(|(_, _, k, _)| k.clone()).collect(); + let want: BTreeSet = [both, src_only, dst_only].into_iter().collect(); + assert_eq!(got, want); + assert!( + out.iter() + .all(|(db, tid, ..)| *db == DB.as_u64() && *tid == TID) + ); + assert!( + RecordHomes::edge(&x_out, &y_out) + .iter() + .all(|h| !inside(h.as_u32())) + ); + } + + /// The per-group slices of every group cover each edge on every group + /// that homes it, and on no other. + #[test] + fn the_group_slices_cover_each_edge_on_its_homes() { + const GROUPS: u32 = 4; + let edges: Vec = (0..128) + .map(|i| edge_key(&format!("u{i}"), &format!("v{}", i * 5 + 1))) + .collect(); + let mut holders: BTreeMap> = BTreeMap::new(); + for g in 0..GROUPS { + let group: HashSet = (0..VShardId::COUNT).filter(|v| v % GROUPS == g).collect(); + let mut out = Vec::new(); + let input: Vec<(String, Vec)> = + edges.iter().map(|k| (k.clone(), Vec::new())).collect(); + push_group_edges(input, DB, TID, &group, &mut out).expect("parseable edges"); + for (_, _, k, _) in out { + holders.entry(k).or_default().insert(g); + } + } + for k in &edges { + let record = StoredRecord::from_edge_key(k).expect("edge key"); + let want: BTreeSet = homes_of_stored(DB, record) + .iter() + .map(|h| h.as_u32() % GROUPS) + .collect(); + assert_eq!(holders.get(k), Some(&want), "edge {k:?}"); + } + } + + /// An unparseable edge key fails the build with a storage error that + /// names the key prefix. The edge never leaves the snapshot silently. + #[test] + fn an_unparseable_edge_key_fails_the_build() { + let group: HashSet = (0..VShardId::COUNT).collect(); + let long_bad_key = format!("malformed-{}", "x".repeat(100)); + let edges = vec![ + (edge_key("a", "b"), vec![1]), + (long_bad_key.clone(), vec![2]), + ]; + let mut out = Vec::new(); + let err = push_group_edges(edges, DB, TID, &group, &mut out) + .expect_err("an unparseable key must fail the build"); + let Error::Storage { engine, detail } = err else { + panic!("expected a storage error, got {err:?}"); + }; + assert_eq!(engine, "graph"); + let prefix: String = long_bad_key.chars().take(EDGE_KEY_PREFIX_CHARS).collect(); + assert!(detail.contains(&format!("{prefix:?}")), "detail: {detail}"); + assert!( + !detail.contains(&long_bad_key), + "the full key must not leak" + ); + } + + /// The group that homes an edge endpoint ships its bind even when the + /// collection homes on another group. A collection without edges ships + /// its binds to its home group only. + #[test] + fn a_group_ships_the_binds_of_its_endpoints() { + let collection = nodedb_types::CollectionKey::from_bare(DB, "follows"); + let collection_home = collection.vshard().as_u32(); + let endpoint = node_where(|v| v % 4 != collection_home % 4); + let endpoint_home = VShardId::from_key(endpoint.as_bytes()).as_u32(); + let group_of = |home: u32| -> HashSet { + (0..VShardId::COUNT).filter(|v| v % 4 == home % 4).collect() + }; + let endpoint_group = group_of(endpoint_home); + let collection_group = group_of(collection_home); + + let pk = endpoint.as_bytes(); + assert!(bind_in_group(collection, pk, true, &endpoint_group)); + assert!(bind_in_group(collection, pk, true, &collection_group)); + assert!(!bind_in_group(collection, pk, false, &endpoint_group)); + assert!(bind_in_group(collection, pk, false, &collection_group)); + assert_eq!( + RecordHomes::edge(&endpoint, "x").owner().as_u32(), + endpoint_home, + "the endpoint's bind ships with the group its edges home on" + ); + } +} diff --git a/nodedb/src/control/cluster/snapshot_cut.rs b/nodedb/src/control/cluster/snapshot_cut.rs new file mode 100644 index 000000000..c28fce096 --- /dev/null +++ b/nodedb/src/control/cluster/snapshot_cut.rs @@ -0,0 +1,103 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The applied index a data-group snapshot is cut at. +//! +//! The builder holds the group's apply gate exclusive, so no entry of the +//! group starts during the capture. Every entry that started before the fence +//! finishes on its core and settles in log order. The cut is the highest +//! started entry, or the applied index the group restored at boot when no +//! entry started since. Once the group settled through the cut, the Data +//! Plane holds exactly the entries at or below it, and so do the write marks +//! and the committed proposal keys. +//! +//! The cut is at or above the Raft snapshot index the snapshot is sent at. +//! The follower that installs it concludes the entries between the two +//! without applying them (see +//! [`crate::control::distributed_applier::propose_tracker::ProposeTracker::cover_through`]). + +use std::time::Duration; + +use nodedb_cluster::WaitOutcome; + +use crate::Error; +use crate::control::state::SharedState; + +/// How long the fenced build waits for the group to settle through the cut. +/// A write its core parks holds the group below the cut. The build then fails +/// and the next heartbeat retries it, so the fence never holds the group's +/// apply for longer than this. +const CUT_SETTLE_TIMEOUT: Duration = Duration::from_secs(5); + +/// The index to cut at: the highest started entry, or the applied index when +/// it is higher. +fn cut_of(started: u64, applied: u64) -> u64 { + started.max(applied) +} + +/// Settle `group_id` through its cut and return the cut. The caller holds the +/// group's apply gate exclusive. +/// +/// Fails when the group does not settle within [`CUT_SETTLE_TIMEOUT`], or when +/// the cut is below `last_included_index`: the Raft log dropped entries this +/// node never applied, and no capture here holds them. +pub(crate) async fn settled_cut( + shared: &SharedState, + group_id: u64, + last_included_index: u64, +) -> Result { + let Some(tracker) = shared.propose_tracker.get() else { + // No apply loop runs, so nothing applies above the Raft index. + return Ok(last_included_index); + }; + let watcher = shared.applied_index_watcher(group_id); + let cut = cut_of(tracker.started_through(group_id), watcher.current()); + if watcher.current() < cut { + let waiting = std::sync::Arc::clone(&watcher); + let outcome = + tokio::task::spawn_blocking(move || waiting.wait_for(cut, CUT_SETTLE_TIMEOUT)) + .await + .map_err(|e| Error::Internal { + detail: format!( + "snapshot build: group {group_id}: the settle wait for cut {cut} \ + did not finish: {e}" + ), + })?; + match outcome { + WaitOutcome::Reached => {} + WaitOutcome::TimedOut | WaitOutcome::GroupGone => { + return Err(Error::Internal { + detail: format!( + "snapshot build: group {group_id} settled through {} of the entries \ + it started through {cut}; the build retries on the next heartbeat", + watcher.current() + ), + }); + } + } + } + if cut < last_included_index { + return Err(Error::Internal { + detail: format!( + "snapshot build: group {group_id} applied through {cut}, below the Raft \ + snapshot index {last_included_index}; this node holds no state to send" + ), + }); + } + Ok(cut) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn the_cut_is_the_higher_of_started_and_applied() { + assert_eq!(cut_of(12, 9), 12, "entries in flight raise the cut"); + assert_eq!( + cut_of(0, 40), + 40, + "a restored group cuts at its applied index" + ); + assert_eq!(cut_of(40, 40), 40); + } +} diff --git a/nodedb/src/control/cluster/snapshot_install/barrier.rs b/nodedb/src/control/cluster/snapshot_install/barrier.rs new file mode 100644 index 000000000..0650c2ab9 --- /dev/null +++ b/nodedb/src/control/cluster/snapshot_install/barrier.rs @@ -0,0 +1,65 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The WAL barrier of a snapshot install. +//! +//! An install writes no WAL records, but this node's WAL still holds the +//! records it applied to the group's collections before the install. Replay +//! re-applies the redb-backed engines' records (documents, full-text, graph +//! edges) with no floor. Replaying a pre-install record will bring back a +//! row the snapshot deleted, or overwrite an installed row with an older +//! value. +//! +//! The barrier is one `CollectionTombstoned` record per cleared collection, +//! all naming the WAL position before the install. Replay then skips every +//! record of those collections below it and replays every later record on +//! top of the installed state. The barrier is fsynced before any core clears, +//! so a crash inside the install leaves no pre-install record to replay over +//! the staged install that boot recovery re-applies. + +use crate::types::{DatabaseId, TenantId}; +use crate::wal::WalManager; +use crate::wal::manager::NO_APPLY_KEY; + +use super::clear::GroupCollection; +use super::error::SnapshotInstallError; + +/// Append and fsync the record that `group_id`'s install completed on this +/// node. +/// +/// The installed rows ride no WAL record, so a point-in-time restore cannot +/// replay across the install from a base taken before it. A restore to a +/// target at or after this record starts from a base taken after it, which +/// holds the installed rows. The record is durable before the install +/// settles, so no write the group applies after the install precedes it. +pub fn append_install_marker(wal: &WalManager, group_id: u64) -> Result<(), SnapshotInstallError> { + wal.appender(NO_APPLY_KEY) + .append_snapshot_installed(group_id) + .map_err(|source| SnapshotInstallError::Barrier { group_id, source })?; + wal.sync() + .map_err(|source| SnapshotInstallError::Barrier { group_id, source }) +} + +/// Append and fsync the install barrier for `collections`. +pub fn append_install_barrier( + wal: &WalManager, + group_id: u64, + collections: &[GroupCollection], +) -> Result<(), SnapshotInstallError> { + if collections.is_empty() { + return Ok(()); + } + let barrier = wal.next_lsn().as_u64(); + let appender = wal.appender(NO_APPLY_KEY); + for coll in collections { + appender + .append_collection_tombstone( + TenantId::new(coll.tenant_id), + DatabaseId::new(coll.database_id), + &coll.name, + barrier, + ) + .map_err(|source| SnapshotInstallError::Barrier { group_id, source })?; + } + wal.sync() + .map_err(|source| SnapshotInstallError::Barrier { group_id, source }) +} diff --git a/nodedb/src/control/cluster/snapshot_install/clear.rs b/nodedb/src/control/cluster/snapshot_install/clear.rs new file mode 100644 index 000000000..1db529e13 --- /dev/null +++ b/nodedb/src/control/cluster/snapshot_install/clear.rs @@ -0,0 +1,134 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The collections a snapshot install clears, per core. +//! +//! A lagging follower's local catalog still lists collections dropped after +//! its lag point, and its engines still hold rows deleted before the snapshot +//! index. Clearing every in-group collection before the install removes both. +//! Every core clears every listed collection: a collection's rows live on its +//! home core, but its graph edges live on the cores of their endpoints. Only +//! the home core reclaims the collection's shared on-disk L1 files, so two +//! cores never race to unlink the same tree. + +use std::collections::HashSet; + +use nodedb_physical::physical_plan::SnapshotClearTarget; +use nodedb_types::id::{CollectionKey, DatabaseId, QualifiedCollection}; + +use crate::control::security::catalog::SystemCatalog; + +use super::error::SnapshotInstallError; +use super::split::CoreMap; + +/// One active collection whose vShard belongs to the installing group. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct GroupCollection { + pub database_id: u64, + pub tenant_id: u64, + /// Bare catalog name. + pub name: String, +} + +/// Every active collection of the local catalog that routes into +/// `group_vshards`, across every database. +pub fn group_collections( + group_id: u64, + catalog: &SystemCatalog, + group_vshards: &HashSet, +) -> Result, SnapshotInstallError> { + if group_vshards.is_empty() { + return Ok(Vec::new()); + } + let collections = catalog + .load_all_collections_across_databases() + .map_err(|source| SnapshotInstallError::Catalog { group_id, source })?; + Ok(collections + .iter() + .filter(|c| { + c.is_active + && group_vshards.contains(&nodedb_cluster::routing::vshard_for_collection( + CollectionKey::from_bare(c.database_id, &c.name), + )) + }) + .map(|c| GroupCollection { + database_id: c.database_id.as_u64(), + tenant_id: c.tenant_id, + name: c.name.clone(), + }) + .collect()) +} + +/// The clear list of each core: every collection on every core, with the L1 +/// reclaim set on the collection's home core only. +pub fn clear_targets_per_core( + group_id: u64, + collections: &[GroupCollection], + cores: &CoreMap, +) -> Result>, SnapshotInstallError> { + let mut per_core: Vec> = vec![Vec::new(); cores.num_cores()]; + for coll in collections { + let db = DatabaseId::new(coll.database_id); + let home = cores.core_of( + group_id, + nodedb_cluster::routing::vshard_for_collection(CollectionKey::from_bare( + db, &coll.name, + )), + )?; + let stored = QualifiedCollection::new(db, &coll.name) + .as_str() + .to_string(); + for (core, targets) in per_core.iter_mut().enumerate() { + targets.push(SnapshotClearTarget { + database_id: coll.database_id, + tenant_id: coll.tenant_id, + collection: stored.clone(), + reclaim_l1_files: core == home, + }); + } + } + Ok(per_core) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::router::vshard::VShardRouter; + + #[test] + fn every_core_clears_and_only_the_home_core_reclaims() { + let cores = CoreMap::from_router(&VShardRouter::round_robin(4)); + let collections: Vec = (0..8) + .map(|i| GroupCollection { + database_id: if i % 2 == 0 { 0 } else { 1025 }, + tenant_id: 1, + name: format!("c{i}"), + }) + .collect(); + + let per_core = clear_targets_per_core(3, &collections, &cores).unwrap(); + assert_eq!(per_core.len(), 4); + for targets in &per_core { + assert_eq!(targets.len(), collections.len()); + } + for (i, coll) in collections.iter().enumerate() { + let reclaiming: Vec = per_core + .iter() + .enumerate() + .filter(|(_, t)| t[i].reclaim_l1_files) + .map(|(core, _)| core) + .collect(); + let db = DatabaseId::new(coll.database_id); + let home = cores + .core_of( + 3, + CollectionKey::from_bare(db, &coll.name).vshard().as_u32(), + ) + .unwrap(); + assert_eq!(reclaiming, vec![home], "collection {}", coll.name); + assert_eq!( + per_core[0][i].collection, + QualifiedCollection::new(db, &coll.name).as_str() + ); + } + } +} diff --git a/nodedb/src/control/cluster/snapshot_install/dispatch.rs b/nodedb/src/control/cluster/snapshot_install/dispatch.rs new file mode 100644 index 000000000..c894970ca --- /dev/null +++ b/nodedb/src/control/cluster/snapshot_install/dispatch.rs @@ -0,0 +1,151 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Send each core its share of a snapshot install over the SPSC bridge and +//! wait for every core to acknowledge. + +use std::time::{Duration, Instant}; + +use futures::future::join_all; + +use crate::bridge::envelope::{ + Admission, ErrorCode, ExemptReason, PhysicalPlan, Priority, Request, Response, Status, +}; +use crate::control::local_dispatch::{DeadlineCollect, collect_under_deadline}; +use crate::control::server::shared::ddl::sync_dispatch::SystemReason; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; + +use super::error::SnapshotInstallError; + +/// Dispatch `plans[i]` to core `i` and wait for every core's response. +/// +/// Returns only after every enqueued share has answered or timed out, so a +/// re-install never runs beside a share of the previous attempt. The first +/// core error is returned. An install is complete only when every core +/// answered `Ok`. +pub async fn install_on_every_core( + state: &SharedState, + group_id: u64, + plans: Vec, + timeout: Duration, +) -> Result<(), SnapshotInstallError> { + let deadline = Instant::now() + timeout; + let max_result_bytes = state.tuning.network.max_query_result_bytes as usize; + let trace_id = TraceId::generate(); + let mut first_error: Option = None; + let mut pending = Vec::with_capacity(plans.len()); + + for (core_id, plan) in plans.into_iter().enumerate() { + let request_id = state.next_request_id(); + let request = Request { + request_id, + tenant_id: TenantId::new(0), + database_id: DatabaseId::DEFAULT, + // `dispatch_to_core` bypasses vShard routing. The share names its + // own database, tenant, and collection per entry. + vshard_id: VShardId::new(core_id as u32), + plan, + deadline, + priority: Priority::Normal, + trace_id, + consistency: ReadConsistency::Strong, + idempotency_key: None, + event_source: SystemReason::ClusterSnapshot.event_source(), + user_roles: Vec::new(), + user_id: None, + statement_digest: None, + txn_id: None, + wal_lsn: None, + resolved_now_ms: None, + commit_hlc: None, + admission: Admission::Exempt(ExemptReason::AlreadyOrdered), + }; + let rx = state.tracker.register(request_id); + let dispatched = state + .dispatcher + .lock() + .unwrap_or_else(|p| p.into_inner()) + .dispatch_to_core(core_id, request); + match dispatched { + Ok(()) => pending.push((core_id, request_id, rx)), + Err(source) => { + state.tracker.cancel(&request_id); + first_error.get_or_insert(SnapshotInstallError::CoreInstall { + group_id, + core_id, + source, + }); + } + } + } + + let node_id = state.node_id; + let outcomes = join_all( + pending + .into_iter() + .map(|(core_id, request_id, mut rx)| async move { + let context = format!("snapshot install on core {core_id}"); + let result = collect_under_deadline( + &mut rx, + DeadlineCollect { + request_id, + deadline, + max_result_bytes, + context: &context, + }, + ) + .await + .and_then(installed); + ( + core_id, + result.and_then(|()| injected_fault(node_id, core_id)), + ) + }), + ) + .await; + + for (core_id, result) in outcomes { + if let Err(source) = result { + first_error.get_or_insert(SnapshotInstallError::CoreInstall { + group_id, + core_id, + source, + }); + } + } + match first_error { + Some(e) => Err(e), + None => Ok(()), + } +} + +/// A core's install response: `Ok` only for a completed install. Unlike a +/// read, a `NotFound` refusal is an error here. +fn installed(resp: Response) -> crate::Result<()> { + match resp.status { + Status::Ok => Ok(()), + Status::Partial | Status::Error => Err(crate::Error::DataPlane( + resp.error_code + .as_deref() + .cloned() + .unwrap_or_else(|| ErrorCode::Internal { + detail: "snapshot install share returned a non-Ok status with no error code" + .into(), + }), + )), + } +} + +/// Fail point `snapshot_install::node::core`: core `C` of node `N` +/// reports its share as failed after it installed it. +fn injected_fault(node_id: u64, core_id: usize) -> crate::Result<()> { + #[cfg(feature = "failpoints")] + if let Some(detail) = + crate::fail_point::eval_fail(&format!("snapshot_install::node{node_id}::core{core_id}")) + { + return Err(crate::Error::DataPlane(ErrorCode::Internal { detail })); + } + #[cfg(not(feature = "failpoints"))] + let _ = (node_id, core_id); + Ok(()) +} diff --git a/nodedb/src/control/cluster/snapshot_install/error.rs b/nodedb/src/control/cluster/snapshot_install/error.rs new file mode 100644 index 000000000..f07b043c5 --- /dev/null +++ b/nodedb/src/control/cluster/snapshot_install/error.rs @@ -0,0 +1,161 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Typed errors of a data-group snapshot install. + +/// The step of an install that runs after every core acknowledged its share. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum SettleStep { + /// Rebinding the snapshot's PK→surrogate identities in the catalog. + SurrogateRebind, + /// Persisting the group's tenant write marks. + WriteMarks, + /// Persisting the proposal keys of the entries the snapshot covers. + ProposalKeys, + /// Holding the snapshot's trigger actions and messages, and raising + /// their cursors. + EventLane, + /// Replacing the group vShards' Calvin applied state and base. + CalvinState, +} + +impl std::fmt::Display for SettleStep { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + Self::SurrogateRebind => "surrogate rebind", + Self::WriteMarks => "write-mark persist", + Self::ProposalKeys => "proposal-key persist", + Self::EventLane => "event lane install", + Self::CalvinState => "Calvin state install", + }) + } +} + +/// Why a data-group snapshot install did not complete. +/// +/// Every variant leaves the Raft boundary and durable applied floor where +/// they were, and the staged install on disk. A re-install of the same +/// snapshot clears each core before it installs, so it converges from any +/// partial state. [`Self::is_retryable`] names the variants a re-install of +/// the same bytes can clear. +#[derive(Debug, thiserror::Error)] +pub enum SnapshotInstallError { + #[error("snapshot install of group {group_id}: the routing table lock is poisoned")] + RoutingPoisoned { group_id: u64 }, + + #[error( + "snapshot install of group {group_id}: the snapshot needs this node's metadata group \ + applied through index {floor}, and it applied through {applied}; the leader resends \ + the snapshot" + )] + MetadataBehind { + group_id: u64, + floor: u64, + applied: u64, + }, + + #[error("snapshot install of group {group_id}: the payload does not decode: {detail}")] + Decode { group_id: u64, detail: String }, + + #[error( + "snapshot install of group {group_id}: {section} key starting {key_prefix:?} names no \ + routable collection or endpoint" + )] + UnroutableKey { + group_id: u64, + section: &'static str, + key_prefix: String, + }, + + #[error("snapshot install of group {group_id}: vShard {vshard} routes to no local core")] + NoCoreForVShard { group_id: u64, vshard: u32 }, + + #[error( + "snapshot install of group {group_id}: the snapshot carries no Calvin cut, so this \ + node cannot tell which Calvin transactions its storage holds; the leader resends a \ + snapshot built with one" + )] + NoCalvinCut { group_id: u64 }, + + #[error("snapshot install of group {group_id}: encode the share of core {core_id}: {detail}")] + Encode { + group_id: u64, + core_id: usize, + detail: String, + }, + + #[error("snapshot install of group {group_id}: read the local catalog: {source}")] + Catalog { + group_id: u64, + #[source] + source: crate::Error, + }, + + #[error("snapshot install of group {group_id}: append the WAL install barrier: {source}")] + Barrier { + group_id: u64, + #[source] + source: crate::Error, + }, + + #[error( + "snapshot install of group {group_id}: core {core_id} did not install its share: {source}" + )] + CoreInstall { + group_id: u64, + core_id: usize, + #[source] + source: crate::Error, + }, + + #[error( + "snapshot install of group {group_id}: {step} after the core installs failed: {source}" + )] + Settle { + group_id: u64, + step: SettleStep, + #[source] + source: crate::Error, + }, +} + +impl SnapshotInstallError { + /// True when a re-install of the same snapshot bytes can succeed: the + /// error came from local state, not from the payload. + pub fn is_retryable(&self) -> bool { + match self { + Self::MetadataBehind { .. } + | Self::Catalog { .. } + | Self::Barrier { .. } + | Self::CoreInstall { .. } + | Self::Settle { .. } => true, + Self::RoutingPoisoned { .. } + | Self::Decode { .. } + | Self::UnroutableKey { .. } + | Self::NoCoreForVShard { .. } + | Self::NoCalvinCut { .. } + | Self::Encode { .. } => false, + } + } +} + +impl From for crate::Error { + fn from(e: SnapshotInstallError) -> Self { + match e { + SnapshotInstallError::Catalog { source, .. } + | SnapshotInstallError::Barrier { source, .. } + | SnapshotInstallError::CoreInstall { source, .. } + | SnapshotInstallError::Settle { source, .. } => source, + other @ (SnapshotInstallError::RoutingPoisoned { .. } + | SnapshotInstallError::MetadataBehind { .. } + | SnapshotInstallError::NoCoreForVShard { .. }) => crate::Error::Dispatch { + detail: other.to_string(), + }, + other @ (SnapshotInstallError::Decode { .. } + | SnapshotInstallError::UnroutableKey { .. } + | SnapshotInstallError::NoCalvinCut { .. } + | SnapshotInstallError::Encode { .. }) => crate::Error::Codec { + detail: other.to_string(), + }, + } + } +} diff --git a/nodedb/src/control/cluster/snapshot_install/mod.rs b/nodedb/src/control/cluster/snapshot_install/mod.rs new file mode 100644 index 000000000..10e49e144 --- /dev/null +++ b/nodedb/src/control/cluster/snapshot_install/mod.rs @@ -0,0 +1,15 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-core install of a data-group Raft snapshot. + +pub mod barrier; +pub mod clear; +pub mod dispatch; +pub mod error; +pub mod split; + +pub use barrier::{append_install_barrier, append_install_marker}; +pub use clear::{GroupCollection, clear_targets_per_core, group_collections}; +pub use dispatch::install_on_every_core; +pub use error::{SettleStep, SnapshotInstallError}; +pub use split::{CoreMap, CoreShares, split_by_core}; diff --git a/nodedb/src/control/cluster/snapshot_install/split.rs b/nodedb/src/control/cluster/snapshot_install/split.rs new file mode 100644 index 000000000..b059a7f97 --- /dev/null +++ b/nodedb/src/control/cluster/snapshot_install/split.rs @@ -0,0 +1,518 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Split a data-group snapshot into one share per owning Data-Plane core. +//! +//! Each core stores only the state its vShards home to, and reads route to +//! that owning core. A row installed on any other core is invisible to reads +//! and duplicated by later writes. Every section therefore routes with the +//! function its live writes route with: +//! +//! - collection-homed sections (documents, indexes, vectors, KV, columnar, +//! timeseries, CRDT): the vShard of the collection's `(database, bare name)` +//! key, the same key [`vshard_of_stored`] resolves; +//! - graph edges: dual-homed on `VShardId::from_key(src)` and +//! `VShardId::from_key(dst)`, the homes an `EdgePut` is written to. +//! +//! The vShard → core step is the dispatcher's own [`VShardRouter`]. + +use nodedb_types::DatabaseId; + +use crate::control::backup::snapshot_keys::vshard_of_stored; +use crate::control::router::vshard::VShardRouter; +use crate::engine::graph::edge_store::parse_versioned_edge_key; +use crate::types::{SurrogateBindEntry, TenantDataSnapshot, VShardId}; + +use super::error::SnapshotInstallError; + +/// Longest key prefix an error carries. Keys hold user data. +const KEY_PREFIX_CHARS: usize = 32; + +/// The local vShard → core map, captured once per install. +pub struct CoreMap { + core_of: Vec>, + num_cores: usize, +} + +impl CoreMap { + /// Capture the dispatcher router's current map. + pub fn from_router(router: &VShardRouter) -> Self { + let core_of = (0..VShardId::COUNT) + .map(|v| router.resolve(VShardId::new(v))) + .collect(); + Self { + core_of, + num_cores: router.num_cores(), + } + } + + pub fn num_cores(&self) -> usize { + self.num_cores + } + + /// The core that owns `vshard`. + pub fn core_of(&self, group_id: u64, vshard: u32) -> Result { + self.core_of + .get(vshard as usize) + .copied() + .flatten() + .ok_or(SnapshotInstallError::NoCoreForVShard { group_id, vshard }) + } + + /// The core that owns the collection `stored` in `database_id`. + pub fn collection_core( + &self, + group_id: u64, + database_id: u64, + stored: &str, + ) -> Result { + self.core_of( + group_id, + vshard_of_stored(DatabaseId::new(database_id), stored), + ) + } +} + +/// A snapshot split for install: one Data-Plane share per core, plus the +/// Control-Plane state the applier binds itself. +pub struct CoreShares { + /// Index `i` is the share of core `i`. + pub per_core: Vec, + pub surrogate_pk: Vec, + pub group_write_marks: Vec<(u64, u64, u8, String, u64)>, + /// `(log_index, proposal_key)` of the committed entries the snapshot + /// covers, for the propose-waiter window. + pub proposal_keys: Vec<(u64, u64)>, + /// Lowest index `proposal_keys` is complete from; `0` when complete. + pub proposal_keys_complete_from: u64, + /// The group's applied index the snapshot was captured at. + pub cut_index: u64, + /// The Event Plane lane state of the group's vShards. + pub event_lane: crate::types::snapshot::GroupEventLane, + /// The Calvin cut of the group's vShards. + pub calvin: Option, +} + +/// `(database, collection)` of a `"{db}:{tid}:{collection}[:suffix]"` key. +/// The collection is the first `':'`- or `'\0'`-delimited token. +fn db_tenant_scoped(key: &str) -> Option<(u64, &str)> { + let mut it = key.splitn(3, ':'); + let db = it.next()?.parse().ok()?; + it.next()?.parse::().ok()?; + let collection = it.next()?.split([':', '\u{0}']).next()?; + (!collection.is_empty()).then_some((db, collection)) +} + +/// `(database, collection)` of a `"{db}:{tid}:{collection}"` key. The +/// collection is the whole remainder and can contain `':'`. +fn db_scoped(key: &str) -> Option<(u64, &str)> { + let mut it = key.splitn(3, ':'); + let db = it.next()?.parse().ok()?; + it.next()?.parse::().ok()?; + let collection = it.next()?; + (!collection.is_empty()).then_some((db, collection)) +} + +fn unroutable(group_id: u64, section: &'static str, key: &str) -> SnapshotInstallError { + SnapshotInstallError::UnroutableKey { + group_id, + section, + key_prefix: key.chars().take(KEY_PREFIX_CHARS).collect(), + } +} + +/// A string-keyed snapshot section: its `(key, value)` entries. +type KeyedSection = Vec<(String, Vec)>; + +/// Selects one string-keyed section of a core's share. +type SectionOf = fn(&mut TenantDataSnapshot) -> &mut KeyedSection; + +struct Splitter<'a> { + group_id: u64, + cores: &'a CoreMap, + per_core: Vec, +} + +impl Splitter<'_> { + /// Route every entry of a string-keyed section to its collection's core. + fn keyed( + &mut self, + section: &'static str, + entries: KeyedSection, + locate: fn(&str) -> Option<(u64, &str)>, + field: SectionOf, + ) -> Result<(), SnapshotInstallError> { + for (key, value) in entries { + let (db, collection) = + locate(&key).ok_or_else(|| unroutable(self.group_id, section, &key))?; + let core = self.cores.collection_core(self.group_id, db, collection)?; + field(&mut self.per_core[core]).push((key, value)); + } + Ok(()) + } + + /// The cores an edge is stored on: the homes of its two endpoints. + fn edge_cores( + &self, + section: &'static str, + key: &str, + ) -> Result<(usize, Option), SnapshotInstallError> { + let (_, src, _, dst, _) = + parse_versioned_edge_key(key).ok_or_else(|| unroutable(self.group_id, section, key))?; + let src_core = self + .cores + .core_of(self.group_id, VShardId::from_key(src.as_bytes()).as_u32())?; + let dst_core = self + .cores + .core_of(self.group_id, VShardId::from_key(dst.as_bytes()).as_u32())?; + Ok((src_core, (dst_core != src_core).then_some(dst_core))) + } +} + +/// Split `snap` into one share per local core. +/// +/// A key that names no routable collection or endpoint fails the split: a +/// row with no owner has no core it can be read back from. +pub fn split_by_core( + group_id: u64, + snap: TenantDataSnapshot, + cores: &CoreMap, +) -> Result { + // Exhaustive destructure: a new section fails to compile here instead of + // being dropped from the install. + let TenantDataSnapshot { + documents, + indexes, + edges, + vectors, + kv_tables, + crdt_state, + crdt_constraints, + timeseries, + flushed_ts_segments, + columnar_engines, + vector_params, + index_configs, + surrogate_pk, + tenant_edges, + edge_hidden, + edge_cuts, + edge_applied, + tenant_edge_cuts, + tenant_edge_applied, + group_write_marks, + group_proposal_keys, + group_proposal_keys_complete_from, + documents_versioned, + indexes_versioned, + vector_multi_documents, + arrays, + // The install waits on the floor before it splits; no core reads it. + metadata_floor: _, + group_cut_index, + group_event_lane, + group_calvin, + } = snap; + + let mut s = Splitter { + group_id, + cores, + per_core: (0..cores.num_cores()) + .map(|_| TenantDataSnapshot::default()) + .collect(), + }; + + s.keyed("documents", documents, db_tenant_scoped, |t| { + &mut t.documents + })?; + s.keyed("indexes", indexes, db_tenant_scoped, |t| &mut t.indexes)?; + s.keyed( + "documents_versioned", + documents_versioned, + db_tenant_scoped, + |t| &mut t.documents_versioned, + )?; + s.keyed( + "indexes_versioned", + indexes_versioned, + db_tenant_scoped, + |t| &mut t.indexes_versioned, + )?; + s.keyed("vectors", vectors, db_tenant_scoped, |t| &mut t.vectors)?; + // Membership goes to the core its index's rows go to, keyed alike. + for (key, documents) in vector_multi_documents { + let (db, collection) = db_tenant_scoped(&key) + .ok_or_else(|| unroutable(group_id, "vector_multi_documents", &key))?; + let core = cores.collection_core(group_id, db, collection)?; + s.per_core[core] + .vector_multi_documents + .push((key, documents)); + } + s.keyed("vector_params", vector_params, db_tenant_scoped, |t| { + &mut t.vector_params + })?; + s.keyed("index_configs", index_configs, db_tenant_scoped, |t| { + &mut t.index_configs + })?; + s.keyed("timeseries", timeseries, db_tenant_scoped, |t| { + &mut t.timeseries + })?; + s.keyed("kv_tables", kv_tables, db_scoped, |t| &mut t.kv_tables)?; + s.keyed("columnar_engines", columnar_engines, db_scoped, |t| { + &mut t.columnar_engines + })?; + + for blob in flushed_ts_segments { + let (db, collection) = db_scoped(&blob.collection_key) + .ok_or_else(|| unroutable(group_id, "flushed_ts_segments", &blob.collection_key))?; + let core = cores.collection_core(group_id, db, collection)?; + s.per_core[core].flushed_ts_segments.push(blob); + } + + for entry in crdt_state { + let core = cores.collection_core(group_id, entry.0, &entry.2)?; + s.per_core[core].crdt_state.push(entry); + } + + for entry in crdt_constraints { + let core = cores.collection_core(group_id, entry.database_id, &entry.collection)?; + s.per_core[core].crdt_constraints.push(entry); + } + + for (key, value) in edges { + let (first, second) = s.edge_cores("edges", &key)?; + if let Some(second) = second { + s.per_core[second].edges.push((key.clone(), value.clone())); + } + s.per_core[first].edges.push((key, value)); + } + + for blob in arrays { + let core = cores.core_of(group_id, blob.vshard)?; + s.per_core[core].arrays.push(blob); + } + + for (db, tid, key, value) in tenant_edges { + let (first, second) = s.edge_cores("tenant_edges", &key)?; + if let Some(second) = second { + s.per_core[second] + .tenant_edges + .push((db, tid, key.clone(), value.clone())); + } + s.per_core[first].tenant_edges.push((db, tid, key, value)); + } + + // A hidden version and an applied ordinal go where their edge goes. + for (key, value) in edge_hidden { + let (first, second) = s.edge_cores("edge_hidden", &key)?; + if let Some(second) = second { + s.per_core[second] + .edge_hidden + .push((key.clone(), value.clone())); + } + s.per_core[first].edge_hidden.push((key, value)); + } + for (key, applied) in edge_applied { + let (first, second) = s.edge_cores("edge_applied", &key)?; + if let Some(second) = second { + s.per_core[second].edge_applied.push((key.clone(), applied)); + } + s.per_core[first].edge_applied.push((key, applied)); + } + for (db, tid, key, applied) in tenant_edge_applied { + let (first, second) = s.edge_cores("tenant_edge_applied", &key)?; + if let Some(second) = second { + s.per_core[second] + .tenant_edge_applied + .push((db, tid, key.clone(), applied)); + } + s.per_core[first] + .tenant_edge_applied + .push((db, tid, key, applied)); + } + // A cut covers every edge of its collection, on whichever core it lives. + for share in &mut s.per_core { + share.edge_cuts.extend(edge_cuts.iter().cloned()); + share + .tenant_edge_cuts + .extend(tenant_edge_cuts.iter().cloned()); + } + + Ok(CoreShares { + per_core: s.per_core, + surrogate_pk, + group_write_marks, + proposal_keys: group_proposal_keys, + proposal_keys_complete_from: group_proposal_keys_complete_from, + cut_index: group_cut_index, + event_lane: group_event_lane, + calvin: group_calvin, + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::engine::graph::edge_store::versioned_edge_key; + use crate::types::snapshot::CrdtConstraintEntry; + use nodedb_types::QualifiedCollection; + + const NUM_CORES: usize = 4; + const GROUP: u64 = 3; + + fn cores() -> CoreMap { + CoreMap::from_router(&VShardRouter::round_robin(NUM_CORES)) + } + + /// The core a live write to `collection` in `db` routes to. + fn home(db: u64, collection: &str) -> usize { + let key = nodedb_types::CollectionKey::from_bare(DatabaseId::new(db), collection); + VShardRouter::round_robin(NUM_CORES) + .resolve(key.vshard()) + .unwrap() + } + + fn endpoint_core(node: &str) -> usize { + VShardRouter::round_robin(NUM_CORES) + .resolve(VShardId::from_key(node.as_bytes())) + .unwrap() + } + + /// Collections whose homes cover more than one core, so a one-core + /// install is caught. + fn spread_collections() -> Vec { + let names: Vec = (0..32).map(|i| format!("coll_{i}")).collect(); + let homes: std::collections::HashSet = names.iter().map(|n| home(0, n)).collect(); + assert!(homes.len() > 1, "test collections must span several cores"); + names + } + + #[test] + fn rows_of_many_vshards_land_on_their_owning_cores() { + let names = spread_collections(); + let mut snap = TenantDataSnapshot::default(); + for name in &names { + for doc in 0..3 { + snap.documents + .push((format!("0:1:{name}:doc{doc}"), b"v".to_vec())); + } + snap.indexes + .push((format!("0:1:{name}:f:x:doc0"), Vec::new())); + snap.kv_tables.push((format!("0:1:{name}"), b"kv".to_vec())); + snap.vectors + .push((format!("0:1:{name}:emb"), b"vec".to_vec())); + snap.crdt_state.push((0, 1, name.clone(), b"loro".to_vec())); + snap.crdt_constraints.push(CrdtConstraintEntry { + database_id: 0, + tenant_id: 1, + collection: name.clone(), + version: 1, + constraints: Vec::new(), + }); + } + + let shares = split_by_core(GROUP, snap, &cores()).unwrap(); + assert_eq!(shares.per_core.len(), NUM_CORES); + + let mut documents = 0; + for (core, share) in shares.per_core.iter().enumerate() { + for (key, _) in &share.documents { + let (_, coll) = db_tenant_scoped(key).unwrap(); + assert_eq!(home(0, coll), core, "document {key} on core {core}"); + documents += 1; + } + for (key, _) in &share.indexes { + assert_eq!(home(0, db_tenant_scoped(key).unwrap().1), core); + } + for (key, _) in &share.kv_tables { + assert_eq!(home(0, db_scoped(key).unwrap().1), core); + } + for (key, _) in &share.vectors { + assert_eq!(home(0, db_tenant_scoped(key).unwrap().1), core); + } + for (_, _, coll, _) in &share.crdt_state { + assert_eq!(home(0, coll), core); + } + for entry in &share.crdt_constraints { + assert_eq!(home(0, &entry.collection), core); + } + } + assert_eq!(documents, names.len() * 3, "no document lost or duplicated"); + } + + #[test] + fn a_named_database_routes_by_its_bare_collection_key() { + let db = 1025; + let stored = QualifiedCollection::new(DatabaseId::new(db), "orders"); + let snap = TenantDataSnapshot { + documents: vec![(format!("{db}:1:{}:doc0", stored.as_str()), b"v".to_vec())], + kv_tables: vec![(format!("{db}:1:{}", stored.as_str()), b"kv".to_vec())], + ..Default::default() + }; + let shares = split_by_core(GROUP, snap, &cores()).unwrap(); + let core = home(db, "orders"); + assert_eq!(shares.per_core[core].documents.len(), 1); + assert_eq!(shares.per_core[core].kv_tables.len(), 1); + } + + #[test] + fn an_edge_lands_on_both_endpoint_homes() { + let (src, dst) = (0..64) + .map(|i| (format!("a{i}"), format!("b{i}"))) + .find(|(s, d)| endpoint_core(s) != endpoint_core(d)) + .unwrap(); + let key = versioned_edge_key("g", &src, "L", &dst, 7).unwrap(); + let snap = TenantDataSnapshot { + tenant_edges: vec![(0, 1, key.clone(), b"p".to_vec())], + ..Default::default() + }; + let shares = split_by_core(GROUP, snap, &cores()).unwrap(); + for (core, share) in shares.per_core.iter().enumerate() { + let expected = usize::from(core == endpoint_core(&src) || core == endpoint_core(&dst)); + assert_eq!(share.tenant_edges.len(), expected, "core {core}"); + } + } + + #[test] + fn control_plane_sections_stay_off_the_cores() { + let snap = TenantDataSnapshot { + surrogate_pk: vec![SurrogateBindEntry { + database_id: 0, + tenant_id: 1, + collection: "c".into(), + pk: b"k".to_vec(), + surrogate: 9, + }], + group_write_marks: vec![(1, 2, 0, "c".into(), 0)], + group_proposal_keys: vec![(5, 0xab)], + group_proposal_keys_complete_from: 4, + ..Default::default() + }; + let shares = split_by_core(GROUP, snap, &cores()).unwrap(); + assert_eq!(shares.surrogate_pk.len(), 1); + assert_eq!(shares.group_write_marks.len(), 1); + assert_eq!(shares.proposal_keys, vec![(5, 0xab)]); + assert_eq!(shares.proposal_keys_complete_from, 4); + assert!(shares.per_core.iter().all(|s| s.surrogate_pk.is_empty() + && s.group_write_marks.is_empty() + && s.group_proposal_keys.is_empty())); + } + + #[test] + fn an_unroutable_key_fails_the_split() { + let snap = TenantDataSnapshot { + documents: vec![("not-a-scoped-key".into(), b"v".to_vec())], + ..Default::default() + }; + let err = split_by_core(GROUP, snap, &cores()) + .err() + .expect("an unroutable key must fail"); + assert!(matches!( + err, + SnapshotInstallError::UnroutableKey { + section: "documents", + .. + } + )); + assert!(!err.is_retryable()); + } +} diff --git a/nodedb/src/control/cluster/spsc_applier.rs b/nodedb/src/control/cluster/spsc_applier.rs index 0effd1025..311099e43 100644 --- a/nodedb/src/control/cluster/spsc_applier.rs +++ b/nodedb/src/control/cluster/spsc_applier.rs @@ -10,6 +10,7 @@ use std::sync::{Arc, Mutex}; +use crate::control::cluster::sequencer_compaction::SequencerCompaction; use crate::control::distributed_applier::DistributedApplier; use nodedb_cluster::calvin::{SEQUENCER_GROUP_ID, SequencerStateMachine}; @@ -22,6 +23,9 @@ use nodedb_cluster::calvin::{SEQUENCER_GROUP_ID, SequencerStateMachine}; pub struct SpscCommitApplier { applier: Arc, sequencer_state_machine: Arc>, + /// Compacts the sequencer log once it grew past its threshold. `None` + /// when the group never compacts. + sequencer_compaction: Option>, } impl SpscCommitApplier { @@ -33,23 +37,43 @@ impl SpscCommitApplier { Self { applier, sequencer_state_machine, + sequencer_compaction: None, } } + + /// Compact the sequencer log through `compaction`. + #[must_use] + pub fn with_sequencer_compaction( + mut self, + compaction: Option>, + ) -> Self { + self.sequencer_compaction = compaction; + self + } } impl nodedb_cluster::CommitApplier for SpscCommitApplier { fn apply_committed(&self, group_id: u64, entries: &[nodedb_raft::message::LogEntry]) -> u64 { if group_id == SEQUENCER_GROUP_ID { - let mut sm = self - .sequencer_state_machine - .lock() - .unwrap_or_else(|p| p.into_inner()); - for entry in entries { - if !entry.data.is_empty() { - sm.apply(entry.index, &entry.data); + { + let mut sm = self + .sequencer_state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()); + for entry in entries { + if !entry.data.is_empty() { + sm.apply(entry.index, &entry.data); + } } } - return entries.last().map(|e| e.index).unwrap_or(0); + let applied = entries.last().map(|e| e.index).unwrap_or(0); + // With the state machine's lock released: the run captures it. + if applied > 0 + && let Some(compaction) = self.sequencer_compaction.as_ref() + { + compaction.after_apply(applied); + } + return applied; } self.applier.apply_committed(group_id, entries) } diff --git a/nodedb/src/control/cluster/start_raft/core.rs b/nodedb/src/control/cluster/start_raft/core.rs index 0f3fd380f..0e4d5d2e7 100644 --- a/nodedb/src/control/cluster/start_raft/core.rs +++ b/nodedb/src/control/cluster/start_raft/core.rs @@ -40,15 +40,23 @@ fn bootstrap_listener_addr( /// dispatcher for the `SpscCommitApplier`). Moves the `MultiRaft` out of /// `handle.multi_raft` into the `RaftLoop`; must be called **exactly /// once** per handle. -pub fn start_raft( +pub async fn start_raft( handle: &ClusterHandle, shared: Arc, data_dir: &std::path::Path, transport_tuning: &ClusterTransportTuning, ) -> crate::Result> { - let (multi_raft, setup) = build_group_setup(handle, &shared, data_dir, transport_tuning)?; - let hooks = build_hooks(handle, &shared, data_dir)?; - let loop_build = build_raft_loop(handle, &shared, data_dir, multi_raft, setup, hooks)?; + let (mut multi_raft, setup) = build_group_setup(handle, &shared, data_dir, transport_tuning)?; + let hooks = build_hooks( + handle, + &shared, + data_dir, + &mut multi_raft, + &setup.token_state, + &setup.sequencer_state_machine, + ) + .await?; + let loop_build = build_raft_loop(handle, &shared, data_dir, multi_raft, setup, hooks).await?; let bootstrap_raft_loop = Arc::clone(&loop_build.raft_loop); let bootstrap_token_state = Arc::clone(&loop_build.token_state); @@ -76,6 +84,12 @@ pub fn start_raft( sequencer_service: loop_build.sequencer_service, }, ); + // `finish_observability` installed the metadata raft handle: releases + // handed off by synchronous code can be proposed from here on. + crate::control::lease::releaser::spawn_lease_releaser(&shared); + // The metadata leader reclaims a DDL preparation lease whose owner died, + // left, or got stuck. It proposes through the same handle. + crate::control::metadata_proposer::ddl_reclaim::spawn_ddl_lease_reclaimer(&shared); if let Some(material) = crate::control::cluster::tls::load_bootstrap_issuer_material(data_dir)? { diff --git a/nodedb/src/control/cluster/start_raft/group_setup.rs b/nodedb/src/control/cluster/start_raft/group_setup.rs index 2fa4da5cc..bdf8aebdf 100644 --- a/nodedb/src/control/cluster/start_raft/group_setup.rs +++ b/nodedb/src/control/cluster/start_raft/group_setup.rs @@ -19,9 +19,9 @@ use nodedb_cluster::calvin::{ use crate::control::cluster::array_executor::DataPlaneArrayExecutor; use crate::control::cluster::calvin::ReadResultEvent; use crate::control::cluster::handle::ClusterHandle; -use crate::control::cluster::metadata_applier::MetadataCommitApplier; +use crate::control::cluster::metadata_applier::{MetadataCommitApplier, seed_metadata_cache}; use crate::control::cluster::spsc_applier::SpscCommitApplier; -use crate::control::cluster::start_raft_helpers::build_vshard_handler; +use crate::control::cluster::vshard_envelope_handler::build_vshard_handler; use crate::control::distributed_applier::{ApplyBatch, ProposeTracker, create_distributed_applier}; use crate::control::state::SharedState; @@ -59,10 +59,78 @@ pub(super) fn build_group_setup( data_dir: &std::path::Path, transport_tuning: &nodedb_types::config::tuning::ClusterTransportTuning, ) -> crate::Result<(nodedb_cluster::multi_raft::MultiRaft, GroupSetup)> { - // Move the MultiRaft constructed by `start_cluster` into this - // function. Rebuilding it here from the routing table would lose - // learner membership for joining nodes and would double-open - // per-group redb log files. + let multi_raft = take_multi_raft(handle)?; + + // Build the propose tracker and distributed applier. + // + // The tracker is wired with the per-group apply watermark registry. The + // apply loop bumps it through the tracker once every entry of a group up + // to an index finished, so proposers and cross-node visibility waits read + // one in-order "data applied on this node" signal. + // Its waiter window is the statement deadline: no waiter outlives it, so + // stored results and committed keys older than it serve nobody. + let tracker = Arc::new( + ProposeTracker::new() + .with_group_watchers(handle.group_watchers.clone()) + .with_waiter_window(std::time::Duration::from_secs( + shared.tuning.network.default_deadline_secs, + )), + ); + let (dist_applier, apply_rx) = create_distributed_applier(tracker.clone()); + let dist_applier = Arc::new(dist_applier); + let calvin = build_calvin_state(handle, shared); + + // Install the propose tracker so CP dispatch paths can await commit. + if shared.propose_tracker.set(tracker.clone()).is_err() { + tracing::warn!("propose_tracker already set — start_raft appears to have run twice"); + } + + let data_applier = SpscCommitApplier::new( + shared.clone(), + dist_applier, + Arc::clone(&calvin.sequencer_state_machine), + ); + + let (metadata_applier, token_state) = build_metadata_applier(handle, shared)?; + + // LocalPlanExecutor is the sole physical-plan execution path. + let plan_executor = Arc::new(crate::control::LocalPlanExecutor::new(shared.clone())); + + let vshard_handler = build_envelope_handler(handle, shared, data_dir)?; + + let tick_interval = Duration::from_millis(transport_tuning.raft_tick_interval_ms); + let (snapshot_chunk_bytes, orphan_partial_max_age_secs) = + load_snapshot_transfer_config(handle)?; + let replication_factor = load_replication_factor(handle)?; + + let setup = GroupSetup { + tracker, + data_applier, + apply_rx, + calvin_completion_registry: calvin.completion_registry, + calvin_verdict_rx: calvin.verdict_rx, + sequencer_state_machine: calvin.sequencer_state_machine, + calvin_read_result_senders: calvin.read_result_senders, + metadata_applier, + token_state, + plan_executor, + vshard_handler, + tick_interval, + snapshot_chunk_bytes, + orphan_partial_max_age_secs, + replication_factor, + }; + + Ok((multi_raft, setup)) +} + +/// Move the `MultiRaft` constructed by `start_cluster` out of `handle`, and +/// add the sequencer group on a bootstrap or restart node. +/// +/// Rebuilding the `MultiRaft` here from the routing table will lose +/// learner membership for joining nodes and will double-open per-group +/// redb log files. +fn take_multi_raft(handle: &ClusterHandle) -> crate::Result { let mut multi_raft = handle .multi_raft .lock() @@ -74,7 +142,7 @@ pub(super) fn build_group_setup( // Bootstrap/restart nodes create the sequencer group here. A fresh joiner // already reconstructed it as a learner from JoinResponse; replacing that - // group with a topology-derived voter set would fork the Raft membership. + // group with a topology-derived voter set will fork the Raft membership. if !multi_raft.contains_group(SEQUENCER_GROUP_ID) { let sequencer_peers: Vec = { let topo = handle.topology.read().unwrap_or_else(|p| p.into_inner()); @@ -89,22 +157,39 @@ pub(super) fn build_group_setup( detail: format!("sequencer raft group add: {e}"), })?; } + Ok(multi_raft) +} - // Build the propose tracker and distributed applier. - // - // The tracker is wired with the per-group apply watermark registry. The - // apply loop bumps it through the tracker once every entry of a group up - // to an index finished, so proposers and cross-node visibility waits read - // one in-order "data applied on this node" signal. - let tracker = - Arc::new(ProposeTracker::new().with_group_watchers(handle.group_watchers.clone())); - let (dist_applier, apply_rx) = create_distributed_applier(tracker.clone()); - let dist_applier = Arc::new(dist_applier); +/// The Calvin state phase 1 builds. +struct CalvinState { + completion_registry: Arc, + verdict_rx: mpsc::Receiver<(TxnId, VerdictOutcome)>, + sequencer_state_machine: Arc>, + read_result_senders: Arc>>>, +} + +/// Build the Calvin completion registry, its verdict channel, and the +/// sequencer state machine. +fn build_calvin_state(handle: &ClusterHandle, shared: &Arc) -> CalvinState { // Verdict signal channel: the registry emits `(txn, commit)` on a complete // vote tally; the sequencer service's leader-guarded arm proposes the // `Verdict`. Bounded to match the scheduler completion channel capacity. let (calvin_verdict_tx, calvin_verdict_rx) = mpsc::channel(512); let calvin_completion_registry = CalvinCompletionRegistry::new(calvin_verdict_tx); + // A coordinator registers its completion waiter within its statement + // deadline, so a terminal entry still waiterless after two deadlines + // never gains one, and is evicted with its ack results. + // A coordinator drains its applied response within the same deadline, so + // a sidecar deposit still undrained after two deadlines never is. + let completion_ttl = std::time::Duration::from_secs( + shared + .tuning + .network + .default_deadline_secs + .saturating_mul(2), + ); + calvin_completion_registry.set_waiterless_ttl(completion_ttl); + shared.calvin.apply_results.set_ttl(completion_ttl); // Escalation for an unrecoverable sequencer epoch regression: a NEW // committed entry re-minted an epoch this replica already consumed, so the // state machine halts rather than alias committed transaction identities. @@ -112,7 +197,7 @@ pub(super) fn build_group_setup( // The escalation is scoped to what was actually lost. Sequencing stops — // the service sheds queued submissions so Calvin writers fail fast instead // of hanging — while reads, metadata, and every non-Calvin write path keep - // serving. Stopping the process instead would turn a subsystem fault into a + // serving. Stopping the process instead will turn a subsystem fault into a // full outage (on a single-node deployment, total unavailability), which is // a strictly worse failure than the one being escalated, and it destroys the // running node an operator needs in order to diagnose the divergence. The @@ -120,7 +205,7 @@ pub(super) fn build_group_setup( // never pass for a healthy node. // // Held weakly — `SharedState` reaches this state machine through the - // compactor closure, and a strong capture here would close that into a + // compactor closure, and a strong capture here will close that into a // reference cycle that pins `SharedState` forever. let shared_for_halt = Arc::downgrade(shared); let node_id_for_halt = handle.node_id; @@ -145,22 +230,32 @@ pub(super) fn build_group_setup( std::collections::HashMap::new(), Arc::clone(&calvin_completion_registry), ) - .with_unrecoverable_hook(unrecoverable_hook), + .with_unrecoverable_hook(unrecoverable_hook) + .with_restore_point_hook(crate::control::pitr::restore_point::sequencer_hook( + Arc::downgrade(shared), + )) + .with_cut_instant_hook(crate::control::state::cut_instant_hook(Arc::downgrade( + shared, + ))), )); - let calvin_read_result_senders = - Arc::new(Mutex::new(BTreeMap::>::new())); - - // Install the propose tracker so CP dispatch paths can await commit. - if shared.propose_tracker.set(tracker.clone()).is_err() { - tracing::warn!("propose_tracker already set — start_raft appears to have run twice"); + let read_result_senders = Arc::new(Mutex::new(BTreeMap::>::new())); + CalvinState { + completion_registry: calvin_completion_registry, + verdict_rx: calvin_verdict_rx, + sequencer_state_machine, + read_result_senders, } +} - let data_applier = SpscCommitApplier::new( - shared.clone(), - dist_applier, - Arc::clone(&sequencer_state_machine), - ); - +/// Build the production metadata applier and the join-token mirror it +/// shares, and re-admit every unexpired enrollment preauthorization. +fn build_metadata_applier( + handle: &ClusterHandle, + shared: &Arc, +) -> crate::Result<( + Arc, + nodedb_cluster::SharedTokenStateMirror, +)> { // Production metadata applier: writes to the shared cache, // writes back to the `SystemCatalog` redb so every non-cache // reader observes the change, bumps the applied-index watcher, @@ -175,6 +270,10 @@ pub(super) fn build_group_setup( .map(|state| (state.token_hash, state)) .collect(), )); + // Leases and the cluster version load from their rows before any entry + // applies. `SharedState::open` already seeded drains, pending DDL records, + // and the DDL preparation owner. + seed_metadata_cache(&handle.metadata_cache, shared.credentials.catalog())?; let metadata_applier_concrete = Arc::new(MetadataCommitApplier::new( handle.metadata_cache.clone(), shared.catalog_change_tx.clone(), @@ -204,12 +303,17 @@ pub(super) fn build_group_setup( // Install the Weak before the raft loop starts // ticking so no commit can reach the applier without it. metadata_applier_concrete.install_shared(Arc::downgrade(shared)); - let metadata_applier: Arc = - metadata_applier_concrete.clone(); - - // LocalPlanExecutor is the C-β physical-plan execution path (C-δ.6: sole execution path). - let plan_executor = Arc::new(crate::control::LocalPlanExecutor::new(shared.clone())); + let metadata_applier: Arc = metadata_applier_concrete; + Ok((metadata_applier, token_state)) +} +/// Build the vshard envelope handler: array shard RPCs and Event-Plane +/// envelopes. +fn build_envelope_handler( + handle: &ClusterHandle, + shared: &Arc, + data_dir: &std::path::Path, +) -> crate::Result { // Build the real ArrayLocalExecutor that bridges incoming array shard RPCs // into the local Data Plane via the SPSC bridge, then the vshard handler // that wraps it. @@ -223,42 +327,56 @@ pub(super) fn build_group_setup( // resolve at message time — an inbound NOTIFY can never be silently // dropped for want of a store it does not use. // - // The HWM store roots under this node's `data_dir` ARGUMENT, not + // The dedup store roots under this node's `data_dir` ARGUMENT, not // `shared.data_dir`: the same path `build_hooks` and `build_raft_loop` // root under. `SharedState`'s field is not that path everywhere a node is // constructed, and rooting a redb file at the wrong one makes two nodes in // a single process open the same file — the second fails to acquire its // lock and the whole node fails to start. + // + // The same store is installed on `SharedState`: every committed redo that + // carries a cross-shard key records it there as it applies. Keys the WAL + // holds but the store missed (a crash between the two) are restored here, + // before this node applies or serves anything. + let cross_shard_dedup = Arc::new(crate::event::cross_shard::CrossShardDedup::open(data_dir)?); + cross_shard_dedup.restore_from_wal(&shared.wal)?; + shared + .cross_shard_dedup + .set(Arc::clone(&cross_shard_dedup)) + .map_err(|_| crate::Error::Internal { + detail: "the cross-shard dedup store was already installed".into(), + })?; let cross_shard_receiver = Arc::new(crate::event::cross_shard::CrossShardReceiver::new( - Arc::new(crate::event::cross_shard::HwmStore::open(data_dir)?), + cross_shard_dedup, Arc::clone(shared), Arc::new(crate::event::cross_shard::CrossShardMetrics::new()), handle.node_id, )); - let vshard_handler = build_vshard_handler(array_executor, cross_shard_receiver); - - let tick_interval = Duration::from_millis(transport_tuning.raft_tick_interval_ms); + Ok(build_vshard_handler(array_executor, cross_shard_receiver)) +} - // Read snapshot-transfer config from the pending subsystem config before - // the raft_loop is constructed (pending is consumed after the loop). - let (snapshot_chunk_bytes, orphan_partial_max_age_secs) = { - let guard = handle - .pending_subsystems - .lock() - .unwrap_or_else(|p| p.into_inner()); - let cfg = guard.as_ref().ok_or_else(|| crate::Error::Config { - detail: "start_raft called twice: pending_subsystems already consumed".into(), - })?; - ( - cfg.config.install_snapshot_chunk_bytes, - cfg.config.orphan_partial_max_age_secs, - ) - }; +/// Read the snapshot-transfer config from the pending subsystem config. +/// It is read before the raft loop is constructed: pending is consumed +/// after the loop. +fn load_snapshot_transfer_config(handle: &ClusterHandle) -> crate::Result<(u64, u64)> { + let guard = handle + .pending_subsystems + .lock() + .unwrap_or_else(|p| p.into_inner()); + let cfg = guard.as_ref().ok_or_else(|| crate::Error::Config { + detail: "start_raft called twice: pending_subsystems already consumed".into(), + })?; + Ok(( + cfg.config.install_snapshot_chunk_bytes, + cfg.config.orphan_partial_max_age_secs, + )) +} - // Load the replication factor from persisted cluster settings. This - // function is always called after bootstrap has written those settings — - // `None` here indicates the node was never bootstrapped, which is an - // invariant violation (not a recoverable condition). +/// Load the replication factor from persisted cluster settings. This +/// function is always called after bootstrap has written those settings — +/// `None` here indicates the node was never bootstrapped, which is an +/// invariant violation (not a recoverable condition). +fn load_replication_factor(handle: &ClusterHandle) -> crate::Result { let replication_factor = match handle.catalog.load_cluster_settings().map_err(|e| { crate::Error::Config { detail: format!("start_raft: failed to load cluster settings: {e}"), @@ -267,7 +385,7 @@ pub(super) fn build_group_setup( Some(s) => s.replication_factor, None => { // Settings not yet persisted on this path — fall back to the - // in-memory config RF (the same value bootstrap would persist). + // in-memory config RF (the same value bootstrap will persist). // Error only if neither source is available. handle .pending_subsystems @@ -280,24 +398,5 @@ pub(super) fn build_group_setup( })? } }; - - let setup = GroupSetup { - tracker, - data_applier, - apply_rx, - calvin_completion_registry, - calvin_verdict_rx, - sequencer_state_machine, - calvin_read_result_senders, - metadata_applier, - token_state, - plan_executor, - vshard_handler, - tick_interval, - snapshot_chunk_bytes, - orphan_partial_max_age_secs, - replication_factor, - }; - - Ok((multi_raft, setup)) + Ok(replication_factor) } diff --git a/nodedb/src/control/cluster/start_raft/hooks.rs b/nodedb/src/control/cluster/start_raft/hooks.rs index a025b3089..e838d80d8 100644 --- a/nodedb/src/control/cluster/start_raft/hooks.rs +++ b/nodedb/src/control/cluster/start_raft/hooks.rs @@ -1,8 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 //! Phase 2 of `start_raft`: build the snapshot quarantine hook, the -//! per-group snapshot builder/applier (including follower boot-restore of -//! any persisted `.snap` files), and the cross-node shuffle / surrogate / +//! per-group snapshot builder/applier (including boot completion of any +//! interrupted snapshot install), and the cross-node shuffle / surrogate / //! Calvin routing hooks that bridge `RaftLoop` callbacks to `SharedState`. use std::sync::Arc; @@ -28,19 +28,22 @@ pub(super) struct Hooks { pub(super) calvin_submit_inbox: Arc, pub(super) reserve_read: Arc, pub(super) release_reservation: Arc, + /// The sequencer group's kept snapshot, which its own compaction writes. + pub(super) sequencer_snapshots: + Arc, } -/// Build every cross-plane hook `RaftLoop` needs, including the follower -/// boot-restore of persisted snapshots (which must run before -/// `run_apply_loop` is spawned, since the leader's log-compaction discards -/// the pre-snapshot log prefix the apply loop would otherwise need to -/// replay). `start_raft` itself is sync, so the async restore call is driven -/// via `block_in_place` + `block_on`, matching the surrounding style used -/// for other cluster subsystems rather than introducing a new runtime entry. -pub(super) fn build_hooks( +/// Build every cross-plane hook `RaftLoop` needs, including the boot +/// completion of interrupted snapshot installs, which must run before +/// `run_apply_loop` is spawned. The recovery is awaited, so the boot runs +/// on any runtime flavor. +pub(super) async fn build_hooks( handle: &ClusterHandle, shared: &Arc, data_dir: &std::path::Path, + multi_raft: &mut nodedb_cluster::multi_raft::MultiRaft, + token_state: &nodedb_cluster::SharedTokenStateMirror, + sequencer_state_machine: &Arc>, ) -> crate::Result { let quarantine_hook = Arc::new( crate::control::cluster::snapshot_hook::RaftSnapshotQuarantineHook { @@ -48,44 +51,75 @@ pub(super) fn build_hooks( }, ); + // The sequencer group's snapshot: its state machine, captured on the + // send path and kept durably on the receive path. A node whose sequencer + // log starts right after an installed snapshot restores the state + // machine from it before any entry applies. + let sequencer_snapshots = Arc::new( + crate::control::cluster::sequencer_snapshot::SequencerSnapshotStore::new( + Arc::clone(sequencer_state_machine), + data_dir, + ), + ); + if sequencer_snapshots.restore_at_boot(multi_raft)? { + info!( + node_id = handle.node_id, + "restored the sequencer state machine from its installed snapshot" + ); + } + + // A data group mounted here takes log entries only once its vShards' + // Calvin state reaches the sequencer log. Until then its leader sends a + // snapshot, which carries the Calvin cut its storage holds. + crate::control::cluster::calvin_snapshot::install_snapshot_requirement( + shared, + Arc::clone(&handle.routing), + multi_raft, + )?; + // Per-group snapshot builder for the SEND path: on the leader, build the // real serialized engine state for a lagging follower's group vshards // (replacing the prior empty stub bytes). let snapshot_builder: Arc = Arc::new( - crate::control::cluster::snapshot_builder::DataPlaneSnapshotBuilder::new(shared.clone()), + crate::control::cluster::snapshot_builder::DataPlaneSnapshotBuilder::new(shared.clone()) + .with_sequencer(Arc::clone(&sequencer_snapshots)), ); // Per-group snapshot applier for the RECEIVE path: on the follower, apply a // received per-group snapshot to the local Data-Plane state machine (via the // existing restore handler with replace_mode = true) before Raft advances. - let snapshot_applier_concrete = - crate::control::cluster::snapshot_applier::DataPlaneSnapshotApplier::new(shared.clone()); - - // Follower boot-restore: re-install any persisted `.snap` snapshots from a - // prior run BEFORE the apply loop is spawned. The leader's log-compaction - // discards the pre-snapshot prefix, so the post-snapshot log tail the apply - // loop will replay can NOT reconstruct that prefix — the persisted snapshot - // is the only source for it. Must precede `run_apply_loop` for that reason. - // Match the surrounding block_in_place style used for other cluster - // subsystems rather than introducing a new runtime entry. - let restored = tokio::task::block_in_place(|| { - tokio::runtime::Handle::current().block_on( - crate::control::cluster::boot_restore::restore_persisted_snapshots( - data_dir, - &snapshot_applier_concrete, - ), - ) + let snapshot_applier: Arc = Arc::new( + crate::control::cluster::snapshot_applier::DataPlaneSnapshotApplier::new(shared.clone()) + .with_metadata( + Arc::clone(&handle.catalog), + crate::control::cluster::metadata_image::RaftOwnedState { + token_state: Arc::clone(token_state), + transport: Some(Arc::clone(&handle.transport)), + }, + ) + .with_sequencer(Arc::clone(&sequencer_snapshots)), + ); + + // Complete every snapshot install a crash interrupted, before the apply + // loop starts: an install whose Raft boundary never moved is applied + // again, then adopted. A finished install is durable on its own and is + // never applied at boot. + let completed = nodedb_cluster::install_snapshot::recover_staged_installs( + data_dir, + multi_raft, + Some(snapshot_applier.as_ref()), + ) + .await + .map_err(|e| crate::Error::Internal { + detail: format!("boot recovery of staged snapshot installs: {e}"), })?; - if restored > 0 { + if completed > 0 { info!( node_id = handle.node_id, - restored, "follower boot-restore re-installed persisted snapshots" + completed, "completed interrupted snapshot installs" ); } - let snapshot_applier: Arc = - Arc::new(snapshot_applier_concrete); - // Cross-node streaming-shuffle receiver (E1): bridge the cluster // `ShufflePush` read-loop to the in-process registry on `SharedState`. let shuffle_receiver: Arc = Arc::new( @@ -160,5 +194,6 @@ pub(super) fn build_hooks( calvin_submit_inbox, reserve_read, release_reservation, + sequencer_snapshots, }) } diff --git a/nodedb/src/control/cluster/start_raft/loop_build.rs b/nodedb/src/control/cluster/start_raft/loop_build.rs index 9607675f3..2c53e7195 100644 --- a/nodedb/src/control/cluster/start_raft/loop_build.rs +++ b/nodedb/src/control/cluster/start_raft/loop_build.rs @@ -58,14 +58,19 @@ pub(super) struct LoopBuild { /// Build the `RaftLoop`, consume `pending_subsystems`, build the sequencer /// service + OLLP orchestrator, spawn the vShard schedulers, and start the /// cluster subsystems that share the loop's `MultiRaft`. -pub(super) fn build_raft_loop( +pub(super) async fn build_raft_loop( handle: &ClusterHandle, shared: &Arc, data_dir: &std::path::Path, - multi_raft: nodedb_cluster::multi_raft::MultiRaft, + mut multi_raft: nodedb_cluster::multi_raft::MultiRaft, setup: GroupSetup, hooks: Hooks, ) -> crate::Result { + // Metadata entries this node stamps as leader carry its HLC. The clock + // starts above every stamp this node applied, so a stamp it takes rises + // above entries compacted out of its log. + crate::control::cluster::metadata_stamp::fold_metadata_stamp_hwm(shared)?; + multi_raft.set_metadata_clock(Arc::clone(&shared.hlc_clock)); let GroupSetup { tracker, data_applier, @@ -112,6 +117,16 @@ pub(super) fn build_raft_loop( // ack against this node's schedulers. calvin_completion_registry.applied_acks.enable(); + // The sequencer log compacts at the threshold every group runs with, + // once its state machine's capture is durable. + let data_applier = data_applier.with_sequencer_compaction( + crate::control::cluster::sequencer_compaction::SequencerCompaction::new( + Arc::clone(&hooks.sequencer_snapshots), + shared, + multi_raft.log_compaction_threshold(), + ), + ); + let raft_loop = Arc::new( nodedb_cluster::RaftLoop::new( multi_raft, @@ -122,8 +137,12 @@ pub(super) fn build_raft_loop( .with_plan_executor(plan_executor) .with_metadata_applier(metadata_applier) .with_metadata_cache(shared.metadata_cache.clone()) + .with_lease_holder_liveness(Arc::clone(&shared.lease_runtime.holder_liveness)) .with_vshard_handler(vshard_handler) .with_tick_interval(tick_interval) + // The routing table, the cluster epoch and a join's topology are + // saved to the same catalog the node restarts from. + .with_catalog(Arc::clone(&handle.catalog)) .with_group_watchers(handle.group_watchers.clone()) .with_snapshot_quarantine_hook(hooks.quarantine_hook) .with_snapshot_builder(hooks.snapshot_builder) @@ -158,8 +177,14 @@ pub(super) fn build_raft_loop( detail: "start_raft called twice: pending_subsystems already consumed".into(), })?; let raft_loop_handle = raft_loop.multi_raft_handle(); + crate::control::pitr::spawn_metadata_log_archiver(shared, raft_loop_handle.clone()); let sequencer_config = SequencerConfig::default(); + sequencer_config + .validate() + .map_err(|e| crate::Error::Config { + detail: e.to_string(), + })?; let (sequencer_inbox, sequencer_inbox_rx) = new_inbox(10_000, &sequencer_config); let (reservation_inbox, reservation_inbox_rx) = new_reservation_inbox(10_000); let ollp_orchestrator = Arc::new(OllpOrchestrator::new(OllpConfig::default())); @@ -194,16 +219,23 @@ pub(super) fn build_raft_loop( scheduler_config: &scheduler_config, })?; - let running = tokio::task::block_in_place(|| { - tokio::runtime::Handle::current().block_on(nodedb_cluster::start_cluster_subsystems( - &pending.config, - Arc::clone(&handle.topology), - Arc::clone(&handle.routing), - Arc::clone(&handle.transport), - raft_loop_handle, - &handle.catalog, - )) - }) + let running = nodedb_cluster::start_cluster_subsystems( + &pending.config, + nodedb_cluster::SubsystemHandles { + topology: Arc::clone(&handle.topology), + routing: Arc::clone(&handle.routing), + transport: Arc::clone(&handle.transport), + raft_multi_raft: raft_loop_handle, + }, + &handle.catalog, + nodedb_cluster::SwimWiring { + transport: Arc::clone(&pending.swim_transport), + subscribers: vec![Arc::clone(&shared.lease_runtime.holder_liveness) + as Arc], + }, + Arc::clone(&handle.migration_tracker), + ) + .await .map_err(|e| crate::Error::Config { detail: format!("cluster subsystem start: {e}"), })?; diff --git a/nodedb/src/control/cluster/start_raft/mod.rs b/nodedb/src/control/cluster/start_raft/mod.rs index b2fc86371..76838459d 100644 --- a/nodedb/src/control/cluster/start_raft/mod.rs +++ b/nodedb/src/control/cluster/start_raft/mod.rs @@ -7,7 +7,7 @@ //! tracker, distributed applier, Calvin state, metadata applier, plan/array //! executors, and vshard handler. //! - [`hooks`]: build the snapshot quarantine hook, per-group snapshot -//! builder/applier (incl. follower boot-restore), and the shuffle/ +//! builder/applier (incl. boot completion of interrupted installs), and the shuffle/ //! surrogate/Calvin routing hooks. //! - [`loop_build`]: construct `RaftLoop`, consume `pending_subsystems`, //! build the Calvin sequencer service, spawn vShard schedulers, and start diff --git a/nodedb/src/control/cluster/start_raft/observability.rs b/nodedb/src/control/cluster/start_raft/observability.rs index 8ca55225a..f17d0f41d 100644 --- a/nodedb/src/control/cluster/start_raft/observability.rs +++ b/nodedb/src/control/cluster/start_raft/observability.rs @@ -57,6 +57,39 @@ pub(super) fn finish_observability( .set(calvin_completion_registry); let _ = shared.ollp_orchestrator.set(ollp_orchestrator); + publish_raft_status(handle, shared, &raft_loop); + spawn_auth_lease_renew(shared); + publish_epoch_and_proposer(shared, &raft_loop); + start_surrogate_refiller(shared); + + // Subscribe to the boot-time readiness watch BEFORE spawning the + // tick loop so we cannot miss the first transition. The receiver + // is returned to `main.rs`, which awaits it before binding any + // client-facing listener. + let ready_rx = raft_loop.subscribe_ready(); + + // Register the raft-tick loop's standardized metrics so the + // `/metrics` route can expose them alongside every other driver. + shared + .loop_metrics_registry + .register(raft_loop.loop_metrics()); + + spawn_raft_services(handle, shared, &raft_loop, sequencer_service); + log_cluster_version_view(handle, shared); + start_health_monitor(handle, shared, transport_tuning); + + info!(node_id = handle.node_id, "raft loop and RPC server started"); + + ready_rx +} + +/// Publish the cluster observer, the live Raft leader status, the metadata +/// term and contact probes, and the read gate onto `SharedState`. +fn publish_raft_status( + handle: &ClusterHandle, + shared: &Arc, + raft_loop: &Arc, +) { // Publish the cluster observability handle to SharedState before // any listener starts serving. let observer = Arc::new(nodedb_cluster::ClusterObserver::new( @@ -82,9 +115,8 @@ pub(super) fn finish_observability( // `group_statuses()` snapshot. // Weak for the same cycle-breaking reason as `cluster_observer` above. // A dropped loop (only reachable post-shutdown) yields an empty status - // snapshot, which every consumer already treats as "no cluster groups" - // — identical to the single-node case where this fn is never installed. - let raft_loop_for_status = Arc::downgrade(&raft_loop); + // snapshot, which every consumer treats as "no cluster groups". + let raft_loop_for_status = Arc::downgrade(raft_loop); if shared .raft_status_fn .set(Arc::new(move || { @@ -98,6 +130,39 @@ pub(super) fn finish_observability( tracing::warn!("raft_status_fn already set — start_raft appears to have run twice"); } + // The lease drainer polls the metadata term every few milliseconds, so it + // reads that one group rather than every group's status. + let multi_raft_for_term = raft_loop.multi_raft_handle(); + if shared + .lease_runtime + .metadata_leader_term + .set(Arc::new(move || { + multi_raft_for_term + .lock() + .unwrap_or_else(|p| p.into_inner()) + .leader_term(nodedb_cluster::METADATA_GROUP_ID) + })) + .is_err() + { + tracing::warn!( + "metadata leader term fn already set — start_raft appears to have run twice" + ); + } + let multi_raft_for_contact = raft_loop.multi_raft_handle(); + if shared + .lease_runtime + .metadata_contact + .set(Arc::new(move |window| { + multi_raft_for_contact + .lock() + .unwrap_or_else(|p| p.into_inner()) + .leader_contact_within(nodedb_cluster::METADATA_GROUP_ID, window) + })) + .is_err() + { + tracing::warn!("metadata contact fn already set — start_raft appears to have run twice"); + } + // Publish the leadership confirmer so a linearizable read served on this // node proves against a quorum that it is still the leader. Holds the // same coordinator mutex the loop ticks, and the loop only weakly, to ask @@ -105,14 +170,16 @@ pub(super) fn finish_observability( let gate: Arc = Arc::new(crate::control::cluster::read_index::MultiRaftReadGate::new( raft_loop.multi_raft_handle(), - Arc::downgrade(&raft_loop), + Arc::downgrade(raft_loop), )); if shared.raft_read_gate.set(gate).is_err() { tracing::warn!("raft_read_gate already set — start_raft appears to have run twice"); } +} - // Renew this node's authorization lease with the metadata leader. Its - // confirmed coverage needs the read gate above. +/// Renew this node's authorization lease with the metadata leader. Its +/// confirmed coverage needs the read gate [`publish_raft_status`] sets. +fn spawn_auth_lease_renew(shared: &Arc) { if let Some(timing) = shared.authorization_fence.timing() { let renew_state = Arc::clone(shared); crate::control::shutdown::spawn_loop( @@ -129,10 +196,13 @@ pub(super) fn finish_observability( }, ); } +} +/// Publish the cluster-epoch state and the metadata proposer handle. +fn publish_epoch_and_proposer(shared: &Arc, raft_loop: &Arc) { // Publish this node's cluster-epoch state so the routing gate can tell // whether this node has missed a topology transition before it coordinates - // work on a view of the cluster that may already be superseded. + // work on a view of the cluster that can already be superseded. if shared .cluster_epoch .set(raft_loop.cluster_epoch_handle()) @@ -149,13 +219,24 @@ pub(super) fn finish_observability( if shared.metadata_raft.set(proposer_handle).is_err() { tracing::warn!("metadata_raft already set — start_raft appears to have run twice"); } +} +/// Install the surrogate assigner's cluster wiring and spawn its +/// reservation refiller. +fn start_surrogate_refiller(shared: &Arc) { // Allow the surrogate assigner's flush path to propose // `SurrogateAlloc` entries to the Raft group so followers advance // their in-memory HWM on every checkpoint. shared .surrogate_assigner .install_shared(Arc::downgrade(shared)); + // A cluster node resolves every key it plans or reads at the key's + // collection home, never with a value of its own. + shared.surrogate_assigner.install_home_authority(Arc::new( + crate::control::server::surrogate_exchange::RoutedHomeAuthority::new(Arc::downgrade( + shared, + )), + )); // Spawn the per-node surrogate reservation refiller. It owns ALL batch // reservation so the latency-critical `assign` insert path never blocks @@ -183,19 +264,15 @@ pub(super) fn finish_observability( info!("surrogate refill loop stopped"); }, ); +} - // Subscribe to the boot-time readiness watch BEFORE spawning the - // tick loop so we cannot miss the first transition. The receiver - // is returned to `main.rs`, which awaits it before binding any - // client-facing listener. - let ready_rx = raft_loop.subscribe_ready(); - - // Register the raft-tick loop's standardized metrics so the - // `/metrics` route can expose them alongside every other driver. - shared - .loop_metrics_registry - .register(raft_loop.loop_metrics()); - +/// Spawn the Raft tick loop, the sequencer service, and the RPC server. +fn spawn_raft_services( + handle: &ClusterHandle, + shared: &Arc, + raft_loop: &Arc, + sequencer_service: nodedb_cluster::calvin::SequencerService, +) { // Start the Raft tick loop. `RaftLoop::run` takes a raw // `watch::Receiver` and drives shutdown internally, so it gets one // from the canonical watch; the `spawn_loop` receiver is unused. Routing @@ -244,25 +321,31 @@ pub(super) fn finish_observability( } }, ); +} - // Wire version of every node is now carried on the live - // `NodeInfo` in `cluster_topology`. Log the derived view for observability. - { - let view = shared.cluster_version_view(); - let compat = crate::control::rolling_upgrade::should_compat_mode(&view); - info!( - node_id = handle.node_id, - nodes = view.node_count, - min_version = view.min_version, - max_version = view.max_version, - mixed = view.is_mixed_version(), - compat_mode = compat, - "cluster version view derived from topology" - ); - } +/// Log the cluster version view. The wire version of every node is carried +/// on the live `NodeInfo` in `cluster_topology`. +fn log_cluster_version_view(handle: &ClusterHandle, shared: &Arc) { + let view = shared.cluster_version_view(); + let compat = crate::control::rolling_upgrade::should_compat_mode(&view); + info!( + node_id = handle.node_id, + nodes = view.node_count, + min_version = view.min_version, + max_version = view.max_version, + mixed = view.is_mixed_version(), + compat_mode = compat, + "cluster version view derived from topology" + ); +} - // Start the health monitor (periodic pings, failure detection, - // topology re-broadcast). +/// Start the health monitor: periodic pings, failure detection, and +/// topology re-broadcast. +fn start_health_monitor( + handle: &ClusterHandle, + shared: &Arc, + transport_tuning: &nodedb_types::config::tuning::ClusterTransportTuning, +) { let health_config = nodedb_cluster::HealthConfig { ping_interval: std::time::Duration::from_secs(transport_tuning.health_ping_interval_secs), failure_threshold: transport_tuning.health_failure_threshold, @@ -290,8 +373,4 @@ pub(super) fn finish_observability( health_monitor.run(health_raw_shutdown).await; }, ); - - info!(node_id = handle.node_id, "raft loop and RPC server started"); - - ready_rx } diff --git a/nodedb/src/control/cluster/start_raft/propose_error.rs b/nodedb/src/control/cluster/start_raft/propose_error.rs index ce7aa5673..35240fa44 100644 --- a/nodedb/src/control/cluster/start_raft/propose_error.rs +++ b/nodedb/src/control/cluster/start_raft/propose_error.rs @@ -76,6 +76,7 @@ pub(super) fn async_propose_error(vshard_id: u32, error: ClusterError) -> crate: | ClusterError::SpatialGather(_) | ClusterError::Bm25Gather(_) | ClusterError::TsGather(_) + | ClusterError::ShufflePush(_) | ClusterError::RemoteUntyped { .. }) => crate::Error::Internal { detail: format!("raft propose (async): {other}"), }, @@ -88,7 +89,10 @@ mod tests { #[test] fn a_missing_leader_is_retryable() { - let error = ClusterError::Raft(RaftError::NotLeader { leader_hint: None }); + let error = ClusterError::Raft(RaftError::NotLeader { + leader_hint: None, + term: 1, + }); assert!(matches!( async_propose_error(3, error), crate::Error::NoLeader { .. } diff --git a/nodedb/src/control/cluster/start_raft/proposer_wiring.rs b/nodedb/src/control/cluster/start_raft/proposer_wiring.rs index d0023522a..9f66a278c 100644 --- a/nodedb/src/control/cluster/start_raft/proposer_wiring.rs +++ b/nodedb/src/control/cluster/start_raft/proposer_wiring.rs @@ -28,6 +28,17 @@ pub(super) fn wire_proposers( calvin_read_result_senders: Arc>>>, sequencer_state_machine: Arc>, ) -> crate::Result<()> { + install_sync_proposer(shared, raft_loop); + install_compactor(shared, raft_loop, sequencer_state_machine); + install_applied_index_sink(shared, raft_loop); + install_apply_gates(shared, raft_loop); + install_async_proposer(shared, raft_loop, &tracker)?; + spawn_apply_loop(shared, tracker, apply_rx, calvin_read_result_senders); + Ok(()) +} + +/// Install the sync `raft_proposer`. +fn install_sync_proposer(shared: &Arc, raft_loop: &Arc) { // Wire the Raft proposer into SharedState so CP dispatch paths // (pgwire, HTTP, array inbound) can route writes through Raft. // Hold `raft_loop` weakly: `SharedState` owns this closure, and the @@ -52,7 +63,14 @@ pub(super) fn wire_proposers( if shared.raft_proposer.set(proposer).is_err() { tracing::warn!("raft_proposer already set — start_raft appears to have run twice"); } +} +/// Install the `raft_compactor`, held below the sequencer's replay range. +fn install_compactor( + shared: &Arc, + raft_loop: &Arc, + sequencer_state_machine: Arc>, +) { // Wire the Raft log-compaction trigger. `run_apply_loop` invokes this // after a committed entry has been durably applied to the Data Plane, // so compaction is gated on the data-plane applied watermark — never @@ -60,7 +78,9 @@ pub(super) fn wire_proposers( // `log_compaction_threshold` is `None`. // Weak for the same cycle-breaking reason as `raft_proposer` above. let raft_loop_for_compact = Arc::downgrade(raft_loop); - let sm_for_compact = Arc::clone(&sequencer_state_machine); + let sm_for_compact = sequencer_state_machine; + // Weak: the compactor lives in `shared`. + let shared_for_compact = Arc::downgrade(shared); let compactor: Arc = Arc::new(move |group_id, applied_index| { let rl = raft_loop_for_compact @@ -79,12 +99,41 @@ pub(super) fn wire_proposers( // cross-shard graph edge means the edge silently vanishes from that // node's index. Floor the compaction boundary strictly below the // lowest armed catch-up so the replay range always survives. + // + // An open multi-part transaction holds the log down the same way: + // a replica that replays the log must meet its header before its + // parts, so the header's index survives until the transaction + // closes. + // + // Two more ranges hold it down. An input a scheduler received + // and has not made durable is gone with a restart unless the log + // keeps it. And a vShard of a group mounted here whose scheduler + // has not started yet replays the log from its Calvin base. let effective_index = if group_id == nodedb_cluster::calvin::SEQUENCER_GROUP_ID { - match sm_for_compact - .lock() - .unwrap_or_else(|p| p.into_inner()) - .min_catch_up_from() - { + let shared = + shared_for_compact + .upgrade() + .ok_or_else(|| crate::Error::Internal { + detail: "raft log compaction: shared state dropped".into(), + })?; + let mirrors = shared.authorization_fence.calvin_mirrors(); + let mut sm = sm_for_compact.lock().unwrap_or_else(|p| p.into_inner()); + let undurable = sm.undurable_floor(|vshard, epoch, position| { + // A vShard with no scheduler here holds no state to lose. + mirrors + .get(vshard) + .is_none_or(|mirror| mirror.is_applied(epoch, position)) + }); + let floor = [ + sm.min_catch_up_from(), + sm.min_open_parts_index(), + undurable, + shared.calvin.bases.replay_floor(), + ] + .into_iter() + .flatten() + .min(); + match floor { // Keep index `m` itself: compaction discards entries at and // below its boundary, and the replay range starts AT `m`. Some(m) => applied_index.min(m.saturating_sub(1)), @@ -106,7 +155,10 @@ pub(super) fn wire_proposers( if shared.raft_compactor.set(compactor).is_err() { tracing::warn!("raft_compactor already set — start_raft appears to have run twice"); } +} +/// Install the durable `raft_applied_index_sink`. +fn install_applied_index_sink(shared: &Arc, raft_loop: &Arc) { // Wire the durable applied-index sink. `run_apply_loop` invokes this for // each committed entry once the write funnel's durable-at-ack barrier has // fsynced that entry's redo record, so the next boot resumes Raft delivery @@ -135,11 +187,36 @@ pub(super) fn wire_proposers( "raft_applied_index_sink already set — start_raft appears to have run twice" ); } +} + +/// Install the per-group apply gates the apply loop takes. +fn install_apply_gates(shared: &Arc, raft_loop: &Arc) { + // The apply loop takes a group's apply gate before each write, so an + // entry a snapshot install covers never reaches the Data Plane after the + // restore. Set before the loop spawns: it reads the gates at start. + let apply_gates = raft_loop + .multi_raft_handle() + .lock() + .unwrap_or_else(|p| p.into_inner()) + .apply_gates(); + if shared.raft_apply_gates.set(apply_gates).is_err() { + tracing::warn!("raft_apply_gates already set — start_raft appears to have run twice"); + } +} +/// Install the `async_raft_proposer` in two phases: propose, then await this +/// node's apply. The admission sequencer holds a vShard's slot across the +/// first phase only. +fn install_async_proposer( + shared: &Arc, + raft_loop: &Arc, + tracker: &Arc, +) -> crate::Result<()> { // Install the async proposer with transparent leader forwarding. // - // Proposes via the data group leader (forwarding to a remote leader if - // needed), then registers a ProposeTracker waiter and awaits apply. + // The first phase proposes via the data group leader (forwarding to a + // remote leader if needed) and registers a ProposeTracker waiter. The + // second phase awaits the apply. // // The ProposeTracker is race-safe: if `run_apply_loop` calls complete() // before register() is called (possible on fast clusters where the entry @@ -151,7 +228,7 @@ pub(super) fn wire_proposers( // Held weakly for the same cycle-breaking reason as `raft_proposer` above: // the proposer lives on `SharedState`. let state_for_proposer = Arc::downgrade(shared); - let async_proposer: Arc = + let async_submit: Arc = Arc::new(move |vshard_id, idempotency_key, data, deadline| { let rl_weak = raft_loop_async.clone(); let tk = tracker_for_proposer.clone(); @@ -178,77 +255,94 @@ pub(super) fn wire_proposers( // `RetryableLeaderChange` instead of leaking a // not-our-payload back to the caller. let rx = tk.register(group_id, log_index, idempotency_key); - let applied = await_local_apply(LocalApplyWait { - state: &state_weak, - tracker: &tk, - group_id, - log_index, - vshard_id, - deadline, - rx, - }) - .await - // Preserve `RetryableLeaderChange` so the gateway - // retry loop can re-propose against the new leader - // — wrapping it in `Dispatch` would hide the - // retryable signal and surface as silent INSERT - // success. Only machinery failures stay wrapped for - // diagnostics; a classified apply verdict keeps its - // client-visible classification. - .map_err(|e| { - if crate::error_classify::is_unclassified_failure(&e) { - crate::Error::Dispatch { - detail: format!("apply error: {e}"), + let applied: crate::control::wal_replication::AppliedWait = Box::pin(async move { + let applied = await_local_apply(LocalApplyWait { + state: &state_weak, + tracker: &tk, + group_id, + log_index, + vshard_id, + deadline, + rx, + }) + .await + // Preserve `RetryableLeaderChange` so the gateway + // retry loop can re-propose against the new leader + // — wrapping it in `Dispatch` will hide the + // retryable signal and surface as silent INSERT + // success. Only machinery failures stay wrapped for + // diagnostics; a classified apply verdict keeps its + // client-visible classification. + .map_err(|e| { + if crate::error_classify::is_unclassified_failure(&e) { + crate::Error::Dispatch { + detail: format!("apply error: {e}"), + } + } else { + e } - } else { - e + }) + // Carry out the write-version the APPLY side stamped, not + // `log_index`. The tracker resolves on the node that applied + // the entry locally, so `write_version` is this replica's own + // post-write `coll_write_lsn` — a WAL LSN, the same domain + // every other feed of that map records in, and the only + // domain the shard-local OCC read validator compares in. The + // raft log index is a per-group counter on a different scale + // entirely; publishing it here made reads validate a WAL LSN + // against a log index. + .map(|applied| (applied.payload, applied.write_version)); + let applied = applied?; + // A write to a vShard homing a permission-tree source is + // acknowledged only once every lease holder covers it, or its + // lease expired. + if let Some(state) = state_weak.upgrade() + && state + .authorization_fence + .sources() + .is_source_vshard(vshard_id) + { + crate::control::security::auth_lease::authorization_barrier( + &state, + vec![nodedb_cluster::GroupCoverage { + group_id, + through: log_index, + }], + ) + .await?; } + Ok(applied) + }); + Ok(crate::control::wal_replication::ProposedWrite { + at: Some(crate::control::wal_replication::ProposedAt { + group_id, + log_index, + }), + applied, }) - // Carry out the write-version the APPLY side stamped, not - // `log_index`. The tracker resolves on the node that applied - // the entry locally, so `write_version` is this replica's own - // post-write `coll_write_lsn` — a WAL LSN, the same domain - // every other feed of that map records in, and the only - // domain the shard-local OCC read validator compares in. The - // raft log index is a per-group counter on a different scale - // entirely; publishing it here made reads validate a WAL LSN - // against a log index. - .map(|applied| (applied.payload, applied.write_version)); - let applied = applied?; - // A write to a vShard homing a permission-tree source is - // acknowledged only once every lease holder covers it, or its - // lease expired. - if let Some(state) = state_weak.upgrade() - && state - .authorization_fence - .sources() - .is_source_vshard(vshard_id) - { - crate::control::security::auth_lease::authorization_barrier( - &state, - vec![nodedb_cluster::GroupCoverage { - group_id, - through: log_index, - }], - ) - .await?; - } - Ok(applied) }) }); - crate::control::vshard_admission::install_async_raft_proposer(shared, async_proposer)?; + crate::control::vshard_admission::install_async_raft_proposer(shared, async_submit) +} +/// Spawn the background apply loop. +fn spawn_apply_loop( + shared: &Arc, + tracker: Arc, + apply_rx: mpsc::Receiver, + calvin_read_result_senders: Arc>>>, +) { // Spawn the background apply loop. It reads from the mpsc channel // pushed by `DistributedApplier::apply_committed`, dispatches to the // Data Plane, and notifies propose waiters. Registered via // `spawn_loop_no_abort` so the Control Plane drain waits for it to exit // (dropping its captured `Arc` deterministically) but NEVER - // force-aborts it — an abort mid-apply would strand + // force-aborts it — an abort mid-apply will strand // committed-but-unapplied entries. It drains at `DrainingControlPlane` // because its applies dispatch to the Data Plane. let apply_state = shared.clone(); - let apply_tracker = tracker.clone(); - let apply_calvin_read_result_senders = Arc::clone(&calvin_read_result_senders); + let apply_tracker = tracker; + let apply_calvin_read_result_senders = calvin_read_result_senders; crate::control::shutdown::spawn_loop_no_abort( &shared.loop_registry, &shared.shutdown, @@ -271,7 +365,6 @@ pub(super) fn wire_proposers( } }, ); - Ok(()) } /// Where a group's pipeline stands, for a propose waiter that timed out: the @@ -366,6 +459,7 @@ async fn await_local_apply( leader_addr: format!( "this node left raft group {group_id} before it applied index {log_index}" ), + leader_term: 0, }); } if tokio::time::Instant::now() >= deadline { diff --git a/nodedb/src/control/cluster/start_raft_helpers.rs b/nodedb/src/control/cluster/start_raft_helpers.rs index ebcd74c32..bfcece5ec 100644 --- a/nodedb/src/control/cluster/start_raft_helpers.rs +++ b/nodedb/src/control/cluster/start_raft_helpers.rs @@ -1,12 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 -use std::pin::Pin; use std::sync::{Arc, Mutex, RwLock}; -use nodedb_cluster::calvin::{CalvinCompletionRegistry, SEQUENCER_GROUP_ID, SequencerStateMachine}; -use nodedb_cluster::distributed_array::{ArrayLocalExecutor, handle_array_shard_rpc}; -use nodedb_cluster::vshard_handler::{DispatchTarget, dispatch_by_type}; -use nodedb_cluster::wire::VShardEnvelope; +use nodedb_cluster::calvin::{CalvinCompletionRegistry, SequencerStateMachine}; use crate::control::cluster::calvin::scheduler::metrics::SchedulerMetrics; use crate::control::cluster::calvin::scheduler::recover_applied; @@ -14,80 +10,9 @@ use crate::control::cluster::calvin::{ RaftSequencerProposer, ReadResultEvent, Scheduler, SchedulerConfig, SchedulerParams, SequencerProposer, }; +use crate::control::cluster::calvin_snapshot; use crate::control::cluster::handle::ClusterHandle; use crate::control::state::SharedState; -use crate::event::cross_shard::CrossShardReceiver; - -/// Build the `VShardEnvelopeHandler` closure used by `RaftLoop`. -/// -/// The closure receives raw envelope bytes from the QUIC transport layer, -/// dispatches based on `msg_type`, and returns a serialized response. -pub(super) fn build_vshard_handler( - array_executor: Arc, - cross_shard_receiver: Arc, -) -> nodedb_cluster::VShardEnvelopeHandler { - Arc::new(move |bytes: Vec| { - let executor = array_executor.clone(); - let receiver = Arc::clone(&cross_shard_receiver); - let fut: Pin< - Box>> + Send>, - > = Box::pin(async move { - let envelope = VShardEnvelope::from_bytes(&bytes).ok_or_else(|| { - nodedb_cluster::error::ClusterError::Codec { - detail: "vshard_handler: failed to deserialize VShardEnvelope".into(), - } - })?; - - let target = dispatch_by_type(&envelope); - match target { - DispatchTarget::ArrayShard => { - let opcode = envelope.msg_type as u32; - let resp_payload = handle_array_shard_rpc( - opcode, - envelope.vshard_id, - &envelope.payload, - &executor, - ) - .await?; - - // Response opcode = request opcode + 1 for all array shard RPCs. - // Resolve the msg_type variant via a minimal scratch envelope parse - // (avoids any unsafe transmute — the `from_bytes` mapping in wire.rs - // is the canonical source of truth for the opcode→variant table). - let resp_opcode = opcode + 1; - let resp_msg_type = resolve_vshard_msg_type(resp_opcode)?; - let resp_envelope = VShardEnvelope::new( - resp_msg_type, - envelope.target_node, - envelope.source_node, - envelope.vshard_id, - resp_payload, - ); - Ok(resp_envelope.to_bytes()) - } - - // `CrossShardEvent` (remote trigger DML) and `NotifyBroadcast` - // (cluster-wide CDC fan-out) both land here. `handle_envelope` - // re-parses the raw bytes and returns a fully-formed response - // envelope — including the error-shaped one for a message type - // that may not arrive as a REQUEST. The `*Ack` variants are such - // a case: every sender reads its Ack as the RESPONSE on the same - // QUIC stream, so an inbound Ack request is a protocol violation, - // not a case to handle. Unlike the ArrayShard arm there is no - // opcode+1 convention to apply: the receiver picks the response - // msg_type per request type itself. - DispatchTarget::EventPlane => Ok(receiver.handle_envelope(bytes).await), - - other => Err(nodedb_cluster::error::ClusterError::Transport { - detail: format!( - "vshard_handler: no handler registered for dispatch target {other:?}" - ), - }), - } - }); - fut - }) -} /// Type alias for the shared per-vShard read-result sender registry. type ReadResultSenders = @@ -109,6 +34,81 @@ fn hosted_vshards(routing: &RwLock, node_id: u64) vshards } +/// Stop the scheduler of every vShard this node served and either no longer +/// hosts or holds a new base for. +/// +/// A node that left a vShard's group holds none of its state. Its scheduler +/// will go on applying the vShard's slices against stale data and acking +/// them. A snapshot install replaced the state a running scheduler started +/// from. Dropping the vShard's sequencer sender closes the scheduler's +/// intake, and the scheduler exits. Its other channels and its lock table go +/// too. The next reconcile that finds the vShard hosted spawns a new +/// scheduler from the vShard's base. +fn retire_vshard_schedulers( + hosted: &[u32], + shared: &Arc, + sequencer_state_machine: &Arc>, + calvin_read_result_senders: &ReadResultSenders, + calvin_completion_registry: &Arc, +) { + let (left, rebased): (Vec, Vec) = { + let mut senders = calvin_read_result_senders + .lock() + .unwrap_or_else(|p| p.into_inner()); + let served: Vec = senders.keys().copied().collect(); + let left: Vec = served + .iter() + .copied() + .filter(|vshard| hosted.binary_search(vshard).is_err()) + .collect(); + let rebased: Vec = calvin_snapshot::rebased(shared, &served) + .into_iter() + .filter(|vshard| !left.contains(vshard)) + .collect(); + for vshard in left.iter().chain(&rebased) { + senders.remove(vshard); + } + (left, rebased) + }; + calvin_snapshot::forget_left(shared, &left); + for &vshard in &rebased { + tracing::info!( + vshard_id = vshard, + "calvin: a snapshot replaced the vShard's state; its scheduler starts again" + ); + } + let left: Vec = left.into_iter().chain(rebased).collect(); + if left.is_empty() { + return; + } + { + let mut sm = sequencer_state_machine + .lock() + .unwrap_or_else(|p| p.into_inner()); + for &vshard in &left { + sm.remove_vshard_sender(vshard); + } + } + for &vshard in &left { + shared + .calvin + .lock_managers + .lock() + .unwrap_or_else(|p| p.into_inner()) + .remove(&vshard); + shared + .calvin + .promotion_senders + .lock() + .unwrap_or_else(|p| p.into_inner()) + .remove(&vshard); + calvin_completion_registry.unregister_verdict_signal_sender(vshard); + // A cut on this node waits only on the schedulers it runs. + shared.calvin.cuts.unregister(vshard); + tracing::info!(vshard_id = vshard, "calvin: the vShard's scheduler stops"); + } +} + /// Parameters for [`reconcile_vshard_schedulers`]. struct ReconcileSchedulersParams<'a> { node_id: u64, @@ -123,18 +123,15 @@ struct ReconcileSchedulersParams<'a> { scheduler_config: &'a SchedulerConfig, } -/// Idempotently ensure a Calvin `Scheduler` is running for every vShard this +/// Idempotently ensure a Calvin `Scheduler` runs for exactly the vShards this /// node currently hosts. /// /// A vShard is considered already-served iff it has a registered read-result -/// sender (the schedulers' presence registry). Only newly-hosted vShards get a -/// fresh scheduler — this pass never double-spawns. Returns the number of NEW +/// sender (the schedulers' presence registry). A served vShard this node no +/// longer hosts has its scheduler stopped first (see +/// [`retire_left_vshard_schedulers`]). Only newly-hosted vShards get a fresh +/// scheduler — this pass never double-spawns. Returns the number of NEW /// schedulers started. -/// -/// This is `add-only`: it never tears down a scheduler for a vShard that has -/// left this node. vShard removal happens via migration / decommission, which -/// own their own teardown path; wiring scheduler removal into that lifecycle is -/// tracked as a separate follow-up. fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate::Result { let ReconcileSchedulersParams { node_id, @@ -148,8 +145,18 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: scheduler_config, } = params; + let hosted = hosted_vshards(routing, node_id); + retire_vshard_schedulers( + &hosted, + shared, + sequencer_state_machine, + calvin_read_result_senders, + calvin_completion_registry, + ); + calvin_snapshot::retain_mounted(shared, raft_loop_handle, routing); let mut spawned = 0usize; - for vshard_id in hosted_vshards(routing, node_id) { + let mut kept = Vec::new(); + for vshard_id in hosted { // Already-served vShards keep their running scheduler untouched. if calvin_read_result_senders .lock() @@ -159,21 +166,58 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: continue; } + // The earliest committed sequencer index still in the retained log — the + // lower bound the spawn-time catch-up (below) arms from. `None` while + // the log has no known start. + let sequencer_start = raft_loop_handle + .lock() + .unwrap_or_else(|p| p.into_inner()) + .sequencer_log_start(); + let group_id = routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .group_for_vshard(vshard_id) + .ok(); + // A scheduler starts only from Calvin state that reaches the log it + // catches up from. The sequencer log keeps that range meanwhile. + if !calvin_snapshot::may_start( + shared, + raft_loop_handle, + vshard_id, + group_id, + sequencer_start, + ) { + continue; + } + let Some(first_available) = sequencer_start else { + continue; + }; + // The applied state the last checkpoint saved, with the markers the // WAL still holds: a checkpoint deletes the segments that held older // markers, and the sequencer log delivers their entries again. - let recovery = recover_applied(&shared.wal, shared.credentials.catalog(), vshard_id)?; + // A failed read leaves the vShard for the next pass. The pass goes + // on, so the schedulers it started keep their bases. + let recovery = match recover_applied( + &shared.wal, + shared.credentials.catalog(), + vshard_id, + group_id, + ) { + Ok(recovery) => recovery, + Err(error) => { + tracing::warn!( + vshard_id, + %error, + "calvin: the vShard's applied state did not load; its scheduler starts \ + on a later pass" + ); + continue; + } + }; let (sequenced_tx, sequenced_rx) = tokio::sync::mpsc::channel(scheduler_config.channel_capacity); - // The earliest committed sequencer index still in the retained log — the - // lower bound the spawn-time catch-up (below) arms from. - let first_available = raft_loop_handle - .lock() - .unwrap_or_else(|p| p.into_inner()) - .first_available_index(SEQUENCER_GROUP_ID) - .unwrap_or(1); - { let mut sm = sequencer_state_machine .lock() @@ -182,10 +226,10 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: // Arm a spawn-time catch-up BEFORE the scheduler runs. A scheduler // subscribes only once this node's membership in the vShard's data // group lands (a late, reconcile-driven event on a forming cluster), - // by which point the sequencer may have already committed — and + // by which point the sequencer can have already committed — and // fanned out to a then-absent sender, i.e. SILENTLY skipped — epochs // for this vShard. A fresh replica has nothing durably applied to - // rebuild from, so it would otherwise consider itself caught up and + // rebuild from, so it will otherwise consider itself caught up and // never receive those txns (cross-shard graph edges among them), // losing them permanently after it becomes leader. Arming from the // first available index makes the scheduler's drain replay every @@ -194,6 +238,8 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: // no-ops). sm.arm_catch_up_from(vshard_id, first_available); } + // The armed catch-up holds the log from here, so the base is kept. + kept.push((vshard_id, first_available)); let (read_result_tx, read_result_rx) = tokio::sync::mpsc::channel(scheduler_config.channel_capacity); @@ -263,7 +309,7 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: // waits for the scheduler to exit (dropping its captured // `Arc` deterministically) but NEVER force-aborts it: a // Calvin `Scheduler` advances a replicated state machine and a - // mid-epoch `.abort()` would diverge this node from its peers. + // mid-epoch `.abort()` will diverge this node from its peers. // `Scheduler::run` already breaks at an epoch-safe boundary (its // `biased` shutdown arm sits at the top of the select loop), so on a // signal it exits well within the shutdown deadline. @@ -287,6 +333,7 @@ fn reconcile_vshard_schedulers(params: ReconcileSchedulersParams<'_>) -> crate:: ); spawned += 1; } + calvin_snapshot::note_kept(shared, &kept); Ok(spawned) } @@ -315,7 +362,8 @@ pub(super) struct SpawnVshardSchedulersParams<'a> { /// and resilient to later ownership changes — this runs an initial reconcile /// (covers the bootstrap node, which already sees its membership) and then /// spawns a background task that re-reconciles on a short interval until -/// shutdown. Reconcile is idempotent and add-only. +/// shutdown. Reconcile is idempotent: it starts schedulers for vShards this +/// node gained and stops those of vShards it left. pub(super) fn spawn_vshard_schedulers( params: SpawnVshardSchedulersParams<'_>, ) -> crate::Result<()> { @@ -403,22 +451,3 @@ pub(super) fn spawn_vshard_schedulers( Ok(()) } - -/// Resolve a raw opcode `u32` to a `VShardMessageType` variant. -/// -/// Uses `VShardEnvelope::from_bytes` as the canonical opcode→variant mapping -/// so this helper stays in sync with the wire format without duplicating the -/// match table. -pub(super) fn resolve_vshard_msg_type( - opcode: u32, -) -> nodedb_cluster::error::Result { - let mut scratch = [0u8; 26]; - scratch[0..2].copy_from_slice(&1u16.to_le_bytes()); // version - scratch[2..4].copy_from_slice(&(opcode as u16).to_le_bytes()); // msg_type - - VShardEnvelope::from_bytes(&scratch) - .map(|e| e.msg_type) - .ok_or_else(|| nodedb_cluster::error::ClusterError::Codec { - detail: format!("resolve_vshard_msg_type: unknown opcode {opcode}"), - }) -} diff --git a/nodedb/src/control/cluster/test_one_node.rs b/nodedb/src/control/cluster/test_one_node.rs new file mode 100644 index 000000000..7e97b2ee7 --- /dev/null +++ b/nodedb/src/control/cluster/test_one_node.rs @@ -0,0 +1,133 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A one-node cluster for unit tests that propose DDL or writes. +//! +//! It boots the way a server with no `[cluster]` section boots: +//! `init_single_node_calvin`, `wire_cluster_handle`, `start_raft`, then the +//! metadata-group readiness wait. Its one Data Plane core acknowledges every +//! request, unless the test supplies its own ([`boot_with_core`]). A test +//! that serves no request and proposes nothing builds a bare `SharedState` +//! instead. +//! +//! The boot runs on any tokio runtime flavor. + +use std::sync::Arc; +use std::time::Duration; + +use nodedb_types::config::tuning::ClusterTransportTuning; + +use crate::bridge::dispatch::{CoreChannelDataSide, Dispatcher}; +use crate::control::security::credential::CredentialStore; +use crate::control::state::SharedState; +use crate::wal::WalManager; + +/// How long each shutdown step can take. +const SHUTDOWN_STEP: Duration = Duration::from_secs(5); + +/// A running one-node cluster and the state it serves. +pub(crate) struct OneNodeCluster { + pub(crate) state: Arc, + core: tokio::task::JoinHandle<()>, + running: Option, + _dir: tempfile::TempDir, +} + +/// Election timeouts for a group whose only voter is this node. A sole voter +/// never waits on a peer, so a short timeout elects it at once. +fn one_node_tuning() -> ClusterTransportTuning { + ClusterTransportTuning { + election_timeout_min_ms: 150, + election_timeout_max_ms: 300, + ..ClusterTransportTuning::default() + } +} + +/// Boot a one-node cluster in a fresh directory and wait until its metadata +/// group applied its first entry. +pub(crate) async fn boot() -> OneNodeCluster { + boot_with(|_| {}).await +} + +/// [`boot`], with `configure` run on the state before it is shared. +pub(crate) async fn boot_with(configure: impl FnOnce(&mut SharedState)) -> OneNodeCluster { + boot_with_core(configure, |state, side| { + tokio::spawn(crate::control::state::test_core::acknowledge_every_request( + state, side, + )) + }) + .await +} + +/// [`boot_with`], with `core` spawning the task that answers the Data +/// Plane side before the cluster starts. The node applies its Raft entries +/// through that core, so a test's core sees and answers them. +pub(crate) async fn boot_with_core( + configure: impl FnOnce(&mut SharedState), + core: impl FnOnce(Arc, CoreChannelDataSide) -> tokio::task::JoinHandle<()>, +) -> OneNodeCluster { + let dir = tempfile::tempdir().expect("create test directory"); + let wal = Arc::new( + WalManager::open_for_testing(&dir.path().join("test.wal")).expect("open test WAL"), + ); + let credentials = Arc::new( + CredentialStore::open(&dir.path().join("system.redb")).expect("open credential store"), + ); + credentials + .catalog() + .bootstrap_default_database() + .expect("bootstrap the default database"); + let (dispatcher, mut data_sides) = Dispatcher::new(1, 64); + let mut state = SharedState::new_with_credentials(dispatcher, wal, credentials, false) + .expect("construct shared state"); + + let tuning = one_node_tuning(); + let handle = super::init_single_node_calvin(dir.path(), &tuning) + .await + .expect("init the one-node cluster"); + { + let wired = Arc::get_mut(&mut state).expect("state is not shared yet"); + configure(wired); + crate::bootstrap::state_wiring::wire_cluster_handle(wired, &handle, dir.path()) + .expect("wire the cluster handle"); + } + super::metadata_applier::seed_host_tables(&state).expect("seed host tables"); + crate::bootstrap::state_wiring::install_gateway(&state).expect("install gateway"); + + let side = data_sides.pop().expect("one data side"); + let core = core(Arc::clone(&state), side); + + let ready = super::start_raft(&handle, Arc::clone(&state), dir.path(), &tuning) + .await + .expect("start raft"); + let running = handle + .running_cluster + .lock() + .unwrap_or_else(|p| p.into_inner()) + .take(); + crate::bootstrap::cluster_ready::await_raft_ready(&state, ready) + .await + .expect("the metadata group applies its first entry"); + + OneNodeCluster { + state, + core, + running, + _dir: dir, + } +} + +impl OneNodeCluster { + /// Stop every task the cluster started and close its transport. + pub(crate) async fn shutdown(mut self) { + self.state.shutdown.signal(); + if let Some(running) = self.running.take() { + let errors = running.shutdown_all(SHUTDOWN_STEP).await; + assert!(errors.is_empty(), "cluster subsystem shutdown: {errors:?}"); + } + if let Some(transport) = self.state.cluster_transport.clone() { + // A one-node cluster has no peer to acknowledge the close. + let _ = transport.close(SHUTDOWN_STEP).await; + } + self.core.abort(); + } +} diff --git a/nodedb/src/control/cluster/tls.rs b/nodedb/src/control/cluster/tls.rs index e981cedad..250790e6a 100644 --- a/nodedb/src/control/cluster/tls.rs +++ b/nodedb/src/control/cluster/tls.rs @@ -26,7 +26,7 @@ use std::fs; use std::io::BufReader; -use std::path::{Path, PathBuf}; +use std::path::Path; use nodedb_cluster::transport::pki_types::{ CertificateDer, CertificateRevocationListDer, PrivateKeyDer, @@ -57,6 +57,9 @@ const CA_KEY_FILE: &str = "ca.key"; /// set. Every CA in this directory is added to the rustls /// RootCertStore for both the server and client configs. pub const CA_TRUST_DIR: &str = "ca.d"; + +use super::ca_trust::load_extra_cas; +pub use super::ca_trust::{remove_trusted_ca, write_trusted_ca}; /// Cluster-wide HMAC key used by the authenticated Raft frame envelope. /// Persisted as raw 32 bytes (no PEM framing) with 0600 perms. const CLUSTER_SECRET_FILE: &str = "cluster_secret.bin"; @@ -250,60 +253,6 @@ fn load_from_data_dir(tls_dir: &Path) -> crate::Result { }) } -/// Load every PEM-encoded CA certificate from `tls_dir/ca.d/*.crt`, -/// sorted by filename for deterministic output. Missing directory is -/// treated as "no overlap CAs" and returns an empty vec. -fn load_extra_cas(tls_dir: &Path) -> crate::Result>> { - let dir = tls_dir.join(CA_TRUST_DIR); - if !dir.exists() { - return Ok(Vec::new()); - } - let mut entries: Vec = fs::read_dir(&dir) - .map_err(|e| crate::Error::Config { - detail: format!("read ca.d {}: {e}", dir.display()), - })? - .filter_map(|r| r.ok()) - .map(|e| e.path()) - .filter(|p| p.extension().and_then(|s| s.to_str()) == Some("crt")) - .collect(); - entries.sort(); - let mut out = Vec::with_capacity(entries.len()); - for p in entries { - out.push(read_single_cert(&p)?); - } - Ok(out) -} - -/// Write a PEM-encoded CA cert into `tls_dir/ca.d/.crt`. -/// Called by the production applier when a `CaTrustChange { add: ... }` -/// entry commits. -pub fn write_trusted_ca(tls_dir: &Path, ca_der: &[u8]) -> crate::Result<[u8; 32]> { - let dir = tls_dir.join(CA_TRUST_DIR); - fs::create_dir_all(&dir).map_err(|e| crate::Error::Config { - detail: format!("create ca.d dir {}: {e}", dir.display()), - })?; - let cert = CertificateDer::from(ca_der.to_vec()); - let fp = nodedb_cluster::ca_fingerprint(&cert); - let name = format!("{}.crt", nodedb_cluster::ca_fingerprint_hex(&fp)); - write_pem_cert(&dir, &name, ca_der)?; - Ok(fp) -} - -/// Delete the overlap-CA file identified by `fp` from `tls_dir/ca.d/`. -/// No-op (and returns `Ok(())`) when the file isn't present — applier -/// behaviour must be idempotent across re-apply and snapshot replay. -pub fn remove_trusted_ca(tls_dir: &Path, fp: &[u8; 32]) -> crate::Result<()> { - let dir = tls_dir.join(CA_TRUST_DIR); - let path = dir.join(format!("{}.crt", nodedb_cluster::ca_fingerprint_hex(fp))); - match fs::remove_file(&path) { - Ok(()) => Ok(()), - Err(e) if e.kind() == std::io::ErrorKind::NotFound => Ok(()), - Err(e) => Err(crate::Error::Config { - detail: format!("remove ca.d entry {}: {e}", path.display()), - }), - } -} - /// L.4 joiner-side helper: connect to `seed`'s bootstrap listener with /// `token`, receive `(ca_cert, node_cert, node_key, cluster_secret)`, /// write the files to `tls_dir/`, and return the loaded credentials. @@ -424,7 +373,7 @@ fn write_cluster_secret(path: &Path, secret: &[u8; CLUSTER_SECRET_LEN]) -> crate }) } -fn read_single_cert(path: &Path) -> crate::Result> { +pub(crate) fn read_single_cert(path: &Path) -> crate::Result> { let bytes = fs::read(path).map_err(|e| crate::Error::Config { detail: format!("read cert {}: {e}", path.display()), })?; @@ -528,6 +477,7 @@ mod tests { log_compaction_threshold: None, join_retry_max_attempts: 8, join_retry_max_backoff_secs: 32, + swim_listen: None, } } diff --git a/nodedb/src/control/cluster/vshard_envelope_handler.rs b/nodedb/src/control/cluster/vshard_envelope_handler.rs new file mode 100644 index 000000000..a3438e907 --- /dev/null +++ b/nodedb/src/control/cluster/vshard_envelope_handler.rs @@ -0,0 +1,127 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The `VShardEnvelopeHandler` `RaftLoop` dispatches vShard envelopes to. + +use std::pin::Pin; +use std::sync::Arc; + +use nodedb_cluster::distributed_array::{ArrayLocalExecutor, handle_array_shard_rpc}; +use nodedb_cluster::vshard_handler::{DispatchTarget, dispatch_by_type}; +use nodedb_cluster::wire::VShardEnvelope; + +use crate::event::cross_shard::CrossShardReceiver; + +/// Build the `VShardEnvelopeHandler` closure used by `RaftLoop`. +/// +/// The closure receives raw envelope bytes from the QUIC transport layer, +/// dispatches based on `msg_type`, and returns a serialized response. +pub(crate) fn build_vshard_handler( + array_executor: Arc, + cross_shard_receiver: Arc, +) -> nodedb_cluster::VShardEnvelopeHandler { + Arc::new(move |bytes: Vec| { + let executor = array_executor.clone(); + let receiver = Arc::clone(&cross_shard_receiver); + let fut: Pin< + Box>> + Send>, + > = Box::pin(async move { + let envelope = VShardEnvelope::from_bytes(&bytes).ok_or_else(|| { + nodedb_cluster::error::ClusterError::Codec { + detail: "vshard_handler: failed to deserialize VShardEnvelope".into(), + } + })?; + + let target = dispatch_by_type(&envelope); + match target { + DispatchTarget::ArrayShard => { + let opcode = envelope.msg_type as u32; + let resp_payload = handle_array_shard_rpc( + opcode, + envelope.vshard_id, + &envelope.payload, + &executor, + ) + .await?; + + // Response opcode = request opcode + 1 for all array shard RPCs. + // Resolve the msg_type variant via a minimal scratch envelope parse + // (avoids any unsafe transmute — the `from_bytes` mapping in wire.rs + // is the canonical source of truth for the opcode→variant table). + let resp_opcode = opcode + 1; + let resp_msg_type = resolve_vshard_msg_type(resp_opcode)?; + let resp_envelope = VShardEnvelope::new( + resp_msg_type, + envelope.target_node, + envelope.source_node, + envelope.vshard_id, + resp_payload, + ); + Ok(resp_envelope.to_bytes()) + } + + // `CrossShardEvent` (remote trigger DML) and `NotifyBroadcast` + // (cluster-wide CDC fan-out) both land here. `handle_envelope` + // re-parses the raw bytes and returns a fully-formed response + // envelope — including the error-shaped one for a message type + // that must not arrive as a REQUEST. The `*Ack` variants are such + // a case: every sender reads its Ack as the RESPONSE on the same + // QUIC stream, so an inbound Ack request is a protocol violation, + // not a case to handle. Unlike the ArrayShard arm there is no + // opcode+1 convention to apply: the receiver picks the response + // msg_type per request type itself. + DispatchTarget::EventPlane => Ok(receiver.handle_envelope(bytes).await), + + other => Err(nodedb_cluster::error::ClusterError::Transport { + detail: format!( + "vshard_handler: no handler registered for dispatch target {other:?}" + ), + }), + } + }); + fut + }) +} + +/// Resolve a raw opcode `u32` to a `VShardMessageType` variant through +/// `VShardMessageType::from_raw`, the wire format's one opcode table. +pub(crate) fn resolve_vshard_msg_type( + opcode: u32, +) -> nodedb_cluster::error::Result { + u16::try_from(opcode) + .ok() + .and_then(nodedb_cluster::wire::VShardMessageType::from_raw) + .ok_or_else(|| nodedb_cluster::error::ClusterError::Codec { + detail: format!("resolve_vshard_msg_type: unknown opcode {opcode}"), + }) +} + +#[cfg(test)] +mod tests { + use nodedb_cluster::wire::VShardMessageType; + + use super::resolve_vshard_msg_type; + + /// Every array shard response opcode resolves, so a shard can answer + /// each array request it serves. + #[test] + fn every_array_response_opcode_resolves() { + let responses = [ + (81, VShardMessageType::ArrayShardSliceResp), + (83, VShardMessageType::ArrayShardAggResp), + (85, VShardMessageType::ArrayShardPutResp), + (87, VShardMessageType::ArrayShardDeleteResp), + (89, VShardMessageType::ArrayShardSurrogateBitmapResp), + ]; + for (opcode, expected) in responses { + let resolved = resolve_vshard_msg_type(opcode) + .unwrap_or_else(|e| panic!("opcode {opcode} must resolve: {e}")); + assert_eq!(resolved, expected); + } + } + + #[test] + fn an_unknown_opcode_is_a_codec_error() { + assert!(resolve_vshard_msg_type(90).is_err()); + assert!(resolve_vshard_msg_type(u32::from(u16::MAX) + 1).is_err()); + } +} diff --git a/nodedb/src/control/cluster/warm_peers/report.rs b/nodedb/src/control/cluster/warm_peers/report.rs index 4354e0152..535b95254 100644 --- a/nodedb/src/control/cluster/warm_peers/report.rs +++ b/nodedb/src/control/cluster/warm_peers/report.rs @@ -40,8 +40,7 @@ impl PeerWarmReport { self.failed.is_empty() && self.succeeded.len() == self.attempted } - /// Empty report — used when the topology has no peers - /// (single-node mode). + /// Empty report — used when the topology has no peers. pub fn empty() -> Self { Self { attempted: 0, diff --git a/nodedb/src/control/cold_tier.rs b/nodedb/src/control/cold_tier.rs index ef2420d99..979fed052 100644 --- a/nodedb/src/control/cold_tier.rs +++ b/nodedb/src/control/cold_tier.rs @@ -5,7 +5,8 @@ //! Each cycle scans `{data_dir}/segments/` for segment files whose modification //! time exceeds `tier_after_secs`. Eligible files are uploaded as raw binary //! objects to the configured cold store under `{prefix}segments/{name}`, then -//! deleted locally on success. +//! deleted locally on success. An upload never overwrites an object a kept +//! base snapshot references: that local file stays until the base is retired. //! //! Runs on the Control Plane (Tokio) — `ColdStorage` is async and `Send + Sync`. @@ -72,7 +73,15 @@ pub fn spawn_cold_tier_task( return; } }; - run_tier_cycle_at(&cold, &segments_dir, tier_after, &prefix, &key).await; + run_tier_cycle_at( + &cold, + &segments_dir, + tier_after, + &prefix, + &key, + shared.pitr.cold_pins(), + ) + .await; } }) } @@ -87,6 +96,7 @@ pub(crate) async fn run_tier_cycle_at( tier_after: Duration, prefix: &str, key: &nodedb_wal::crypto::WalEncryptionKey, + pins: &tokio::sync::RwLock, ) { let now = SystemTime::now(); @@ -134,7 +144,19 @@ pub(crate) async fn run_tier_cycle_at( let object_path = format!("{}segments/{}", prefix, segment_name); let entry_path_clone = entry_path.clone(); - match upload_raw_segment(cold, &entry_path_clone, &object_path, key).await { + // Held across the upload, so no base pins the key mid-overwrite. + let held = pins.read().await; + if held.is_pinned(&object_path) { + debug!( + object_path = %object_path, + "cold tier: a kept base snapshot references this key, skipping" + ); + continue; + } + let uploaded = upload_raw_segment(cold, &entry_path_clone, &object_path, key).await; + drop(held); + + match uploaded { Ok(()) => { info!( segment = %segment_name, @@ -275,4 +297,55 @@ mod tests { let object = object_store::path::Path::from("segments/forged.seg"); assert!(cold.object_store().head(&object).await.is_err()); } + + #[tokio::test] + async fn a_pinned_key_is_never_overwritten_until_its_base_is_retired() { + use crate::control::pitr::ColdPins; + use crate::storage::segment::{SegmentFooter, encrypt_untrusted_segment_bytes}; + use crate::types::Lsn; + + let segments = tempfile::tempdir().expect("local segment directory"); + let cold_dir = tempfile::tempdir().expect("cold object directory"); + let key = nodedb_wal::crypto::WalEncryptionKey::from_bytes(&[0xA5; 32]) + .expect("test encryption key"); + let footer = SegmentFooter::new("n", 0, Lsn::new(1), Lsn::new(1)); + let bytes = encrypt_untrusted_segment_bytes(b"new", &footer, &key).expect("encrypt"); + let local = segments.path().join("s.seg"); + std::fs::write(&local, &bytes).expect("write segment"); + + let cold = crate::storage::cold::ColdStorage::new(ColdStorageConfig { + local_dir: Some(cold_dir.path().to_path_buf()), + ..ColdStorageConfig::default() + }) + .expect("cold storage"); + let object = object_store::path::Path::from("p/segments/s.seg"); + cold.object_store() + .put( + &object, + object_store::PutPayload::from_static(b"referenced"), + ) + .await + .expect("seed object"); + let pins = tokio::sync::RwLock::new(ColdPins::default()); + pins.write() + .await + .pin("snap-1", vec!["p/segments/s.seg".to_string()]); + + run_tier_cycle_at(&cold, segments.path(), Duration::ZERO, "p/", &key, &pins).await; + let kept = cold.object_store().get(&object).await.expect("get"); + assert_eq!(kept.bytes().await.expect("bytes").as_ref(), b"referenced"); + assert!( + local.exists(), + "the local copy stays while the key is pinned" + ); + + pins.write().await.unpin("snap-1"); + run_tier_cycle_at(&cold, segments.path(), Duration::ZERO, "p/", &key, &pins).await; + let replaced = cold.object_store().get(&object).await.expect("get"); + assert_eq!( + replaced.bytes().await.expect("bytes").as_ref(), + bytes.as_slice() + ); + assert!(!local.exists(), "the local copy goes once the upload lands"); + } } diff --git a/nodedb/src/control/crdt_admission.rs b/nodedb/src/control/crdt_admission.rs index 8dc7b0988..c366b5e0b 100644 --- a/nodedb/src/control/crdt_admission.rs +++ b/nodedb/src/control/crdt_admission.rs @@ -98,7 +98,8 @@ struct CrdtAdmissionWorkflow<'a> { } /// Whether an operation changes the Loro frontier and must serialize with an -/// admission preview when executed directly on a single-node Data Plane. +/// admission preview when it is dispatched to this node's cores without a +/// proposal. pub fn changes_crdt_frontier(op: &CrdtOp) -> bool { match op { CrdtOp::Apply { .. } @@ -201,6 +202,15 @@ pub(crate) async fn dispatch_crdt_apply_admitted_outcome( event_source, policy, } = request; + // Every CRDT apply converges here; the lease gates it on a drained + // collection and lives until the apply's outcome. + let _lease = crate::control::server::shared::clone_write::write_lease( + state, + tenant_id, + database_id, + &plan, + ) + .await?; let (document_id, delta) = match &plan { PhysicalPlan::Crdt( CrdtOp::Apply { @@ -249,9 +259,13 @@ pub(crate) async fn dispatch_crdt_apply_admitted_outcome( }; tokio::time::timeout( timeout, - state.vshard_admission_sequencer.run(vshard_id, || { - admit_apply_locked(&workflow, plan, &document_id, &delta) - }), + state.vshard_admission_sequencer.run_after_proposed( + vshard_id, + |last_proposed| async move { + caught_up(&workflow, last_proposed).await?; + admit_apply_locked(&workflow, plan, &document_id, &delta).await + }, + ), ) .await .map_err(|_| crate::Error::CrdtAdmissionTimeout { @@ -260,6 +274,21 @@ pub(crate) async fn dispatch_crdt_apply_admitted_outcome( })? } +/// Wait until this node applied every write admitted to the workflow's +/// vShard before this admission took the slot, so its preview reads them. +async fn caught_up( + workflow: &CrdtAdmissionWorkflow<'_>, + last_proposed: Option, +) -> crate::Result<()> { + crate::control::vshard_admission::await_admitted_applies( + workflow.state, + workflow.vshard_id, + last_proposed, + std::time::Instant::now() + workflow.timeout, + ) + .await +} + async fn admit_apply_locked( workflow: &CrdtAdmissionWorkflow<'_>, plan: PhysicalPlan, @@ -426,6 +455,15 @@ pub(crate) async fn dispatch_crdt_restore_admitted( event_source, policy, } = request; + // The restore writes a delta like any apply; the lease gates it on a + // drained collection and lives until its outcome. + let _lease = crate::control::server::shared::clone_write::collections_write_lease( + state, + tenant_id, + database_id, + [collection.to_owned()], + ) + .await?; // `collection` is the plan's database-qualified name. let vshard_id = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); @@ -441,39 +479,50 @@ pub(crate) async fn dispatch_crdt_restore_admitted( }; tokio::time::timeout( timeout, - state.vshard_admission_sequencer.run(vshard_id, || async { - for _attempt in 0..FRONTIER_RETRY_LIMIT { - let delta = - generate_restore_delta(&workflow, document_id, target_version_json, surrogate) - .await?; - if delta.is_empty() { - return Ok(None); - } - let plan = PhysicalPlan::Crdt(CrdtOp::Apply { - collection: nodedb_types::QualifiedCollection::from_stored( - collection.to_owned(), - ), - document_id: document_id.to_owned(), - delta: delta.clone(), - peer_id, - mutation_id: 0, - surrogate, - provenance: None, - constraint_version_required: 0, - expected_frontier_digest: None, - }); - let preview = preview(&workflow, document_id, &delta).await?; - workflow.policy.evaluate(&preview)?; - match apply_fenced(&workflow, stamp_fence(plan, preview.frontier_digest)?).await { - Err(crate::Error::DataPlane(ErrorCode::CrdtFrontierMismatch { .. })) => {} - result => return result.map(Some), + state.vshard_admission_sequencer.run_after_proposed( + vshard_id, + |last_proposed| async move { + caught_up(&workflow, last_proposed).await?; + for _attempt in 0..FRONTIER_RETRY_LIMIT { + let delta = generate_restore_delta( + &workflow, + document_id, + target_version_json, + surrogate, + ) + .await?; + if delta.is_empty() { + return Ok(None); + } + let plan = PhysicalPlan::Crdt(CrdtOp::Apply { + collection: nodedb_types::QualifiedCollection::from_stored( + collection.to_owned(), + ), + document_id: document_id.to_owned(), + delta: delta.clone(), + peer_id, + mutation_id: 0, + surrogate, + provenance: None, + constraint_version_required: 0, + expected_frontier_digest: None, + }); + let preview = preview(&workflow, document_id, &delta).await?; + workflow.policy.evaluate(&preview)?; + match apply_fenced(&workflow, stamp_fence(plan, preview.frontier_digest)?).await + { + Err(crate::Error::DataPlane(ErrorCode::CrdtFrontierMismatch { + .. + })) => {} + result => return result.map(Some), + } } - } - Err(crate::Error::CrdtAdmissionRetriesExhausted { - vshard_id, - attempts: FRONTIER_RETRY_LIMIT, - }) - }), + Err(crate::Error::CrdtAdmissionRetriesExhausted { + vshard_id, + attempts: FRONTIER_RETRY_LIMIT, + }) + }, + ), ) .await .map_err(|_| crate::Error::CrdtAdmissionTimeout { @@ -518,61 +567,32 @@ async fn apply_fenced( workflow: &CrdtAdmissionWorkflow<'_>, plan: PhysicalPlan, ) -> crate::Result { - if let Some(raw) = workflow.state.raw_async_raft_proposer() { - let entry = to_replicated_entry( - workflow.tenant_id, - workflow.database_id, - workflow.vshard_id, - &ReplicableWrite::decide_for_replication(&plan)?, - )? - .ok_or(crate::Error::CrdtAdmissionInvalidPlan { - reason: "admitted CRDT Apply has no replicated form", - })? - // Every replica gives the write the source this node dispatches it - // with. - .with_event_source(workflow.event_source); - let outcome = tokio::time::timeout( - workflow.timeout, - crate::control::wal_replication::propose_replicated_entry(workflow.state, raw, entry), - ) - .await - .map_err(|_| crate::Error::CrdtAdmissionTimeout { - vshard_id: workflow.vshard_id, - timeout_ms: timeout_ms(workflow.timeout), - })??; - // This node's apply of the entry recorded its commit HLC. - return Ok(CrdtAdmissionOutcome { - payload: outcome.0, - write_version: outcome.1, - trimmed_ops: 0, - }); - } - let response = tokio::time::timeout( + let raw = workflow.state.raw_async_raft_proposer()?; + let entry = to_replicated_entry( + workflow.tenant_id, + workflow.database_id, + workflow.vshard_id, + &ReplicableWrite::decide_for_replication(&plan)?, + )? + .ok_or(crate::Error::CrdtAdmissionInvalidPlan { + reason: "admitted CRDT Apply has no replicated form", + })? + // Every replica gives the write the source this node dispatches it + // with. + .with_event_source(workflow.event_source); + let outcome = tokio::time::timeout( workflow.timeout, - crate::control::server::dispatch_utils::dispatch_autocommit_write( - workflow.state, - crate::control::server::dispatch_utils::AutocommitWrite { - tenant_id: workflow.tenant_id, - database_id: workflow.database_id, - vshard_id: workflow.vshard_id, - plan, - trace_id: crate::types::TraceId::ZERO, - event_source: workflow.event_source, - txn_id: None, - }, - ), + crate::control::wal_replication::propose_replicated_entry(workflow.state, raw, entry), ) .await .map_err(|_| crate::Error::CrdtAdmissionTimeout { vshard_id: workflow.vshard_id, timeout_ms: timeout_ms(workflow.timeout), })??; - if response.status != Status::Ok { - return Err(response_error(&response)); - } + // This node's apply of the entry recorded its commit HLC. Ok(CrdtAdmissionOutcome { - payload: response.payload.to_vec(), - write_version: response.read_version_lsn, + payload: outcome.0, + write_version: outcome.1, trimmed_ops: 0, }) } @@ -630,7 +650,7 @@ mod tests { delta: vec![0x91, 0x01], peer_id: 7, mutation_id: 9, - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), provenance: None, constraint_version_required: 0, expected_frontier_digest: None, @@ -709,7 +729,7 @@ mod tests { } #[tokio::test] - async fn admitted_local_apply_previews_exact_post_image_before_fenced_apply() { + async fn admitted_apply_previews_exact_post_image_before_fenced_apply() { let (state, side, _directory) = fixture(); let digest = [0x4d; 32]; let post_image = vec![0x81, 0xa2, b'o', b'k', 0xc3]; @@ -721,14 +741,12 @@ mod tests { }) .expect("preview payload"); let preview_fields = Arc::new(Mutex::new(Vec::new())); - let apply_fences = Arc::new(Mutex::new(Vec::new())); let fields = Arc::clone(&preview_fields); - let fences = Arc::clone(&apply_fences); let responder = tokio::spawn(respond_n( Arc::clone(&state), side, - 2, + 1, move |request| match request.plan { PhysicalPlan::Crdt(CrdtOp::PreviewApply { collection, @@ -741,19 +759,36 @@ mod tests { .push((collection, document_id, delta)); response(request.request_id, preview_payload.clone()) } - PhysicalPlan::Crdt(CrdtOp::Apply { + other => panic!("unexpected request: {other:?}"), + }, + )); + let apply_fences = Arc::new(Mutex::new(Vec::new())); + let fences = Arc::clone(&apply_fences); + let raw: Arc = + Arc::new(move |_shard, _key, bytes, _deadline| { + let fences = Arc::clone(&fences); + Box::pin(async move { + let entry = + crate::control::wal_replication::ReplicatedEntry::from_bytes(&bytes) + .expect("replicated entry"); + if let crate::control::wal_replication::ReplicatedWrite::CrdtApplyFenced { expected_frontier_digest, .. - }) => { + } = entry.write + { fences .lock() .expect("fences lock") .push(expected_frontier_digest); - response(request.request_id, Vec::new()) } - other => panic!("unexpected request: {other:?}"), - }, - )); + Ok((Vec::new(), Lsn::ZERO)) + }) + }); + crate::control::vshard_admission::install_async_raft_proposer( + &state, + crate::control::vshard_admission::applying_submit(raw), + ) + .expect("install raw/sequenced proposer pair"); let seen = Arc::new(Mutex::new(Vec::new())); let policy = RecordingPolicy { seen: Arc::clone(&seen), @@ -761,14 +796,11 @@ mod tests { }; dispatch_crdt_apply_admitted(&state, admission_request(&policy)) .await - .expect("admitted local apply"); + .expect("admitted apply"); responder.await.expect("responder completes"); assert_eq!(*seen.lock().expect("seen lock"), vec![post_image]); assert_eq!(preview_fields.lock().expect("fields lock").len(), 1); - assert_eq!( - *apply_fences.lock().expect("fences lock"), - vec![Some(digest)] - ); + assert_eq!(*apply_fences.lock().expect("fences lock"), vec![digest]); } #[tokio::test] @@ -851,8 +883,11 @@ mod tests { Ok((Vec::new(), Lsn::ZERO)) }) }); - crate::control::vshard_admission::install_async_raft_proposer(&state, raw) - .expect("install raw/sequenced proposer pair"); + crate::control::vshard_admission::install_async_raft_proposer( + &state, + crate::control::vshard_admission::applying_submit(raw), + ) + .expect("install raw/sequenced proposer pair"); let seen = Arc::new(Mutex::new(Vec::new())); let policy = RecordingPolicy { seen, @@ -899,8 +934,11 @@ mod tests { } }) }); - crate::control::vshard_admission::install_async_raft_proposer(&state, raw) - .expect("install raw/sequenced proposer pair"); + crate::control::vshard_admission::install_async_raft_proposer( + &state, + crate::control::vshard_admission::applying_submit(raw), + ) + .expect("install raw/sequenced proposer pair"); let seen = Arc::new(Mutex::new(Vec::new())); let policy = RecordingPolicy { seen: Arc::clone(&seen), @@ -964,8 +1002,11 @@ mod tests { }) }) }; - crate::control::vshard_admission::install_async_raft_proposer(&state, raw) - .expect("install proposer"); + crate::control::vshard_admission::install_async_raft_proposer( + &state, + crate::control::vshard_admission::applying_submit(raw), + ) + .expect("install proposer"); let policy = RecordingPolicy { seen: Arc::new(Mutex::new(Vec::new())), reject: false, @@ -978,7 +1019,7 @@ mod tests { collection: "docs", document_id: "doc-1", target_version_json: "{}", - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), peer_id: 1, timeout: Duration::from_secs(1), event_source: EventSource::User, @@ -1028,8 +1069,11 @@ mod tests { })) }) }); - crate::control::vshard_admission::install_async_raft_proposer(&state, raw) - .expect("install raw/sequenced proposer pair"); + crate::control::vshard_admission::install_async_raft_proposer( + &state, + crate::control::vshard_admission::applying_submit(raw), + ) + .expect("install raw/sequenced proposer pair"); let seen = Arc::new(Mutex::new(Vec::new())); let policy = RecordingPolicy { seen: Arc::clone(&seen), @@ -1055,8 +1099,9 @@ mod tests { async fn signed_delta_collection_rejects_plain_external_apply_before_preview() { let (state, _side, _directory) = fixture(); let tenant_id = TenantId::new(1); - let mut collection = - crate::control::security::catalog::StoredCollection::new(1, "docs", "owner"); + let mut collection = crate::control::security::catalog::StoredCollection::stamped_for_test( + 1, "docs", "owner", + ); collection.crdt = true; collection.crdt_signing_required = true; state @@ -1150,9 +1195,7 @@ mod tests { &state, "docs", authorized, - Duration::from_millis(10), EventSource::User, - None, ) .await; assert!(matches!( @@ -1171,7 +1214,7 @@ mod tests { #[test] fn frontier_classifier_covers_every_crdt_operation_category() { - let surrogate = Surrogate::ZERO; + let surrogate = Surrogate::new(1); assert_frontier_mutation(CrdtOp::Apply { collection: QualifiedCollection::new(DatabaseId::DEFAULT, &collection()), document_id: "id".into(), @@ -1230,7 +1273,7 @@ mod tests { assert_frontier_mutation(CrdtOp::DocDelete { collection: QualifiedCollection::new(DatabaseId::DEFAULT, &collection()), document_id: "id".into(), - surrogate, + surrogate: Some(surrogate), returning: None, rls_filters: Vec::new(), }); diff --git a/nodedb/src/control/database/allocate.rs b/nodedb/src/control/database/allocate.rs new file mode 100644 index 000000000..661b15034 --- /dev/null +++ b/nodedb/src/control/database/allocate.rs @@ -0,0 +1,28 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The one database-id allocation entry point for every DDL path. + +use nodedb_types::DatabaseId; + +use crate::control::metadata_proposer::propose_database_id_reserve; +use crate::control::state::SharedState; + +/// Allocate a fresh database id. +/// +/// The id comes from a replicated `DatabaseIdReserve` entry, so every node +/// agrees on it. The applier persists the new hwm before the id is +/// returned, and a persist error returns `Err`. +pub async fn allocate_database_id(state: &SharedState) -> crate::Result { + let registry = &state.database_registry; + let request_id = registry.begin_request(); + let proposed = propose_database_id_reserve(state, state.node_id, request_id).await; + // Always clear the request so an error path leaves no pending slot. + let id = registry.finish_request(request_id); + let log_index = proposed?; + id.ok_or_else(|| crate::Error::Internal { + detail: format!( + "database id reservation at metadata log index {log_index} \ + (request {request_id}) applied without issuing an id to this node" + ), + }) +} diff --git a/nodedb/src/control/database/mod.rs b/nodedb/src/control/database/mod.rs index 23ebbae0c..c36b6c5eb 100644 --- a/nodedb/src/control/database/mod.rs +++ b/nodedb/src/control/database/mod.rs @@ -1,10 +1,9 @@ // SPDX-License-Identifier: BUSL-1.1 +pub mod allocate; pub mod persist; pub mod registry; -pub use persist::{DatabaseHwmPersist, SystemCatalogDatabaseHwm}; -pub use registry::{ - DatabaseAllocError, DatabaseRegistry, FLUSH_ELAPSED_THRESHOLD, FLUSH_OPS_THRESHOLD, - USER_DB_START, -}; +pub use allocate::allocate_database_id; +pub use persist::DatabaseHwmPersist; +pub use registry::{DatabaseAllocError, DatabaseRegistry, USER_DB_START}; diff --git a/nodedb/src/control/database/persist.rs b/nodedb/src/control/database/persist.rs index 7c459a3e6..3f8846ca6 100644 --- a/nodedb/src/control/database/persist.rs +++ b/nodedb/src/control/database/persist.rs @@ -1,48 +1,37 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Database hwm persistence trait + concrete `SystemCatalog`-backed impl. +//! Database hwm persistence trait and its `SystemCatalog` impl. //! -//! Mirrors `nodedb::control::surrogate::persist` for the database -//! allocator. The trait separates the registry's allocation logic from -//! the storage layer so tests can substitute an in-memory impl. - -use std::sync::Arc; +//! The trait separates the registry's allocation logic from the storage +//! layer so tests can substitute an in-memory impl. use crate::control::security::catalog::SystemCatalog; -/// Pluggable persistence boundary for `DatabaseRegistry`. -/// Tests substitute an in-memory store; production wires -/// [`SystemCatalogDatabaseHwm`]. +/// Pluggable persistence boundary for `DatabaseRegistry`. Production uses +/// the `SystemCatalog` impl over `_system.database_hwm`. pub trait DatabaseHwmPersist: Send + Sync { - /// Persist the current high-watermark. Called by - /// `DatabaseRegistry::flush` whenever periodic-flush thresholds - /// (64 ops or 200 ms) are tripped. - fn checkpoint(&self, hwm: u64) -> crate::Result<()>; + /// Persist the hwm and the log index of the reservation that produced + /// it, atomically. + fn checkpoint_reserve(&self, hwm: u64, reserve_index: u64) -> crate::Result<()>; - /// Load the persisted high-watermark, or `0` if none recorded yet - /// (fresh database). + /// Load the persisted hwm, or `0` on a fresh catalog. fn load(&self) -> crate::Result; -} -/// `SystemCatalog`-backed persistence — delegates to -/// `put_database_hwm` / `get_database_hwm`. -pub struct SystemCatalogDatabaseHwm { - catalog: Arc, + /// Load the applied-reservation cursor, or `0` on a fresh catalog. + fn load_reserve_index(&self) -> crate::Result; } -impl SystemCatalogDatabaseHwm { - pub fn new(catalog: Arc) -> Self { - Self { catalog } +impl DatabaseHwmPersist for SystemCatalog { + fn checkpoint_reserve(&self, hwm: u64, reserve_index: u64) -> crate::Result<()> { + self.put_database_reserve_state(hwm, reserve_index) } -} -impl DatabaseHwmPersist for SystemCatalogDatabaseHwm { - fn checkpoint(&self, hwm: u64) -> crate::Result<()> { - self.catalog.put_database_hwm(hwm) + fn load(&self) -> crate::Result { + self.get_database_hwm() } - fn load(&self) -> crate::Result { - self.catalog.get_database_hwm() + fn load_reserve_index(&self) -> crate::Result { + self.get_database_reserve_index() } } @@ -51,12 +40,12 @@ mod tests { use super::*; #[test] - fn handle_roundtrip_via_catalog() { + fn roundtrip_via_catalog() { let dir = tempfile::tempdir().unwrap(); - let catalog = Arc::new(SystemCatalog::open(&dir.path().join("system.redb")).unwrap()); - let p = SystemCatalogDatabaseHwm::new(catalog); - assert_eq!(p.load().unwrap(), 0); - p.checkpoint(1024).unwrap(); - assert_eq!(p.load().unwrap(), 1024); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).unwrap(); + assert_eq!(catalog.load().unwrap(), 0); + catalog.checkpoint_reserve(1025, 9).unwrap(); + assert_eq!(catalog.load().unwrap(), 1025); + assert_eq!(catalog.load_reserve_index().unwrap(), 9); } } diff --git a/nodedb/src/control/database/registry.rs b/nodedb/src/control/database/registry.rs index 0d51a90ea..4a5ecf721 100644 --- a/nodedb/src/control/database/registry.rs +++ b/nodedb/src/control/database/registry.rs @@ -1,171 +1,152 @@ // SPDX-License-Identifier: BUSL-1.1 -//! `DatabaseRegistry` — thread-safe monotonic database-id allocator. +//! `DatabaseRegistry` — monotonic database-id allocator. //! -//! ## Counter semantics +//! ## Id ranges //! -//! The internal `AtomicU64` counter stores the **next** database id to -//! be handed out. `alloc_one()` does `fetch_add(1, AcqRel)`; the returned -//! value is the previous counter (i.e. the id the caller now owns). After -//! every successful allocation, `current_hwm()` returns the highest id -//! ever issued — equivalently, `counter - 1`. +//! `DatabaseId(0)` is the built-in `default` database. `1..=1023` is +//! reserved for system databases. User databases start at +//! [`USER_DB_START`]. //! -//! ## Reserved range +//! ## Allocation //! -//! `DatabaseId(0)` is permanently reserved for the built-in `default` -//! database. `DatabaseId(1..=1023)` is reserved for future system -//! databases; none are assigned in v1. User-created databases start -//! at `DatabaseId(1024)`. `from_persisted_hwm` enforces this floor: -//! if the persisted hwm is less than 1023, the counter is initialized -//! to 1024 so the first allocation cannot invade the reserved range. +//! Every id comes from a `DatabaseIdReserve` log entry of the metadata Raft +//! group, a one-node cluster included. [`DatabaseRegistry::reserve_at_index`] +//! runs at apply time on every node, in log order, so every node computes +//! the same id. It persists the hwm together with the entry's log index. //! -//! ## Restart semantics +//! The hwm is persisted before the id leaves the registry. A persist error +//! returns `Err` and leaves the counter unchanged, so a restart never +//! reissues an id. //! -//! `from_persisted_hwm(hwm)` initializes `counter = max(hwm + 1, 1024)`. +//! ## Replay //! -//! ## Raft routing -//! -//! The atomic counter is a local cache only. The authoritative -//! allocation goes through Raft metadata group 0 via -//! `crate::control::metadata_proposer::propose_database_hwm`. The -//! `install_shared` hook (mirroring `SurrogateAssigner`) wires the -//! weak `SharedState` handle so the flush path can propose. -//! -//! ## Width -//! -//! Database IDs are `u64`. With user databases starting at 1024 and -//! u64::MAX ≈ 1.8 × 10^19, overflow is not a practical concern; the -//! registry does not implement an `Exhausted` error path. +//! The metadata log replays from its first entry on every boot. The +//! registry is seeded with the persisted `(hwm, reserve_index)` pair, and +//! `reserve_at_index` skips every entry at or below `reserve_index`. Those +//! reservations are already folded into the seeded hwm. +use std::collections::HashMap; use std::sync::Mutex; -use std::sync::atomic::{AtomicU64, Ordering}; -use std::time::{Duration, Instant}; use nodedb_types::DatabaseId; use super::persist::DatabaseHwmPersist; -/// Periodic flush trigger: every N allocations, regardless of elapsed time. -pub const FLUSH_OPS_THRESHOLD: u64 = 64; - -/// Periodic flush trigger: every T elapsed since the last flush. -pub const FLUSH_ELAPSED_THRESHOLD: Duration = Duration::from_millis(200); - /// First user-assignable database id. `0..=1023` reserved. pub const USER_DB_START: u64 = 1024; -/// Allocation errors. Surfaced to the caller; `From` impl wires this into -/// the crate's central `Error` enum. +/// Allocation errors. `From` wires this into the crate's central `Error`. #[derive(Debug, thiserror::Error)] pub enum DatabaseAllocError { - #[error("database hwm flush failed: {detail}")] - FlushFailed { detail: String }, + #[error("database hwm persist failed: {detail}")] + PersistFailed { detail: String }, +} + +/// Counter state guarded as one unit so the id and its cursor move together. +struct Counter { + /// Next id to issue. Always `>= USER_DB_START`. + next: u64, + /// Highest metadata log index whose reservation is folded into `next`. + reserve_index: u64, } /// Thread-safe database-id allocator. -/// -/// The `Mutex` for `last_flush_at` is uncontended on the hot -/// path (`alloc_one` only touches atomics); only `should_flush` and -/// `flush` take the lock, which run at most once per ~200 ms or per 64 -/// allocations. pub struct DatabaseRegistry { - /// Next id to hand out. Always >= `USER_DB_START`. - counter: AtomicU64, - /// Allocations since the last flush. Reset by `flush()`. - allocs_since_flush: AtomicU64, - /// Wall-clock anchor for the elapsed-time flush trigger. - last_flush_at: Mutex, + counter: Mutex, + /// In-flight replicated reservations of this node: request id -> the id + /// the applier carved for it, once applied. + pending: Mutex>>, } impl DatabaseRegistry { - /// Create an empty registry — first allocation returns `DatabaseId(1024)`. + /// Create an empty registry — the first id issued is `USER_DB_START`. pub fn new() -> Self { - Self::from_persisted_hwm(0) + Self::from_persisted(0, 0) } - /// Restore from a persisted high-watermark. Next allocation returns - /// `max(hwm + 1, USER_DB_START)`. - pub fn from_persisted_hwm(hwm: u64) -> Self { - let next = (hwm + 1).max(USER_DB_START); + /// Restore from the persisted hwm and applied-reservation cursor. + pub fn from_persisted(hwm: u64, reserve_index: u64) -> Self { Self { - counter: AtomicU64::new(next), - allocs_since_flush: AtomicU64::new(0), - last_flush_at: Mutex::new(Instant::now()), + counter: Mutex::new(Counter { + next: hwm.saturating_add(1).max(USER_DB_START), + reserve_index, + }), + pending: Mutex::new(HashMap::new()), } } - /// Allocate a single database id. - pub fn alloc_one(&self) -> DatabaseId { - let prev = self.counter.fetch_add(1, Ordering::AcqRel); - self.allocs_since_flush.fetch_add(1, Ordering::AcqRel); - DatabaseId::new(prev) + /// Reset the counter to the persisted hwm and applied-reservation + /// cursor, after the catalog rows were replaced by a metadata snapshot. + /// Waiting requests stay registered: their entries fall inside the + /// snapshot, so they time out and retry. + pub fn restore_persisted(&self, hwm: u64, reserve_index: u64) { + let mut counter = self.lock_counter(); + counter.next = hwm.saturating_add(1).max(USER_DB_START); + counter.reserve_index = reserve_index; + } + + /// Apply the `DatabaseIdReserve` entry at `raft_index`. + /// + /// Returns `None` for an entry already folded into the seeded hwm + /// (replay or duplicate delivery). Otherwise issues the next id and + /// persists it with `raft_index` before returning it. + pub fn reserve_at_index( + &self, + raft_index: u64, + persist: &dyn DatabaseHwmPersist, + ) -> Result, DatabaseAllocError> { + let mut counter = self.lock_counter(); + if raft_index <= counter.reserve_index { + return Ok(None); + } + let id = counter.next; + persist + .checkpoint_reserve(id, raft_index) + .map_err(persist_failed)?; + counter.next = id + 1; + counter.reserve_index = raft_index; + Ok(Some(DatabaseId::new(id))) } - /// Highest database id ever issued — `USER_DB_START - 1` if no user - /// allocations yet (meaning no user database has been created). + /// Highest id ever issued, or `USER_DB_START - 1` before the first one. pub fn current_hwm(&self) -> u64 { - let next = self.counter.load(Ordering::Acquire); - next.saturating_sub(1) + self.lock_counter().next - 1 } - /// Idempotently raise the high-watermark to at least `new_hwm`. - /// Used by WAL replay or Raft follower catch-up. Never lowers. - pub fn restore_hwm(&self, new_hwm: u64) { - let target = new_hwm + 1; - let mut current = self.counter.load(Ordering::Acquire); + /// Register an in-flight replicated reservation. The request id is + /// random, so a request of this process never matches an entry a + /// previous process of this node proposed. + pub fn begin_request(&self) -> u64 { + let mut pending = self.lock_pending(); loop { - if target <= current { - return; - } - match self.counter.compare_exchange_weak( - current, - target, - Ordering::AcqRel, - Ordering::Acquire, - ) { - Ok(_) => return, - Err(actual) => current = actual, + let request_id = rand::random::(); + if let std::collections::hash_map::Entry::Vacant(slot) = pending.entry(request_id) { + slot.insert(None); + return request_id; } } } - /// True if the periodic-flush thresholds (ops or elapsed) are tripped. - pub fn should_flush(&self) -> bool { - if self.allocs_since_flush.load(Ordering::Acquire) >= FLUSH_OPS_THRESHOLD { - return true; + /// Record the id the applier carved for `request_id`. A request this + /// process does not wait on is ignored. + pub fn complete_request(&self, request_id: u64, id: DatabaseId) { + if let Some(slot) = self.lock_pending().get_mut(&request_id) { + *slot = Some(id); } - if let Ok(last) = self.last_flush_at.lock() { - return last.elapsed() >= FLUSH_ELAPSED_THRESHOLD; - } - false } - /// Persist the current high-watermark and reset flush counters. - /// Idempotent: calling on an unmodified registry just rewrites the - /// same hwm. - pub fn flush(&self, persist: &dyn DatabaseHwmPersist) -> Result<(), DatabaseAllocError> { - let hwm = self.current_hwm(); - persist - .checkpoint(hwm) - .map_err(|e| DatabaseAllocError::FlushFailed { - detail: e.to_string(), - })?; - self.allocs_since_flush.store(0, Ordering::Release); - if let Ok(mut guard) = self.last_flush_at.lock() { - *guard = Instant::now(); - } - Ok(()) + /// Stop waiting on `request_id` and return its id, if it was applied. + pub fn finish_request(&self, request_id: u64) -> Option { + self.lock_pending().remove(&request_id).flatten() } - /// Test-only: force the elapsed-flush trigger by rewinding the - /// wall-clock anchor. - #[cfg(test)] - fn rewind_flush_clock(&self, by: Duration) { - if let Ok(mut guard) = self.last_flush_at.lock() - && let Some(earlier) = guard.checked_sub(by) - { - *guard = earlier; - } + fn lock_counter(&self) -> std::sync::MutexGuard<'_, Counter> { + self.counter.lock().unwrap_or_else(|p| p.into_inner()) + } + + fn lock_pending(&self) -> std::sync::MutexGuard<'_, HashMap>> { + self.pending.lock().unwrap_or_else(|p| p.into_inner()) } } @@ -175,10 +156,16 @@ impl Default for DatabaseRegistry { } } +fn persist_failed(e: crate::Error) -> DatabaseAllocError { + DatabaseAllocError::PersistFailed { + detail: e.to_string(), + } +} + impl From for crate::Error { fn from(e: DatabaseAllocError) -> Self { match e { - DatabaseAllocError::FlushFailed { detail } => crate::Error::Storage { + DatabaseAllocError::PersistFailed { detail } => crate::Error::Storage { engine: "database_registry".into(), detail, }, @@ -188,148 +175,145 @@ impl From for crate::Error { #[cfg(test)] mod tests { - use std::sync::Arc; - use std::sync::atomic::AtomicU32; + use std::sync::atomic::{AtomicBool, Ordering}; use super::*; + use crate::control::security::catalog::SystemCatalog; + #[derive(Default)] struct MemPersist { - last: std::sync::Mutex>, - calls: AtomicU32, + state: Mutex<(u64, u64)>, + fail: AtomicBool, } impl MemPersist { - fn new() -> Self { - Self { - last: std::sync::Mutex::new(None), - calls: AtomicU32::new(0), + fn check_fail(&self) -> crate::Result<()> { + if self.fail.load(Ordering::Acquire) { + return Err(crate::Error::Storage { + engine: "test".into(), + detail: "injected".into(), + }); } - } - - fn last(&self) -> Option { - *self.last.lock().unwrap() - } - - fn calls(&self) -> u32 { - self.calls.load(Ordering::Acquire) + Ok(()) } } impl DatabaseHwmPersist for MemPersist { - fn checkpoint(&self, hwm: u64) -> crate::Result<()> { - *self.last.lock().unwrap() = Some(hwm); - self.calls.fetch_add(1, Ordering::AcqRel); + fn checkpoint_reserve(&self, hwm: u64, reserve_index: u64) -> crate::Result<()> { + self.check_fail()?; + *self.state.lock().unwrap() = (hwm, reserve_index); Ok(()) } fn load(&self) -> crate::Result { - Ok(self.last().unwrap_or(0)) + Ok(self.state.lock().unwrap().0) } - } - #[test] - fn first_alloc_returns_user_db_start() { - let reg = DatabaseRegistry::new(); - let d = reg.alloc_one(); - assert_eq!(d.as_u64(), USER_DB_START); + fn load_reserve_index(&self) -> crate::Result { + Ok(self.state.lock().unwrap().1) + } } - #[test] - fn monotonic_100() { - let reg = DatabaseRegistry::new(); - let mut prev = 0u64; - for _ in 0..100 { - let d = reg.alloc_one(); - assert!(d.as_u64() > prev); - prev = d.as_u64(); - } + fn reopen(persist: &dyn DatabaseHwmPersist) -> DatabaseRegistry { + DatabaseRegistry::from_persisted( + persist.load().unwrap(), + persist.load_reserve_index().unwrap(), + ) } #[test] - fn restart_respects_hwm() { - let reg = DatabaseRegistry::from_persisted_hwm(5000); - let d = reg.alloc_one(); - assert_eq!(d.as_u64(), 5001); - assert_eq!(reg.current_hwm(), 5001); + fn first_reservation_is_user_db_start_and_persisted() { + let persist = MemPersist::default(); + let reg = DatabaseRegistry::new(); + let id = reg.reserve_at_index(1, &persist).unwrap(); + assert_eq!(id, Some(DatabaseId::new(USER_DB_START))); + assert_eq!(persist.load().unwrap(), USER_DB_START); + assert_eq!(persist.load_reserve_index().unwrap(), 1); } #[test] - fn restart_below_user_start_floored() { - // hwm=0 → counter starts at USER_DB_START - let reg = DatabaseRegistry::from_persisted_hwm(0); - let d = reg.alloc_one(); - assert_eq!(d.as_u64(), USER_DB_START); - // hwm=500 → still floored to USER_DB_START - let reg2 = DatabaseRegistry::from_persisted_hwm(500); - let d2 = reg2.alloc_one(); - assert_eq!(d2.as_u64(), USER_DB_START); + fn hwm_below_user_start_is_floored() { + let persist = MemPersist::default(); + let reg = DatabaseRegistry::from_persisted(500, 0); + assert_eq!( + reg.reserve_at_index(1, &persist).unwrap(), + Some(DatabaseId::new(USER_DB_START)) + ); } #[test] - fn restore_hwm_monotonic() { + fn persist_error_is_returned_and_leaves_the_counter() { + let persist = MemPersist::default(); let reg = DatabaseRegistry::new(); - reg.restore_hwm(9000); - let d = reg.alloc_one(); - assert_eq!(d.as_u64(), 9001); - // Lowering is a no-op - reg.restore_hwm(100); - let d2 = reg.alloc_one(); - assert_eq!(d2.as_u64(), 9002); + persist.fail.store(true, Ordering::Release); + assert!(reg.reserve_at_index(1, &persist).is_err()); + persist.fail.store(false, Ordering::Release); + assert_eq!( + reg.reserve_at_index(1, &persist).unwrap(), + Some(DatabaseId::new(USER_DB_START)) + ); } + /// Two nodes applying the same log compute the same ids. #[test] - fn concurrent_32x50_unique() { - let reg = Arc::new(DatabaseRegistry::new()); - let mut handles = Vec::with_capacity(32); - for _ in 0..32 { - let r = reg.clone(); - handles.push(std::thread::spawn(move || { - (0..50).map(|_| r.alloc_one().as_u64()).collect::>() - })); + fn reservations_agree_across_nodes() { + let (pa, pb) = (MemPersist::default(), MemPersist::default()); + let (a, b) = (DatabaseRegistry::new(), DatabaseRegistry::new()); + for index in [3, 8, 9] { + assert_eq!( + a.reserve_at_index(index, &pa).unwrap(), + b.reserve_at_index(index, &pb).unwrap() + ); } - let mut all: Vec = handles - .into_iter() - .flat_map(|h| h.join().unwrap()) - .collect(); - all.sort(); - all.dedup(); - assert_eq!(all.len(), 1600); + assert_eq!(a.current_hwm(), USER_DB_START + 2); } + /// A full-log replay after a restart skips every folded reservation and + /// issues fresh ids only for entries past the cursor. #[test] - fn flush_ops_threshold() { + fn replay_after_restart_skips_folded_reservations() { + let persist = MemPersist::default(); let reg = DatabaseRegistry::new(); - for _ in 0..(FLUSH_OPS_THRESHOLD - 1) { - reg.alloc_one(); - } - assert!(!reg.should_flush()); - reg.alloc_one(); - assert!(reg.should_flush()); - - let persist = MemPersist::new(); - reg.flush(&persist).unwrap(); - assert_eq!(persist.calls(), 1); - assert!(!reg.should_flush()); + let first = reg.reserve_at_index(4, &persist).unwrap(); + let second = reg.reserve_at_index(6, &persist).unwrap(); + assert_eq!(first, Some(DatabaseId::new(USER_DB_START))); + assert_eq!(second, Some(DatabaseId::new(USER_DB_START + 1))); + + let reg = reopen(&persist); + assert_eq!(reg.reserve_at_index(4, &persist).unwrap(), None); + assert_eq!(reg.reserve_at_index(6, &persist).unwrap(), None); + assert_eq!( + reg.reserve_at_index(11, &persist).unwrap(), + Some(DatabaseId::new(USER_DB_START + 2)) + ); } + /// The hwm and cursor survive a reopen of the real catalog. #[test] - fn flush_elapsed_threshold() { - let reg = DatabaseRegistry::new(); - reg.alloc_one(); - assert!(!reg.should_flush()); - reg.rewind_flush_clock(FLUSH_ELAPSED_THRESHOLD * 2); - assert!(reg.should_flush()); - let persist = MemPersist::new(); - reg.flush(&persist).unwrap(); - assert!(!reg.should_flush()); + fn catalog_backed_hwm_survives_reopen() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("system.redb"); + let before = { + let catalog = SystemCatalog::open(&path).unwrap(); + let reg = reopen(&catalog); + reg.reserve_at_index(5, &catalog).unwrap(); + reg.reserve_at_index(7, &catalog).unwrap().unwrap() + }; + let catalog = SystemCatalog::open(&path).unwrap(); + let reg = reopen(&catalog); + assert_eq!(reg.current_hwm(), before.as_u64()); + assert_eq!(reg.reserve_at_index(7, &catalog).unwrap(), None); + let after = reg.reserve_at_index(12, &catalog).unwrap().unwrap(); + assert!(after.as_u64() > before.as_u64()); } #[test] - fn flush_idempotent() { + fn request_routing_only_completes_waited_requests() { let reg = DatabaseRegistry::new(); - let persist = MemPersist::new(); - reg.flush(&persist).unwrap(); - reg.flush(&persist).unwrap(); - assert_eq!(persist.calls(), 2); + let request = reg.begin_request(); + reg.complete_request(request.wrapping_add(1), DatabaseId::new(9)); + reg.complete_request(request, DatabaseId::new(2000)); + assert_eq!(reg.finish_request(request), Some(DatabaseId::new(2000))); + assert_eq!(reg.finish_request(request), None); } } diff --git a/nodedb/src/control/distributed_applier/applied_index.rs b/nodedb/src/control/distributed_applier/applied_index.rs index 818197504..6ab3c790c 100644 --- a/nodedb/src/control/distributed_applier/applied_index.rs +++ b/nodedb/src/control/distributed_applier/applied_index.rs @@ -19,7 +19,7 @@ //! delivery at `applied_index + 1`, and the two never overlap. Maintaining it //! is entirely about WHERE the index advances: only after the write funnel's //! durable-at-ack barrier has fsynced that entry's redo record. Advancing at -//! engine-apply instead would leave a crash window between the engine's commit +//! engine-apply instead leaves a crash window between the engine's commit //! and the index write in which the entry is applied but not covered — the //! double-apply this exists to close. @@ -34,8 +34,8 @@ use crate::control::state::SharedState; /// returned `Ok` with an ok status, whose durable-at-ack barrier has performed /// that fsync. It deliberately does NOT fsync again. /// -/// A failed apply must NOT advance the floor, and neither may any entry BEHIND -/// one: leaving the floor below the first failure is what keeps that entry +/// A failed apply must NOT advance the floor, and no entry BEHIND +/// one advances it: leaving the floor below the first failure is what keeps that entry /// replayable on the next boot instead of silently skipped. Use /// [`AppliedPrefix`] to compute the index rather than passing a bare success. /// @@ -46,18 +46,21 @@ pub fn save_applied_index(state: &Arc, group_id: u64, applied_index let Some(sink) = state.raft_applied_index_sink.get() else { return; }; - if let Err(e) = sink(group_id, applied_index) { - // A failed save costs correctness only in the safe direction: the floor - // stays behind, so the next boot re-delivers entries WAL replay also - // covers — the pre-existing double-apply — rather than skipping any. - // The next successful apply in this group re-saves a higher index and - // closes the gap. - tracing::warn!( - group_id, - applied_index, - error = %e, - "failed to persist durable raft applied index" - ); + match sink(group_id, applied_index) { + Ok(()) => state.pitr.note_durable_applied(group_id, applied_index), + Err(e) => { + // A failed save costs correctness only in the safe direction: the + // floor stays behind, so the next boot re-delivers entries WAL + // replay also covers — the pre-existing double-apply — rather + // than skipping any. The next successful apply in this group + // re-saves a higher index and closes the gap. + tracing::warn!( + group_id, + applied_index, + error = %e, + "failed to persist durable raft applied index" + ); + } } } @@ -109,8 +112,8 @@ impl AppliedPrefix { /// Note an entry that carries no durable state — it neither advances the /// prefix nor breaks it. /// - /// Advancing on it would assert a redo record that was never written; - /// breaking on it would stall the floor and force the batch's later, + /// Advancing on it asserts a redo record that was never written; + /// breaking on it stalls the floor and forces the batch's later, /// genuinely durable writes to be re-delivered and applied twice. A /// no-op is the only correct answer, and it is spelled out rather than /// left implicit so every branch of the apply loop is deliberate. @@ -141,8 +144,8 @@ mod tests { prefix.record(1, true); prefix.record(2, true); prefix.record(3, false); - // Entry 3 never applied; 4 and 5 did. Saving 5 would make the next boot - // resume at 6 and drop 3 forever. + // Entry 3 never applied; 4 and 5 did. Saving 5 makes the next boot + // resume at 6 and drops 3 forever. prefix.record(4, true); prefix.record(5, true); assert_eq!(prefix.floor(), Some(2)); diff --git a/nodedb/src/control/distributed_applier/apply_loop/array_cell_route.rs b/nodedb/src/control/distributed_applier/apply_loop/array_cell_route.rs new file mode 100644 index 000000000..e1c1aee57 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/array_cell_route.rs @@ -0,0 +1,168 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Route a committed array cell write to its array incarnation before it +//! applies. +//! +//! The write names the key its proposer planned against. This replica can +//! have applied a MOVE TENANT or a DROP of that array since. The write then +//! applies under the key its incarnation moved to, or concludes as superseded +//! with a final refusal: the array it named no longer exists, so it has +//! nothing to mutate. It never fails for a key its catalog no longer holds. +//! +//! Both the SQL cell writes (`ArrayCellPut` / `ArrayCellDelete`) and the Lite +//! sync ops (`ArrayOp`) route here. + +use nodedb_raft::message::LogEntry; + +use crate::bridge::envelope::ErrorCode; +use crate::control::array_catalog::cell_route::{CellRoute, gate_incarnation, route}; +use crate::control::array_sync::raft_apply::AppliedPosition; +use crate::control::wal_replication::ReplicatedEntry; +use crate::control::write_gate::{self, GateKey, SharedGate}; +use crate::types::{DatabaseId, TenantId}; + +use super::context::{ApplyContext, FinishedApply}; +use super::proposal_gate::EntryOutcome; +use super::start::Prepared; +use super::write_dispatch::{EntryScope, prepare_generic_entry}; + +/// The array a committed cell write names, and the incarnation its proposer +/// stamped. +pub(super) struct CellWrite { + pub tenant_id: TenantId, + pub array: String, + pub incarnation: nodedb_types::Hlc, +} + +/// Prepare a committed SQL array cell write: route it, then apply it as an +/// exclusive generic entry while the incarnation's gate stays shared. +pub(super) fn prepare_array_cell_entry<'a>( + ctx: ApplyContext<'a>, + pos: AppliedPosition, + entry: LogEntry, + scope: EntryScope, + cell: CellWrite, +) -> Prepared<'a> { + Prepared::Exclusive(Box::pin(async move { + let (_gate, database_id) = match route_cell_write(ctx, pos, scope.database_id, &cell).await + { + Ok(routed) => routed, + Err(finished) => return *finished, + }; + let (entry, scope) = if database_id == scope.database_id { + (entry, scope) + } else { + match rekeyed(entry, database_id) { + Ok(entry) => ( + entry, + EntryScope { + database_id, + ..scope + }, + ), + Err(error) => { + ctx.tracker + .complete(pos.group_id, pos.log_index, pos.applied_key, Err(error)); + return concluded(pos, false); + } + } + }; + match prepare_generic_entry(ctx, pos, entry, scope, true) { + Prepared::Exclusive(apply) => apply.await, + Prepared::Concluded(outcome) => FinishedApply { + group_id: pos.group_id, + log_index: pos.log_index, + outcome, + }, + Prepared::Enqueue(_) | Prepared::Barrier => { + ctx.tracker.complete( + pos.group_id, + pos.log_index, + pos.applied_key, + Err(crate::Error::Internal { + detail: "an array cell write prepared as a non-exclusive entry".into(), + }), + ); + concluded(pos, false) + } + } + })) +} + +/// Route a committed array cell write. Returns the incarnation's gate, held +/// shared until the write is on its core, with the database it applies +/// under; or the concluded apply of a superseded write, boxed because a +/// superseded write is rare and a concluded apply is large. +pub(super) async fn route_cell_write( + ctx: ApplyContext<'_>, + pos: AppliedPosition, + database_id: DatabaseId, + cell: &CellWrite, +) -> Result<(SharedGate, DatabaseId), Box> { + let read_mirror = || match ctx.state.array_catalog.read() { + Ok(mirror) => mirror, + Err(poisoned) => poisoned.into_inner(), + }; + let incarnation = gate_incarnation( + &read_mirror(), + cell.tenant_id, + database_id, + &cell.array, + cell.incarnation, + ); + let Some(incarnation) = incarnation else { + return Err(Box::new(superseded(ctx, pos))); + }; + let gate = write_gate::shared(GateKey::Array(incarnation)).await; + let decision = route( + &read_mirror(), + cell.tenant_id, + database_id, + &cell.array, + incarnation, + ); + match decision { + CellRoute::Here => Ok((gate, database_id)), + CellRoute::Moved(to) => Ok((gate, to)), + CellRoute::Superseded => Err(Box::new(superseded(ctx, pos))), + } +} + +/// Conclude a write whose array incarnation no longer exists: a final +/// refusal, durable because nothing is left to apply. +fn superseded(ctx: ApplyContext<'_>, pos: AppliedPosition) -> FinishedApply { + ctx.tracker.complete( + pos.group_id, + pos.log_index, + pos.applied_key, + Err(crate::Error::DataPlane(ErrorCode::NotFound)), + ); + concluded(pos, true) +} + +fn concluded(pos: AppliedPosition, durable: bool) -> FinishedApply { + FinishedApply { + group_id: pos.group_id, + log_index: pos.log_index, + outcome: EntryOutcome::Applied { + durable, + result: None, + }, + } +} + +/// `entry` with its write moved to `database_id`. +fn rekeyed(entry: LogEntry, database_id: DatabaseId) -> crate::Result { + let mut replicated = + ReplicatedEntry::from_bytes(&entry.data).ok_or_else(|| crate::Error::Internal { + detail: format!( + "array cell entry {} does not decode for its reroute", + entry.index + ), + })?; + replicated.database_id = database_id.as_u64(); + Ok(LogEntry { + data: replicated.encode()?, + ..entry + }) +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/collection_route.rs b/nodedb/src/control/distributed_applier/apply_loop/collection_route.rs new file mode 100644 index 000000000..386093aca --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/collection_route.rs @@ -0,0 +1,211 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Apply a committed collection write only to the collection incarnation its +//! proposer planned against. +//! +//! The proposer stamps, for each collection the write names, the incarnation +//! its catalog held. On this replica: +//! +//! - The collection still holds that incarnation: the write applies. +//! - The collection is gone, or holds another incarnation: the write is +//! superseded. It concludes with a final refusal counted as durable, and +//! changes nothing. This covers a purge followed by a same-name create, and +//! the source of a MOVE TENANT, whose cutover re-issued every row into the +//! target before it moved the catalog row. Applying the write to the target +//! applies it twice. +//! - The write names no incarnation (`Hlc::ZERO`): it applies by key alone. +//! Only a write can be unstamped. Every committed collection row holds a +//! non-zero incarnation. +//! +//! The decision needs no WAL tombstone, so it holds after tombstone GC. +//! +//! The write routes and reaches its core's queue under each collection's +//! gate held shared. A purge or a MOVE TENANT reclaim of the key holds the +//! gate exclusive, so a routed write is on its core before the reclaim. + +use nodedb_types::{CollectionKey, Hlc}; + +use crate::control::security::catalog::SystemCatalog; +use crate::control::state::SharedState; +use crate::control::wal_replication::CollectionIncarnation; +use crate::control::write_gate::{self, GateKey, SharedGate}; +use crate::types::DatabaseId; + +/// Where a committed collection write goes. +pub(super) enum CollectionRoute { + /// The write applies. The gates stay held until it is on its core. + Apply(Vec), + /// A collection the write names no longer holds its incarnation. + Superseded, +} + +fn key_of<'a>(database_id: DatabaseId, named: &'a str) -> CollectionKey<'a> { + CollectionKey::from_qualified_str(database_id, named) + .unwrap_or_else(|_| CollectionKey::from_bare(database_id, named)) +} + +/// Route a write of `tenant_id` in `database_id` that names `named`. +pub(super) async fn route( + state: &SharedState, + tenant_id: u64, + database_id: DatabaseId, + named: &[CollectionIncarnation], +) -> crate::Result { + if named.is_empty() { + return Ok(CollectionRoute::Apply(Vec::new())); + } + let keys = named + .iter() + .map(|entry| { + let key = key_of(database_id, &entry.collection); + GateKey::Collection { + database_id: key.database_id().as_u64(), + tenant_id, + name: key.name().to_string(), + } + }) + .collect(); + let gates = write_gate::shared_all(keys).await; + if applies(state.credentials.catalog(), tenant_id, database_id, named)? { + Ok(CollectionRoute::Apply(gates)) + } else { + Ok(CollectionRoute::Superseded) + } +} + +/// Whether every collection of `named` still holds the incarnation its +/// proposer stamped, in this node's committed catalog. +pub(super) fn applies( + catalog: &SystemCatalog, + tenant_id: u64, + database_id: DatabaseId, + named: &[CollectionIncarnation], +) -> crate::Result { + for entry in named { + if entry.incarnation == Hlc::ZERO { + continue; + } + if !catalog.holds_incarnation( + database_id, + tenant_id, + &entry.collection, + entry.incarnation, + )? { + // Refusing a moved collection's write is correct: the cutover + // drained the source and re-issued every row into the target. + return Ok(false); + } + } + Ok(true) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + use crate::control::catalog_entry::CatalogEntry; + use crate::control::catalog_entry::descriptor_stamp::stamp; + use crate::control::security::catalog::StoredCollection; + use crate::control::security::credential::CredentialStore; + use nodedb_types::HlcClock; + + fn open() -> (Arc, tempfile::TempDir) { + let tmp = tempfile::tempdir().expect("tmpdir"); + let store = Arc::new(CredentialStore::open(&tmp.path().join("system.redb")).expect("open")); + (store, tmp) + } + + /// Stamp and write a put of `row`, as the metadata applier does. + fn put(catalog: &SystemCatalog, clock: &HlcClock, row: StoredCollection) -> StoredCollection { + let CatalogEntry::PutCollection(stamped) = + stamp(CatalogEntry::PutCollection(Box::new(row)), clock, catalog).expect("stamp") + else { + panic!("a collection put stamps as a collection put"); + }; + catalog + .put_collection(DatabaseId::DEFAULT, &stamped) + .expect("write the row"); + *stamped + } + + fn naming(incarnation: Hlc) -> Vec { + vec![CollectionIncarnation { + collection: "orders".into(), + incarnation, + }] + } + + /// A write planned against a purged incarnation never lands on a + /// same-name recreate, with no WAL tombstone left to fence it. A redo + /// entry names its collections the same way and is refused alike. An + /// ALTER keeps the incarnation, so a write planned before it applies. + #[test] + fn a_stale_write_never_lands_on_a_recreated_collection() { + let (store, _tmp) = open(); + let catalog = store.catalog(); + let clock = HlcClock::new(); + + let first = put(catalog, &clock, StoredCollection::new(1, "orders", "admin")); + assert_ne!( + first.incarnation, + Hlc::ZERO, + "a create names an incarnation" + ); + let stale = naming(first.incarnation); + assert!(applies(catalog, 1, DatabaseId::DEFAULT, &stale).expect("route")); + + let altered = put(catalog, &clock, first.clone()); + assert_eq!(altered.incarnation, first.incarnation, "ALTER keeps it"); + assert!(applies(catalog, 1, DatabaseId::DEFAULT, &stale).expect("route")); + + // Purge, with the tombstone already collected: only the row goes. + catalog + .delete_collection(DatabaseId::DEFAULT, 1, "orders") + .expect("purge the row"); + assert!(!applies(catalog, 1, DatabaseId::DEFAULT, &stale).expect("route")); + + let recreated = put(catalog, &clock, StoredCollection::new(1, "orders", "admin")); + assert_ne!(recreated.incarnation, first.incarnation); + assert!( + !applies(catalog, 1, DatabaseId::DEFAULT, &stale).expect("route"), + "the stale write is superseded" + ); + + // A redo entry that writes the collection among others routes on every + // collection it names. + let redo = |incarnation: Hlc| { + vec![ + CollectionIncarnation { + collection: "orders".into(), + incarnation, + }, + CollectionIncarnation { + collection: "ledger".into(), + incarnation: Hlc::ZERO, + }, + ] + }; + assert!( + !applies(catalog, 1, DatabaseId::DEFAULT, &redo(first.incarnation)).expect("route"), + "the stale redo entry is superseded" + ); + assert!( + applies( + catalog, + 1, + DatabaseId::DEFAULT, + &redo(recreated.incarnation) + ) + .expect("route") + ); + } + + #[test] + fn an_unstamped_write_applies_by_key() { + let (store, _tmp) = open(); + assert!( + applies(store.catalog(), 1, DatabaseId::DEFAULT, &naming(Hlc::ZERO)).expect("route") + ); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/driver.rs b/nodedb/src/control/distributed_applier/apply_loop/driver.rs index 1d515a0a6..865cf9ffe 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/driver.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/driver.rs @@ -6,7 +6,7 @@ use std::sync::Arc; -use tokio::sync::mpsc; +use tokio::sync::{mpsc, watch}; use crate::control::cluster::calvin::ReadResultEvent; use crate::control::distributed_applier::applier::ApplyBatch; @@ -38,7 +38,7 @@ pub async fn run_apply_loop( Ok(records) => records, Err(error) => { // Without the keys an entry re-delivered above the durable floor, - // or a second committed copy of a proposal, would apply a second + // or a second committed copy of a proposal, applies a second // time. Refuse to apply anything rather than risk it: the loop // stops, and every propose waiter surfaces the stall. tracing::error!( @@ -51,6 +51,8 @@ pub async fn run_apply_loop( }; let ledger = ProposalLedger::from_records(&records, PROPOSAL_LEDGER_CAPACITY); drop(records); + // Change events of this node's writes take Raft log positions from here on. + state.cdc_router.positions().mark_replicated(); let ctx = ApplyContext { state: &state, @@ -58,19 +60,22 @@ pub async fn run_apply_loop( calvin_read_result_senders: &calvin_read_result_senders, }; let mut pipeline = Pipeline::new(ctx, ledger); + let mut install_released = state.raft_apply_gates.get().map(|g| g.subscribe_released()); let mut accepting = true; loop { if !accepting && !pipeline.has_running() { - // The channel closed and every started entry concluded. The - // pump started every entry that can start, so none is queued. + // The channel closed and every started entry concluded. An entry + // still queued waits on a snapshot install that shutdown ends. return; } let running = pipeline.has_running(); + let blocked = pipeline.install_blocked(); tokio::select! { biased; Some(event) = pipeline.next_event(), if running => { pipeline.handle(event); } + () = wait_install_released(&mut install_released), if blocked => {} batch = apply_rx.recv(), if accepting => match batch { Some(batch) => { pipeline.accept(batch); @@ -87,3 +92,17 @@ pub async fn run_apply_loop( pipeline.settle(); } } + +/// Resolve once a snapshot install releases a group's apply gate. Never +/// resolves before `start_raft` installs the gates, when no install runs. +async fn wait_install_released(released: &mut Option>) { + match released { + Some(rx) => { + if rx.changed().await.is_err() { + // The gates are gone with the Raft loop: no install follows. + std::future::pending::<()>().await; + } + } + None => std::future::pending::<()>().await, + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/group_watch.rs b/nodedb/src/control/distributed_applier/apply_loop/group_watch.rs index 4bb3367d9..9d85000cc 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/group_watch.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/group_watch.rs @@ -1,21 +1,34 @@ // SPDX-License-Identifier: BUSL-1.1 //! Per-group state the apply loop keeps across batches: the highest index it -//! applied, to report a second apply of a committed entry, and the backup cut -//! floor, which raises the commit HLC of every entry after a cut barrier. +//! applied, to report a second apply of a committed entry, and the cut +//! floors, which raise the commit HLC of every entry after a cut barrier. -use std::collections::HashMap; +use std::collections::{BTreeMap, HashMap}; + +use crate::control::pitr::restore_point::RecordedCut; /// Per-group apply state. #[derive(Debug, Default)] pub(super) struct GroupWatch { highest_applied: HashMap, - /// Lowest commit HLC an entry after the group's latest cut barrier - /// records: one above that barrier's watermark. - cut_floor: HashMap, + /// Per group, barrier log index to the lowest commit HLC an entry after + /// that barrier records: one above the barrier's watermark. + cut_floors: HashMap>, } impl GroupWatch { + /// A watch that knows the restore point cuts this node applied before a + /// restart: the entries it applies again after one of them record above + /// it, as they did the first time. + pub(super) fn with_cuts(cuts: &[RecordedCut]) -> Self { + let mut watch = Self::default(); + for cut in cuts { + watch.raise_cut(cut.group_id, cut.barrier_index, cut.hlc); + } + watch + } + /// Note that the apply loop applies `(group_id, log_index)`. Reports an /// index at or below one it already applied as a second apply. pub(super) fn note_apply(&mut self, group_id: u64, log_index: u64) { @@ -25,29 +38,58 @@ impl GroupWatch { return; } *highest = log_index; + self.fold_cuts_below(group_id, log_index); } - /// Raise `group_id`'s cut floor above the watermark `cut_hlc` of a - /// backup's cut barrier. - pub(super) fn raise_cut(&mut self, group_id: u64, cut_hlc: u64) { - let floor = self.cut_floor.entry(group_id).or_insert(0); + /// Raise `group_id`'s floor above the watermark `cut_hlc` of the cut + /// barrier at `barrier_index`. + pub(super) fn raise_cut(&mut self, group_id: u64, barrier_index: u64, cut_hlc: u64) { + let floor = self + .cut_floors + .entry(group_id) + .or_default() + .entry(barrier_index) + .or_insert(0); *floor = (*floor).max(cut_hlc.saturating_add(1)); } - /// The commit HLC an entry of `group_id` stamped `write_hlc` records. + /// The commit HLC the entry of `group_id` at `log_index` stamped + /// `write_hlc` records. /// /// An entry the log places after a cut barrier records at least the cut - /// floor, however early its proposer stamped it: the backup that placed - /// the barrier did not contain it, so a restore of that backup refuses - /// it. `0` means the entry carries no stamp; its apply stamps its own - /// append, which already follows every barrier before it. - pub(super) fn commit_hlc(&self, group_id: u64, write_hlc: u64) -> u64 { + /// floor, however early its proposer stamped it: the cut did not contain + /// it, so a restore of that cut refuses it. `0` means the entry carries no + /// stamp; its apply stamps its own append, which already follows every + /// barrier before it. + pub(super) fn commit_hlc(&self, group_id: u64, log_index: u64, write_hlc: u64) -> u64 { if write_hlc == 0 { return 0; } - self.cut_floor + let floor = self + .cut_floors .get(&group_id) - .map_or(write_hlc, |floor| write_hlc.max(*floor)) + .and_then(|floors| floors.range(..log_index).map(|(_, floor)| *floor).max()); + floor.map_or(write_hlc, |floor| write_hlc.max(floor)) + } + + /// Fold every barrier below `log_index` into the highest of them: entries + /// apply in log order, so no later entry sits between two of them. + fn fold_cuts_below(&mut self, group_id: u64, log_index: u64) { + let Some(floors) = self.cut_floors.get_mut(&group_id) else { + return; + }; + let below: Vec<(u64, u64)> = floors.range(..log_index).map(|(i, f)| (*i, *f)).collect(); + let Some(&(last_index, _)) = below.last() else { + return; + }; + if below.len() < 2 { + return; + } + let folded = below.iter().map(|(_, floor)| *floor).max().unwrap_or(0); + for (index, _) in &below { + floors.remove(index); + } + floors.insert(last_index, folded); } } @@ -58,18 +100,116 @@ mod tests { #[test] fn an_entry_after_a_cut_records_above_the_cut() { let mut watch = GroupWatch::default(); - assert_eq!(watch.commit_hlc(1, 50), 50); - watch.raise_cut(1, 100); - assert_eq!(watch.commit_hlc(1, 50), 101); - assert_eq!(watch.commit_hlc(1, 200), 200); - assert_eq!(watch.commit_hlc(2, 50), 50, "a cut binds only its group"); + assert_eq!(watch.commit_hlc(1, 4, 50), 50); + watch.raise_cut(1, 5, 100); + assert_eq!(watch.commit_hlc(1, 6, 50), 101); + assert_eq!(watch.commit_hlc(1, 6, 200), 200); assert_eq!( - watch.commit_hlc(1, 0), + watch.commit_hlc(1, 5, 50), + 50, + "the barrier's own place is below it" + ); + assert_eq!(watch.commit_hlc(2, 6, 50), 50, "a cut binds only its group"); + assert_eq!( + watch.commit_hlc(1, 6, 0), 0, "an unstamped entry stamps its own append" ); } + #[test] + fn an_entry_applied_again_after_a_restart_records_as_before() { + let cuts = [ + RecordedCut { + group_id: 1, + barrier_index: 10, + hlc: 100, + }, + RecordedCut { + group_id: 1, + barrier_index: 20, + hlc: 200, + }, + ]; + let mut watch = GroupWatch::with_cuts(&cuts); + // Delivery resumes between the two barriers. + watch.note_apply(1, 15); + assert_eq!(watch.commit_hlc(1, 15, 50), 101); + watch.note_apply(1, 25); + assert_eq!(watch.commit_hlc(1, 25, 50), 201); + // Folding kept the highest floor below the applied index. + assert_eq!(watch.cut_floors[&1].len(), 1); + } + + /// A checkpoint truncates the WAL past the barrier's record, the node + /// restarts, and the floor still binds the entries after the barrier. + #[test] + fn a_floor_outlives_the_wal_record_of_its_barrier() { + use crate::control::pitr::restore_point::load_recorded_cuts; + use crate::control::security::catalog::SystemCatalog; + use crate::control::security::catalog::cut_floors::StoredBarrier; + use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + use crate::wal::WalManager; + use crate::wal::manager::NO_APPLY_KEY; + use nodedb_wal::record::{RecordType, RestorePointPayload}; + + let dir = tempfile::tempdir().unwrap(); + let catalog_path = dir.path().join("system.redb"); + let wal = WalManager::open_for_testing(&dir.path().join("wal")).unwrap(); + { + // The barrier at index 10 of group 1 applies with watermark 100. + let catalog = SystemCatalog::open(&catalog_path).unwrap(); + wal.appender(NO_APPLY_KEY) + .append_restore_point(&RestorePointPayload { + id: 9, + hlc: 100, + group_id: 1, + applied_index: 10, + term: 1, + next_epoch: 0, + epoch_system_ms: 0, + vshards: vec![3], + }) + .unwrap(); + catalog + .put_cut_floor( + 1, + StoredBarrier { + index: 10, + watermark: 100, + }, + 0, + ) + .unwrap(); + } + wal.seal_active_segment().unwrap(); + wal.appender(NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) + .append_put( + TenantId::new(1), + VShardId::new(3), + DatabaseId::DEFAULT, + b"row", + ) + .unwrap(); + wal.sync().unwrap(); + wal.truncate_before(Lsn::new(wal.next_lsn().as_u64())) + .unwrap(); + let barrier_left = wal.replay().unwrap().iter().any(|record| { + RecordType::from_raw(record.logical_record_type()) == Some(RecordType::RestorePoint) + }); + assert!( + !barrier_left, + "the checkpoint truncated the barrier's record" + ); + + // The restart reads the floors back from the catalog. + let catalog = SystemCatalog::open(&catalog_path).unwrap(); + let mut watch = GroupWatch::with_cuts(&load_recorded_cuts(&catalog).unwrap()); + watch.note_apply(1, 11); + assert_eq!(watch.commit_hlc(1, 11, 50), 101); + } + #[test] fn a_second_apply_of_an_index_is_counted() { let before = crate::diag::raft_entries_reapplied(); diff --git a/nodedb/src/control/distributed_applier/apply_loop/lane.rs b/nodedb/src/control/distributed_applier/apply_loop/lane.rs index be896f36f..71728e455 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/lane.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/lane.rs @@ -18,7 +18,7 @@ use crate::control::distributed_applier::applied_index::AppliedPrefix; use crate::control::distributed_applier::propose_tracker::{ApplyingEntry, ProposeTracker}; use crate::control::server::shared::write_admission::plan_writes_user_data; use crate::control::state::tenant_marks::{MarkSite, TenantMarks}; -use crate::control::wal_replication::{ReplicatedEntry, ReplicatedWrite, from_replicated_entry}; +use crate::control::wal_replication::{ReplicatedEntry, ReplicatedWrite, decode_replicated_entry}; use super::proposal_gate::PrefixStep; @@ -51,20 +51,23 @@ impl QueuedEntry { self.decoded.as_ref().map_or(0, |e| e.metadata_floor) } - /// `(tenant_id, write_hlc)` of an entry that writes a tenant's data. A - /// cut barrier and a Calvin read result write nothing, and an entry with - /// no proposer stamp has no commit HLC to record. - pub fn write_stamp(&self) -> Option<(u64, u64)> { + /// `(tenant_id, write_hlc, restore_id)` of an entry that writes a + /// tenant's data. A cut barrier, a Calvin read result and a surrogate + /// bind write no data, and an entry with no proposer stamp has no commit + /// HLC to record. + pub fn write_stamp(&self) -> Option<(u64, u64, u64)> { let decoded = self.decoded.as_ref()?; if decoded.write_hlc == 0 || matches!( decoded.write, - ReplicatedWrite::CutBarrier { .. } | ReplicatedWrite::CalvinReadResult { .. } + ReplicatedWrite::CutBarrier { .. } + | ReplicatedWrite::CalvinReadResult { .. } + | ReplicatedWrite::SurrogateBind { .. } ) { return None; } - Some((decoded.tenant_id, decoded.write_hlc)) + Some((decoded.tenant_id, decoded.write_hlc, decoded.restore_id)) } /// Whether the entry's plan writes user data, as the write funnel decides @@ -81,9 +84,10 @@ impl QueuedEntry { | ReplicatedWrite::TransactionRedo { .. } => true, ReplicatedWrite::ArraySchema { .. } | ReplicatedWrite::CutBarrier { .. } - | ReplicatedWrite::CalvinReadResult { .. } => false, + | ReplicatedWrite::CalvinReadResult { .. } + | ReplicatedWrite::SurrogateBind { .. } => false, _ => matches!( - from_replicated_entry(&self.entry.data, None), + decode_replicated_entry(&self.entry.data), Ok(Some((_, _, plan, _))) if plan_writes_user_data(&plan) ), } @@ -92,7 +96,10 @@ impl QueuedEntry { /// Whether the entry must apply with nothing else of its group in /// flight. The array paths await their own write inside the apply, so /// the loop cannot fix their arrival order at the core any other way, - /// and a schema import must follow every earlier entry's apply. + /// and a schema import must follow every earlier entry's apply. A topic + /// publication awaits its proposal marker's fsync. A cut barrier + /// persists its floor, and a capturing one snapshots its tenants, after + /// every earlier entry and before every later one. pub fn is_exclusive(&self) -> bool { self.decoded.as_ref().is_some_and(|e| { matches!( @@ -101,6 +108,8 @@ impl QueuedEntry { | ReplicatedWrite::ArraySchema { .. } | ReplicatedWrite::ArrayCellPut { .. } | ReplicatedWrite::ArrayCellDelete { .. } + | ReplicatedWrite::TopicPublish { .. } + | ReplicatedWrite::CutBarrier { .. } ) }) } @@ -125,9 +134,10 @@ pub(super) struct Slot { pub proposal_key: u64, /// The collection the entry writes, when its apply named one. pub collection: Option, - /// `(tenant_id, commit_hlc)` the entry records on its tenant's mark in - /// this group once it settles, when it carries a proposer stamp. - pub write_mark: Option<(u64, u64)>, + /// `(tenant_id, commit_hlc, restore_id)` the entry records on its + /// tenant's mark in this group once it settles, when it carries a + /// proposer stamp. A non-zero `restore_id` raises the restore mark. + pub write_mark: Option<(u64, u64, u64)>, /// Whether the entry's plan writes user data. Only such an entry raises /// its tenant's mark. pub user_write: bool, @@ -144,10 +154,12 @@ pub(super) struct Lane { pub blocking: Option, /// The durable prefix over every entry this process settled for the /// group. A break holds for the life of the process: an index saved past - /// a non-durable entry would let the next boot skip it. + /// a non-durable entry lets the next boot skip it. prefix: AppliedPrefix, /// The floor last saved for the group. saved_floor: Option, + /// The entry the last settle ended at. + last_settled: Option, } impl Lane { @@ -159,9 +171,20 @@ impl Lane { blocking: None, prefix: AppliedPrefix::new(), saved_floor: None, + last_settled: None, } } + /// The first started entry not yet settled. + pub fn front_index(&self) -> Option { + self.slots.front().map(|slot| slot.log_index) + } + + /// The entry the last settle ended at. + pub fn last_settled(&self) -> Option { + self.last_settled + } + /// Whether any started entry of the group has not concluded. pub fn has_running(&self) -> bool { self.slots @@ -246,7 +269,7 @@ impl Lane { SlotState::Starting | SlotState::Running => break, SlotState::Barrier => { // Every entry before the barrier finished; a waiting - // backup may snapshot this group now. + // backup can snapshot this group now. tracker.complete( self.group_id, front.log_index, @@ -263,15 +286,25 @@ impl Lane { }; let log_index = front.log_index; if front.user_write - && let Some((tenant_id, commit_hlc)) = front.write_mark + && let Some((tenant_id, commit_hlc, restore_id)) = front.write_mark { - marks.raise( - self.group_id, - tenant_id, - commit_hlc, - MarkSite::ReplicatedApply, - front.collection.as_deref(), - ); + if restore_id == 0 { + marks.raise( + self.group_id, + tenant_id, + commit_hlc, + MarkSite::ReplicatedApply, + front.collection.as_deref(), + ); + } else { + marks.raise_restore( + self.group_id, + tenant_id, + commit_hlc, + front.collection.as_deref(), + restore_id, + ); + } } self.slots.pop_front(); match step { @@ -279,6 +312,7 @@ impl Lane { PrefixStep::Record(durable) => self.prefix.record(log_index, durable), } tracker.note_applied(self.group_id, log_index); + self.last_settled = Some(log_index); settled += 1; } tracker.note_applying( @@ -403,11 +437,11 @@ mod tests { let marks = TenantMarks::default(); let mut lane = Lane::new(4); let mut write = slot(1, SlotState::Concluded(PrefixStep::Record(true))); - write.write_mark = Some((7, 500)); + write.write_mark = Some((7, 500, 0)); write.user_write = true; write.collection = Some("docs".to_owned()); let mut index_change = slot(2, SlotState::Concluded(PrefixStep::Record(true))); - index_change.write_mark = Some((7, 900)); + index_change.write_mark = Some((7, 900, 0)); lane.push(write); lane.push(index_change); @@ -420,13 +454,30 @@ mod tests { assert_eq!(mark.collection.as_deref(), Some("docs")); } + #[test] + fn a_restore_write_raises_the_restore_mark_under_its_id() { + let tracker = ProposeTracker::new(); + let marks = TenantMarks::default(); + let mut lane = Lane::new(4); + let mut write = slot(1, SlotState::Concluded(PrefixStep::Record(true))); + write.write_mark = Some((7, 600, 42)); + write.user_write = true; + lane.push(write); + + lane.settle(&tracker, &marks); + assert_eq!(marks.get(4, 7), None, "a restore write raises no user mark"); + let all = marks.get_all(4, 7); + assert_eq!(all.len(), 1); + assert_eq!((all[0].hlc, all[0].restore_id), (600, 42)); + } + #[test] fn a_refused_user_write_raises_no_mark() { let tracker = ProposeTracker::new(); let marks = TenantMarks::default(); let mut lane = Lane::new(4); let mut refused = slot(1, SlotState::Running); - refused.write_mark = Some((7, 500)); + refused.write_mark = Some((7, 500, 0)); refused.user_write = true; refused.collection = Some("docs".to_owned()); lane.push(refused); diff --git a/nodedb/src/control/distributed_applier/apply_loop/metadata_floor.rs b/nodedb/src/control/distributed_applier/apply_loop/metadata_floor.rs index d182027b2..d0b870dc8 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/metadata_floor.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/metadata_floor.rs @@ -128,3 +128,60 @@ fn conclude_unapplied( result: Some(applied), } } + +#[cfg(test)] +mod tests { + use std::sync::atomic::{AtomicBool, Ordering}; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::wal::WalManager; + + /// A follower whose metadata apply lags a collection's creation holds a + /// data entry for that collection until the creation applied. The + /// metadata watcher bumps only after the applier returned, and its + /// return includes the creation's post-apply storage clear. The write + /// therefore lands after the clear, and survives it. + #[tokio::test(flavor = "multi_thread")] + async fn a_lagging_followers_write_waits_for_the_create() { + let directory = tempfile::tempdir().expect("temporary WAL directory"); + let wal = Arc::new( + WalManager::open_for_testing(&directory.path().join("hold.wal")).expect("test WAL"), + ); + let (dispatcher, _sides) = Dispatcher::new(1, 64); + let state = SharedState::new(dispatcher, wal).expect("shared state"); + let tracker = Arc::new(ProposeTracker::new()); + let watcher = state.applied_index_watcher(METADATA_GROUP_ID); + let create_index = watcher.current() + 2; + + let enqueued = Arc::new(AtomicBool::new(false)); + let flag = Arc::clone(&enqueued); + let write = Prepared::Enqueue(Box::pin(async move { + flag.store(true, Ordering::SeqCst); + StartedEntry::concluded(EntryOutcome::Skipped) + })); + let held = HeldEntry { + group_id: 7, + log_index: 1, + proposal_key: 0, + metadata_floor: create_index, + }; + let Prepared::Enqueue(mut hold) = hold_for_metadata(&state, &tracker, held, write) else { + panic!("a write below its floor stays an enqueue behind the hold"); + }; + + assert!( + tokio::time::timeout(Duration::from_millis(200), &mut hold) + .await + .is_err(), + "the write waits while the create is unapplied" + ); + assert!(!enqueued.load(Ordering::SeqCst)); + + watcher.bump(create_index); + tokio::time::timeout(Duration::from_secs(10), hold) + .await + .expect("the write proceeds once the create applied"); + assert!(enqueued.load(Ordering::SeqCst)); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/mod.rs b/nodedb/src/control/distributed_applier/apply_loop/mod.rs index f6566ef51..00d25c0f8 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/mod.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/mod.rs @@ -26,16 +26,26 @@ //! - [`write_dispatch`]: the generic decode + write-funnel enqueue path. //! - [`transaction_redo`]: a committed transaction's redo, stamped with its //! Raft entry and applied through the WAL replay arms. +//! - [`topic_publish`]: a committed durable-topic publication, appended to +//! this replica's topic log at the entry's position. //! - [`proposal_gate`]: skips a second committed copy of an applied proposal //! and records each applied proposal in the ledger. +//! - [`snapshot_gate`]: skips an entry an installed snapshot covers, and +//! orders each write before a later snapshot restore. //! - [`group_watch`]: per-group second-apply detection and backup cut floors. //! - [`bookkeeping`]: applied-floor persistence + Raft log compaction trigger. //! - [`helpers`]: shared response/result classification helpers. //! - [`metadata_floor`]: holds a write until this node's catalog reached the //! one its proposer planned it against. +//! - [`array_cell_route`]: routes an array cell write to the incarnation its +//! proposer wrote against. +//! - [`collection_route`]: applies a collection write only while each +//! collection it names holds the incarnation its proposer planned against. +mod array_cell_route; mod bookkeeping; mod calvin_read_result; +mod collection_route; mod context; mod driver; mod group_watch; @@ -44,7 +54,10 @@ mod lane; mod metadata_floor; mod pipeline; mod proposal_gate; +mod snapshot_gate; mod start; +mod surrogate_bind; +mod topic_publish; mod transaction_redo; mod write_dispatch; diff --git a/nodedb/src/control/distributed_applier/apply_loop/pipeline.rs b/nodedb/src/control/distributed_applier/apply_loop/pipeline.rs index 7f22690fe..89dccd45d 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/pipeline.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/pipeline.rs @@ -11,12 +11,15 @@ //! wait for it to finish. use std::collections::HashMap; +use std::sync::Arc; use futures::FutureExt; use futures::stream::{FuturesUnordered, StreamExt}; use crate::control::distributed_applier::applier::ApplyBatch; use crate::control::distributed_applier::proposal_ledger::ProposalLedger; +use crate::control::server::dispatch_utils::publish_settled_changes; +use crate::control::state::SharedState; use super::bookkeeping::record_durable_apply; use super::context::{ApplyContext, ApplyFuture, LoopEvent, LoopFuture, Started, StartedEntry}; @@ -24,6 +27,7 @@ use super::group_watch::GroupWatch; use super::lane::{Lane, QueuedEntry, Slot, SlotState}; use super::metadata_floor::{HeldEntry, hold_for_metadata}; use super::proposal_gate::{EntryOutcome, ProposalGate}; +use super::snapshot_gate::{EntryAdmission, admit_entry, hold_permit}; use super::start::{Prepared, prepare_entry}; /// Every group's lane, and the enqueues and applies that run. @@ -33,6 +37,9 @@ pub(super) struct Pipeline<'a> { gate: ProposalGate, watch: GroupWatch, running: FuturesUnordered>, + /// Set by a pump that left an entry queued because a snapshot install + /// held its group's apply gate. + install_blocked: bool, } impl<'a> Pipeline<'a> { @@ -41,11 +48,18 @@ impl<'a> Pipeline<'a> { ctx, lanes: HashMap::new(), gate: ProposalGate::new(ledger), - watch: GroupWatch::default(), + watch: GroupWatch::with_cuts(ctx.state.pitr.recorded_cuts()), running: FuturesUnordered::new(), + install_blocked: false, } } + /// Whether the last pump left an entry queued behind a snapshot install. + /// The loop pumps again once an install releases its gate. + pub fn install_blocked(&self) -> bool { + self.install_blocked + } + /// Queue a batch the applier handed off, behind its group's earlier /// entries. pub fn accept(&mut self, batch: ApplyBatch) { @@ -53,8 +67,17 @@ impl<'a> Pipeline<'a> { .lanes .entry(batch.group_id) .or_insert_with(|| Lane::new(batch.group_id)); + let first = lane.backlog.len(); lane.backlog .extend(batch.entries.into_iter().map(QueuedEntry::new)); + // Every handed-off entry is committed. A snapshot this node builds + // carries their keys. + self.ctx.tracker.note_committed( + batch.group_id, + lane.backlog + .range(first..) + .map(|queued| (queued.entry.index, queued.proposal_key())), + ); } /// Whether any enqueue or apply runs. @@ -88,6 +111,9 @@ impl<'a> Pipeline<'a> { self.enqueued(group_id, log_index, entry), ), LoopEvent::Finished(finished) => { + self.ctx + .tracker + .note_concluded(finished.group_id, finished.log_index); let gate = &mut self.gate; let wrote_rows = finished.outcome.wrote_rows(); let handled = self.lanes.get_mut(&finished.group_id).is_some_and(|lane| { @@ -127,6 +153,7 @@ impl<'a> Pipeline<'a> { Started::Concluded(outcome) => { // The write concluded without reaching its core. It leaves // its enqueue and concludes in one step. + self.ctx.tracker.note_concluded(group_id, log_index); let gate = &mut self.gate; let wrote_rows = outcome.wrote_rows(); lane.enqueued(log_index, SlotState::Running, collection, user_write) @@ -139,6 +166,12 @@ impl<'a> Pipeline<'a> { /// Start every entry each group can start now, in log order. pub fn pump(&mut self) { + self.install_blocked = false; + // Keys a snapshot install carried, before any entry starts: a later + // copy of one of those proposals is a duplicate here too. + for key in self.ctx.tracker.take_restored_keys() { + self.gate.note_restored(key); + } let groups: Vec = self .lanes .iter() @@ -151,16 +184,66 @@ impl<'a> Pipeline<'a> { } fn pump_group(&mut self, group_id: u64) { + let shared: &'a Arc = self.ctx.state; + let gates = shared.raft_apply_gates.get().map(|gates| &**gates); while let Some(queued) = self.next_startable(group_id) { let log_index = queued.entry.index; let proposal_key = queued.proposal_key(); + let covered_through = self.ctx.tracker.covered_through(group_id); + let permit = match admit_entry(gates, group_id, log_index, covered_through) { + EntryAdmission::Installing => { + if let Some(lane) = self.lanes.get_mut(&group_id) { + lane.backlog.push_front(queued); + } + self.install_blocked = true; + return; + } + EntryAdmission::Start(permit) => permit, + // At or below the snapshot's Raft index: the installed state + // and the Raft boundary both hold the entry. + EntryAdmission::Covered => { + if !self.conclude_covered( + group_id, + log_index, + proposal_key, + EntryOutcome::Skipped, + ) { + return; + } + continue; + } + // Above the Raft index and at or below the cut: only the + // installed state holds the entry. It concludes without + // applying and extends the durable prefix, so a restart never + // delivers it again. + EntryAdmission::CoveredByCut => { + if !self.conclude_covered( + group_id, + log_index, + proposal_key, + EntryOutcome::Covered, + ) { + return; + } + continue; + } + }; + // Every entry that leaves the backlog for good starts. A snapshot + // builder cuts at the highest one. + self.ctx.tracker.note_started(group_id, log_index); // Stamped before the entry is prepared: a barrier raises the cut // floor only for the entries after it. - let write_mark = queued.write_stamp().map(|(tenant_id, write_hlc)| { - (tenant_id, self.watch.commit_hlc(group_id, write_hlc)) - }); + let write_mark = queued + .write_stamp() + .map(|(tenant_id, write_hlc, restore_id)| { + ( + tenant_id, + self.watch.commit_hlc(group_id, log_index, write_hlc), + restore_id, + ) + }); // A second copy of an applied proposal never reaches the funnel, - // so its plan is classified here. Its first copy may sit above + // so its plan is classified here. Its first copy can sit above // the saved floor, and this copy then carries the mark again. let repeat_writes = self.gate.prior_wrote_rows(proposal_key) && queued.plan_writes_user_data(); @@ -188,6 +271,8 @@ impl<'a> Pipeline<'a> { Prepared::Barrier => (SlotState::Barrier, false, false), Prepared::Enqueue(enqueue) => { self.gate.open(proposal_key); + self.ctx.tracker.note_dispatched(group_id, log_index); + let enqueue = hold_permit(permit, enqueue); self.running .push(Box::pin(enqueue.map(move |entry| LoopEvent::Enqueued { group_id, @@ -200,7 +285,9 @@ impl<'a> Pipeline<'a> { Prepared::Exclusive(apply) => { // An array op or cell write: user data. self.gate.open(proposal_key); - self.running.push(finished_event(apply)); + self.ctx.tracker.note_dispatched(group_id, log_index); + self.running + .push(finished_event(hold_permit(permit, apply))); (SlotState::Running, true, true) } }; @@ -221,7 +308,36 @@ impl<'a> Pipeline<'a> { } } - /// Take the next entry of `group_id` when it may start now. + /// Conclude an entry an installed snapshot holds, without reaching a + /// core. It counts as started. Its waiter learns the write committed + /// without a result. Returns `false` when the group has no lane. + fn conclude_covered( + &mut self, + group_id: u64, + log_index: u64, + proposal_key: u64, + outcome: EntryOutcome, + ) -> bool { + self.ctx.tracker.note_started(group_id, log_index); + self.ctx + .tracker + .complete_covered(group_id, log_index, proposal_key); + let concluded = SlotState::Concluded(self.gate.conclude(proposal_key, false, outcome)); + let Some(lane) = self.lanes.get_mut(&group_id) else { + return false; + }; + lane.push(Slot { + log_index, + proposal_key, + collection: None, + write_mark: None, + user_write: false, + state: concluded, + }); + true + } + + /// Take the next entry of `group_id` when it can start now. /// /// It waits while the group's previous write is in its enqueue or an /// exclusive entry of the group runs, while it is exclusive and an @@ -253,9 +369,15 @@ impl<'a> Pipeline<'a> { let state = self.ctx.state; let tracker = self.ctx.tracker; for (group_id, lane) in &mut self.lanes { + let first = lane.front_index(); let settled = lane.settle(tracker, &state.tenant_marks); if settled > 0 { tracker.window().release(*group_id, settled); + // The entries' change events publish in log order, at their + // log positions, on every replica. + if let (Some(first), Some(last)) = (first, lane.last_settled()) { + publish_settled_changes(state, *group_id, first, last); + } } if !lane.floor_pending() { continue; @@ -274,6 +396,8 @@ impl<'a> Pipeline<'a> { continue; } if let Some(floor) = lane.take_floor_to_save() { + // Every entry the floor covers keeps its changes on disk. + state.change_stream.journal_group(*group_id, floor); record_durable_apply(state, *group_id, floor); } } diff --git a/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs b/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs index 2d1325ee5..7eda3df9d 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/proposal_gate.rs @@ -24,6 +24,10 @@ pub(super) enum EntryOutcome { /// first copy's outcome is durable, so it extends the prefix. The ledger /// already holds the proposal. Repeat, + /// A snapshot this node installed holds the entry's effects: the + /// snapshot was cut above its Raft index. The installed state is durable, + /// so the entry extends the prefix without applying. + Covered, /// The entry was applied. `durable` says its outcome survives a restart. /// `result` is what its waiter received, when the apply produced one. Applied { @@ -39,7 +43,7 @@ impl EntryOutcome { /// [`ProposalGate::prior_wrote_rows`]). pub fn wrote_rows(&self) -> bool { match self { - Self::Skipped | Self::Repeat => false, + Self::Skipped | Self::Repeat | Self::Covered => false, Self::Applied { result: Some(outcome), .. @@ -137,6 +141,12 @@ impl ProposalGate { true } + /// Note a proposal a snapshot install covered. It committed and applied + /// through the snapshot, with no outcome kept on this node. + pub fn note_restored(&mut self, proposal_key: u64) { + self.ledger.note(proposal_key, None); + } + /// Note that a copy of `proposal_key` started and has not concluded. pub fn open(&mut self, proposal_key: u64) { if proposal_key != 0 { @@ -168,6 +178,13 @@ impl ProposalGate { match outcome { EntryOutcome::Skipped => PrefixStep::Neutral, EntryOutcome::Repeat => PrefixStep::Record(true), + EntryOutcome::Covered => { + // A later copy of the proposal is a duplicate here too. + if proposal_key != 0 { + self.ledger.note(proposal_key, None); + } + PrefixStep::Record(true) + } EntryOutcome::Applied { durable, result } => { if durable { self.ledger.note(proposal_key, result); @@ -221,4 +238,16 @@ mod tests { assert!(!gate.in_flight(7)); assert!(!gate.skip_duplicate(&tracker, 1, 9, 7)); } + + /// An entry an installed snapshot's cut covers extends the durable + /// prefix, and a later copy of its proposal never applies. + #[test] + fn a_cut_covered_entry_extends_the_prefix_and_skips_a_later_copy() { + let tracker = ProposeTracker::new(); + let mut gate = ProposalGate::new(ProposalLedger::new(PROPOSAL_LEDGER_CAPACITY)); + let step = gate.conclude(11, false, EntryOutcome::Covered); + assert_eq!(step, PrefixStep::Record(true)); + assert!(!EntryOutcome::Covered.wrote_rows()); + assert!(gate.skip_duplicate(&tracker, 1, 14, 11)); + } } diff --git a/nodedb/src/control/distributed_applier/apply_loop/snapshot_gate.rs b/nodedb/src/control/distributed_applier/apply_loop/snapshot_gate.rs new file mode 100644 index 000000000..32c8e1327 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/snapshot_gate.rs @@ -0,0 +1,189 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Keeps an entry a snapshot install covers off the Data Plane. +//! +//! Entries wait in the apply channel and the group lanes after the Raft tick +//! handed them off. A snapshot can install for the group meanwhile. Each +//! entry therefore takes the group's apply gate before it starts, and is +//! skipped when the gate reports an installed snapshot at or above its index. +//! The permit is held until the entry's write reached its core. An install +//! waits for it, so a core always receives the write before the restore. + +use std::future::Future; +use std::pin::Pin; + +use nodedb_cluster::{ApplyPermit, GroupApplyGates}; + +/// Whether an entry can start. +pub(super) enum EntryAdmission { + /// Start the entry. The permit, when gates are wired, must be held until + /// the entry's write reached its core. + Start(Option), + /// An installed snapshot already holds the entry's effects: the entry is + /// at or below the snapshot's Raft index. + Covered, + /// An installed snapshot holds the entry's effects: the entry is above + /// the snapshot's Raft index and at or below the applied index the + /// snapshot was cut at. + CoveredByCut, + /// An install holds or awaits the gate. Retry once it releases. + Installing, +} + +/// Admit entry `log_index` of `group_id`. `gates` is `None` before +/// `start_raft` installs them, when no snapshot installs. `covered_through` +/// is the cut of the last snapshot this node installed for the group. +pub(super) fn admit_entry( + gates: Option<&GroupApplyGates>, + group_id: u64, + log_index: u64, + covered_through: u64, +) -> EntryAdmission { + let permit = match gates { + None => None, + Some(gates) => match gates.try_apply(group_id) { + None => return EntryAdmission::Installing, + Some(permit) if log_index <= permit.installed_through() => { + return EntryAdmission::Covered; + } + Some(permit) => Some(permit), + }, + }; + if log_index <= covered_through { + return EntryAdmission::CoveredByCut; + } + EntryAdmission::Start(permit) +} + +/// `work`, holding `permit` until it completes. +pub(super) fn hold_permit<'a, T: 'a>( + permit: Option, + work: Pin + Send + 'a>>, +) -> Pin + Send + 'a>> { + match permit { + None => work, + Some(permit) => Box::pin(async move { + let output = work.await; + drop(permit); + output + }), + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + + #[test] + fn a_node_without_raft_starts_every_entry() { + assert!(matches!( + admit_entry(None, 1, 7, 0), + EntryAdmission::Start(None) + )); + } + + /// Entries queued before an install skip what the snapshot covers and + /// start the rest. + #[tokio::test] + async fn queued_entries_at_or_below_an_installed_snapshot_are_covered() { + let gates = Arc::new(GroupApplyGates::new()); + gates.mount(1); + assert!(matches!( + admit_entry(Some(&gates), 1, 5, 0), + EntryAdmission::Start(Some(_)) + )); + + let mut install = gates.install(1).await; + assert!(matches!( + admit_entry(Some(&gates), 1, 5, 0), + EntryAdmission::Installing + )); + install.adopted(5); + drop(install); + + assert!(matches!( + admit_entry(Some(&gates), 1, 4, 0), + EntryAdmission::Covered + )); + assert!(matches!( + admit_entry(Some(&gates), 1, 5, 0), + EntryAdmission::Covered + )); + assert!(matches!( + admit_entry(Some(&gates), 1, 6, 0), + EntryAdmission::Start(Some(_)) + )); + assert!(matches!( + admit_entry(Some(&gates), 2, 5, 0), + EntryAdmission::Start(Some(_)) + )); + } + + /// A snapshot cut at applied index 12 and sent at Raft index 9 holds the + /// effects of entries 10 through 12. They reach the follower from the log + /// after the install and never start, so none applies twice. Entry 13 + /// starts. + #[tokio::test] + async fn entries_between_the_raft_index_and_the_cut_never_start() { + let gates = Arc::new(GroupApplyGates::new()); + gates.mount(1); + gates.install(1).await.adopted(9); + let cut = 12; + assert!(matches!( + admit_entry(Some(&gates), 1, 9, cut), + EntryAdmission::Covered + )); + for log_index in 10..=cut { + assert!( + matches!( + admit_entry(Some(&gates), 1, log_index, cut), + EntryAdmission::CoveredByCut + ), + "entry {log_index} is in the installed state and must not apply again" + ); + } + assert!(matches!( + admit_entry(Some(&gates), 1, 13, cut), + EntryAdmission::Start(Some(_)) + )); + assert!(matches!( + admit_entry(None, 1, 11, cut), + EntryAdmission::CoveredByCut + )); + } + + /// An install cannot start while a started write still holds its permit. + #[tokio::test] + async fn an_install_waits_for_a_started_write() { + let gates = Arc::new(GroupApplyGates::new()); + gates.mount(1); + let EntryAdmission::Start(permit) = admit_entry(Some(&gates), 1, 6, 0) else { + panic!("open gate must start the entry"); + }; + let (release, released) = tokio::sync::oneshot::channel::<()>(); + let write = hold_permit( + permit, + Box::pin(async move { + let _ = released.await; + }), + ); + let write = tokio::spawn(write); + + let install = { + let gates = Arc::clone(&gates); + tokio::spawn(async move { gates.install(1).await.adopted(6) }) + }; + tokio::task::yield_now().await; + assert!(!install.is_finished(), "install must wait for the write"); + + let _ = release.send(()); + write.await.expect("write task"); + install.await.expect("install task"); + assert!(matches!( + admit_entry(Some(&gates), 1, 6, 0), + EntryAdmission::Covered + )); + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/start.rs b/nodedb/src/control/distributed_applier/apply_loop/start.rs index 2edb3efe2..87b8f7ff3 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/start.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/start.rs @@ -12,7 +12,9 @@ use crate::control::array_sync::ArrayOpTarget; use crate::control::array_sync::raft_apply::{ AppliedPosition, ArraySchemaPayload, apply_array_op, apply_array_schema, }; -use crate::control::wal_replication::ReplicatedWrite; +use nodedb_raft::message::LogEntry; + +use crate::control::wal_replication::{ReplicatedEntry, ReplicatedWrite}; use crate::types::{DatabaseId, TenantId}; use super::calvin_read_result::{CalvinReadResultFields, forward_calvin_read_result}; @@ -20,6 +22,7 @@ use super::context::{ApplyContext, ApplyFuture, EnqueueFuture, FinishedApply}; use super::group_watch::GroupWatch; use super::lane::QueuedEntry; use super::proposal_gate::{EntryOutcome, ProposalGate}; +use super::topic_publish::{TopicPublishEntry, prepare_topic_publish_entry}; use super::transaction_redo::prepare_transaction_redo_entry; use super::write_dispatch::{EntryScope, prepare_generic_entry}; @@ -46,30 +49,13 @@ pub(super) fn prepare_entry<'a>( group_id: u64, queued: QueuedEntry, ) -> Prepared<'a> { - let QueuedEntry { entry, decoded } = queued; + let QueuedEntry { entry, mut decoded } = queued; let log_index = entry.index; + let log_term = entry.term; watch.note_apply(group_id, log_index); - // A leader-change no-op committed where a proposer may wait. The - // proposer's data is gone; firing an empty success would tell it the - // write applied. `RetryableLeaderChange` makes the gateway re-propose. if entry.data.is_empty() { - tracing::error!( - group_id, - log_index, - "leader-change no-op committed at index where a proposer was waiting; \ - surfacing RetryableLeaderChange so the gateway re-proposes" - ); - ctx.tracker.complete( - group_id, - log_index, - 0, - Err(crate::Error::RetryableLeaderChange { - group_id, - log_index, - }), - ); - return Prepared::Concluded(EntryOutcome::Skipped); + return conclude_leader_change_noop(ctx, group_id, log_index); } // `0` for unparseable / pre-key entries; the tracker treats 0 as "no key" @@ -79,22 +65,12 @@ pub(super) fn prepare_entry<'a>( // before this entry. The entry's mark carries it, not the instant this // replica applies, so a late apply never records a write as newer than a // backup taken after its ack. - let commit_hlc = watch.commit_hlc(group_id, decoded.as_ref().map_or(0, |e| e.write_hlc)); - // Database scope for the entry, read from the wire. The generic decode - // path returns no scope, so it is taken from the entry itself: a redo - // appended under the wrong scope replays into the wrong namespace. - let database_id = decoded - .as_ref() - .map_or(DatabaseId::DEFAULT, |e| DatabaseId::new(e.database_id)); - // The source the proposer stamped. An entry that does not decode applies - // nothing, so its source is never read. - let event_source = decoded - .as_ref() - .map_or(crate::event::EventSource::User, |e| e.event_source.into()); - let scope = EntryScope { - database_id, - event_source, - }; + let commit_hlc = watch.commit_hlc( + group_id, + log_index, + decoded.as_ref().map_or(0, |e| e.write_hlc), + ); + let scope = entry_scope(&mut decoded); // A second committed copy of a proposal this node already applied (a // re-proposal after a leader change whose first copy also committed) @@ -112,42 +88,94 @@ pub(super) fn prepare_entry<'a>( let Some(replicated) = decoded else { return prepare_generic_entry(ctx, pos, entry, scope, false); }; + prepare_replicated(ctx, watch, pos, log_term, entry, scope, replicated) +} + +/// A leader-change no-op committed where a proposer can wait. The +/// proposer's data is gone; firing an empty success tells it the +/// write applied. `RetryableLeaderChange` makes the gateway re-propose. +fn conclude_leader_change_noop<'a>( + ctx: ApplyContext<'a>, + group_id: u64, + log_index: u64, +) -> Prepared<'a> { + tracing::error!( + group_id, + log_index, + "leader-change no-op committed at index where a proposer was waiting; \ + surfacing RetryableLeaderChange so the gateway re-proposes" + ); + ctx.tracker.complete( + group_id, + log_index, + 0, + Err(crate::Error::RetryableLeaderChange { + group_id, + log_index, + }), + ); + Prepared::Concluded(EntryOutcome::Skipped) +} + +/// The scope a decoded entry applies under. The entry's incarnations move +/// into the scope, so the entry no longer holds them. +fn entry_scope(decoded: &mut Option) -> EntryScope { + // Database scope for the entry, read from the wire. The generic decode + // path returns no scope, so it is taken from the entry itself: a redo + // appended under the wrong scope replays into the wrong namespace. + let database_id = decoded + .as_ref() + .map_or(DatabaseId::DEFAULT, |e| DatabaseId::new(e.database_id)); + // The source the proposer stamped. An entry that does not decode applies + // nothing, so its source is never read. + let event_source = decoded + .as_ref() + .map_or(crate::event::EventSource::User, |e| e.event_source.into()); + EntryScope { + database_id, + event_source, + incarnations: decoded + .as_mut() + .map(|e| std::mem::take(&mut e.incarnations)) + .unwrap_or_default(), + } +} + +/// Route a decoded entry to the apply path of its write. +fn prepare_replicated<'a>( + ctx: ApplyContext<'a>, + watch: &mut GroupWatch, + pos: AppliedPosition, + log_term: u64, + entry: LogEntry, + scope: EntryScope, + replicated: ReplicatedEntry, +) -> Prepared<'a> { let tenant_id = TenantId::new(replicated.tenant_id); let entry_database = DatabaseId::new(replicated.database_id); match replicated.write { ReplicatedWrite::ArrayOp { array, op_bytes, + cell_surrogate, provenance, + incarnation, .. - } => { - // The op path submits through the write funnel, so its redo is - // durable before it reports success. A failure breaks the - // prefix: the entry must stay replayable. - Prepared::Exclusive(Box::pin(async move { - let applied_ok = apply_array_op( - ctx.state, - ctx.tracker, - pos, - ArrayOpTarget { - tenant_id, - database_id: entry_database, - array: &array, - }, - &op_bytes, - provenance.as_deref(), - ) - .await; - FinishedApply { - group_id, - log_index, - outcome: EntryOutcome::Applied { - durable: applied_ok, - result: None, - }, - } - })) - } + } => prepare_array_op( + ctx, + pos, + entry_database, + ArrayOpWrite { + cell: super::array_cell_route::CellWrite { + tenant_id, + array, + incarnation, + }, + op_bytes, + cell_surrogate, + provenance, + }, + ), ReplicatedWrite::ArraySchema { ref array, ref snapshot_payload, @@ -174,16 +202,69 @@ pub(super) fn prepare_entry<'a>( result: None, }) } - ReplicatedWrite::ArrayCellPut { .. } | ReplicatedWrite::ArrayCellDelete { .. } => { - prepare_generic_entry(ctx, pos, entry, scope, true) + ReplicatedWrite::ArrayCellPut { + array, incarnation, .. } + | ReplicatedWrite::ArrayCellDelete { + array, incarnation, .. + } => super::array_cell_route::prepare_array_cell_entry( + ctx, + pos, + entry, + scope, + super::array_cell_route::CellWrite { + tenant_id, + array, + incarnation, + }, + ), ReplicatedWrite::TransactionRedo { .. } => { - prepare_transaction_redo_entry(ctx, pos, &replicated) + prepare_transaction_redo_entry(ctx, pos, &replicated, scope.incarnations) } - ReplicatedWrite::CutBarrier { hlc } => { - // Every entry after the barrier records above the cut. - watch.raise_cut(group_id, hlc); - Prepared::Barrier + ReplicatedWrite::SurrogateBind { ref identities } => { + Prepared::Concluded(super::surrogate_bind::apply_surrogate_bind( + ctx, + pos, + tenant_id, + entry_database, + identities, + )) + } + ReplicatedWrite::TopicPublish { + topic, + payload, + event_time, + origin, + } => prepare_topic_publish_entry( + ctx, + pos, + TopicPublishEntry { + database_id: entry_database, + tenant_id, + vshard_id: replicated.vshard_id, + topic, + payload, + event_time, + origin, + }, + ), + ReplicatedWrite::CutBarrier { + hlc, + restore_point, + capture, + } => { + // Every entry after the barrier records above the cut, on this + // life and on every later one. + watch.raise_cut(pos.group_id, pos.log_index, hlc); + let barrier = CutBarrierPoint { + hlc, + restore_point, + log_term, + }; + match capture { + Some(request) => prepare_capture_barrier(ctx, pos, barrier, request), + None => prepare_plain_barrier(ctx, pos, barrier), + } } ReplicatedWrite::CalvinReadResult { epoch, @@ -208,9 +289,204 @@ pub(super) fn prepare_entry<'a>( // A read result is forwarded to an in-memory Calvin scheduler and // writes nothing durable, so it neither advances the prefix nor // breaks it. The epoch it belongs to does not survive a restart, - // so a re-delivery could not usefully replay it. + // so a re-delivery cannot usefully replay it. Prepared::Concluded(EntryOutcome::Skipped) } _ => prepare_generic_entry(ctx, pos, entry, scope, false), } } + +/// The parts of a `ReplicatedWrite::ArrayOp` its apply takes. +struct ArrayOpWrite { + cell: super::array_cell_route::CellWrite, + op_bytes: Vec, + cell_surrogate: Option, + provenance: Option>, +} + +/// Prepare an array op. The op path submits through the write funnel, so +/// its redo is durable before it reports success. A failure breaks the +/// prefix: the entry must stay replayable. +fn prepare_array_op<'a>( + ctx: ApplyContext<'a>, + pos: AppliedPosition, + entry_database: DatabaseId, + write: ArrayOpWrite, +) -> Prepared<'a> { + Prepared::Exclusive(Box::pin(async move { + let ArrayOpWrite { + cell, + op_bytes, + cell_surrogate, + provenance, + } = write; + let (_gate, database_id) = match super::array_cell_route::route_cell_write( + ctx, + pos, + entry_database, + &cell, + ) + .await + { + Ok(routed) => routed, + Err(finished) => return *finished, + }; + let applied_ok = apply_array_op( + ctx.state, + ctx.tracker, + pos, + ArrayOpTarget { + tenant_id: cell.tenant_id, + database_id, + array: &cell.array, + }, + &op_bytes, + cell_surrogate, + provenance.as_deref(), + ) + .await; + FinishedApply { + group_id: pos.group_id, + log_index: pos.log_index, + outcome: EntryOutcome::Applied { + durable: applied_ok, + result: None, + }, + } + })) +} + +/// Where a cut barrier cuts its group. +#[derive(Clone, Copy)] +struct CutBarrierPoint { + hlc: u64, + restore_point: u64, + log_term: u64, +} + +/// Prepare a backup capture's cut barrier. The lane starts a barrier with +/// nothing else of its group in flight, and starts no later entry until it +/// finishes: the floor is durable, and the capture holds every entry at or +/// below it and none above. +fn prepare_capture_barrier<'a>( + ctx: ApplyContext<'a>, + pos: AppliedPosition, + barrier: CutBarrierPoint, + request: nodedb_physical::physical_plan::CutCaptureRequest, +) -> Prepared<'a> { + let AppliedPosition { + group_id, + log_index, + applied_key, + .. + } = pos; + let CutBarrierPoint { + hlc, + restore_point, + log_term, + } = barrier; + Prepared::Exclusive(Box::pin(async move { + persist_floor_durably(ctx, group_id, log_index, hlc).await; + record_restore_point(ctx, group_id, restore_point, hlc, log_index, log_term); + crate::control::backup::cut_capture::apply::capture_at_barrier( + ctx.state, group_id, &request, + ) + .await; + finish_barrier(ctx, group_id, log_index, applied_key) + })) +} + +/// Prepare a cut barrier with no capture. A failed floor write holds the +/// group until the floor is durable. +fn prepare_plain_barrier<'a>( + ctx: ApplyContext<'a>, + pos: AppliedPosition, + barrier: CutBarrierPoint, +) -> Prepared<'a> { + let AppliedPosition { + group_id, + log_index, + applied_key, + .. + } = pos; + let CutBarrierPoint { + hlc, + restore_point, + log_term, + } = barrier; + let persisted = + crate::control::pitr::restore_point::persist_cut_floor(ctx.state, group_id, log_index, hlc); + if persisted.is_ok() { + record_restore_point(ctx, group_id, restore_point, hlc, log_index, log_term); + return Prepared::Barrier; + } + Prepared::Exclusive(Box::pin(async move { + persist_floor_durably(ctx, group_id, log_index, hlc).await; + record_restore_point(ctx, group_id, restore_point, hlc, log_index, log_term); + finish_barrier(ctx, group_id, log_index, applied_key) + })) +} + +/// Record the group's place at the cluster restore point `restore_point` a +/// cut barrier takes. `0` takes none. +fn record_restore_point( + ctx: ApplyContext<'_>, + group_id: u64, + restore_point: u64, + hlc: u64, + log_index: u64, + log_term: u64, +) { + if restore_point == 0 { + return; + } + crate::control::pitr::restore_point::record_group_point( + ctx.state, + nodedb_wal::record::RestorePointPayload { + id: restore_point, + hlc, + group_id, + applied_index: log_index, + term: log_term, + next_epoch: 0, + epoch_system_ms: 0, + vshards: Vec::new(), + }, + ); +} + +/// Persist the floor of the barrier at `log_index`, retrying until it holds. +/// A floor write that keeps failing wedges the node until it succeeds. +async fn persist_floor_durably(ctx: ApplyContext<'_>, group_id: u64, log_index: u64, hlc: u64) { + crate::control::pitr::restore_point::persist_until_durable( + &ctx.state.metadata_apply_wedge, + group_id, + log_index, + || { + crate::control::pitr::restore_point::persist_cut_floor( + ctx.state, group_id, log_index, hlc, + ) + }, + ) + .await; +} + +/// Resolve a barrier's waiter once every step of the barrier is durable. +fn finish_barrier( + ctx: ApplyContext<'_>, + group_id: u64, + log_index: u64, + applied_key: u64, +) -> FinishedApply { + ctx.tracker.complete( + group_id, + log_index, + applied_key, + Ok(crate::control::distributed_applier::AppliedWrite::unversioned(Vec::new())), + ); + FinishedApply { + group_id, + log_index, + outcome: EntryOutcome::Skipped, + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/surrogate_bind.rs b/nodedb/src/control/distributed_applier/apply_loop/surrogate_bind.rs new file mode 100644 index 000000000..67525a30f --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/surrogate_bind.rs @@ -0,0 +1,56 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Apply path for a committed `ReplicatedWrite::SurrogateBind` entry. +//! +//! The leader of a key's collection home mints the key's surrogate by +//! proposing this entry (`surrogate_exchange::authority`). Every replica binds +//! each key first-wins, in Raft log order. A key an earlier entry of the log +//! already bound keeps that binding on every replica alike, so the proposer +//! reads the winner back after its own entry applies. The binds land in the +//! fsync-committed catalog before the apply reports. + +use crate::control::array_sync::raft_apply::AppliedPosition; +use crate::control::distributed_applier::propose_tracker::AppliedWrite; +use crate::control::wal_replication::ReplicatedIdentity; +use crate::types::{DatabaseId, TenantId}; + +use super::context::ApplyContext; +use super::proposal_gate::EntryOutcome; + +/// Bind every identity of the entry and resolve its waiter. +pub(super) fn apply_surrogate_bind( + ctx: ApplyContext<'_>, + pos: AppliedPosition, + tenant_id: TenantId, + database_id: DatabaseId, + identities: &[ReplicatedIdentity], +) -> EntryOutcome { + let AppliedPosition { + group_id, + log_index, + applied_key, + .. + } = pos; + let result = identities.iter().try_for_each(|identity| { + ctx.state + .surrogate_assigner + .bind( + nodedb_types::CollectionKey::from_bare(database_id, &identity.collection), + tenant_id, + &identity.pk_bytes, + nodedb_types::Surrogate::new(identity.surrogate), + ) + .map(|_| ()) + }); + let durable = result.is_ok(); + ctx.tracker.complete( + group_id, + log_index, + applied_key, + result.map(|()| AppliedWrite::unversioned(Vec::new())), + ); + EntryOutcome::Applied { + durable, + result: None, + } +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/topic_publish.rs b/nodedb/src/control/distributed_applier/apply_loop/topic_publish.rs new file mode 100644 index 000000000..d5c6b8ba3 --- /dev/null +++ b/nodedb/src/control/distributed_applier/apply_loop/topic_publish.rs @@ -0,0 +1,131 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Apply path for a committed `ReplicatedWrite::TopicPublish` entry. +//! +//! The message is appended to this replica's catalog in the same durable +//! redb transaction that advances the topic's applied position, so an entry +//! re-delivered at the same position appends nothing. A second committed copy +//! of the proposal, at another position, is caught by the proposal ledger: +//! the apply writes a `ProposalApplied` WAL marker under the entry's key and +//! waits for it to be durable before it reports. + +use crate::control::array_sync::raft_apply::AppliedPosition; +use crate::control::distributed_applier::propose_tracker::AppliedWrite; +use crate::event::topic::{ReplicatedPublish, ReplicatedPublishOutcome, apply_replicated_publish}; +use crate::types::{DatabaseId, TenantId, VShardId}; + +use super::context::{ApplyContext, FinishedApply}; +use super::proposal_gate::{EntryOutcome, ledger_outcome}; +use super::start::Prepared; + +/// The fields of one committed publication. +pub(super) struct TopicPublishEntry { + pub database_id: DatabaseId, + pub tenant_id: TenantId, + pub vshard_id: u32, + pub topic: String, + pub payload: String, + pub event_time: u64, + pub origin: Option, +} + +/// Apply one publication, with nothing else of its group in flight, and +/// resolve its waiter with the assigned sequence. +pub(super) fn prepare_topic_publish_entry<'a>( + ctx: ApplyContext<'a>, + pos: AppliedPosition, + entry: TopicPublishEntry, +) -> Prepared<'a> { + Prepared::Exclusive(Box::pin(async move { + let outcome = apply_topic_publish(ctx, pos, &entry).await; + FinishedApply { + group_id: pos.group_id, + log_index: pos.log_index, + outcome, + } + })) +} + +async fn apply_topic_publish( + ctx: ApplyContext<'_>, + pos: AppliedPosition, + entry: &TopicPublishEntry, +) -> EntryOutcome { + let outcome = apply_replicated_publish( + ctx.state, + ReplicatedPublish { + database_id: entry.database_id, + tenant_id: entry.tenant_id.as_u64(), + topic: &entry.topic, + payload: &entry.payload, + event_time: entry.event_time, + position: pos.change_position(ctx.state, entry.vshard_id), + origin: entry.origin, + }, + ); + let outcome = match outcome { + Ok(ReplicatedPublishOutcome::Appended(sequence)) => mark_applied(ctx, pos, entry) + .await + .map(|()| ReplicatedPublishOutcome::Appended(sequence)), + other => other, + }; + // A missing topic is every replica's verdict at this position, so it is + // the entry's final outcome. A storage failure is retried on re-delivery. + let durable = match &outcome { + Ok(_) | Err(crate::Error::UndefinedObject { .. }) => true, + Err(_) => false, + }; + let result = outcome.and_then(|outcome| { + let sequence = match outcome { + ReplicatedPublishOutcome::Appended(sequence) => sequence, + // A re-delivery after a restart repeats an entry, and no waiter + // of the first delivery survives it. A committed message the + // topic already holds has no sequence of its own to report. + ReplicatedPublishOutcome::AlreadyApplied => 0, + }; + zerompk::to_msgpack_vec(&sequence) + .map(AppliedWrite::unversioned) + .map_err(|error| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("topic publish result: {error}"), + }) + }); + if let Err(error) = &result { + tracing::warn!( + group_id = pos.group_id, + index = pos.log_index, + topic = %entry.topic, + %error, + "applying committed topic publication failed" + ); + } + let applied = ledger_outcome(&result); + ctx.tracker + .complete(pos.group_id, pos.log_index, pos.applied_key, result); + EntryOutcome::Applied { + durable, + result: Some(applied), + } +} + +/// Make the entry's proposal key durable in the WAL, so the proposal ledger +/// recovers it after a restart. +async fn mark_applied( + ctx: ApplyContext<'_>, + pos: AppliedPosition, + entry: &TopicPublishEntry, +) -> crate::Result<()> { + let Some(lsn) = ctx + .state + .wal + .appender(pos.applied_key) + .append_proposal_applied( + entry.tenant_id, + VShardId::new(entry.vshard_id), + entry.database_id, + )? + else { + return Ok(()); + }; + ctx.state.wal.wait_durable(lsn).await +} diff --git a/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs b/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs index c33e7d669..da0299849 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/transaction_redo.rs @@ -19,10 +19,12 @@ use crate::bridge::envelope::Status; use crate::control::array_sync::raft_apply::AppliedPosition; use crate::control::distributed_applier::propose_tracker::{AppliedWrite, ProposeTracker}; -use crate::control::server::dispatch_utils::{SubmitOutcome, refusal_is_final}; -use crate::control::wal_replication::ReplicatedEntry; +use crate::control::server::dispatch_utils::{ChangeFeedOwner, SubmitOutcome, refusal_is_final}; use crate::control::wal_replication::decode::transaction_redo_payload; -use crate::control::wal_replication::transaction_redo::{RedoTarget, enqueue_transaction_redo}; +use crate::control::wal_replication::transaction_redo::{ + RedoTarget, enqueue_transaction_redo, record_cross_shard_key, +}; +use crate::control::wal_replication::{CollectionIncarnation, ReplicatedEntry}; use crate::types::{DatabaseId, TenantId, VShardId}; use super::context::{ApplyContext, FinishedApply, Started, StartedEntry}; @@ -34,10 +36,14 @@ use super::start::Prepared; /// record and hands it to its core. The apply that follows resolves the /// propose waiter. Its outcome says whether the entry's effect is durable on /// this node, which is what the group's applied prefix records. +/// +/// `incarnations` are the entry's collection incarnations, moved out of the +/// decoded entry. pub(super) fn prepare_transaction_redo_entry<'a>( ctx: ApplyContext<'a>, pos: AppliedPosition, entry: &ReplicatedEntry, + incarnations: Vec, ) -> Prepared<'a> { let ApplyContext { state, tracker, .. } = ctx; let payload = match transaction_redo_payload(&entry.write) { @@ -60,17 +66,60 @@ pub(super) fn prepare_transaction_redo_entry<'a>( }; Prepared::Enqueue(Box::pin(async move { let collection = payload.collections.first().cloned(); + // A redo for a collection incarnation this node no longer holds has + // nothing to mutate. The gates stay held until the redo is enqueued. + let routed = super::collection_route::route( + state, + target.tenant_id.as_u64(), + target.database_id, + &incarnations, + ) + .await; + let _gates = match routed { + Ok(super::collection_route::CollectionRoute::Apply(gates)) => gates, + Ok(super::collection_route::CollectionRoute::Superseded) => { + tracker.complete( + pos.group_id, + pos.log_index, + pos.applied_key, + Err(crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::NotFound, + )), + ); + return StartedEntry::concluded(EntryOutcome::Applied { + durable: true, + result: None, + }); + } + Err(error) => { + tracker.complete(pos.group_id, pos.log_index, pos.applied_key, Err(error)); + return StartedEntry::concluded(EntryOutcome::Applied { + durable: false, + result: None, + }); + } + }; let enqueued = enqueue_transaction_redo( state, target, &payload, pos.applied_key, pos.carried_commit_hlc(), + Some(pos.change_position(state, target.vshard_id.as_u32())), + // Every replica stages the redo's row changes under the entry and + // publishes them at its log position once the entry settles. + ChangeFeedOwner::Replicated { + group_id: pos.group_id, + log_index: pos.log_index, + }, ) .await; let started = match enqueued { Ok(pending) => Started::Running(Box::pin(async move { let submitted = pending.finish(state).await; + if let Ok(outcome) = &submitted { + record_cross_shard_key(state, &payload, outcome); + } FinishedApply { group_id: pos.group_id, log_index: pos.log_index, diff --git a/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs index 4f55e6686..967287026 100644 --- a/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs +++ b/nodedb/src/control/distributed_applier/apply_loop/write_dispatch.rs @@ -32,13 +32,16 @@ use super::proposal_gate::{EntryOutcome, ledger_outcome}; use super::start::Prepared; /// What a generic entry's apply takes from its decoded envelope. -#[derive(Debug, Clone, Copy)] +#[derive(Debug, Clone)] pub(super) struct EntryScope { /// Database scope of the entry. pub database_id: DatabaseId, /// The source the proposer stamped. Every replica gives the write's /// events this source. pub event_source: crate::event::EventSource, + /// The collections the write names, with the incarnations its proposer + /// planned against. + pub incarnations: Vec, } /// Prepare a generic entry. `exclusive` marks a Raft-native array cell write: @@ -80,6 +83,7 @@ async fn enqueue_generic_entry<'a>( let EntryScope { database_id, event_source, + incarnations, } = scope; let ApplyContext { state, tracker, .. } = ctx; let AppliedPosition { @@ -88,11 +92,11 @@ async fn enqueue_generic_entry<'a>( applied_key, .. } = pos; - let decoded = from_replicated_entry(&entry.data, Some(state.surrogate_assigner.as_ref())); + let decoded = from_replicated_entry(&entry.data, state.surrogate_assigner.as_ref()); let (tenant_id, vshard_id, plan, resolved_now_ms) = match decoded { Ok(Some(t)) => t, Ok(None) => { - // Couldn't deserialize — might be a different format or corrupted. + // Couldn't deserialize — a different format or corrupted. debug!( group_id, index = entry.index, @@ -139,6 +143,36 @@ async fn enqueue_generic_entry<'a>( } }; + // A write for a collection incarnation this node no longer holds has + // nothing to mutate. The gates stay held until the write is enqueued. + let _gates = + match super::collection_route::route(state, tenant_id.as_u64(), database_id, &incarnations) + .await + { + Ok(super::collection_route::CollectionRoute::Apply(gates)) => gates, + Ok(super::collection_route::CollectionRoute::Superseded) => { + tracker.complete( + group_id, + entry.index, + applied_key, + Err(crate::Error::DataPlane( + crate::bridge::envelope::ErrorCode::NotFound, + )), + ); + return StartedEntry::concluded(EntryOutcome::Applied { + durable: true, + result: None, + }); + } + Err(error) => { + tracker.complete(group_id, entry.index, applied_key, Err(error)); + return StartedEntry::concluded(EntryOutcome::Applied { + durable: false, + result: None, + }); + } + }; + // Raft-native array cell writes (`ArrayCellPut` / `ArrayCellDelete`) // decode to `PhysicalPlan::Array(Put | Delete)`. A follower must // OPEN the array on the Data Plane before applying, so these route @@ -160,6 +194,7 @@ async fn enqueue_generic_entry<'a>( database_id, vshard: vshard_id, resolved_now_ms, + event_source, }, plan, ) @@ -209,22 +244,26 @@ async fn enqueue_generic_entry<'a>( // write's `expire_at_ms` to the instant the proposing node // resolved, so this replica's redo record and its live apply // install the byte-identical value every other replica does. + // `change_position` gives the write's change events the entry's + // log position, which every replica shares. durability: WalDurability::AppendHere { now_override: resolved_now_ms, apply_key: applied_key, commit_hlc: pos.carried_commit_hlc(), + change_position: Some(pos.change_position(state, vshard_id.as_u32())), }, // Raft committed this entry at a fixed log index; every // replica applies it in that order. Re-entering the - // write-admission gate would re-decide an ordering that is + // write-admission gate re-decides an ordering that is // already final. ordering: WriteOrdering::AlreadyOrdered, - // This loop runs on EVERY replica, so it must not publish: - // the node that proposed this entry already published the - // write's change event once, after commit + apply. Emitting - // here would give each subscriber one copy per replica plus - // a NOTIFY fan-out from each. See [`ChangeFeedOwner`]. - change_feed: ChangeFeedOwner::Unowned, + // Every replica stages the write's change events under the + // entry, and publishes them at its log position once the entry + // settles. See [`ChangeFeedOwner`]. + change_feed: ChangeFeedOwner::Replicated { + group_id, + log_index, + }, }, ) .await; diff --git a/nodedb/src/control/distributed_applier/proposal_ledger.rs b/nodedb/src/control/distributed_applier/proposal_ledger.rs index 85374ffea..515ff9a55 100644 --- a/nodedb/src/control/distributed_applier/proposal_ledger.rs +++ b/nodedb/src/control/distributed_applier/proposal_ledger.rs @@ -25,6 +25,8 @@ //! counts as the entry's outcome. //! - An apply that writes no record of its own (a `wal=false` timeseries //! ingest) appends a payload-free `ProposalApplied` record in its place. +//! - A snapshot install appends a `ProposalApplied` record per key the +//! snapshot carried: those entries never apply on this node. //! //! A duplicate delivered after a restart is recognised as long as the WAL //! still retains a record of its original. diff --git a/nodedb/src/control/distributed_applier/propose_tracker.rs b/nodedb/src/control/distributed_applier/propose_tracker.rs deleted file mode 100644 index 8d06a918f..000000000 --- a/nodedb/src/control/distributed_applier/propose_tracker.rs +++ /dev/null @@ -1,405 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Propose tracker — slot map keyed by `(group_id, log_index)` that lets -//! proposers wait for a Raft entry to commit and execute, with race-safe -//! resolution if the apply path beats the proposer's `register()` call. - -use std::collections::HashMap; -use std::collections::hash_map::Entry; -use std::sync::{Arc, Mutex}; - -use super::apply_window::ApplyWindow; - -use tokio::sync::oneshot; - -use nodedb_cluster::GroupAppliedWatchers; - -use crate::bridge::envelope::Response; -use crate::types::Lsn; - -/// What a committed entry produced on the replica that applied it. -/// -/// Carries the write's per-collection version alongside the payload because the -/// proposer has no other way to learn it: the version is minted inside the apply -/// path (the write funnel's WAL append), never on the wire, and the propose -/// tracker resolves on the very node that applied locally — so the version this -/// carries is that node's own, which is exactly what shard-local OCC validates -/// against. -#[derive(Debug, Clone)] -pub struct AppliedWrite { - /// The Data Plane's response payload, verbatim. - pub payload: Vec, - /// The written collection's `coll_write_lsn` AFTER this write, in the local - /// WAL-LSN domain — the one domain OCC's read validator compares in. It is - /// NOT the Raft log index: the log index is a per-group counter that shares - /// no scale with the WAL LSNs every other feed of that map records, and - /// mixing the two silently breaks both directions of the comparison. - pub write_version: Lsn, -} - -impl AppliedWrite { - /// Take both fields off the Data Plane's response to a committed write. - /// - /// `Response::read_version_lsn` is stamped by the core loop from the written - /// collection's `coll_write_lsn`, read AFTER the handler recorded this - /// write's LSN into the version index — so on a write response it is the - /// post-write version. It is `Lsn::ZERO` for a plan that maps to no single - /// user collection (see [`AppliedWrite::write_version`]). - pub fn from_response(response: &Response) -> Self { - Self { - payload: response.payload.to_vec(), - write_version: response.read_version_lsn, - } - } - - /// An applied entry that publishes no per-collection write-version: it wrote - /// no Data-Plane collection state (a decode skip, a forwarded read result, a - /// schema snapshot), or it was deduplicated before reaching the funnel. There - /// is no version to floor a later read at, and `Lsn::ZERO` is the read-set - /// capture's "no own-write floor" value — never a fabricated stand-in for a - /// version that exists but was not read back. - pub fn unversioned(payload: Vec) -> Self { - Self { - payload, - write_version: Lsn::ZERO, - } - } -} - -/// Result sent back to the proposer after commit + execution. -pub type ProposeResult = std::result::Result; - -/// Slot in the propose tracker — either a pending waiter or a completed result -/// that arrived before the waiter was registered. -enum TrackerSlot { - /// Waiter registered by the proposer; awaiting `complete()`. - /// - /// `expected_key` is the proposer's idempotency key. The apply path - /// passes the applied entry's key to `complete`; if they differ the - /// proposer's reservation was overwritten by a different proposer's - /// entry under a leader change and we surface - /// `RetryableLeaderChange` instead of the (success-shaped) result - /// that would otherwise leak the wrong entry's payload back to the - /// proposer. `expected_key == 0` is a wildcard accepting any key - /// (used for legacy synthetic registrations). - Waiting { - tx: oneshot::Sender, - expected_key: u64, - }, - /// `complete()` was called before `register()`. Stored so `register()` - /// can resolve the channel immediately. - Completed(ProposeResult), -} - -/// Tracks pending proposals awaiting Raft commit. -/// -/// Keyed by `(group_id, log_index)`. The proposer calls `register()` after -/// the proposal returns the log index; `run_apply_loop` calls `complete()` -/// after the entry is applied. Either side may win the race — `complete()` -/// stores the result if no waiter exists yet, and `register()` picks it up -/// immediately if `complete()` already fired. -pub struct ProposeTracker { - slots: Mutex>, - /// Per-Raft-group apply watermark registry. Bumped by - /// [`Self::note_applied`] once every entry of the group up to the index - /// finished, so the watcher reflects "data applied on this node up to - /// index N" — the only semantic that's useful for cross-node visibility - /// waits. Tick-loop bumps cover the metadata group (sync redb apply); - /// this tracker covers data groups (async SPSC dispatch through - /// `run_apply_loop`). `None` only in tests that don't exercise the - /// watcher. - group_watchers: Option>, - /// Per group, the oldest committed entry the apply loop has not finished. - applying: Mutex>, - /// Per-group bound on entries between hand-off and settle, shared by the - /// applier that hands entries off and the loop that settles them. - window: Arc, -} - -/// The oldest committed entry of a group the apply loop has not finished: -/// what a propose waiter that timed out names as the entry its group's -/// applied index waits behind. -#[derive(Debug, Clone, PartialEq, Eq)] -pub struct ApplyingEntry { - pub group_id: u64, - pub log_index: u64, - /// The collection the entry's plan writes, once the apply decoded it. - pub collection: Option, -} - -impl std::fmt::Display for ApplyingEntry { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - write!(f, "group {} index {}", self.group_id, self.log_index)?; - if let Some(collection) = &self.collection { - write!(f, " writing '{collection}'")?; - } - Ok(()) - } -} - -impl Default for ProposeTracker { - fn default() -> Self { - Self::new() - } -} - -impl ProposeTracker { - pub fn new() -> Self { - Self { - slots: Mutex::new(HashMap::new()), - group_watchers: None, - applying: Mutex::new(HashMap::new()), - window: Arc::new(ApplyWindow::default()), - } - } - - /// Wire the per-group apply watermark registry. Called by - /// `start_raft` after `SharedState` is constructed. - pub fn with_group_watchers(mut self, watchers: Arc) -> Self { - self.group_watchers = Some(watchers); - self - } - - /// The per-group apply window. - pub fn window(&self) -> &Arc { - &self.window - } - - /// Record the oldest entry of `group_id` the apply loop has not finished, - /// or `None` once every entry it holds for the group finished. - pub fn note_applying(&self, group_id: u64, entry: Option) { - let mut applying = self.applying.lock().unwrap_or_else(|p| p.into_inner()); - match entry { - Some(entry) => { - applying.insert(group_id, entry); - } - None => { - applying.remove(&group_id); - } - } - } - - /// The oldest entry of `group_id` the apply loop has not finished, if any. - pub fn applying(&self, group_id: u64) -> Option { - self.applying - .lock() - .unwrap_or_else(|p| p.into_inner()) - .get(&group_id) - .cloned() - } - - /// Advance `group_id`'s applied watermark to `log_index`. The apply loop - /// calls it once every entry of the group up to `log_index` finished. - pub fn note_applied(&self, group_id: u64, log_index: u64) { - if let Some(w) = &self.group_watchers { - w.bump(group_id, log_index); - } - } - - /// Register a waiter for a proposed entry. Returns a receiver that - /// resolves when the entry is committed and executed. - /// - /// If `complete()` was called first (the entry was applied before this - /// node could register), the receiver is pre-resolved and ready - /// immediately. - pub fn register( - &self, - group_id: u64, - log_index: u64, - expected_key: u64, - ) -> oneshot::Receiver { - let (tx, rx) = oneshot::channel(); - let mut slots = self.slots.lock().unwrap_or_else(|p| p.into_inner()); - match slots.entry((group_id, log_index)) { - Entry::Vacant(e) => { - e.insert(TrackerSlot::Waiting { tx, expected_key }); - } - Entry::Occupied(e) => { - match e.get() { - TrackerSlot::Completed(_) => { - // complete() already fired — extract the result, resolve - // the receiver immediately, and clean up the slot. - if let TrackerSlot::Completed(result) = e.remove() { - let _ = tx.send(result); - } - } - TrackerSlot::Waiting { .. } => { - // Duplicate register — shouldn't happen. Insert the new - // sender; the old receiver will see channel-closed. - *e.into_mut() = TrackerSlot::Waiting { tx, expected_key }; - } - } - } - } - rx - } - - /// Drop the waiter a proposer registered at `(group_id, log_index)` and - /// stopped waiting on: its deadline passed, or this node left the group - /// and will never apply the index. A result already stored there stays. - pub fn abandon(&self, group_id: u64, log_index: u64) { - let mut slots = self.slots.lock().unwrap_or_else(|p| p.into_inner()); - if let Entry::Occupied(e) = slots.entry((group_id, log_index)) - && matches!(e.get(), TrackerSlot::Waiting { .. }) - { - e.remove(); - } - } - - /// Complete a waiter after the entry has been committed and executed. - /// - /// If the proposer has already called `register()`, the result is sent - /// immediately. If not, the result is stored so the next `register()` - /// call picks it up without waiting. - /// - /// Entries of one group complete in any order. The applied watermark - /// moves only through [`Self::note_applied`], in log order. - /// - /// Returns true if a live waiter was found and notified, false otherwise. - pub fn complete( - &self, - group_id: u64, - log_index: u64, - applied_key: u64, - result: ProposeResult, - ) -> bool { - let mut slots = self.slots.lock().unwrap_or_else(|p| p.into_inner()); - match slots.entry((group_id, log_index)) { - Entry::Vacant(e) => { - // No waiter yet — store result for the upcoming register(). - e.insert(TrackerSlot::Completed(result)); - false - } - Entry::Occupied(e) => { - match e.get() { - TrackerSlot::Waiting { expected_key, .. } => { - // Idempotency-key gate: the entry that committed - // at this (group_id, log_index) must be the one - // the proposer reserved. If the keys disagree, - // a leader change overwrote the proposer's entry - // with a different one — surface the retryable - // signal instead of the (success-shaped) result - // that belongs to a different proposer. A zero - // applied_key means "no key carried" (empty - // entry / legacy); a zero expected_key means the - // registration is wildcard (legacy callers). - let mismatch = - applied_key != 0 && *expected_key != 0 && applied_key != *expected_key; - let final_result = if mismatch { - tracing::warn!( - group_id, - log_index, - applied_key, - expected_key = *expected_key, - "raft entry at proposer's index was overwritten by \ - a different proposal (idempotency_key mismatch); \ - surfacing RetryableLeaderChange" - ); - Err(crate::Error::RetryableLeaderChange { - group_id, - log_index, - }) - } else { - result - }; - if let TrackerSlot::Waiting { tx, .. } = e.remove() { - let _ = tx.send(final_result); - return true; - } - } - TrackerSlot::Completed(_) => { - // Already completed — overwrite with newer result. - // Duplicate completes should not occur in practice; - // last write wins. - *e.into_mut() = TrackerSlot::Completed(result); - } - } - false - } - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn propose_tracker_register_and_complete() { - let tracker = ProposeTracker::new(); - let mut rx = tracker.register(1, 5, 0xdead_beef); - - // Waiter must receive both payload and coll_write_lsn — its only channel - // for a version minted on the apply path. - assert!(tracker.complete( - 1, - 5, - 0xdead_beef, - Ok(AppliedWrite { - payload: b"result".to_vec(), - write_version: Lsn::new(137), - }), - )); - - let result = rx.try_recv().unwrap().unwrap(); - assert_eq!(result.payload, b"result"); - assert_eq!(result.write_version, Lsn::new(137)); - } - - #[test] - fn propose_tracker_no_waiter_returns_false() { - let tracker = ProposeTracker::new(); - assert!(!tracker.complete(1, 99, 0, Ok(AppliedWrite::unversioned(Vec::new())))); - } - - #[test] - fn propose_tracker_key_mismatch_surfaces_retryable_leader_change() { - let tracker = ProposeTracker::new(); - let mut rx = tracker.register(1, 5, 0xaaaa); - - // A different proposer's entry committed at the same (group_id, - // log_index); waiter must see RetryableLeaderChange, not its result. - assert!(tracker.complete( - 1, - 5, - 0xbbbb, - Ok(AppliedWrite::unversioned( - b"other-proposers-payload".to_vec() - )), - )); - - let result = rx.try_recv().unwrap(); - match result { - Err(crate::Error::RetryableLeaderChange { - group_id, - log_index, - }) => { - assert_eq!(group_id, 1); - assert_eq!(log_index, 5); - } - other => panic!("expected RetryableLeaderChange, got {other:?}"), - } - } - - #[test] - fn propose_tracker_zero_applied_key_passes_through_explicit_error() { - // applied_key = 0 (leader-change no-op) must forward the explicit - // RetryableLeaderChange, not be treated as a key mismatch. - let tracker = ProposeTracker::new(); - let mut rx = tracker.register(1, 5, 0xaaaa); - assert!(tracker.complete( - 1, - 5, - 0, - Err(crate::Error::RetryableLeaderChange { - group_id: 1, - log_index: 5, - }), - )); - let result = rx.try_recv().unwrap(); - assert!(matches!( - result, - Err(crate::Error::RetryableLeaderChange { .. }) - )); - } -} diff --git a/nodedb/src/control/distributed_applier/propose_tracker/applied_write.rs b/nodedb/src/control/distributed_applier/propose_tracker/applied_write.rs new file mode 100644 index 000000000..667d6c89d --- /dev/null +++ b/nodedb/src/control/distributed_applier/propose_tracker/applied_write.rs @@ -0,0 +1,58 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! What a committed entry produced, as its propose waiter receives it. + +use crate::bridge::envelope::Response; +use crate::types::Lsn; + +/// What a committed entry produced on the replica that applied it. +/// +/// Carries the write's per-collection version alongside the payload because the +/// proposer has no other way to learn it: the version is minted inside the apply +/// path (the write funnel's WAL append), never on the wire, and the propose +/// tracker resolves on the very node that applied locally — so the version this +/// carries is that node's own, which is exactly what shard-local OCC validates +/// against. +#[derive(Debug, Clone)] +pub struct AppliedWrite { + /// The Data Plane's response payload, verbatim. + pub payload: Vec, + /// The written collection's `coll_write_lsn` AFTER this write, in the local + /// WAL-LSN domain — the one domain OCC's read validator compares in. It is + /// NOT the Raft log index: the log index is a per-group counter that shares + /// no scale with the WAL LSNs every other feed of that map records, and + /// mixing the two silently breaks both directions of the comparison. + pub write_version: Lsn, +} + +impl AppliedWrite { + /// Take both fields off the Data Plane's response to a committed write. + /// + /// `Response::read_version_lsn` is stamped by the core loop from the written + /// collection's `coll_write_lsn`, read AFTER the handler recorded this + /// write's LSN into the version index — so on a write response it is the + /// post-write version. It is `Lsn::ZERO` for a plan that maps to no single + /// user collection (see [`AppliedWrite::write_version`]). + pub fn from_response(response: &Response) -> Self { + Self { + payload: response.payload.to_vec(), + write_version: response.read_version_lsn, + } + } + + /// An applied entry that publishes no per-collection write-version: it wrote + /// no Data-Plane collection state (a decode skip, a forwarded read result, a + /// schema snapshot), or it was deduplicated before reaching the funnel. There + /// is no version to floor a later read at, and `Lsn::ZERO` is the read-set + /// capture's "no own-write floor" value — never a fabricated stand-in for a + /// version that exists but was not read back. + pub fn unversioned(payload: Vec) -> Self { + Self { + payload, + write_version: Lsn::ZERO, + } + } +} + +/// Result sent back to the proposer after commit + execution. +pub type ProposeResult = std::result::Result; diff --git a/nodedb/src/control/distributed_applier/propose_tracker/applying.rs b/nodedb/src/control/distributed_applier/propose_tracker/applying.rs new file mode 100644 index 000000000..d85aeb6a4 --- /dev/null +++ b/nodedb/src/control/distributed_applier/propose_tracker/applying.rs @@ -0,0 +1,24 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The oldest committed entry of a group the apply loop has not finished. + +/// The oldest committed entry of a group the apply loop has not finished: +/// what a propose waiter that timed out names as the entry its group's +/// applied index waits behind. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ApplyingEntry { + pub group_id: u64, + pub log_index: u64, + /// The collection the entry's plan writes, once the apply decoded it. + pub collection: Option, +} + +impl std::fmt::Display for ApplyingEntry { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "group {} index {}", self.group_id, self.log_index)?; + if let Some(collection) = &self.collection { + write!(f, " writing '{collection}'")?; + } + Ok(()) + } +} diff --git a/nodedb/src/control/distributed_applier/propose_tracker/committed_keys.rs b/nodedb/src/control/distributed_applier/propose_tracker/committed_keys.rs new file mode 100644 index 000000000..08ba96868 --- /dev/null +++ b/nodedb/src/control/distributed_applier/propose_tracker/committed_keys.rs @@ -0,0 +1,178 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Proposal keys of the committed entries this node handed to its apply +//! loop, per group, for the window a propose waiter can exist in. +//! +//! A data-group snapshot carries the keys at or below its index. The follower +//! that installs it never applies those entries, and the keys are the only +//! way it learns which proposals they were. + +use std::collections::{HashMap, VecDeque}; +use std::time::{Duration, Instant}; + +use crate::control::distributed_applier::proposal_ledger::PROPOSAL_LEDGER_CAPACITY; + +/// One committed entry's key. +#[derive(Debug, Clone, Copy)] +struct CommittedKey { + log_index: u64, + key: u64, + at: Instant, +} + +/// One group's recent keys, oldest first, and where they are complete. +#[derive(Debug, Default)] +struct GroupKeys { + keys: VecDeque, + /// First index this process received. Keys below it were never seen here. + first_noted: u64, + /// Highest index whose key the count bound dropped. + cap_dropped_through: u64, +} + +/// The keys a snapshot carries, and the lowest index from which they are +/// complete. `complete_from` is `0` when every key in the window is present. +#[derive(Debug, Clone, PartialEq, Eq, Default)] +pub struct CarriedKeys { + pub keys: Vec<(u64, u64)>, + pub complete_from: u64, +} + +/// Recent committed keys per group, oldest first. +/// +/// Bounded by age (the waiter window) and by count per group. The count bound +/// is the proposal ledger's capacity. A key older than the window serves no +/// waiter. A key the count bound drops leaves a gap, which +/// [`CarriedKeys::complete_from`] reports. +#[derive(Debug)] +pub(super) struct CommittedKeys { + groups: HashMap, + window: Duration, +} + +impl CommittedKeys { + pub fn new(window: Duration) -> Self { + Self { + groups: HashMap::new(), + window, + } + } + + pub fn set_window(&mut self, window: Duration) { + self.window = window; + } + + /// Record that entry `log_index` of `group_id` committed with proposal + /// `key`. Key `0` names no proposal and is not kept. + pub fn note(&mut self, group_id: u64, log_index: u64, key: u64, now: Instant) { + let group = self.groups.entry(group_id).or_default(); + if group.first_noted == 0 { + group.first_noted = log_index; + } + if key == 0 { + return; + } + group.keys.push_back(CommittedKey { + log_index, + key, + at: now, + }); + while group + .keys + .front() + .is_some_and(|oldest| now.saturating_duration_since(oldest.at) > self.window) + { + group.keys.pop_front(); + } + while group.keys.len() > PROPOSAL_LEDGER_CAPACITY { + if let Some(dropped) = group.keys.pop_front() { + group.cap_dropped_through = group.cap_dropped_through.max(dropped.log_index); + } + } + } + + /// `group_id`'s keys at or below `through`, committed within the window, + /// and the lowest index from which they are complete. + pub fn through(&self, group_id: u64, through: u64, now: Instant) -> CarriedKeys { + let Some(group) = self.groups.get(&group_id) else { + // Nothing of the group was seen here: no index is known. + return CarriedKeys { + keys: Vec::new(), + complete_from: through.saturating_add(1), + }; + }; + let keys = group + .keys + .iter() + .filter(|entry| { + entry.log_index <= through && now.saturating_duration_since(entry.at) <= self.window + }) + .map(|entry| (entry.log_index, entry.key)) + .collect(); + let from = group + .first_noted + .max(group.cap_dropped_through.saturating_add(1)); + // Complete from the group's first index is complete. + let complete_from = if from <= 1 { 0 } else { from }; + CarriedKeys { + keys, + complete_from, + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn keys_at_or_below_the_index_within_the_window_are_returned() { + let window = Duration::from_secs(30); + let mut keys = CommittedKeys::new(window); + let start = Instant::now(); + keys.note(1, 4, 0xa, start); + keys.note(1, 5, 0, start); + keys.note(1, 6, 0xb, start); + keys.note(2, 3, 0xc, start); + keys.note(1, 9, 0xd, start); + + assert_eq!(keys.through(1, 6, start).keys, vec![(4, 0xa), (6, 0xb)]); + assert_eq!(keys.through(2, 6, start).keys, vec![(3, 0xc)]); + assert!(keys.through(1, 6, start + window * 2).keys.is_empty()); + } + + /// Keys from a group's first index are complete. A process that first saw + /// the group at a later index, or a count bound that dropped keys, leaves + /// the keys complete only above the gap. + #[test] + fn complete_from_names_the_gap() { + let window = Duration::from_secs(30); + let start = Instant::now(); + + let mut keys = CommittedKeys::new(window); + keys.note(1, 1, 0xa, start); + assert_eq!(keys.through(1, 1, start).complete_from, 0); + + keys.note(2, 40, 0xb, start); + assert_eq!(keys.through(2, 50, start).complete_from, 40); + + assert_eq!(keys.through(3, 50, start).complete_from, 51); + + let mut capped = CommittedKeys::new(window); + let last = PROPOSAL_LEDGER_CAPACITY as u64 + 3; + for index in 1..=last { + capped.note(1, index, index, start); + } + assert_eq!(capped.through(1, last, start).complete_from, 4); + } + + #[test] + fn keys_past_the_window_are_evicted_on_the_next_note() { + let window = Duration::from_secs(30); + let mut keys = CommittedKeys::new(window); + let start = Instant::now(); + keys.note(1, 1, 0xa, start); + keys.note(1, 2, 0xb, start + window * 2); + assert_eq!(keys.groups.get(&1).map(|group| group.keys.len()), Some(1)); + } +} diff --git a/nodedb/src/control/distributed_applier/propose_tracker/cut.rs b/nodedb/src/control/distributed_applier/propose_tracker/cut.rs new file mode 100644 index 000000000..d434e8d0d --- /dev/null +++ b/nodedb/src/control/distributed_applier/propose_tracker/cut.rs @@ -0,0 +1,71 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-group indexes a data-group snapshot is cut at. +//! +//! - `started`: the highest entry the apply loop took off its backlog. Entries +//! start in log order, so every entry at or below it started. A snapshot +//! builder fences the group's apply and captures at this index once the +//! group settled through it. +//! - `covered`: the highest entry a snapshot this node installed holds. The +//! snapshot is cut at or above its Raft index, so the entries between the +//! two reach the apply loop from the log. Their effects are in the +//! installed state already, so the loop concludes them without applying. +//! +//! One pair of indexes per group, bounded by the groups of the cluster. + +use std::collections::HashMap; + +/// Both indexes of one group. +#[derive(Debug, Default, Clone, Copy)] +struct GroupCut { + started: u64, + covered: u64, +} + +/// Every group's cut indexes. +#[derive(Debug, Default)] +pub(super) struct GroupCuts { + groups: HashMap, +} + +impl GroupCuts { + /// Raise `group_id`'s started index to `log_index`. + pub fn note_started(&mut self, group_id: u64, log_index: u64) { + let cut = self.groups.entry(group_id).or_default(); + cut.started = cut.started.max(log_index); + } + + /// Highest entry of `group_id` the apply loop started. + pub fn started_through(&self, group_id: u64) -> u64 { + self.groups.get(&group_id).map_or(0, |cut| cut.started) + } + + /// Raise `group_id`'s covered index to `log_index`. + pub fn cover_through(&mut self, group_id: u64, log_index: u64) { + let cut = self.groups.entry(group_id).or_default(); + cut.covered = cut.covered.max(log_index); + } + + /// Highest entry of `group_id` an installed snapshot holds. + pub fn covered_through(&self, group_id: u64) -> u64 { + self.groups.get(&group_id).map_or(0, |cut| cut.covered) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn indexes_only_rise_and_stay_per_group() { + let mut cuts = GroupCuts::default(); + cuts.note_started(1, 9); + cuts.note_started(1, 4); + cuts.cover_through(1, 12); + cuts.cover_through(1, 7); + assert_eq!(cuts.started_through(1), 9); + assert_eq!(cuts.covered_through(1), 12); + assert_eq!(cuts.started_through(2), 0); + assert_eq!(cuts.covered_through(2), 0); + } +} diff --git a/nodedb/src/control/distributed_applier/propose_tracker/mod.rs b/nodedb/src/control/distributed_applier/propose_tracker/mod.rs new file mode 100644 index 000000000..3c71bc5d7 --- /dev/null +++ b/nodedb/src/control/distributed_applier/propose_tracker/mod.rs @@ -0,0 +1,16 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Propose tracker: lets proposers wait for a Raft entry to commit and +//! execute on this node. + +pub mod applied_write; +pub mod applying; +mod committed_keys; +mod cut; +mod slots; +pub mod tracker; + +pub use applied_write::{AppliedWrite, ProposeResult}; +pub use applying::ApplyingEntry; +pub use committed_keys::CarriedKeys; +pub use tracker::{DEFAULT_WAITER_WINDOW, ProposeTracker}; diff --git a/nodedb/src/control/distributed_applier/propose_tracker/slots.rs b/nodedb/src/control/distributed_applier/propose_tracker/slots.rs new file mode 100644 index 000000000..b0523fd51 --- /dev/null +++ b/nodedb/src/control/distributed_applier/propose_tracker/slots.rs @@ -0,0 +1,561 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Propose waiters and early results, keyed by `(group_id, log_index)`. +//! +//! A proposer registers its waiter after the proposal returns its log index. +//! The apply loop can finish the entry first, so a result with no waiter is +//! stored for the `register` that follows. Most stored results are never +//! claimed: a follower applies entries proposed on other nodes. Stored +//! results are therefore bounded by the waiter window and by count, and the +//! oldest are evicted. + +use std::collections::hash_map::Entry; +use std::collections::{HashMap, HashSet, VecDeque}; +use std::time::{Duration, Instant}; + +use tokio::sync::oneshot; + +use super::applied_write::ProposeResult; + +/// Results stored before their waiter registered, kept at most this many. +pub(super) const MAX_STORED_RESULTS: usize = 1 << 16; + +/// A pending waiter, or a result that arrived before its waiter. +enum Slot { + /// `expected_key` is the proposer's idempotency key. `0` accepts any key. + Waiting { + tx: oneshot::Sender, + expected_key: u64, + }, + /// Stored at `at`. Evicted once older than the waiter window. + Completed { result: ProposeResult, at: Instant }, +} + +/// The committed proposal keys a snapshot install covered in one group. +struct Covered { + through: u64, + /// Empty once the install is older than the waiter window. + keys: HashSet, + /// Lowest index the keys are complete from. Below it a missing key + /// proves nothing. [`KEYS_EXPIRED`] once the keys are cleared. + complete_from: u64, + at: Instant, +} + +/// `Covered::complete_from` of an install whose keys expired: no index is +/// complete, so every covered waiter gets an unknown outcome. +const KEYS_EXPIRED: u64 = u64::MAX; + +/// Every waiter and stored result, plus what an install covered. +pub(super) struct WaiterSlots { + slots: HashMap<(u64, u64), Slot>, + /// Stored results in store order, for eviction by age and count. + stored: VecDeque<(Instant, (u64, u64))>, + /// Entries the apply loop sent toward a core and has not concluded. Their + /// own `complete` answers their waiters, so an install never does. + dispatched: HashSet<(u64, u64)>, + covered: HashMap, + window: Duration, +} + +/// Whether the entry that committed carries a different proposal than the +/// waiter reserved. `0` on either side means no key to compare. +fn key_mismatch(applied_key: u64, expected_key: u64) -> bool { + applied_key != 0 && expected_key != 0 && applied_key != expected_key +} + +fn retryable_leader_change(group_id: u64, log_index: u64) -> ProposeResult { + Err(crate::Error::RetryableLeaderChange { + group_id, + log_index, + }) +} + +fn proposal_outcome_unknown(group_id: u64, log_index: u64) -> ProposeResult { + Err(crate::Error::ProposalOutcomeUnknown { + group_id, + log_index, + }) +} + +fn committed_result_unavailable(group_id: u64, log_index: u64) -> ProposeResult { + Err(crate::Error::CommittedResultUnavailable { + group_id, + log_index, + }) +} + +impl WaiterSlots { + pub fn new(window: Duration) -> Self { + Self { + slots: HashMap::new(), + stored: VecDeque::new(), + dispatched: HashSet::new(), + covered: HashMap::new(), + window, + } + } + + pub fn set_window(&mut self, window: Duration) { + self.window = window; + } + + /// Number of stored results not yet claimed. + #[cfg(test)] + pub fn stored_len(&self) -> usize { + self.slots + .values() + .filter(|slot| matches!(slot, Slot::Completed { .. })) + .count() + } + + /// The answer for a waiter at a covered index. A key the snapshot carried + /// committed. A missing key means overwritten only at an index the keys + /// are complete from; below it the outcome is unknown. A waiter with no + /// key cannot be matched and is told it committed. + fn covered_outcome(&self, group_id: u64, log_index: u64, expected_key: u64) -> ProposeResult { + let Some(covered) = self.covered.get(&group_id) else { + return proposal_outcome_unknown(group_id, log_index); + }; + if covered.keys.contains(&expected_key) { + return committed_result_unavailable(group_id, log_index); + } + if log_index < covered.complete_from { + return proposal_outcome_unknown(group_id, log_index); + } + if expected_key == 0 { + committed_result_unavailable(group_id, log_index) + } else { + retryable_leader_change(group_id, log_index) + } + } + + /// Clear the keys of every install older than the waiter window. Every + /// waiter the keys can answer has passed its deadline, so holding them + /// serves nobody. A later waiter in the covered range gets an unknown + /// outcome, the safe answer once the keys are gone. + fn expire_covered(&mut self, now: Instant) { + let window = self.window; + for covered in self.covered.values_mut() { + if covered.complete_from != KEYS_EXPIRED + && now.saturating_duration_since(covered.at) > window + { + covered.keys = HashSet::new(); + covered.complete_from = KEYS_EXPIRED; + } + } + } + + fn is_covered(&self, group_id: u64, log_index: u64) -> bool { + self.covered + .get(&group_id) + .is_some_and(|covered| log_index <= covered.through) + && !self.dispatched.contains(&(group_id, log_index)) + } + + pub fn register( + &mut self, + group_id: u64, + log_index: u64, + expected_key: u64, + tx: oneshot::Sender, + now: Instant, + ) { + self.expire_covered(now); + let key = (group_id, log_index); + if !self.slots.contains_key(&key) && self.is_covered(group_id, log_index) { + // No apply on this node will complete a covered entry. + let _ = tx.send(self.covered_outcome(group_id, log_index, expected_key)); + return; + } + match self.slots.entry(key) { + Entry::Vacant(e) => { + e.insert(Slot::Waiting { tx, expected_key }); + } + Entry::Occupied(mut e) => match e.get() { + Slot::Completed { .. } => { + if let Slot::Completed { result, .. } = e.remove() { + let _ = tx.send(result); + } + } + Slot::Waiting { .. } => { + // A second register for one index. The older receiver + // sees its channel close. + e.insert(Slot::Waiting { tx, expected_key }); + } + }, + } + } + + pub fn abandon(&mut self, group_id: u64, log_index: u64) { + if let Entry::Occupied(e) = self.slots.entry((group_id, log_index)) + && matches!(e.get(), Slot::Waiting { .. }) + { + e.remove(); + } + } + + /// Answer the waiter at `(group_id, log_index)` with `result`, or store + /// the result for a `register` still to come. Returns whether a waiter + /// was answered. + pub fn complete( + &mut self, + group_id: u64, + log_index: u64, + applied_key: u64, + result: ProposeResult, + now: Instant, + ) -> bool { + let key = (group_id, log_index); + match self.slots.entry(key) { + Entry::Vacant(e) => { + e.insert(Slot::Completed { result, at: now }); + self.stored.push_back((now, key)); + self.evict(now); + false + } + Entry::Occupied(mut e) => match e.get() { + Slot::Waiting { expected_key, .. } => { + let final_result = if key_mismatch(applied_key, *expected_key) { + tracing::warn!( + group_id, + log_index, + applied_key, + expected_key = *expected_key, + "raft entry at proposer's index was overwritten by \ + a different proposal (idempotency_key mismatch); \ + surfacing RetryableLeaderChange" + ); + retryable_leader_change(group_id, log_index) + } else { + result + }; + if let Slot::Waiting { tx, .. } = e.remove() { + let _ = tx.send(final_result); + return true; + } + false + } + Slot::Completed { .. } => { + // A second completion of one index: the latest wins and + // keeps the original store time. + if let Slot::Completed { at, .. } = e.get() { + let at = *at; + e.insert(Slot::Completed { result, at }); + } + false + } + }, + } + } + + /// Drop stored results older than the waiter window, then the oldest past + /// the count bound. A waiter cannot register for them anymore. Install + /// keys older than the window go too. + fn evict(&mut self, now: Instant) { + self.expire_covered(now); + while let Some(&(at, key)) = self.stored.front() { + let expired = now.saturating_duration_since(at) > self.window; + if !expired && self.stored.len() <= MAX_STORED_RESULTS { + break; + } + self.stored.pop_front(); + // Only the result this record names: a claimed slot or a newer + // store at the same index stays. + if let Entry::Occupied(e) = self.slots.entry(key) + && matches!(e.get(), Slot::Completed { at: stored, .. } if *stored == at) + { + e.remove(); + } + } + } + + pub fn note_dispatched(&mut self, group_id: u64, log_index: u64) { + self.dispatched.insert((group_id, log_index)); + } + + pub fn note_concluded(&mut self, group_id: u64, log_index: u64) { + self.dispatched.remove(&(group_id, log_index)); + } + + /// Answer every waiter of `group_id` at or below `through` whose entry + /// the apply loop has not dispatched, by its key against `keys`, which + /// are complete from `complete_from`. Later registrations at or below + /// `through` are answered the same way. + pub fn resolve_covered( + &mut self, + group_id: u64, + through: u64, + keys: HashSet, + complete_from: u64, + now: Instant, + ) { + let window = self.window; + match self.covered.entry(group_id) { + Entry::Vacant(e) => { + e.insert(Covered { + through, + keys, + complete_from, + at: now, + }); + } + Entry::Occupied(mut e) => { + let covered = e.get_mut(); + // Keys of an install past the window serve no waiter. + if now.saturating_duration_since(covered.at) > window { + covered.keys = keys; + covered.complete_from = complete_from; + } else { + covered.keys.extend(keys); + // A gap in either install stays a gap. + covered.complete_from = covered.complete_from.max(complete_from); + } + covered.through = covered.through.max(through); + covered.at = now; + } + } + let answer: Vec<(u64, u64)> = self + .slots + .iter() + .filter(|((gid, index), slot)| { + *gid == group_id + && *index <= through + && matches!(slot, Slot::Waiting { .. }) + && !self.dispatched.contains(&(*gid, *index)) + }) + .map(|(key, _)| *key) + .collect(); + for (gid, index) in answer { + if let Some(Slot::Waiting { tx, expected_key }) = self.slots.remove(&(gid, index)) { + let _ = tx.send(self.covered_outcome(gid, index, expected_key)); + } + } + } + + /// Answer the waiter of an entry the apply loop skipped as covered. The + /// entry's own key is known here. Stores nothing when no waiter exists. + pub fn complete_covered(&mut self, group_id: u64, log_index: u64, applied_key: u64) -> bool { + let Entry::Occupied(e) = self.slots.entry((group_id, log_index)) else { + return false; + }; + let Slot::Waiting { expected_key, .. } = e.get() else { + return false; + }; + let result = if key_mismatch(applied_key, *expected_key) { + retryable_leader_change(group_id, log_index) + } else { + committed_result_unavailable(group_id, log_index) + }; + if let Slot::Waiting { tx, .. } = e.remove() { + let _ = tx.send(result); + return true; + } + false + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::distributed_applier::propose_tracker::AppliedWrite; + + const WINDOW: Duration = Duration::from_secs(30); + + fn waiter( + slots: &mut WaiterSlots, + log_index: u64, + key: u64, + ) -> oneshot::Receiver { + waiter_at(slots, log_index, key, Instant::now()) + } + + fn waiter_at( + slots: &mut WaiterSlots, + log_index: u64, + key: u64, + now: Instant, + ) -> oneshot::Receiver { + let (tx, rx) = oneshot::channel(); + slots.register(1, log_index, key, tx, now); + rx + } + + /// Once the waiter window passes, an install's keys are released and a + /// waiter in its range gets an unknown outcome, even for a carried key. + #[test] + fn expired_install_keys_are_released_and_answer_unknown() { + let mut slots = WaiterSlots::new(WINDOW); + let installed = Instant::now(); + slots.resolve_covered(1, 8, HashSet::from([0xa, 0xb]), 0, installed); + + let mut fresh = waiter_at(&mut slots, 5, 0xa, installed); + assert!(matches!( + fresh.try_recv().expect("answered"), + Err(crate::Error::CommittedResultUnavailable { .. }) + )); + + let later = installed + WINDOW * 2; + let mut late = waiter_at(&mut slots, 6, 0xb, later); + assert!(matches!( + late.try_recv().expect("answered"), + Err(crate::Error::ProposalOutcomeUnknown { log_index: 6, .. }) + )); + let mut late_unkeyed = waiter_at(&mut slots, 7, 0, later); + assert!(matches!( + late_unkeyed.try_recv().expect("answered"), + Err(crate::Error::ProposalOutcomeUnknown { log_index: 7, .. }) + )); + let covered = slots.covered.get(&1).expect("coverage kept"); + assert_eq!(covered.keys.capacity(), 0, "the key set is released"); + } + + /// A stored result past the window releases install keys too, with no + /// register needed. + #[test] + fn a_later_completion_releases_expired_install_keys() { + let mut slots = WaiterSlots::new(WINDOW); + let installed = Instant::now(); + slots.resolve_covered(1, 8, HashSet::from([0xa]), 0, installed); + slots.complete(1, 20, 0, applied(), installed + WINDOW * 2); + let covered = slots.covered.get(&1).expect("coverage kept"); + assert!(covered.keys.is_empty()); + assert_eq!(covered.complete_from, KEYS_EXPIRED); + } + + fn applied() -> ProposeResult { + Ok(AppliedWrite::unversioned(b"real".to_vec())) + } + + /// A follower applies entries nobody on it waits for. Their stored + /// results never grow past the count bound, and age out of the window. + #[test] + fn unclaimed_results_do_not_grow() { + let mut slots = WaiterSlots::new(WINDOW); + let start = Instant::now(); + let total = MAX_STORED_RESULTS as u64 + 1000; + for index in 1..=total { + slots.complete(1, index, 0, applied(), start); + } + assert_eq!(slots.stored_len(), MAX_STORED_RESULTS); + + slots.complete(1, total + 1, 0, applied(), start + WINDOW * 2); + assert_eq!(slots.stored_len(), 1); + } + + /// A result stored before its waiter registered reaches the waiter. + #[test] + fn a_late_register_claims_its_stored_result() { + let mut slots = WaiterSlots::new(WINDOW); + slots.complete(1, 5, 0xa, applied(), Instant::now()); + let mut rx = waiter(&mut slots, 5, 0xa); + let result = rx.try_recv().expect("stored result delivered"); + assert_eq!(result.expect("applied").payload, b"real"); + assert_eq!(slots.stored_len(), 0); + } + + /// Under a covering snapshot, a waiter whose key the snapshot carries + /// committed; one whose key it lacks was overwritten and never commits. + #[test] + fn covered_waiters_are_answered_by_their_key() { + let mut slots = WaiterSlots::new(WINDOW); + let mut committed = waiter(&mut slots, 5, 0xa); + let mut overwritten = waiter(&mut slots, 6, 0xb); + slots.resolve_covered(1, 7, HashSet::from([0xa]), 0, Instant::now()); + + assert!(matches!( + committed.try_recv().expect("answered"), + Err(crate::Error::CommittedResultUnavailable { log_index: 5, .. }) + )); + assert!(matches!( + overwritten.try_recv().expect("answered"), + Err(crate::Error::RetryableLeaderChange { log_index: 6, .. }) + )); + + let mut late_committed = waiter(&mut slots, 4, 0xa); + assert!(matches!( + late_committed.try_recv().expect("answered"), + Err(crate::Error::CommittedResultUnavailable { .. }) + )); + let mut late_overwritten = waiter(&mut slots, 3, 0xc); + assert!(matches!( + late_overwritten.try_recv().expect("answered"), + Err(crate::Error::RetryableLeaderChange { .. }) + )); + } + + /// A snapshot whose keys the count bound capped answers an old waiter + /// with an unknown outcome, and a waiter at a complete index exactly. + #[test] + fn a_capped_snapshot_answers_old_waiters_with_an_unknown_outcome() { + let mut slots = WaiterSlots::new(WINDOW); + let mut old = waiter(&mut slots, 3, 0xa); + let mut committed = waiter(&mut slots, 6, 0xb); + let mut overwritten = waiter(&mut slots, 7, 0xc); + // Keys complete from index 5: the cap dropped everything below it. + slots.resolve_covered(1, 8, HashSet::from([0xb]), 5, Instant::now()); + + assert!(matches!( + old.try_recv().expect("answered"), + Err(crate::Error::ProposalOutcomeUnknown { log_index: 3, .. }) + )); + assert!(matches!( + committed.try_recv().expect("answered"), + Err(crate::Error::CommittedResultUnavailable { log_index: 6, .. }) + )); + assert!(matches!( + overwritten.try_recv().expect("answered"), + Err(crate::Error::RetryableLeaderChange { log_index: 7, .. }) + )); + + let mut late_old = waiter(&mut slots, 4, 0xd); + assert!(matches!( + late_old.try_recv().expect("answered"), + Err(crate::Error::ProposalOutcomeUnknown { .. }) + )); + // A carried key proves the commit even below the gap. + let mut late_carried = waiter(&mut slots, 2, 0xb); + assert!(matches!( + late_carried.try_recv().expect("answered"), + Err(crate::Error::CommittedResultUnavailable { .. }) + )); + } + + /// An entry already sent to a core before the install keeps its waiter: + /// its own completion answers it with the real result. + #[test] + fn a_dispatched_entry_keeps_its_waiter_through_an_install() { + let mut slots = WaiterSlots::new(WINDOW); + let mut rx = waiter(&mut slots, 5, 0xa); + slots.note_dispatched(1, 5); + slots.resolve_covered(1, 7, HashSet::from([0xa]), 0, Instant::now()); + assert!(rx.try_recv().is_err(), "the install must not answer it"); + + assert!(slots.complete(1, 5, 0xa, applied(), Instant::now())); + slots.note_concluded(1, 5); + let result = rx.try_recv().expect("real result delivered"); + assert_eq!(result.expect("applied").payload, b"real"); + } + + /// A skipped covered entry answers its own waiter by the entry's key. + #[test] + fn a_skipped_covered_entry_answers_by_its_own_key() { + let mut slots = WaiterSlots::new(WINDOW); + let mut own = waiter(&mut slots, 5, 0xa); + assert!(slots.complete_covered(1, 5, 0xa)); + assert!(matches!( + own.try_recv().expect("answered"), + Err(crate::Error::CommittedResultUnavailable { .. }) + )); + + let mut replaced = waiter(&mut slots, 6, 0xa); + assert!(slots.complete_covered(1, 6, 0xb)); + assert!(matches!( + replaced.try_recv().expect("answered"), + Err(crate::Error::RetryableLeaderChange { .. }) + )); + + assert!(!slots.complete_covered(1, 7, 0xc)); + assert_eq!(slots.stored_len(), 0, "no waiter, nothing stored"); + } +} diff --git a/nodedb/src/control/distributed_applier/propose_tracker/tracker.rs b/nodedb/src/control/distributed_applier/propose_tracker/tracker.rs new file mode 100644 index 000000000..7f543d0c8 --- /dev/null +++ b/nodedb/src/control/distributed_applier/propose_tracker/tracker.rs @@ -0,0 +1,453 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `ProposeTracker`: lets proposers wait for a Raft entry to commit and +//! execute on this node, with race-safe resolution if the apply path beats +//! the proposer's `register()` call. + +use std::collections::{HashMap, HashSet}; +use std::sync::{Arc, Mutex}; +use std::time::{Duration, Instant}; + +use tokio::sync::oneshot; + +use nodedb_cluster::GroupAppliedWatchers; + +use super::applied_write::ProposeResult; +use super::applying::ApplyingEntry; +use super::committed_keys::{CarriedKeys, CommittedKeys}; +use super::cut::GroupCuts; +use super::slots::WaiterSlots; +use crate::control::distributed_applier::apply_window::ApplyWindow; + +/// How long a propose waiter can exist: the default statement deadline. +/// `start_raft` sets the configured deadline through +/// [`ProposeTracker::with_waiter_window`]. +pub const DEFAULT_WAITER_WINDOW: Duration = Duration::from_secs(30); + +/// Tracks pending proposals awaiting Raft commit. +/// +/// Keyed by `(group_id, log_index)`. The proposer calls `register()` after +/// the proposal returns the log index; `run_apply_loop` calls `complete()` +/// after the entry is applied. Either side can win the race — `complete()` +/// stores the result if no waiter exists yet, and `register()` picks it up +/// immediately if `complete()` already fired. Stored results are bounded by +/// the waiter window. +pub struct ProposeTracker { + slots: Mutex, + /// Keys of recently committed entries, carried by a data-group snapshot. + committed: Mutex, + /// Keys a snapshot install restored, per group, until the install adopts. + installed_keys: Mutex>, + /// Keys a snapshot install restored, until the apply loop takes them into + /// its proposal ledger. + restored_for_ledger: Mutex>, + /// Per-Raft-group apply watermark registry. Bumped by + /// [`Self::note_applied`] once every entry of the group up to the index + /// finished, so the watcher reflects "data applied on this node up to + /// index N" — the only semantic that's useful for cross-node visibility + /// waits. Tick-loop bumps cover the metadata group (sync redb apply); + /// this tracker covers data groups (async SPSC dispatch through + /// `run_apply_loop`). `None` only in tests that don't exercise the + /// watcher. + group_watchers: Option>, + /// Per group, the oldest committed entry the apply loop has not finished. + applying: Mutex>, + /// Per-group bound on entries between hand-off and settle, shared by the + /// applier that hands entries off and the loop that settles them. + window: Arc, + /// Per group, the indexes a data-group snapshot is cut at. + cuts: Mutex, +} + +/// The keys one install carried, and the lowest index they are complete from. +#[derive(Default)] +struct InstalledKeys { + keys: HashSet, + complete_from: u64, +} + +impl Default for ProposeTracker { + fn default() -> Self { + Self::new() + } +} + +impl ProposeTracker { + pub fn new() -> Self { + Self { + slots: Mutex::new(WaiterSlots::new(DEFAULT_WAITER_WINDOW)), + committed: Mutex::new(CommittedKeys::new(DEFAULT_WAITER_WINDOW)), + installed_keys: Mutex::new(HashMap::new()), + restored_for_ledger: Mutex::new(Vec::new()), + group_watchers: None, + applying: Mutex::new(HashMap::new()), + window: Arc::new(ApplyWindow::default()), + cuts: Mutex::new(GroupCuts::default()), + } + } + + /// Wire the per-group apply watermark registry. Called by + /// `start_raft` after `SharedState` is constructed. + pub fn with_group_watchers(mut self, watchers: Arc) -> Self { + self.group_watchers = Some(watchers); + self + } + + /// Set how long a propose waiter can exist: the statement deadline. + /// Stored results and committed keys are kept this long. + pub fn with_waiter_window(self, window: Duration) -> Self { + self.slots + .lock() + .unwrap_or_else(|p| p.into_inner()) + .set_window(window); + self.committed + .lock() + .unwrap_or_else(|p| p.into_inner()) + .set_window(window); + self + } + + /// The per-group apply window. + pub fn window(&self) -> &Arc { + &self.window + } + + fn slots(&self) -> std::sync::MutexGuard<'_, WaiterSlots> { + self.slots.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Record the oldest entry of `group_id` the apply loop has not finished, + /// or `None` once every entry it holds for the group finished. + pub fn note_applying(&self, group_id: u64, entry: Option) { + let mut applying = self.applying.lock().unwrap_or_else(|p| p.into_inner()); + match entry { + Some(entry) => { + applying.insert(group_id, entry); + } + None => { + applying.remove(&group_id); + } + } + } + + /// The oldest entry of `group_id` the apply loop has not finished, if any. + pub fn applying(&self, group_id: u64) -> Option { + self.applying + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&group_id) + .cloned() + } + + /// Advance `group_id`'s applied watermark to `log_index`. The apply loop + /// calls it once every entry of the group up to `log_index` finished. + pub fn note_applied(&self, group_id: u64, log_index: u64) { + if let Some(w) = &self.group_watchers { + w.bump(group_id, log_index); + } + } + + /// Register a waiter for a proposed entry. Returns a receiver that + /// resolves when the entry is committed and executed. + /// + /// If `complete()` was called first, the receiver is pre-resolved. At an + /// index an installed snapshot covers, it is answered at once by the + /// proposal's key. + pub fn register( + &self, + group_id: u64, + log_index: u64, + expected_key: u64, + ) -> oneshot::Receiver { + let (tx, rx) = oneshot::channel(); + self.slots() + .register(group_id, log_index, expected_key, tx, Instant::now()); + rx + } + + /// Drop the waiter a proposer registered at `(group_id, log_index)` and + /// stopped waiting on: its deadline passed, or this node left the group + /// and will never apply the index. A result already stored there stays. + pub fn abandon(&self, group_id: u64, log_index: u64) { + self.slots().abandon(group_id, log_index); + } + + /// Complete a waiter after the entry has been committed and executed. + /// + /// If the proposer has already called `register()`, the result is sent + /// immediately. If not, the result is stored, within the waiter window, + /// so the next `register()` picks it up without waiting. + /// + /// When the applied entry's key differs from the proposer's, a leader + /// change overwrote the proposal, and the waiter gets + /// `RetryableLeaderChange` instead of another proposal's result. + /// + /// Returns true if a live waiter was found and notified, false otherwise. + pub fn complete( + &self, + group_id: u64, + log_index: u64, + applied_key: u64, + result: ProposeResult, + ) -> bool { + self.slots() + .complete(group_id, log_index, applied_key, result, Instant::now()) + } + + /// Note that the apply loop sent entry `log_index` of `group_id` toward a + /// core. Its own `complete` answers its waiter, so an install does not. + pub fn note_dispatched(&self, group_id: u64, log_index: u64) { + self.slots().note_dispatched(group_id, log_index); + } + + /// Note that a dispatched entry concluded. + pub fn note_concluded(&self, group_id: u64, log_index: u64) { + self.slots().note_concluded(group_id, log_index); + } + + /// Record the keys of committed entries the apply loop received, for the + /// snapshots this node builds. + pub fn note_committed(&self, group_id: u64, entries: impl IntoIterator) { + let now = Instant::now(); + let mut committed = self.committed.lock().unwrap_or_else(|p| p.into_inner()); + for (log_index, key) in entries { + committed.note(group_id, log_index, key, now); + } + } + + fn cuts(&self) -> std::sync::MutexGuard<'_, GroupCuts> { + self.cuts.lock().unwrap_or_else(|p| p.into_inner()) + } + + /// Note that the apply loop took entry `log_index` of `group_id` off its + /// backlog. Entries start in log order. + pub fn note_started(&self, group_id: u64, log_index: u64) { + self.cuts().note_started(group_id, log_index); + } + + /// Highest entry of `group_id` the apply loop started. With the group's + /// apply fenced, a snapshot is cut here once the group settled through it. + pub fn started_through(&self, group_id: u64) -> u64 { + self.cuts().started_through(group_id) + } + + /// Note that a snapshot this node installed holds `group_id`'s entries + /// through `log_index`. + pub fn cover_through(&self, group_id: u64, log_index: u64) { + self.cuts().cover_through(group_id, log_index); + } + + /// Highest entry of `group_id` an installed snapshot holds. The apply + /// loop concludes an entry at or below it without applying it. + pub fn covered_through(&self, group_id: u64) -> u64 { + self.cuts().covered_through(group_id) + } + + /// `group_id`'s committed keys at or below `through`, within the waiter + /// window, and the lowest index they are complete from. A snapshot at + /// `through` carries them. + pub fn committed_keys_through(&self, group_id: u64, through: u64) -> CarriedKeys { + self.committed + .lock() + .unwrap_or_else(|p| p.into_inner()) + .through(group_id, through, Instant::now()) + } + + /// Take the committed keys a snapshot install carried, complete from + /// index `complete_from`. The install's adopt answers waiters by them, + /// and the apply loop adds them to its proposal ledger. + pub fn restore_committed_keys(&self, group_id: u64, keys: &[(u64, u64)], complete_from: u64) { + { + let mut installed = self + .installed_keys + .lock() + .unwrap_or_else(|p| p.into_inner()); + let group = installed.entry(group_id).or_default(); + group.keys.extend(keys.iter().map(|&(_, key)| key)); + group.complete_from = group.complete_from.max(complete_from); + } + self.restored_for_ledger + .lock() + .unwrap_or_else(|p| p.into_inner()) + .extend(keys.iter().map(|&(_, key)| key)); + } + + /// Keys restored by snapshot installs since the last call. + pub fn take_restored_keys(&self) -> Vec { + std::mem::take( + &mut *self + .restored_for_ledger + .lock() + .unwrap_or_else(|p| p.into_inner()), + ) + } + + /// Answer every waiter of `group_id` at or below `through`, the index of + /// a snapshot this node adopted, whose entry the apply loop has not + /// dispatched. A proposal key the snapshot carried committed: + /// [`crate::Error::CommittedResultUnavailable`]. A key it lacks was + /// overwritten and never commits: `RetryableLeaderChange`. Below the + /// index the keys are complete from, a missing key proves nothing: + /// [`crate::Error::ProposalOutcomeUnknown`]. A later `register` at or + /// below `through` is answered the same way at once. + /// + /// An install that restored no keys (an empty stub) leaves every covered + /// index unknown. + pub fn resolve_covered(&self, group_id: u64, through: u64) { + let installed = self + .installed_keys + .lock() + .unwrap_or_else(|p| p.into_inner()) + .remove(&group_id) + .unwrap_or(InstalledKeys { + keys: HashSet::new(), + complete_from: through.saturating_add(1), + }); + self.slots().resolve_covered( + group_id, + through, + installed.keys, + installed.complete_from, + Instant::now(), + ); + } + + /// Answer the waiter at `(group_id, log_index)`, an entry the apply loop + /// skipped because an installed snapshot covers it. The entry's own + /// `applied_key` decides the answer. Stores nothing when no waiter exists. + pub fn complete_covered(&self, group_id: u64, log_index: u64, applied_key: u64) -> bool { + self.slots() + .complete_covered(group_id, log_index, applied_key) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::distributed_applier::propose_tracker::AppliedWrite; + use crate::types::Lsn; + + #[test] + fn propose_tracker_register_and_complete() { + let tracker = ProposeTracker::new(); + let mut rx = tracker.register(1, 5, 0xdead_beef); + + // Waiter must receive both payload and coll_write_lsn — its only channel + // for a version minted on the apply path. + assert!(tracker.complete( + 1, + 5, + 0xdead_beef, + Ok(AppliedWrite { + payload: b"result".to_vec(), + write_version: Lsn::new(137), + }), + )); + + let result = rx.try_recv().unwrap().unwrap(); + assert_eq!(result.payload, b"result"); + assert_eq!(result.write_version, Lsn::new(137)); + } + + #[test] + fn propose_tracker_no_waiter_returns_false() { + let tracker = ProposeTracker::new(); + assert!(!tracker.complete(1, 99, 0, Ok(AppliedWrite::unversioned(Vec::new())))); + } + + #[test] + fn propose_tracker_key_mismatch_surfaces_retryable_leader_change() { + let tracker = ProposeTracker::new(); + let mut rx = tracker.register(1, 5, 0xaaaa); + + // A different proposer's entry committed at the same (group_id, + // log_index); waiter must see RetryableLeaderChange, not its result. + assert!(tracker.complete( + 1, + 5, + 0xbbbb, + Ok(AppliedWrite::unversioned( + b"other-proposers-payload".to_vec() + )), + )); + + match rx.try_recv().unwrap() { + Err(crate::Error::RetryableLeaderChange { + group_id, + log_index, + }) => { + assert_eq!(group_id, 1); + assert_eq!(log_index, 5); + } + other => panic!("expected RetryableLeaderChange, got {other:?}"), + } + } + + #[test] + fn propose_tracker_zero_applied_key_passes_through_explicit_error() { + // applied_key = 0 (leader-change no-op) must forward the explicit + // RetryableLeaderChange, not be treated as a key mismatch. + let tracker = ProposeTracker::new(); + let mut rx = tracker.register(1, 5, 0xaaaa); + assert!(tracker.complete( + 1, + 5, + 0, + Err(crate::Error::RetryableLeaderChange { + group_id: 1, + log_index: 5, + }), + )); + assert!(matches!( + rx.try_recv().unwrap(), + Err(crate::Error::RetryableLeaderChange { .. }) + )); + } + + /// The keys a leader carries in its snapshot decide each covered waiter + /// on the follower that installs it: a committed proposal gets + /// `CommittedResultUnavailable`, an overwritten one `RetryableLeaderChange`. + #[test] + fn snapshot_keys_decide_covered_waiters() { + let leader = ProposeTracker::new(); + leader.note_committed(1, [(5, 0xa), (6, 0xc), (9, 0xd)]); + let carried = leader.committed_keys_through(1, 7); + assert_eq!(carried.keys, vec![(5, 0xa), (6, 0xc)]); + assert_eq!(carried.complete_from, 5, "first index this leader saw"); + + let follower = ProposeTracker::new(); + let mut committed = follower.register(1, 5, 0xa); + let mut overwritten = follower.register(1, 6, 0xb); + let mut unknown = follower.register(1, 4, 0xe); + follower.restore_committed_keys(1, &carried.keys, carried.complete_from); + follower.resolve_covered(1, 7); + + assert!(matches!( + committed.try_recv().expect("answered"), + Err(crate::Error::CommittedResultUnavailable { log_index: 5, .. }) + )); + assert!(matches!( + overwritten.try_recv().expect("answered"), + Err(crate::Error::RetryableLeaderChange { log_index: 6, .. }) + )); + assert!(matches!( + unknown.try_recv().expect("answered"), + Err(crate::Error::ProposalOutcomeUnknown { log_index: 4, .. }) + )); + assert_eq!(follower.take_restored_keys(), vec![0xa, 0xc]); + } + + /// An install that restored no keys, an empty stub, leaves every covered + /// waiter's outcome unknown. + #[test] + fn a_stub_install_leaves_covered_outcomes_unknown() { + let tracker = ProposeTracker::new(); + let mut rx = tracker.register(1, 5, 0xa); + tracker.resolve_covered(1, 7); + assert!(matches!( + rx.try_recv().expect("answered"), + Err(crate::Error::ProposalOutcomeUnknown { log_index: 5, .. }) + )); + assert!(tracker.take_restored_keys().is_empty()); + } +} diff --git a/nodedb/src/control/event_action_error.rs b/nodedb/src/control/event_action_error.rs index 7e801f648..1b01d2836 100644 --- a/nodedb/src/control/event_action_error.rs +++ b/nodedb/src/control/event_action_error.rs @@ -9,9 +9,9 @@ use crate::control::system_txn::SystemTxnError; -/// Why a DEFINE EVENT THEN action template could not become executable SQL. +/// Why a DEFINE EVENT THEN action template did not become executable SQL. /// -/// Both cases mean the template is malformed in a way that could change what +/// Both cases mean the template is malformed in a way that can change what /// the rendered statement does, so rendering refuses rather than guessing. #[derive(Debug, Clone, thiserror::Error, PartialEq, Eq)] pub enum TriggerRenderError { @@ -34,7 +34,7 @@ pub enum TriggerRenderError { /// Why one DEFINE EVENT THEN action did not run to completion. #[derive(Debug, thiserror::Error)] pub enum TriggerActionError { - /// The action template could not be rendered into executable SQL. + /// The action template cannot be rendered into executable SQL. #[error("trigger action rejected: {source}")] Rejected { #[source] @@ -61,6 +61,14 @@ pub enum TriggerActionError { #[source] source: SystemTxnError, }, + + /// Whether an earlier firing of the event already applied the action + /// was unreadable, so the action did not run. + #[error("trigger action applied-key lookup failed: {source}")] + AppliedLookup { + #[source] + source: crate::Error, + }, } impl From for crate::Error { @@ -72,9 +80,9 @@ impl From for crate::Error { TriggerActionError::Rejected { source } => crate::Error::BadRequest { detail: source.to_string(), }, - TriggerActionError::Plan { source } | TriggerActionError::LeaseAdmission { source } => { - source - } + TriggerActionError::Plan { source } + | TriggerActionError::LeaseAdmission { source } + | TriggerActionError::AppliedLookup { source } => source, // The transaction error keeps its class: its statement or commit // error, or the Data-Plane verdict that aborted the commit. TriggerActionError::Transaction { source } => source.into(), @@ -94,7 +102,10 @@ impl TriggerActionError { match self { // A failed transaction applied nothing at all, so there is never // a partial application to duplicate by running it again. - Self::Plan { .. } | Self::LeaseAdmission { .. } | Self::Transaction { .. } => true, + Self::Plan { .. } + | Self::LeaseAdmission { .. } + | Self::Transaction { .. } + | Self::AppliedLookup { .. } => true, Self::Rejected { .. } => false, } } diff --git a/nodedb/src/control/event_trigger.rs b/nodedb/src/control/event_trigger.rs index 717f1198c..0105857e1 100644 --- a/nodedb/src/control/event_trigger.rs +++ b/nodedb/src/control/event_trigger.rs @@ -40,7 +40,7 @@ pub async fn process_write_event( // fire triggers fire event definitions. A restored row fired its event // definitions when it was first written. let fires = match event.source { - EventSource::User | EventSource::Deferred => true, + EventSource::User | EventSource::ImplicitClient | EventSource::Deferred => true, EventSource::Trigger | EventSource::RaftFollower | EventSource::CrdtSync @@ -96,6 +96,14 @@ pub async fn process_write_event( event.tenant_id, sql, &event_def.name, + event_action_key( + event.vshard_id.as_u32(), + event.lsn.as_u64(), + event.sequence, + event.database_id, + &event_def.name, + index, + ), ) .await } @@ -139,8 +147,8 @@ pub async fn process_write_event( "event trigger action failed" ); // Only an action that applied nothing can be re-run. A malformed - // template will never render, and a part-applied action would - // duplicate the tasks that already landed. + // template will never render, and a part-applied action + // duplicates the tasks that already landed. if let (true, Ok(sql)) = (error.is_retryable(), &rendered) { queue.enqueue(FailedAction { key: ActionKey { @@ -174,6 +182,7 @@ fn event_operation(op: WriteOp) -> &'static str { WriteOp::Update => "UPDATE", WriteOp::Delete | WriteOp::BulkDelete { .. } => "DELETE", WriteOp::Heartbeat => "HEARTBEAT", + WriteOp::Publish => "PUBLISH", } } @@ -321,13 +330,31 @@ fn render_then_action_sql(action: &str, event: &WriteEvent) -> Result, database_id: DatabaseId, tenant_id: TenantId, sql: &str, trigger_name: &str, + applied_key: crate::wal::CrossShardAppliedKey, ) -> Result<(), TriggerActionError> { + if let Some(dedup) = shared.cross_shard_dedup.get() + && dedup + .is_applied(&applied_key) + .map_err(|source| TriggerActionError::AppliedLookup { source })? + { + debug!( + trigger = trigger_name, + "event trigger action already applied for this event; not run again" + ); + return Ok(()); + } + let key_vshard = applied_key.source_vshard; let query_ctx = QueryContext::for_state(&shared); // A trigger action is database-defined code with no external requester, so // it plans as the system — the same SECURITY DEFINER model the trigger @@ -351,13 +378,19 @@ pub async fn run_event_action_sql( // Keep the Arc and lease scope alive through the whole action. Admission // is fail-closed while a descriptor drains. let lease_scope = Arc::new( - Arc::clone(&shared) + shared .acquire_plan_lease_scope(&versions) + .await .map_err(|source| TriggerActionError::LeaseAdmission { source })?, ); + // The action writes, so it is checked only before dispatch: cancelling + // committing tasks leaves their outcome unknown. + lease_scope + .check_not_revoked() + .map_err(|source| TriggerActionError::LeaseAdmission { source })?; // The action's tasks commit as one transaction. An action that dispatched - // its tasks one by one could stop half-applied, and re-running a + // its tasks one by one can stop half-applied, and re-running a // half-applied action repeats the tasks that already landed — which is // what makes a retry queue unsafe to point at it. let identity = event_action_identity(tenant_id); @@ -367,6 +400,7 @@ pub async fn run_event_action_sql( tasks, lease_scope, crate::event::EventSource::Trigger, + Some((applied_key, key_vshard)), ) .await .map_err(|source| TriggerActionError::Transaction { source })?; @@ -379,6 +413,25 @@ pub async fn run_event_action_sql( Ok(()) } +/// The key of THEN clause `index` of event `event_name`, fired by the source +/// event `(source_vshard, source_lsn, source_sequence)`: the event's +/// replicated identity, which every replica shares. +pub fn event_action_key( + source_vshard: u32, + source_lsn: u64, + source_sequence: u64, + database_id: DatabaseId, + event_name: &str, + index: usize, +) -> crate::wal::CrossShardAppliedKey { + crate::wal::CrossShardAppliedKey { + source_vshard, + source_lsn, + source_sequence, + origin: format!("event/{}/{event_name}/{index}", database_id.as_u64()), + } +} + /// Identity a DEFINE EVENT action executes under. /// /// A THEN action is database-defined code with no external requester, so it @@ -427,6 +480,7 @@ mod tests { valid_time_ms: None, user_id: None, statement_digest: None, + commit_hlc: Some(crate::event::test_utils::test_commit_hlc()), } } diff --git a/nodedb/src/control/exec_receiver/backup_cut.rs b/nodedb/src/control/exec_receiver/backup_cut.rs index 1a44bb81a..88dbac18d 100644 --- a/nodedb/src/control/exec_receiver/backup_cut.rs +++ b/nodedb/src/control/exec_receiver/backup_cut.rs @@ -7,16 +7,21 @@ //! That node takes the same cut on its replicas before it snapshots, so every //! write committed below `W` has its final outcome in the snapshot, and every //! write at or above `W` refuses a restore of the envelope. +//! +//! A database backup's request also carries a capture request. The node then +//! takes the cut with capturing barriers and answers with the captures it +//! parked, never with a snapshot of its live state. -use nodedb_cluster::rpc_codec::TypedClusterError; +use nodedb_cluster::rpc_codec::{ExecuteResponse, TypedClusterError}; use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; use crate::control::state::SharedState; -use super::support::execution_error_to_typed; +use super::support::{PLAN_DECODE_FAILED, execution_error_to_typed}; /// Take the cut a tenant snapshot plan asks for, then return the plan with -/// the request cleared. Every other plan passes through unchanged. +/// the request cleared. A capture request passes through unchanged for +/// [`answer_capture_plan`]. Every other plan passes through unchanged. pub(super) async fn take_backup_cut( state: &std::sync::Arc, plan: PhysicalPlan, @@ -24,6 +29,8 @@ pub(super) async fn take_backup_cut( let PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id, cut_watermark: Some(watermark), + cut_capture: None, + arrays, }) = plan else { return Ok(plan); @@ -34,5 +41,50 @@ pub(super) async fn take_backup_cut( Ok(PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id, cut_watermark: None, + cut_capture: None, + arrays, })) } + +/// Whether `plan` asks this node for a database backup's cut captures. +pub(super) fn is_capture_plan(plan: &PhysicalPlan) -> bool { + matches!( + plan, + PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + cut_capture: Some(_), + .. + }) + ) +} + +/// The answer to `plan` when it asks for a database backup's cut captures, +/// else `None`: this node takes the cut with capturing barriers and answers +/// with every capture it parked for the request, encoded for the wire. +pub(super) async fn answer_capture_plan( + state: &SharedState, + plan: &PhysicalPlan, +) -> Option { + let PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { + tenant_id, + cut_watermark, + cut_capture: Some(request), + .. + }) = plan + else { + return None; + }; + let Some(watermark) = *cut_watermark else { + return Some(ExecuteResponse::err(TypedClusterError::Internal { + code: PLAN_DECODE_FAILED, + message: "a cut capture request carries no cut watermark".into(), + })); + }; + let answer = crate::control::backup::cut_capture::collect::cut_and_reply( + state, *tenant_id, watermark, request, + ) + .await; + Some(match answer { + Ok(payload) => ExecuteResponse::ok(vec![payload], 0, 0), + Err(error) => ExecuteResponse::err(execution_error_to_typed(error)), + }) +} diff --git a/nodedb/src/control/exec_receiver/executor.rs b/nodedb/src/control/exec_receiver/executor.rs index e1dfe540a..df46cd15c 100644 --- a/nodedb/src/control/exec_receiver/executor.rs +++ b/nodedb/src/control/exec_receiver/executor.rs @@ -3,8 +3,9 @@ //! Local execution of incoming `ExecuteRequest` / `ExecuteStreamRequest` RPCs. //! //! When this node leads the target vShard, [`LocalPlanExecutor`] validates -//! descriptor versions, decodes the `PhysicalPlan`, and fans it across all -//! local Data-Plane cores before returning the merged result. +//! descriptor versions, decodes the `PhysicalPlan`, and runs it through +//! `execute_received_plan`: a vShard-scoped plan on its one owning core, every +//! other plan across all local cores. use std::sync::Arc; use std::time::{Duration, SystemTime}; @@ -16,13 +17,14 @@ use nodedb_cluster::forward::{ChunkSink, PlanExecutor}; use nodedb_cluster::rpc_codec::{ExecuteRequest, ExecuteResponse, TypedClusterError}; use crate::bridge::envelope::PhysicalPlan; -use crate::control::server::exchange::execute_plan_all_local_cores; +use crate::control::server::exchange::execute_received_plan; use crate::control::state::SharedState; use crate::control::trace_export::EmitSpanParams; use crate::types::DatabaseId; use super::backup_cut::take_backup_cut; use super::plan_decode::decode_plan; +use super::read_leg::confirm_read_leg; use super::request_validation::validate_request; use super::support::{PLAN_DECODE_FAILED, SinkOutcome, execution_error_to_typed}; @@ -138,146 +140,43 @@ impl LocalPlanExecutor { let tenant_id = crate::types::TenantId::new(req.tenant_id); let trace_id = nodedb_types::TraceId(req.trace_id); - if let PhysicalPlan::ClusterEvent( - nodedb_physical::physical_plan::ClusterEventOp::TenantWriteMarks { - tenant_id: marks_tenant, - group_ids, - }, - ) = &plan + if let Some(response) = super::backup_cut::answer_capture_plan(&self.state, &plan).await { + return response; + } + if let Some(response) = + super::tenant_marks::answer_marks_plan(&self.state, &plan, deadline).await { - return super::tenant_marks::answer_tenant_marks( - &self.state, - *marks_tenant, - group_ids, - deadline, - ) - .await; + return response; } - - if let PhysicalPlan::ClusterEvent( - nodedb_physical::physical_plan::ClusterEventOp::PublishTopic { - database_id: topic_database_id, - topic_name, - payload, - }, - ) = &plan + if let Some(response) = super::surrogate_binds::answer_binds_plan(&self.state, &plan) { + return response; + } + if let Some(response) = super::surrogate_binds::answer_holders_plan(&self.state, &plan) { + return response; + } + if let Some(response) = + super::metadata_applied::answer_applied_plan(&self.state, &plan, deadline).await { - if *topic_database_id != database_id { - return ExecuteResponse::err(TypedClusterError::Internal { - code: PLAN_DECODE_FAILED, - message: "topic publish database does not match RPC database".into(), - }); - } - return match crate::event::topic::publish::publish_to_topic( - &self.state, - database_id, - req.tenant_id, - topic_name, - payload, - ) - .await - { - Ok(sequence) => match zerompk::to_msgpack_vec(&sequence) { - Ok(payload) => ExecuteResponse::ok(vec![payload], 0, 0), - Err(error) => ExecuteResponse::err(TypedClusterError::Internal { - code: PLAN_DECODE_FAILED, - message: format!("topic response encoding failed: {error}"), - }), - }, - Err(error) => ExecuteResponse::err(TypedClusterError::Internal { - code: PLAN_DECODE_FAILED, - message: error.to_string(), - }), - }; + return response; } - if let PhysicalPlan::ClusterEvent( - nodedb_physical::physical_plan::ClusterEventOp::ConsumeStream { - database_id: stream_database_id, - stream_name, - group_name, - partition, - limit, - committed_offsets, - }, - ) = &plan - { - if let Err(error) = reject_consume_database_mismatch(*stream_database_id, database_id) { - return ExecuteResponse::err(error); - } - let limit = match usize::try_from(*limit) { - Ok(limit) => limit, - Err(_) => { - return ExecuteResponse::err(TypedClusterError::Internal { - code: PLAN_DECODE_FAILED, - message: "CDC consume limit exceeds platform range".into(), - }); - } - }; - let params = crate::event::cdc::consume::ConsumeParams { - database_id: *stream_database_id, - tenant_id: req.tenant_id, - stream_name, - group_name, - partition: *partition, - limit, - }; - if let Err(error) = - crate::event::cdc::consume::validate_consume_identity(&self.state, ¶ms) - { - return ExecuteResponse::err(TypedClusterError::Internal { - code: PLAN_DECODE_FAILED, - message: error.to_string(), - }); - } - let committed_offsets = - match crate::event::cdc::consume::decode_remote_committed_offsets(committed_offsets) - { - Ok(offsets) => offsets, - Err(error) => { - return ExecuteResponse::err(TypedClusterError::Internal { - code: PLAN_DECODE_FAILED, - message: error.to_string(), - }); - } - }; - // Events go to an authenticated peer node, not a subscriber; the - // requesting node applies its caller's redaction at the delivery - // surface (SELECT / HTTP poll / SSE), using the same replicated - // catalog policies on both sides. - return match crate::event::cdc::consume::consume_local_with_offsets( - &self.state, - ¶ms, - Some(&committed_offsets), - ) { - Ok(result) => { - let events = result - .events - .iter() - .map(|event| event.as_ref().clone()) - .collect::>(); - match zerompk::to_msgpack_vec(&events) { - Ok(payload) => ExecuteResponse::ok(vec![payload], 0, 0), - Err(error) => ExecuteResponse::err(TypedClusterError::Internal { - code: PLAN_DECODE_FAILED, - message: format!("CDC response encoding failed: {error}"), - }), - } - } - Err(error) => ExecuteResponse::err(TypedClusterError::Internal { - code: PLAN_DECODE_FAILED, - message: error.to_string(), - }), - }; + if let Some(response) = super::stream_events::answer_stream_event_plan( + &self.state, + &plan, + database_id, + req.tenant_id, + ) { + return response; } // Replicable write: drive through Raft, not local cores. Fanning it - // across local cores only would commit here without proposing to the + // across local cores only commits here without proposing to the // Raft group — silent write loss. Propose through the same proposer // the local pgwire write path uses. Reads / non-replicable plans fall - // through to `execute_plan_all_local_cores` unchanged. + // through to `execute_received_plan` unchanged. // - // The vshard is not carried on the wire; re-derive it as a pure + // Only a vShard-scoped plan carries its vShard on the wire. For a + // replicable write, re-derive the vShard as a pure // function of the plan's primary collection, matching the gateway // router's `CollectionHomed` arm (`vshard_for_collection`). The plan // carries the database-qualified name, de-qualified into the @@ -300,10 +199,30 @@ impl LocalPlanExecutor { return ExecuteResponse::err(error); } - if let Some(proposer) = self.state.async_raft_proposer() { + { + let proposer = match self.state.async_raft_proposer() { + Ok(proposer) => proposer, + Err(e) => return ExecuteResponse::err(execution_error_to_typed(e)), + }; + // The entry carries resolved rows: a timeseries ingest resolves + // here, on the proposer, before the entry exists. + let resolved = match crate::control::write_resolve::resolve_for_log( + &self.state, + crate::control::write_resolve::WriteResolveContext { + tenant_id, + database_id, + }, + vshard_id, + &plan, + ) + .await + { + Ok(resolved) => resolved, + Err(e) => return ExecuteResponse::err(execution_error_to_typed(e)), + }; let replicable = match crate::control::wal_replication::ReplicableWrite::decide_for_replication( - &plan, + resolved.as_ref().unwrap_or(&plan), ) { Ok(replicable) => replicable, Err(e) => { @@ -335,21 +254,7 @@ impl LocalPlanExecutor { { // Replicated writes carry no read watermark → 0: it floors a // session's later reads, and this RPC seam has no session. - Ok((payload, write_version)) => { - // Replicas apply with `ChangeFeedOwner::Unowned`. This - // node proposed the write once, so it publishes the - // change event. - crate::control::server::dispatch_utils::publish_change_set_with_lsn( - &self.state, - tenant_id, - database_id, - crate::control::server::dispatch_utils::extract_write_change_set( - &plan, tenant_id, - ), - write_version, - ); - ExecuteResponse::ok(vec![payload], 0, 0) - } + Ok((payload, _write_version)) => ExecuteResponse::ok(vec![payload], 0, 0), // A replicated write's apply verdict is a Data-Plane // verdict: carry its code, never flatten to internal. Err(e) => ExecuteResponse::err(execution_error_to_typed(e)), @@ -359,15 +264,23 @@ impl LocalPlanExecutor { } } + if let Err(error) = + confirm_read_leg(&self.state, database_id, &plan, &req.read_groups, deadline).await + { + return ExecuteResponse::err(error); + } + // A vShard-scoped plan runs on its one owning core. Every other plan + // fans across all local cores. match tokio::time::timeout( deadline, - execute_plan_all_local_cores( + execute_received_plan( &self.state, tenant_id, database_id, plan, trace_id, req.txn_id, + req.vshard_id, ), ) .await @@ -407,7 +320,29 @@ impl LocalPlanExecutor { message: "ClusterEvent operations do not support streaming RPC".into(), }); } + // A capture request is answered from parked captures, never from a + // live read of the cores. + if super::backup_cut::is_capture_plan(&plan) { + return Some(TypedClusterError::Internal { + code: PLAN_DECODE_FAILED, + message: "a cut capture request does not support streaming RPC".into(), + }); + } + // The stream fans across every core, so it cannot serve a plan that + // must run on one owning core. + if req.vshard_id.is_some() || crate::control::gateway::router::is_task_vshard_scoped(&plan) + { + return Some(TypedClusterError::Internal { + code: PLAN_DECODE_FAILED, + message: "vShard-scoped plans do not support streaming RPC".into(), + }); + } + if let Err(error) = + confirm_read_leg(&self.state, database_id, &plan, &req.read_groups, deadline).await + { + return Some(error); + } let tenant_id = crate::types::TenantId::new(req.tenant_id); let trace_id = nodedb_types::TraceId(req.trace_id); @@ -461,53 +396,11 @@ impl LocalPlanExecutor { } } -fn reject_consume_database_mismatch( - stream_database_id: crate::types::DatabaseId, - envelope_database_id: crate::types::DatabaseId, -) -> Result<(), TypedClusterError> { - if stream_database_id == envelope_database_id { - Ok(()) - } else { - Err(TypedClusterError::Internal { - code: PLAN_DECODE_FAILED, - message: "CDC consume database does not match RPC database".into(), - }) - } -} - #[cfg(test)] mod tests { use super::*; use nodedb_physical::physical_plan::CrdtOp; - #[test] - fn consume_stream_rejects_database_mismatch() { - let op = nodedb_physical::physical_plan::ClusterEventOp::ConsumeStream { - database_id: crate::types::DatabaseId::new(7), - stream_name: "topic:orders".into(), - group_name: "analytics".into(), - partition: Some(0), - limit: 1, - committed_offsets: vec![(0, 0, 0)], - }; - let nodedb_physical::physical_plan::ClusterEventOp::ConsumeStream { database_id, .. } = op - else { - panic!("expected typed consume operation"); - }; - assert!(matches!( - reject_consume_database_mismatch(database_id, crate::types::DatabaseId::new(8)), - Err(TypedClusterError::Internal { .. }) - )); - } - - #[test] - fn consume_stream_rejects_duplicate_caller_offsets() { - assert!(matches!( - crate::event::cdc::consume::decode_remote_committed_offsets(&[(3, 7, 1), (3, 8, 1)]), - Err(crate::event::cdc::consume::ConsumeError::InvalidRemoteOffsets(_)) - )); - } - #[test] fn every_remote_execution_mode_rejects_unadmitted_crdt_apply() { let plan = PhysicalPlan::Crdt(CrdtOp::Apply { @@ -516,7 +409,7 @@ mod tests { delta: Vec::new(), peer_id: 1, mutation_id: 1, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), provenance: None, constraint_version_required: 0, expected_frontier_digest: None, diff --git a/nodedb/src/control/exec_receiver/metadata_applied.rs b/nodedb/src/control/exec_receiver/metadata_applied.rs new file mode 100644 index 000000000..79b0c1d1f --- /dev/null +++ b/nodedb/src/control/exec_receiver/metadata_applied.rs @@ -0,0 +1,50 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Answer a restoring node once this node applied the metadata log through +//! the index it names. + +use std::time::Duration; + +use nodedb_cluster::METADATA_GROUP_ID; +use nodedb_cluster::rpc_codec::{ExecuteResponse, TypedClusterError}; +use nodedb_physical::physical_plan::{ClusterEventOp, PhysicalPlan}; + +use crate::control::state::SharedState; + +use super::support::PLAN_DECODE_FAILED; + +/// The answer to `plan` when it is a metadata-applied request, else `None`: +/// one empty payload once this node applied the metadata log through the +/// requested index, or an error when `budget` runs out first. +pub(super) async fn answer_applied_plan( + state: &SharedState, + plan: &PhysicalPlan, + budget: Duration, +) -> Option { + let PhysicalPlan::ClusterEvent(ClusterEventOp::MetadataApplied { index }) = plan else { + return None; + }; + let watcher = state.applied_index_watcher(METADATA_GROUP_ID); + let reached = match crate::control::metadata_proposer::wait::wait_applied( + watcher, *index, budget, + ) + .await + { + Ok(outcome) => outcome.is_reached(), + Err(error) => { + tracing::warn!(index = *index, %error, "metadata-applied wait did not finish"); + false + } + }; + Some(if reached { + ExecuteResponse::ok(vec![Vec::new()], 0, 0) + } else { + ExecuteResponse::err(TypedClusterError::Internal { + code: PLAN_DECODE_FAILED, + message: format!( + "node {} did not apply the metadata log through index {index} in time", + state.node_id + ), + }) + }) +} diff --git a/nodedb/src/control/exec_receiver/mod.rs b/nodedb/src/control/exec_receiver/mod.rs index 477f8e6b8..926e49a30 100644 --- a/nodedb/src/control/exec_receiver/mod.rs +++ b/nodedb/src/control/exec_receiver/mod.rs @@ -4,9 +4,13 @@ mod backup_cut; pub mod executor; +mod metadata_applied; mod plan_decode; +mod read_leg; mod request_validation; +mod stream_events; mod support; +mod surrogate_binds; mod tenant_marks; pub use executor::LocalPlanExecutor; diff --git a/nodedb/src/control/exec_receiver/plan_decode.rs b/nodedb/src/control/exec_receiver/plan_decode.rs index 83f22d209..1c510d7e2 100644 --- a/nodedb/src/control/exec_receiver/plan_decode.rs +++ b/nodedb/src/control/exec_receiver/plan_decode.rs @@ -14,8 +14,8 @@ use nodedb_physical::physical_plan::wire as plan_wire; use super::support::{PLAN_DECODE_FAILED, plan_contains_exchange}; /// Decodes `plan_bytes` into a [`PhysicalPlan`], re-resolves a -/// `DocumentOp::PointGet` surrogate when the coordinator shipped -/// `Surrogate::ZERO`, and rejects a plan that still contains an +/// `DocumentOp::PointGet` surrogate when the coordinator shipped none, and +/// rejects a plan that still contains an /// unresolved Exchange node. pub(super) fn decode_plan( state: &SharedState, @@ -39,15 +39,15 @@ pub(super) fn decode_plan( // The query coordinator resolves `WHERE pk = ` → surrogate against // ITS OWN local catalog. The surrogate↔PK map is sharded to the // collection's data-group members, so a coordinator that is NOT a - // member of that group misses the binding and ships `Surrogate::ZERO`. - // We (the owner) ARE a group member, so our local catalog HAS the - // binding — re-resolve here before the plan reaches the Data Plane. + // member of that group misses the binding and ships `None`. We (the + // owner) ARE a group member, so our local catalog HAS the binding — + // re-resolve here before the plan reaches the Data Plane. // // Scope is intentionally tight: only `DocumentOp::PointGet` reads, only - // when the carried surrogate is ZERO and `pk_bytes` is non-empty. A - // non-ZERO carried surrogate is authoritative (immutable first-wins - // bind) and is left untouched; a genuinely-absent PK stays ZERO and - // correctly resolves to not-found. + // when the carried surrogate is `None` and `pk_bytes` is non-empty. A + // carried surrogate is authoritative (immutable first-wins bind) and is + // left untouched; a genuinely-absent PK stays `None` and correctly + // resolves to not-found. let catalog_ref = state.credentials.catalog(); if let nodedb_physical::physical_plan::PhysicalPlan::Document( nodedb_physical::physical_plan::DocumentOp::PointGet { @@ -57,13 +57,13 @@ pub(super) fn decode_plan( .. }, ) = &mut plan - && *surrogate == nodedb_types::Surrogate::ZERO + && surrogate.is_none() && !pk_bytes.is_empty() && let Ok(key) = nodedb_types::CollectionKey::from_qualified(database_id, collection) && let Ok(Some(resolved)) = catalog_ref.get_surrogate_for_pk(key, crate::types::TenantId::new(tenant_id), pk_bytes) { - *surrogate = resolved; + *surrogate = Some(resolved); } // ── 3b. Reject unresolved Exchange nodes ────────────────────────────── diff --git a/nodedb/src/control/exec_receiver/read_leg.rs b/nodedb/src/control/exec_receiver/read_leg.rs new file mode 100644 index 000000000..de6ed5a73 --- /dev/null +++ b/nodedb/src/control/exec_receiver/read_leg.rs @@ -0,0 +1,35 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The serving half of a linearizable read leg sent by another node. + +use std::time::Duration; + +use nodedb_cluster::rpc_codec::TypedClusterError; + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::gateway::read_leg::confirm_local_read; +use crate::control::server::shared::write_admission::plan_is_write; +use crate::control::state::SharedState; +use crate::types::DatabaseId; + +use super::support::execution_error_to_typed; + +/// Confirm a linearizable read leg on this node before it reads. +/// +/// `read_groups` comes from the coordinator. It is empty unless the leg is a +/// linearizable read, and a write ignores it: Raft orders writes. +pub(super) async fn confirm_read_leg( + state: &SharedState, + database_id: DatabaseId, + plan: &PhysicalPlan, + read_groups: &[u64], + budget: Duration, +) -> Result<(), TypedClusterError> { + if read_groups.is_empty() || plan_is_write(plan) { + return Ok(()); + } + let budget_ms = u64::try_from(budget.as_millis()).unwrap_or(u64::MAX); + confirm_local_read(state, database_id, plan, read_groups, budget_ms) + .await + .map_err(execution_error_to_typed) +} diff --git a/nodedb/src/control/exec_receiver/stream_events.rs b/nodedb/src/control/exec_receiver/stream_events.rs new file mode 100644 index 000000000..fea2f8028 --- /dev/null +++ b/nodedb/src/control/exec_receiver/stream_events.rs @@ -0,0 +1,180 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Receiver side of the event-stream `ClusterEvent` RPC: CDC consume. +//! +//! The consume runs no `confirm_read_leg`, and needs none. +//! +//! - A CDC consume reads this node's change-stream buffer. Every replica +//! routes each committed entry's change events into its own buffer at the +//! entry's Raft log position, so a lagging replica holds a prefix of the +//! same sequence. It returns fewer events, never different ones. +//! - The consumer cursor belongs to the caller. It arrives in +//! `committed_offsets` and is never read from or written to this node's +//! `OffsetStore`. Each read starts after the caller's committed position, so +//! a stale serving node cannot make a consumer re-read an event. + +use nodedb_cluster::rpc_codec::{ExecuteResponse, TypedClusterError}; +use nodedb_physical::physical_plan::ClusterEventOp; + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::state::SharedState; +use crate::types::DatabaseId; + +use super::support::PLAN_DECODE_FAILED; + +/// The decoded fields of a `ClusterEventOp::ConsumeStream` plan. +struct ConsumeStreamRequest<'a> { + stream_database_id: DatabaseId, + stream_name: &'a str, + group_name: &'a str, + partition: Option, + limit: u64, + committed_offsets: &'a [(u32, u64, u64, u64)], +} + +/// Answer `plan` when it is a CDC consume. +/// +/// Returns `None` for every other plan. +pub(super) fn answer_stream_event_plan( + state: &SharedState, + plan: &PhysicalPlan, + database_id: DatabaseId, + tenant_id: u64, +) -> Option { + match plan { + PhysicalPlan::ClusterEvent(ClusterEventOp::ConsumeStream { + database_id: stream_database_id, + stream_name, + group_name, + partition, + limit, + committed_offsets, + }) => { + let request = ConsumeStreamRequest { + stream_database_id: *stream_database_id, + stream_name, + group_name, + partition: *partition, + limit: *limit, + committed_offsets, + }; + Some(answer_consume_stream( + state, + database_id, + tenant_id, + request, + )) + } + _ => None, + } +} + +/// Read this node's CDC buffer from the caller's committed cursor. +fn answer_consume_stream( + state: &SharedState, + database_id: DatabaseId, + tenant_id: u64, + request: ConsumeStreamRequest<'_>, +) -> ExecuteResponse { + if let Err(error) = reject_consume_database_mismatch(request.stream_database_id, database_id) { + return ExecuteResponse::err(error); + } + let limit = match usize::try_from(request.limit) { + Ok(limit) => limit, + Err(_) => { + return ExecuteResponse::err(TypedClusterError::Internal { + code: PLAN_DECODE_FAILED, + message: "CDC consume limit exceeds platform range".into(), + }); + } + }; + let params = crate::event::cdc::consume::ConsumeParams { + database_id: request.stream_database_id, + tenant_id, + stream_name: request.stream_name, + group_name: request.group_name, + partition: request.partition, + limit, + }; + if let Err(error) = crate::event::cdc::consume::validate_consume_identity(state, ¶ms) { + return ExecuteResponse::err(TypedClusterError::Internal { + code: PLAN_DECODE_FAILED, + message: error.to_string(), + }); + } + let committed_offsets = match crate::event::cdc::consume::decode_remote_committed_offsets( + request.committed_offsets, + ) { + Ok(offsets) => offsets, + Err(error) => { + return ExecuteResponse::err(TypedClusterError::Internal { + code: PLAN_DECODE_FAILED, + message: error.to_string(), + }); + } + }; + // Events go to an authenticated peer node, not a subscriber. The + // requesting node applies its caller's redaction at the delivery surface + // (SELECT / HTTP poll / SSE), using the same replicated catalog policies + // on both sides. + // A cursor below the events this node holds answers a typed reply, so + // the caller surfaces `OffsetOutOfRange` rather than an opaque error. + let reply = crate::event::cdc::consume::RemoteConsumeReply::from_result( + crate::event::cdc::consume::consume_local_with_offsets( + state, + ¶ms, + Some(&committed_offsets), + ), + ); + match reply { + Ok(reply) => match zerompk::to_msgpack_vec(&reply) { + Ok(payload) => ExecuteResponse::ok(vec![payload], 0, 0), + Err(error) => ExecuteResponse::err(TypedClusterError::Internal { + code: PLAN_DECODE_FAILED, + message: format!("CDC response encoding failed: {error}"), + }), + }, + Err(error) => ExecuteResponse::err(TypedClusterError::Internal { + code: PLAN_DECODE_FAILED, + message: error.to_string(), + }), + } +} + +fn reject_consume_database_mismatch( + stream_database_id: DatabaseId, + envelope_database_id: DatabaseId, +) -> Result<(), TypedClusterError> { + if stream_database_id == envelope_database_id { + Ok(()) + } else { + Err(TypedClusterError::Internal { + code: PLAN_DECODE_FAILED, + message: "CDC consume database does not match RPC database".into(), + }) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn consume_stream_rejects_database_mismatch() { + assert!(matches!( + reject_consume_database_mismatch(DatabaseId::new(7), DatabaseId::new(8)), + Err(TypedClusterError::Internal { .. }) + )); + } + + #[test] + fn consume_stream_rejects_duplicate_caller_offsets() { + assert!(matches!( + crate::event::cdc::consume::decode_remote_committed_offsets(&[ + (3, 0, 7, 1), + (3, 0, 8, 1) + ]), + Err(crate::event::cdc::consume::ConsumeError::InvalidRemoteOffsets(_)) + )); + } +} diff --git a/nodedb/src/control/exec_receiver/surrogate_binds.rs b/nodedb/src/control/exec_receiver/surrogate_binds.rs new file mode 100644 index 000000000..f65562c54 --- /dev/null +++ b/nodedb/src/control/exec_receiver/surrogate_binds.rs @@ -0,0 +1,63 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Answer a backup or MOVE TENANT coordinator's request for this node's +//! PK→surrogate binds with a home among the vShards this node is the source +//! for. + +use std::collections::HashSet; + +use nodedb_cluster::rpc_codec::ExecuteResponse; +use nodedb_physical::physical_plan::{ClusterEventOp, PhysicalPlan}; + +use crate::control::backup::bind_capture::{encode_binds, local_binds}; +use crate::control::backup::restore::bind_conflicts::{encode_holders, local_holders}; +use crate::control::state::SharedState; + +use super::support::execution_error_to_typed; + +/// The answer to `plan` when it is a surrogate bind request, else `None`: +/// this node's binds of the tenant's collections with a home among the +/// requested vShards, encoded for the wire. +pub(super) fn answer_binds_plan( + state: &SharedState, + plan: &PhysicalPlan, +) -> Option { + let PhysicalPlan::ClusterEvent(ClusterEventOp::SurrogateBinds { + tenant_id, + database_id, + vshards, + collections, + }) = plan + else { + return None; + }; + let vshards: HashSet = vshards.iter().copied().collect(); + let encoded = + local_binds(state, *tenant_id, *database_id, &vshards, collections).and_then(encode_binds); + Some(match encoded { + Ok(payload) => ExecuteResponse::ok(vec![payload], 0, 0), + Err(error) => ExecuteResponse::err(execution_error_to_typed(error)), + }) +} + +/// The answer to `plan` when it is a surrogate holders request, else `None`: +/// the key this node binds each requested `(collection, surrogate)` to, +/// encoded for the wire. +pub(super) fn answer_holders_plan( + state: &SharedState, + plan: &PhysicalPlan, +) -> Option { + let PhysicalPlan::ClusterEvent(ClusterEventOp::SurrogateHolders { + tenant_id, + database_id, + entries, + }) = plan + else { + return None; + }; + let encoded = local_holders(state, *tenant_id, *database_id, entries).and_then(encode_holders); + Some(match encoded { + Ok(payload) => ExecuteResponse::ok(vec![payload], 0, 0), + Err(error) => ExecuteResponse::err(execution_error_to_typed(error)), + }) +} diff --git a/nodedb/src/control/exec_receiver/tenant_marks.rs b/nodedb/src/control/exec_receiver/tenant_marks.rs index 49a769d63..82855c4fa 100644 --- a/nodedb/src/control/exec_receiver/tenant_marks.rs +++ b/nodedb/src/control/exec_receiver/tenant_marks.rs @@ -10,6 +10,7 @@ use std::sync::Arc; use std::time::Duration; use nodedb_cluster::rpc_codec::{ExecuteResponse, TypedClusterError}; +use nodedb_physical::physical_plan::{ClusterEventOp, PhysicalPlan}; use crate::control::backup::restore::guard::{encode_marks, local_tenant_marks}; use crate::control::security::auth_fence::cluster::hosts_group; @@ -17,6 +18,22 @@ use crate::control::state::SharedState; use super::support::execution_error_to_typed; +/// The answer to `plan` when it is a tenant write-mark request, else `None`. +pub(super) async fn answer_marks_plan( + state: &Arc, + plan: &PhysicalPlan, + budget: Duration, +) -> Option { + let PhysicalPlan::ClusterEvent(ClusterEventOp::TenantWriteMarks { + tenant_id, + group_ids, + }) = plan + else { + return None; + }; + Some(answer_tenant_marks(state, *tenant_id, group_ids, budget).await) +} + /// This node's marks of `tenant_id` in `group_ids`, encoded for the wire. /// /// A group this node does not replicate is refused with @@ -35,6 +52,8 @@ pub(super) async fn answer_tenant_marks( group_id, leader_node_id: replica_hint(state, group_id), leader_addr: None, + // The hint names a replica, not a leader this node observed, so + // it carries no term and never moves a termed routing hint. term: 0, }); } diff --git a/nodedb/src/control/fail_gate.rs b/nodedb/src/control/fail_gate.rs index eb7bb6dec..d8ed2319c 100644 --- a/nodedb/src/control/fail_gate.rs +++ b/nodedb/src/control/fail_gate.rs @@ -22,10 +22,18 @@ const GATE_POLL: Duration = Duration::from_millis(20); /// Park until the file armed for `name` with `wait_file()` exists. No-op /// when nothing is armed for `name`. +/// +/// On arrival the gate creates `.parked`, so a test can wait until the +/// task has reached the gate before it acts. pub(crate) async fn wait(name: &str) { let Some(FailAction::WaitForFile(path)) = lookup(name) else { return; }; + let mut parked = path.clone().into_os_string(); + parked.push(".parked"); + if let Err(e) = std::fs::write(&parked, name) { + tracing::warn!(gate = name, error = %e, "fail gate parked marker not created"); + } while !path.exists() { tokio::time::sleep(GATE_POLL).await; } @@ -35,7 +43,9 @@ pub(crate) async fn wait(name: &str) { /// before any core holds the request. Named per collection, /// `funnel::before_dispatch::`, and only a write carrying a WAL /// LSN reaches it. A committed redo parks on the gate of each collection it -/// writes. +/// writes. A replicated write parks inside its apply entry's enqueue, so +/// every later entry of its Raft group waits for it. Only a write of another +/// group applies while it is parked. /// /// `funnel::before_dispatch::node::` parks the write only on /// node `N`. An in-process cluster test shares one fail-point registry across @@ -53,6 +63,16 @@ pub(crate) async fn before_dispatch(node_id: u64, plan: &PhysicalPlan, wal_lsn: } } +/// The Calvin submit gate: after a transaction's collection incarnations are +/// stamped and before the sequencer admits it. Named per collection, +/// `calvin::after_stamp::`. A test parks a transaction here, then +/// changes a collection it names before the transaction reaches a replica. +pub(crate) async fn after_calvin_stamp(tx_class: &nodedb_cluster::calvin::types::TxClass) { + for named in &tx_class.incarnations { + wait(&format!("calvin::after_stamp::{}", named.collection)).await; + } +} + /// The Calvin scheduler's flush gate: after a committed transaction's redo /// record is appended and before its flush reaches a core. Named per /// collection, `calvin::before_flush::`. @@ -69,3 +89,69 @@ pub(crate) fn holds_flush(plan: &PhysicalPlan) -> bool { ) }) } + +/// The committed-message delivery gates of node `node_id`: +/// `publish::before_delivery::node` before a delivery pass sends anything, +/// and `publish::before_cursor_commit::node` after a partition's messages +/// are sent and before its cursor is committed. `point` names the gate. +/// +/// Returns `false` when the node shuts down while parked: the pass then stops +/// where it is, as a node killed at that point does. +pub(crate) async fn publish_delivery( + node_id: u64, + point: &str, + shutdown: tokio::sync::watch::Receiver, +) -> bool { + delivery_gate(&format!("publish::{point}::node{node_id}"), shutdown).await +} + +/// The trigger action lane's firing gates of node `node_id`: +/// `trigger::before_firing::node` before a firing pass fires anything, +/// and `trigger::before_cursor_commit::node` after a partition's actions +/// fired and before its cursor is committed. `point` names the gate. +/// +/// Returns `false` when the node shuts down while parked. +pub(crate) async fn action_firing( + node_id: u64, + point: &str, + shutdown: tokio::sync::watch::Receiver, +) -> bool { + delivery_gate(&format!("trigger::{point}::node{node_id}"), shutdown).await +} + +/// Park on gate `name` until it is released or the node shuts down. +async fn delivery_gate(name: &str, mut shutdown: tokio::sync::watch::Receiver) -> bool { + if *shutdown.borrow() { + return false; + } + tokio::select! { + () = wait(name) => true, + _ = shutdown.changed() => false, + } +} + +/// The metadata apply gate of node `node_id`, named by +/// [`crate::control::cluster::metadata_applier::metadata_apply_hold_point`]. +/// +/// The metadata applier runs on the Raft loop and cannot park, so the gate +/// answers without waiting: `true` while the file armed for it is absent. +/// On its first hold the gate creates `.parked`, so a test can wait +/// until the node's apply is held. +pub(crate) fn holds_metadata_apply(node_id: u64) -> bool { + let name = crate::control::cluster::metadata_applier::metadata_apply_hold_point(node_id); + let Some(FailAction::WaitForFile(path)) = lookup(&name) else { + return false; + }; + if path.exists() { + return false; + } + let mut parked = path.into_os_string(); + parked.push(".parked"); + let parked = std::path::PathBuf::from(parked); + if !parked.exists() + && let Err(e) = std::fs::write(&parked, &name) + { + tracing::warn!(gate = %name, error = %e, "fail gate parked marker not created"); + } + true +} diff --git a/nodedb/src/control/gateway/cache_miss.rs b/nodedb/src/control/gateway/cache_miss.rs index 942356bb0..2d8cc7cf7 100644 --- a/nodedb/src/control/gateway/cache_miss.rs +++ b/nodedb/src/control/gateway/cache_miss.rs @@ -4,7 +4,7 @@ //! //! When the planner returns `Error::RetryableSchemaChanged { descriptor }`, //! the gateway: -//! 1. Fetches a fresh descriptor lease via the Phase B.3 lease machinery. +//! 1. Fetches a fresh descriptor lease via the descriptor lease machinery. //! 2. Calls the supplied `plan_fn` once more to re-plan against fresh state. //! 3. Proceeds to dispatch with the new plan. //! @@ -22,7 +22,7 @@ use crate::control::state::SharedState; /// /// `plan_fn` — closure that produces a `PhysicalPlan` or an error. Called /// at most twice. On the second call the lease for the affected descriptor -/// has been refreshed so the catalog adapter should return a fresh version. +/// has been refreshed so the catalog adapter returns a fresh version. /// /// `database_id` and `tenant_id` — used when acquiring the descriptor lease. pub async fn plan_with_cache_miss_retry( @@ -52,20 +52,12 @@ where /// Acquire (or renew) the lease for a descriptor, forcing the catalog adapter /// to re-read from the replicated metadata store. -/// -/// In single-node mode (no metadata raft handle) this is a no-op — the -/// catalog is always fresh. async fn refresh_descriptor_lease( shared: &SharedState, database_id: nodedb_types::DatabaseId, tenant_id: u64, descriptor: &str, ) -> Result<(), Error> { - if shared.metadata_raft.get().is_none() { - // Single-node: no lease infrastructure, catalog always fresh. - return Ok(()); - } - let descriptor_id = nodedb_cluster::DescriptorId::new( database_id.as_u64(), tenant_id, @@ -89,12 +81,7 @@ async fn refresh_descriptor_lease( }); }; - // `acquire_lease` is synchronous (parks on a Condvar internally) and - // must be wrapped in `block_in_place` so the Tokio reactor is not - // starved while the raft propose + apply happens. - tokio::task::block_in_place(|| { - acquire_lease(shared, descriptor_id, version, DEFAULT_LEASE_DURATION) - })?; + acquire_lease(shared, descriptor_id, version, DEFAULT_LEASE_DURATION).await?; Ok(()) } @@ -135,10 +122,7 @@ mod tests { fn ok_path_calls_plan_fn_once() { let call_count = std::cell::Cell::new(0usize); let rt = tokio::runtime::Runtime::new().unwrap(); - // We can't build a real SharedState here — test the logic path - // without a raft handle (single-node branch). - // - // Use a mock approach: test the retry branches directly. + // Test the retry branches directly, with no SharedState. let mut attempts = 0usize; let result: Result = rt.block_on(async { // Simulate plan_with_cache_miss_retry with an always-ok plan_fn. diff --git a/nodedb/src/control/gateway/cluster_error.rs b/nodedb/src/control/gateway/cluster_error.rs new file mode 100644 index 000000000..66c2c7210 --- /dev/null +++ b/nodedb/src/control/gateway/cluster_error.rs @@ -0,0 +1,129 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Map a remote node's [`TypedClusterError`] to the gateway's [`Error`]. + +use nodedb_cluster::rpc_codec::TypedClusterError; + +use crate::Error; +use crate::types::VShardId; + +/// Map a [`TypedClusterError`] to an internal [`Error`]. +/// +/// `NotLeader` is mapped such that the gateway retry loop can extract the +/// hinted leader from `Error::NotLeader.leader_node` and update the routing +/// table before the next attempt. +pub(super) fn map_typed_cluster_error(err: TypedClusterError, vshard_id: u64) -> Error { + match err { + TypedClusterError::NotLeader { + leader_node_id, + leader_addr, + term, + .. + } => Error::NotLeader { + vshard_id: VShardId::new((vshard_id % VShardId::COUNT as u64) as u32), + leader_node: leader_node_id.unwrap_or(0), + leader_addr: leader_addr.unwrap_or_default(), + leader_term: term, + }, + TypedClusterError::DescriptorMismatch { + collection, + expected_version, + actual_version, + } => { + // A repeating mismatch means the planner and leaseholder disagree + // persistently — a bug, not the transient race the retry assumes. + tracing::debug!( + %collection, + expected_version, + actual_version, + "gateway: descriptor version mismatch at leaseholder" + ); + Error::RetryableSchemaChanged { + descriptor: collection, + } + } + TypedClusterError::DeadlineExceeded { .. } => Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(0), + }, + // Remote Data-Plane verdict: keep the code so the client sees the + // SQLSTATE local execution renders, not a generic internal error. + TypedClusterError::DataPlane { code } => Error::DataPlane(code.into()), + // Remote constraint refusal: keep the kind so the client sees 23502 + // vs 23505, exactly as a local refusal on this node renders. + TypedClusterError::RejectedConstraint { + collection, + constraint, + detail, + } => Error::RejectedConstraint { + collection, + constraint, + detail, + }, + // A numeric class crosses as `Error::RemoteTyped`, so the client sees + // the SQLSTATE the executing node gave it. Only a code of 0 (no class) + // decodes as `Error::Internal`. + internal @ TypedClusterError::Internal { .. } => Error::from(internal), + // A Calvin abort keeps the error a local submit returns. + aborted @ TypedClusterError::CalvinAborted { .. } => Error::from(aborted), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn map_not_leader() { + let err = TypedClusterError::NotLeader { + group_id: 0, + leader_node_id: Some(5), + leader_addr: Some("10.0.0.5:9400".into()), + term: 3, + }; + match map_typed_cluster_error(err, 7) { + Error::NotLeader { + leader_node, + leader_term, + .. + } => assert_eq!((leader_node, leader_term), (5, 3)), + other => panic!("expected NotLeader, got {other:?}"), + } + } + + #[test] + fn map_descriptor_mismatch() { + let err = TypedClusterError::DescriptorMismatch { + collection: "orders".into(), + expected_version: 1, + actual_version: 2, + }; + match map_typed_cluster_error(err, 0) { + Error::RetryableSchemaChanged { descriptor } => assert_eq!(descriptor, "orders"), + other => panic!("expected RetryableSchemaChanged, got {other:?}"), + } + } + + /// A remote error with a numeric class keeps it, never `Internal`. + #[test] + fn map_internal_keeps_its_numeric_class() { + let err = TypedClusterError::Internal { + code: u32::from(nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED.0), + message: "permission denied on orders".into(), + }; + match map_typed_cluster_error(err, 0) { + Error::RemoteTyped { code, .. } => { + assert_eq!(code, nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED); + } + other => panic!("expected RemoteTyped, got {other:?}"), + } + } + + #[test] + fn map_deadline_exceeded() { + let err = TypedClusterError::DeadlineExceeded { elapsed_ms: 100 }; + assert!(matches!( + map_typed_cluster_error(err, 0), + Error::DeadlineExceeded { .. } + )); + } +} diff --git a/nodedb/src/control/gateway/core.rs b/nodedb/src/control/gateway/core.rs index 506257fcf..1a4913e2a 100644 --- a/nodedb/src/control/gateway/core.rs +++ b/nodedb/src/control/gateway/core.rs @@ -34,11 +34,12 @@ use nodedb_physical::physical_plan::PhysicalPlan; use super::dispatcher::{DispatchRouteParams, dispatch_route, statement_deadline_ms}; use super::fuser::fuse_payloads; use super::key_extractor::UnwiredKeyExtractor; +use super::live_leaders::resolve_live_decision; use super::outcome::GatewayOutcome; use super::plan_cache::PlanCache; use super::retry::retry_not_leader; use super::route::TaskRoute; -use super::router::{resolve_decision, route_plan}; +use super::router::route_plan; use super::version_set::GatewayVersionSet; /// Context passed to [`Gateway::execute`]. @@ -55,13 +56,21 @@ pub struct QueryContext { /// forwarding on remote `ExecuteRequest`. `None` for autocommit and /// non-interactive callers. pub txn_id: Option, + /// Every read this query dispatches must observe all writes committed + /// before it began. The node that serves each read leg confirms the leg's + /// group first (see `control::cluster::linearizable_read`). Writes ignore + /// it: Raft orders them. + pub linearizable: bool, } +/// The authorized plan and its write lease. The caller holds the lease until +/// the plan's outcome returns. pub(super) fn authorized_plan_for_context( ctx: &QueryContext, checked: CloneCheckedTask, -) -> Result { - let task = checked.into_authorized().into_physical_task(); +) -> Result<(PhysicalPlan, crate::control::lease::QueryLeaseScope), Error> { + let (authorized, lease) = checked.into_parts(); + let task = authorized.into_physical_task(); if task.tenant_id != ctx.tenant_id || task.database_id != ctx.database_id || task.txn_id != ctx.txn_id @@ -70,7 +79,7 @@ pub(super) fn authorized_plan_for_context( detail: "authorized task scope does not match gateway query context".into(), }); } - Ok(task.plan) + Ok((task.plan, lease)) } /// The gateway: routes, dispatches, retries, and caches physical plans. @@ -78,7 +87,7 @@ pub struct Gateway { /// `Weak` back-reference to the owning [`SharedState`]. /// /// `SharedState` owns this `Gateway` via its strong `Option>` - /// field, so a strong `Arc` here would form a reference cycle + /// field, so a strong `Arc` here forms a reference cycle /// that keeps `SharedState` alive forever (its clone count never reaches /// zero on shutdown). Holding it `Weak` breaks the cycle: while the node /// runs some other owner always keeps `SharedState` alive, so @@ -176,13 +185,13 @@ impl Gateway { /// committed LSN (local SPSC response watermark or the remote's /// `ExecuteResponse.watermark_lsn`). The cross-node gather consumer folds /// these into the transaction read-set so a remote-homed read records the - /// remote's actual LSN instead of the former hardcoded `Lsn::ZERO`. + /// remote's actual LSN. pub async fn execute_with_watermarks( &self, ctx: &QueryContext, checked: CloneCheckedTask, ) -> Result<(Vec>, Vec<(VShardId, Lsn)>, Lsn), Error> { - let plan = authorized_plan_for_context(ctx, checked)?; + let (plan, _lease) = authorized_plan_for_context(ctx, checked)?; self.execute_plan_outcome(ctx, plan) .await .map(GatewayOutcome::into_parts) @@ -278,35 +287,10 @@ impl Gateway { let database_id = ctx.database_id; let trace_id = ctx.trace_id; let txn_id = ctx.txn_id; + let linearizable = ctx.linearizable; let version_set = version_set_for_route.clone(); async move { - let decision = { - let routing_guard = shared - .cluster_routing - .as_ref() - .map(|rw| rw.read().unwrap_or_else(|p| p.into_inner())); - let raft_snapshot: Vec = - shared.raft_status_fn.get().map(|f| f()).unwrap_or_default(); - let live_leader = move |group_id: u64| -> u64 { - raft_snapshot - .iter() - .find(|gs| gs.group_id == group_id) - .map(|gs| gs.leader_id) - .unwrap_or(0) - }; - let live_lookup: Option<&dyn Fn(u64) -> u64> = - if shared.raft_status_fn.get().is_some() { - Some(&live_leader) - } else { - None - }; - resolve_decision( - vshard_id_u32, - shared.node_id, - routing_guard.as_deref(), - live_lookup, - ) - }; + let decision = resolve_live_decision(&shared, vshard_id_u32); let route = TaskRoute { plan, decision, @@ -321,6 +305,7 @@ impl Gateway { deadline_ms, version_set: &version_set, txn_id, + linearizable, }) .await } @@ -388,11 +373,26 @@ impl Gateway { plan: PhysicalPlan, ctx: &QueryContext, ) -> Result, Error> { + // A `ClusterArray` plan runs through this node's array coordinator: it + // has no wire encoding and no Data-Plane handler. Array DDL proposes a + // replicated catalog entry: a route opens the array on one node only. + let local_only = if matches!(plan, PhysicalPlan::ClusterArray(_)) { + Some("a ClusterArray plan runs through the array coordinator") + } else if crate::control::array_catalog::ddl::is_array_ddl(&plan) { + Some("array DDL runs through the replicated array catalog") + } else { + None + }; + if let Some(reason) = local_only { + return Err(Error::Internal { + detail: format!("gateway: {reason}, never a gateway route"), + }); + } // Fail-closed safety floor: refuse a cross-collection write whose source // and target are not co-resident on one Data-Plane core. This runs in // BOTH single-node and cluster mode — the single-node early-return in // `route_plan` bypasses `route_single_collection`, which is exactly the - // multi-core scenario that triggers the silent-wrong-result bug. + // multi-core scenario that returns silently wrong results. let shared = self.shared()?; super::colocation_guard::guard_cross_collection_write(&shared, ctx.database_id, &plan)?; @@ -426,7 +426,7 @@ impl Gateway { /// /// `tenant_id` must match the authenticated tenant of the query so that /// the catalog key lookup (`"{tenant_id}:{collection_name}"`) finds the - /// correct descriptor version. Using tenant 0 here would return version 0 + /// correct descriptor version. Using tenant 0 here returns version 0 /// for every collection stored under any other tenant, causing spurious /// `DescriptorMismatch` rejections at the leader. /// @@ -575,7 +575,7 @@ mod tests { // Ensure the counter trick works: simulate "plan_fn called N times". let plan_fn_calls = Arc::new(AtomicUsize::new(0)); - let _ = plan_fn_calls; // just a placeholder — real test is in integration tests + let _ = plan_fn_calls; // placeholder — real test is in integration tests } /// Simulate the full two-phase execute_sql flow using only PlanCache APIs. @@ -595,7 +595,7 @@ mod tests { // Helper: simulates what execute_sql does on every call. // - // `version_of_widgets` is the version the catalog would return. + // `version_of_widgets` is the version the catalog returns. // `expect_hit` controls whether we assert a hit or miss. let simulate_call = |cache: &PlanCache, plan_fn_calls: &Arc, diff --git a/nodedb/src/control/gateway/dispatch_local.rs b/nodedb/src/control/gateway/dispatch_local.rs new file mode 100644 index 000000000..53c4b43e3 --- /dev/null +++ b/nodedb/src/control/gateway/dispatch_local.rs @@ -0,0 +1,209 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Local route dispatch: a plan this node applies on its own cores, over +//! the SPSC bridge or through a Raft proposal. + +use std::sync::Arc; + +use crate::Error; +use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; +use crate::control::local_dispatch::reject_data_plane_error; +use crate::control::server::dispatch_utils::{ + AutocommitWrite, dispatch_autocommit_write, dispatch_routed_read_to_data_plane, +}; +use crate::control::server::shared::write_admission::plan_is_write; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, Lsn, TenantId, TraceId, TxnId, VShardId}; + +use super::dispatcher::DispatchOutcome; +use super::route::TaskRoute; +use super::router::is_task_vshard_scoped; +use super::version_check::check_local_descriptor_versions; +use super::version_set::GatewayVersionSet; + +/// Session and trace context of one local dispatch. +pub(super) struct LocalContext<'a> { + pub(super) shared: &'a Arc, + pub(super) tenant_id: TenantId, + pub(super) database_id: DatabaseId, + pub(super) trace_id: TraceId, + pub(super) txn_id: Option, + pub(super) version_set: &'a GatewayVersionSet, +} + +/// Local dispatch via SPSC bridge. +/// +/// Carries `txn_id` so the Data Plane can resolve this session transaction's +/// staging overlay (read-your-own-writes) for in-block SQL and direct ops. +pub(super) async fn dispatch_local( + route: TaskRoute, + ctx: LocalContext<'_>, +) -> Result { + let LocalContext { + shared, + tenant_id, + database_id, + trace_id, + txn_id, + version_set, + } = ctx; + // Staying on this node does not make the plan fresh: a DDL can bump a + // descriptor between planning and dispatch, and a drain that times out is + // force-ended. Fence the local plan against this node's catalog exactly as + // the leaseholder fences a forwarded one. + check_local_descriptor_versions(shared, tenant_id, database_id, version_set)?; + + let vshard_id = VShardId::new(route.vshard_id); + + if txn_id.is_some() + && matches!( + &route.plan, + PhysicalPlan::Crdt( + nodedb_physical::physical_plan::CrdtOp::Apply { .. } + | nodedb_physical::physical_plan::CrdtOp::ApplyAuthenticated { .. } + ) + ) + { + return Err(Error::CrdtApplyForbiddenInTransaction); + } + + // A proposal carries resolved rows: the proposer resolves a timeseries + // ingest before its entry exists. + let resolved = if txn_id.is_none() { + crate::control::write_resolve::resolve_for_log( + shared, + crate::control::write_resolve::WriteResolveContext { + tenant_id, + database_id, + }, + vshard_id, + &route.plan, + ) + .await? + } else { + None + }; + let plan = resolved.unwrap_or(route.plan); + + if txn_id.is_none() + && let Some(entry) = crate::control::wal_replication::to_replicated_entry( + tenant_id, + database_id, + vshard_id, + &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&plan)?, + )? + { + let proposer = shared.async_raft_proposer()?; + let (payload, write_version) = + crate::control::wal_replication::propose_replicated_entry(shared, proposer, entry) + .await?; + return Ok(DispatchOutcome { + payloads: vec![payload], + // A write carries no read watermark (Lsn::ZERO); its post-write + // `coll_write_lsn` is surfaced via `read_version_lsn` instead. + shard_watermarks: vec![(vshard_id, Lsn::ZERO)], + read_version_lsn: write_version, + not_found: false, + }); + } + + let resp = dispatch_local_plan(LocalPlan { + shared, + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + txn_id, + }) + .await?; + // The remote sibling turns `ExecuteResponse.error` into `Err`; the local + // route must reject its own error status the same way. Keeping only the + // payload hands a post-scan operator's failed expression back as an + // empty success — an error status is not an empty result set. + reject_data_plane_error(&resp)?; + Ok(DispatchOutcome { + payloads: vec![resp.payload.to_vec()], + shard_watermarks: vec![(vshard_id, resp.watermark_lsn)], + read_version_lsn: resp.read_version_lsn, + not_found: is_not_found(&resp), + }) +} + +/// One plan this node applies on its own cores. +struct LocalPlan<'a> { + shared: &'a Arc, + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + plan: PhysicalPlan, + trace_id: TraceId, + txn_id: Option, +} + +/// Dispatch a plan to this node's cores on the route its class needs. +/// +/// A base-state write enters the funnel with `AppendHere`, which appends its +/// redo record under the write-admission guard. It reaches here when no Raft +/// proposal carries it: a plan with no replicated encoding, or a write a +/// transaction cannot buffer. The transaction meta-ops +/// own their durability, and a staged write is logged at COMMIT, so both take +/// the read route with everything else. +/// +/// An autocommit CRDT op that moves the Loro frontier runs inside the vShard's +/// admission sequencer. A replicated CRDT write is serialized by the +/// sequenced proposer, so this covers the frontier ops with no replicated +/// form. Without it, such an op can interleave with an admission preview +/// and its fenced apply. +async fn dispatch_local_plan(local: LocalPlan<'_>) -> Result { + let LocalPlan { + shared, + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + txn_id, + } = local; + if plan_is_write(&plan) && !is_task_vshard_scoped(&plan) { + let frontier_mutation = txn_id.is_none() + && matches!( + &plan, + PhysicalPlan::Crdt(op) if crate::control::crdt_admission::changes_crdt_frontier(op) + ); + let write = AutocommitWrite { + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + event_source: crate::event::EventSource::User, + txn_id, + }; + if frontier_mutation { + return shared + .vshard_admission_sequencer + .run(vshard_id, || dispatch_autocommit_write(shared, write)) + .await; + } + return dispatch_autocommit_write(shared, write).await; + } + dispatch_routed_read_to_data_plane( + shared, + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + txn_id, + ) + .await +} + +/// Whether the core refused the task with `ErrorCode::NotFound`. +/// +/// `reject_data_plane_error` passes this refusal as an empty success. +/// The flag keeps the verdict for a caller that needs it. +fn is_not_found(resp: &Response) -> bool { + resp.status == Status::Error && resp.error_code.as_deref() == Some(&ErrorCode::NotFound) +} diff --git a/nodedb/src/control/gateway/dispatch_remote.rs b/nodedb/src/control/gateway/dispatch_remote.rs index bdf0d253e..50fd74339 100644 --- a/nodedb/src/control/gateway/dispatch_remote.rs +++ b/nodedb/src/control/gateway/dispatch_remote.rs @@ -21,7 +21,8 @@ use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, TenantId, TraceId, TxnId, VShardId}; use nodedb_physical::physical_plan::wire as plan_wire; -use super::dispatcher::{DispatchOutcome, map_typed_cluster_error}; +use super::cluster_error::map_typed_cluster_error; +use super::dispatcher::DispatchOutcome; use super::version_set::GatewayVersionSet; /// Arguments for a remote dispatch call (bundles the parameters to stay @@ -39,6 +40,12 @@ pub(super) struct RemoteDispatchArgs<'a> { /// Session-transaction context forwarded to the remote executor, or `None` /// for non-transactional dispatch. pub txn_id: Option, + /// The route is a leg of a linearizable read. Exchange nodes resolved here + /// before the plan ships inherit it. + pub linearizable: bool, + /// Groups the remote node confirms before it reads. Empty for a write or + /// a weaker read. + pub read_groups: Vec, } /// Remote dispatch via `ExecuteRequest` RPC. @@ -56,6 +63,8 @@ pub(super) async fn dispatch_remote( deadline_ms, version_set, txn_id, + linearizable, + read_groups, } = args; let transport = shared.cluster_transport.as_ref().ok_or(Error::Internal { detail: "gateway: cluster transport not available for remote dispatch".into(), @@ -73,13 +82,15 @@ pub(super) async fn dispatch_remote( // Cluster remote-dispatch: no session-transaction context crosses this // boundary yet, so `None`. TRACKED: cross-node in-transaction reads are a // known gap (see resolve/exchange.rs). - let plan = match Box::pin(crate::control::server::exchange::resolve_exchange_in_plan( - shared, + let scope = crate::control::server::exchange::ReadScope { database_id, tenant_id, - plan, trace_id, - None, + txn_id: None, + linearizable, + }; + let plan = match Box::pin(crate::control::server::exchange::resolve_exchange_in_plan( + shared, plan, scope, )) .await? { @@ -103,10 +114,8 @@ pub(super) async fn dispatch_remote( } crate::control::server::exchange::Resolved::Plan(p) => *p, // Gateway path returns collected bytes: materialize the stream into one - // merged-array payload. (Single-node streaming never reaches the gateway - // — `state.gateway.is_none()` gates the Stream branch — but handle it - // exhaustively and behaviour-preservingly regardless.) Key the collected - // watermark to the collection's owning vShard this route dispatched to. + // merged-array payload. Key the collected watermark to the collection's + // owning vShard this route dispatched to. crate::control::server::exchange::Resolved::Stream(s) => { let (merged, lsn) = crate::control::server::result_stream::materialize(s).await?; return Ok(DispatchOutcome { @@ -140,6 +149,7 @@ pub(super) async fn dispatch_remote( ) .collect(); + let scoped_vshard = scoped_vshard(&plan, vshard_id)?; let req = RaftRpc::ExecuteRequest(ExecuteRequest { plan_bytes, tenant_id: tenant_id.as_u64(), @@ -148,6 +158,8 @@ pub(super) async fn dispatch_remote( trace_id: trace_id.0, descriptor_versions, txn_id, + vshard_id: scoped_vshard, + read_groups, }); debug!( @@ -168,6 +180,7 @@ pub(super) async fn dispatch_remote( vshard_id: VShardId::new((vshard_id % VShardId::COUNT as u64) as u32), leader_node: 0, leader_addr: format!("node-{node_id} (transport error: {e})"), + leader_term: 0, } })?; @@ -212,7 +225,7 @@ pub(super) async fn dispatch_remote( /// [`map_typed_cluster_error`] to a retryable [`Error`] and propagated to the /// gateway's existing not-leader retry loop. Once at least one chunk has been /// observed, any subsequent error is TERMINAL — it is surfaced as a stream -/// `Err` and never retried (re-running the plan would duplicate the rows +/// `Err` and never retried (re-running the plan duplicates the rows /// already streamed to the client). /// /// The returned stream re-emits the buffered first batch followed by the rest. @@ -230,6 +243,8 @@ pub(super) async fn dispatch_remote_stream( deadline_ms, version_set, txn_id, + linearizable, + read_groups, } = args; let transport = shared.cluster_transport.as_ref().ok_or(Error::Internal { detail: "gateway: cluster transport not available for remote stream dispatch".into(), @@ -237,13 +252,15 @@ pub(super) async fn dispatch_remote_stream( // Resolve Exchange nodes before shipping (symmetric with `dispatch_remote`). // No session-transaction context crosses this boundary yet, so `None`. - let plan = match Box::pin(crate::control::server::exchange::resolve_exchange_in_plan( - shared, + let scope = crate::control::server::exchange::ReadScope { database_id, tenant_id, - plan, trace_id, - None, + txn_id: None, + linearizable, + }; + let plan = match Box::pin(crate::control::server::exchange::resolve_exchange_in_plan( + shared, plan, scope, )) .await? { @@ -281,6 +298,7 @@ pub(super) async fn dispatch_remote_stream( ) .collect(); + let scoped_vshard = scoped_vshard(&plan, vshard_id)?; let req = RaftRpc::ExecuteStreamRequest(ExecuteRequest { plan_bytes, tenant_id: tenant_id.as_u64(), @@ -289,6 +307,8 @@ pub(super) async fn dispatch_remote_stream( trace_id: trace_id.0, descriptor_versions, txn_id, + vshard_id: scoped_vshard, + read_groups, }); debug!( @@ -308,6 +328,7 @@ pub(super) async fn dispatch_remote_stream( vshard_id: VShardId::new((vshard_id % VShardId::COUNT as u64) as u32), leader_node: 0, leader_addr: format!("node-{node_id} (stream open error: {e})"), + leader_term: 0, })?; // The `async_stream` body is `!Unpin`; pin it on the heap so we can pull // the eager first frame and then keep the tail around for `.chain`. @@ -397,10 +418,63 @@ fn map_stream_cluster_error(err: ClusterError, vshard_id: u64) -> Error { | ClusterError::SpatialGather(_) | ClusterError::Bm25Gather(_) | ClusterError::TsGather(_) + | ClusterError::ShufflePush(_) | ClusterError::RemoteUntyped { .. }) => Error::NotLeader { vshard_id: VShardId::new((vshard_id % VShardId::COUNT as u64) as u32), leader_node: 0, leader_addr: format!("stream dispatch error: {other}"), + leader_term: 0, }, } } + +/// The vShard a remote receiver must run `plan` on, or `None` when the plan +/// fans across the receiver's local cores. +/// +/// A transaction meta-op names no collection, so only the route's own vShard +/// can pick the core that holds the transaction's staging overlay. +fn scoped_vshard( + plan: &nodedb_physical::physical_plan::PhysicalPlan, + vshard_id: u64, +) -> Result, Error> { + if !super::router::is_task_vshard_scoped(plan) { + return Ok(None); + } + let raw = u32::try_from(vshard_id) + .ok() + .filter(|raw| *raw < VShardId::COUNT) + .ok_or_else(|| Error::Internal { + detail: format!( + "gateway: vShard-scoped plan routed to out-of-range vShard {vshard_id}" + ), + })?; + Ok(Some(VShardId::new(raw))) +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; + + #[test] + fn a_transaction_meta_op_carries_its_route_vshard() { + let plan = PhysicalPlan::Meta(MetaOp::DropTxnOverlay { + txn_id: TxnId::new(4), + }); + assert_eq!(scoped_vshard(&plan, 77).unwrap(), Some(VShardId::new(77))); + } + + #[test] + fn a_fanned_plan_carries_no_vshard() { + let plan = PhysicalPlan::Meta(MetaOp::Checkpoint); + assert_eq!(scoped_vshard(&plan, 77).unwrap(), None); + } + + #[test] + fn an_out_of_range_vshard_is_refused() { + let plan = PhysicalPlan::Meta(MetaOp::DropTxnOverlay { + txn_id: TxnId::new(4), + }); + assert!(scoped_vshard(&plan, u64::from(VShardId::COUNT)).is_err()); + } +} diff --git a/nodedb/src/control/gateway/dispatcher.rs b/nodedb/src/control/gateway/dispatcher.rs index 915fcfc63..552b01918 100644 --- a/nodedb/src/control/gateway/dispatcher.rs +++ b/nodedb/src/control/gateway/dispatcher.rs @@ -9,24 +9,18 @@ use std::sync::Arc; -use nodedb_cluster::rpc_codec::TypedClusterError; - use crate::Error; -use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; -use crate::control::local_dispatch::reject_data_plane_error; -use crate::control::server::dispatch_utils::{ - AutocommitWrite, dispatch_autocommit_write, dispatch_to_data_plane_with_txn, - extract_write_change_set, publish_change_set_with_lsn, -}; +use crate::bridge::envelope::PhysicalPlan; use crate::control::server::result_stream::ResultStream; -use crate::control::server::shared::write_admission::plan_is_write; +use crate::control::server::shared::session::served_reads; use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, TenantId, TraceId, TxnId, VShardId}; +use super::dispatch_local::{LocalContext, dispatch_local}; use super::dispatch_remote::{RemoteDispatchArgs, dispatch_remote, dispatch_remote_stream}; +use super::read_leg::{confirm_local_read, linearizable_read_groups}; use super::route::{RouteDecision, TaskRoute}; -use super::router::is_task_vshard_scoped; -use super::version_check::check_descriptor_versions; +use super::version_check::check_local_descriptor_versions; use super::version_set::GatewayVersionSet; /// Result of dispatching a single route: the raw payload bytes plus the @@ -63,6 +57,9 @@ pub struct DispatchRouteParams<'a> { pub deadline_ms: u64, pub version_set: &'a GatewayVersionSet, pub txn_id: Option, + /// The route is a leg of a linearizable read (see + /// `QueryContext::linearizable`). + pub linearizable: bool, } /// Dispatch a single route and return the raw payload bytes. @@ -78,23 +75,32 @@ pub(crate) async fn dispatch_route( deadline_ms, version_set, txn_id, + linearizable, } = params; reject_unadmitted_crdt_apply(&route.plan)?; + let read_groups = linearizable_read_groups(shared, &route, linearizable)?; + let route_vshard = route.vshard_id; match route.decision { RouteDecision::Local => { - dispatch_local( + confirm_local_read(shared, database_id, &route.plan, &read_groups, deadline_ms).await?; + let outcome = dispatch_local( route, - shared, - tenant_id, - database_id, - trace_id, - txn_id, - version_set, + LocalContext { + shared, + tenant_id, + database_id, + trace_id, + txn_id, + version_set, + }, ) - .await + .await?; + // The read's versions are this node's WAL positions. + served_reads::note(route_vshard, shared.node_id); + Ok(outcome) } RouteDecision::Remote { node_id, vshard_id } => { - dispatch_remote(RemoteDispatchArgs { + let outcome = dispatch_remote(RemoteDispatchArgs { plan: route.plan, shared, node_id, @@ -105,12 +111,17 @@ pub(crate) async fn dispatch_route( deadline_ms, version_set, txn_id, + linearizable, + read_groups, }) - .await + .await?; + // The read's versions are the remote node's WAL positions. + served_reads::note(route_vshard, node_id); + Ok(outcome) } RouteDecision::Broadcast { .. } => { // Split into individual Local/Remote routes by the router before - // dispatch; this arm should not be reached. + // dispatch; this arm is unreachable. Err(Error::Internal { detail: "dispatcher: Broadcast route reached dispatch — should have been split" .into(), @@ -123,6 +134,7 @@ pub(crate) async fn dispatch_route( vshard_id: VShardId::new(vshard_id as u32), leader_node: 0, leader_addr: String::new(), + leader_term: 0, }) } } @@ -137,11 +149,11 @@ pub struct DispatchRouteStreamParams<'a> { pub trace_id: TraceId, pub deadline_ms: u64, pub version_set: &'a GatewayVersionSet, + /// The route is a leg of a linearizable read. + pub linearizable: bool, } -/// Streaming sibling of [`dispatch_route`]: `Local` fans to all local cores, -/// `Remote` uses eager-first-frame dispatch, `Broadcast` is unreachable -/// (pre-split by the router), `LeaderUnknown` returns `NotLeader`. +/// Refuse a CRDT apply or snapshot import that did not pass CRDT admission. fn reject_unadmitted_crdt_apply(plan: &PhysicalPlan) -> Result<(), Error> { if matches!( plan, @@ -156,6 +168,9 @@ fn reject_unadmitted_crdt_apply(plan: &PhysicalPlan) -> Result<(), Error> { Ok(()) } +/// Streaming sibling of [`dispatch_route`]: `Local` fans to all local cores, +/// `Remote` uses eager-first-frame dispatch, `Broadcast` is unreachable +/// (pre-split by the router), `LeaderUnknown` returns `NotLeader`. pub(crate) async fn dispatch_route_stream( args: DispatchRouteStreamParams<'_>, ) -> Result { @@ -167,8 +182,10 @@ pub(crate) async fn dispatch_route_stream( trace_id, deadline_ms, version_set, + linearizable, } = args; reject_unadmitted_crdt_apply(&route.plan)?; + let read_groups = linearizable_read_groups(shared, &route, linearizable)?; match route.decision { // Cluster gateway route dispatch: no session-transaction context // crosses this boundary yet, so `None`. TRACKED: cross-node @@ -177,6 +194,7 @@ pub(crate) async fn dispatch_route_stream( // Same fence as the one-shot local path: a streaming read planned // against a superseded descriptor must not reach the cores. check_local_descriptor_versions(shared, tenant_id, database_id, version_set)?; + confirm_local_read(shared, database_id, &route.plan, &read_groups, deadline_ms).await?; crate::control::server::exchange::gather::gather_all_cores_stream( shared, tenant_id, @@ -200,6 +218,8 @@ pub(crate) async fn dispatch_route_stream( // No session-transaction context crosses the streaming gateway // boundary yet (see `resolve/exchange.rs`), so `None`. txn_id: None, + linearizable, + read_groups, }) .await } @@ -211,270 +231,11 @@ pub(crate) async fn dispatch_route_stream( vshard_id: VShardId::new(vshard_id as u32), leader_node: 0, leader_addr: String::new(), + leader_term: 0, }), } } -/// Re-compare a plan's stamped descriptor versions against this node's own -/// catalog. A mismatch surfaces as [`Error::RetryableSchemaChanged`], which the -/// gateway's cache-miss retry absorbs by re-planning against fresh state. -fn check_local_descriptor_versions( - shared: &Arc, - tenant_id: TenantId, - database_id: DatabaseId, - version_set: &GatewayVersionSet, -) -> Result<(), Error> { - check_descriptor_versions( - shared.credentials.catalog(), - database_id, - tenant_id.as_u64(), - version_set - .iter() - .map(|(collection, version)| (collection.as_str(), *version)), - )?; - Ok(()) -} - -/// Local dispatch via SPSC bridge. -/// -/// Carries `txn_id` so the Data Plane can resolve this session transaction's -/// staging overlay (read-your-own-writes) for in-block SQL and direct ops. -async fn dispatch_local( - route: TaskRoute, - shared: &Arc, - tenant_id: TenantId, - database_id: DatabaseId, - trace_id: TraceId, - txn_id: Option, - version_set: &GatewayVersionSet, -) -> Result { - // Staying on this node does not make the plan fresh: a DDL can bump a - // descriptor between planning and dispatch, and a drain that times out is - // force-ended. Fence the local plan against this node's catalog exactly as - // the leaseholder fences a forwarded one. - check_local_descriptor_versions(shared, tenant_id, database_id, version_set)?; - - let vshard_id = VShardId::new(route.vshard_id); - - if txn_id.is_some() - && matches!( - &route.plan, - PhysicalPlan::Crdt( - nodedb_physical::physical_plan::CrdtOp::Apply { .. } - | nodedb_physical::physical_plan::CrdtOp::ApplyAuthenticated { .. } - ) - ) - { - return Err(Error::CrdtApplyForbiddenInTransaction); - } - - // In local mode, frontier-changing CRDT operations have no Raft ordering. - // Serialize their complete Data Plane dispatch; replicated/transactional - // paths rely on the fenced-apply retry rather than this local mutex. - if txn_id.is_none() - && shared.async_raft_proposer().is_none() - && let PhysicalPlan::Crdt(op) = &route.plan - && crate::control::crdt_admission::changes_crdt_frontier(op) - { - let resp = shared - .vshard_admission_sequencer - .run(vshard_id, || async { - dispatch_local_plan(LocalPlan { - shared, - tenant_id, - database_id, - vshard_id, - plan: route.plan, - trace_id, - txn_id: None, - }) - .await - }) - .await?; - reject_data_plane_error(&resp)?; - return Ok(DispatchOutcome { - payloads: vec![resp.payload.to_vec()], - shard_watermarks: vec![(vshard_id, resp.watermark_lsn)], - read_version_lsn: resp.read_version_lsn, - not_found: is_not_found(&resp), - }); - } - - if txn_id.is_none() - && let Some(proposer) = shared.async_raft_proposer() - && let Some(entry) = crate::control::wal_replication::to_replicated_entry( - tenant_id, - database_id, - vshard_id, - &crate::control::wal_replication::ReplicableWrite::decide_for_replication(&route.plan)?, - )? - { - let (payload, write_version) = - crate::control::wal_replication::propose_replicated_entry(shared, proposer, entry) - .await?; - // Replicas apply with `ChangeFeedOwner::Unowned`. This node proposed - // the write once, so it publishes the change event. - publish_change_set_with_lsn( - shared, - tenant_id, - database_id, - extract_write_change_set(&route.plan, tenant_id), - write_version, - ); - return Ok(DispatchOutcome { - payloads: vec![payload], - // A write carries no read watermark (Lsn::ZERO); its post-write - // `coll_write_lsn` is surfaced via `read_version_lsn` instead. - shard_watermarks: vec![(vshard_id, Lsn::ZERO)], - read_version_lsn: write_version, - not_found: false, - }); - } - - let resp = dispatch_local_plan(LocalPlan { - shared, - tenant_id, - database_id, - vshard_id, - plan: route.plan, - trace_id, - txn_id, - }) - .await?; - // The remote sibling turns `ExecuteResponse.error` into `Err`; the local - // route must reject its own error status the same way. Keeping only the - // payload would hand a post-scan operator's failed expression back as an - // empty success — an error status is not an empty result set. - reject_data_plane_error(&resp)?; - Ok(DispatchOutcome { - payloads: vec![resp.payload.to_vec()], - shard_watermarks: vec![(vshard_id, resp.watermark_lsn)], - read_version_lsn: resp.read_version_lsn, - not_found: is_not_found(&resp), - }) -} - -/// One plan this node applies on its own cores. -struct LocalPlan<'a> { - shared: &'a Arc, - tenant_id: TenantId, - database_id: DatabaseId, - vshard_id: VShardId, - plan: PhysicalPlan, - trace_id: TraceId, - txn_id: Option, -} - -/// Dispatch a plan to this node's cores on the route its class needs. -/// -/// A base-state write enters the funnel with `AppendHere`, which appends its -/// redo record under the write-admission guard. It reaches here when no Raft -/// proposal carries it: a standalone node, a plan with no replicated -/// encoding, or a write a transaction cannot buffer. The transaction meta-ops -/// own their durability, and a staged write is logged at COMMIT, so both take -/// the read route with everything else. -async fn dispatch_local_plan(local: LocalPlan<'_>) -> Result { - let LocalPlan { - shared, - tenant_id, - database_id, - vshard_id, - plan, - trace_id, - txn_id, - } = local; - if plan_is_write(&plan) && !is_task_vshard_scoped(&plan) { - return dispatch_autocommit_write( - shared, - AutocommitWrite { - tenant_id, - database_id, - vshard_id, - plan, - trace_id, - event_source: crate::event::EventSource::User, - txn_id, - }, - ) - .await; - } - dispatch_to_data_plane_with_txn( - shared, - tenant_id, - database_id, - vshard_id, - plan, - trace_id, - txn_id, - ) - .await -} - -/// Whether the core refused the task with `ErrorCode::NotFound`. -/// -/// `reject_data_plane_error` passes this refusal as an empty success. -/// The flag keeps the verdict for a caller that needs it. -fn is_not_found(resp: &Response) -> bool { - resp.status == Status::Error && resp.error_code.as_deref() == Some(&ErrorCode::NotFound) -} - -/// Map a [`TypedClusterError`] to an internal [`Error`]. -/// -/// `NotLeader` is mapped such that the gateway retry loop can extract the -/// hinted leader from `Error::NotLeader.leader_node` and update the routing -/// table before the next attempt. -pub(super) fn map_typed_cluster_error(err: TypedClusterError, vshard_id: u64) -> Error { - match err { - TypedClusterError::NotLeader { - leader_node_id, - leader_addr, - .. - } => Error::NotLeader { - vshard_id: VShardId::new((vshard_id % VShardId::COUNT as u64) as u32), - leader_node: leader_node_id.unwrap_or(0), - leader_addr: leader_addr.unwrap_or_default(), - }, - TypedClusterError::DescriptorMismatch { - collection, - expected_version, - actual_version, - } => { - // A repeating mismatch means the planner and leaseholder disagree - // persistently — a bug, not the transient race the retry assumes. - tracing::debug!( - %collection, - expected_version, - actual_version, - "gateway: descriptor version mismatch at leaseholder" - ); - Error::RetryableSchemaChanged { - descriptor: collection, - } - } - TypedClusterError::DeadlineExceeded { .. } => Error::DeadlineExceeded { - request_id: crate::types::RequestId::new(0), - }, - // Remote Data-Plane verdict: keep the code so the client sees the - // SQLSTATE local execution renders, not a generic internal error. - TypedClusterError::DataPlane { code } => Error::DataPlane(code.into()), - // Remote constraint refusal: keep the kind so the client sees 23502 - // vs 23505, exactly as a local refusal on this node would render. - TypedClusterError::RejectedConstraint { - collection, - constraint, - detail, - } => Error::RejectedConstraint { - collection, - constraint, - detail, - }, - // A numeric class crosses as `Error::RemoteTyped`, so the client sees - // the SQLSTATE the executing node gave it. Only a code of 0 (no class) - // decodes as `Error::Internal`. - internal @ TypedClusterError::Internal { .. } => Error::from(internal), - } -} - /// Milliseconds left on the running statement, for a remote hop's /// `ExecuteRequest.deadline_remaining_ms`. /// @@ -490,58 +251,6 @@ pub fn statement_deadline_ms(shared: &SharedState) -> u64 { #[cfg(test)] mod tests { use super::*; - use nodedb_cluster::rpc_codec::TypedClusterError; - - #[test] - fn map_not_leader() { - let err = TypedClusterError::NotLeader { - group_id: 0, - leader_node_id: Some(5), - leader_addr: Some("10.0.0.5:9400".into()), - term: 3, - }; - match map_typed_cluster_error(err, 7) { - Error::NotLeader { leader_node, .. } => assert_eq!(leader_node, 5), - other => panic!("expected NotLeader, got {other:?}"), - } - } - - #[test] - fn map_descriptor_mismatch() { - let err = TypedClusterError::DescriptorMismatch { - collection: "orders".into(), - expected_version: 1, - actual_version: 2, - }; - match map_typed_cluster_error(err, 0) { - Error::RetryableSchemaChanged { descriptor } => assert_eq!(descriptor, "orders"), - other => panic!("expected RetryableSchemaChanged, got {other:?}"), - } - } - - /// A remote error with a numeric class keeps it, never `Internal`. - #[test] - fn map_internal_keeps_its_numeric_class() { - let err = TypedClusterError::Internal { - code: u32::from(nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED.0), - message: "permission denied on orders".into(), - }; - match map_typed_cluster_error(err, 0) { - Error::RemoteTyped { code, .. } => { - assert_eq!(code, nodedb_types::error::ErrorCode::AUTHORIZATION_DENIED); - } - other => panic!("expected RemoteTyped, got {other:?}"), - } - } - - #[test] - fn map_deadline_exceeded() { - let err = TypedClusterError::DeadlineExceeded { elapsed_ms: 100 }; - assert!(matches!( - map_typed_cluster_error(err, 0), - Error::DeadlineExceeded { .. } - )); - } #[test] fn gateway_rejects_unadmitted_crdt_apply_before_route_selection() { @@ -551,7 +260,7 @@ mod tests { delta: vec![1], peer_id: 1, mutation_id: 1, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), provenance: None, constraint_version_required: 0, expected_frontier_digest: None, diff --git a/nodedb/src/control/gateway/error_map/class_parity.rs b/nodedb/src/control/gateway/error_map/class_parity.rs deleted file mode 100644 index a6fd98a39..000000000 --- a/nodedb/src/control/gateway/error_map/class_parity.rs +++ /dev/null @@ -1,1186 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Every Data-Plane `ErrorCode`, and each Control-Plane error a client acts -//! on, answers one class on native, pgwire and HTTP. -//! -//! pgwire renders a Data-Plane verdict as its SQLSTATE. A native client reads -//! the numeric `nodedb_types` code on the frame. The two agree when the -//! numeric code renders, through the numeric-code SQLSTATE table, in the same -//! SQLSTATE class (the first two characters) as the pgwire SQLSTATE. That same -//! table renders a code that crossed a node as a bare number, so agreement -//! also keeps a verdict's class across nodes. - -use nodedb_types::error::sqlstate; -use nodedb_types::sync::violation::ViolationType; -use nodedb_types::sync::wire::SyncProvenance; - -use crate::bridge::envelope::{CounterFault, ErrorCode, SyncHold}; -use crate::control::server::native::dispatch::{error_code_to_native, native_error_fields}; -use crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate; -use crate::control::server::pgwire::types::error_to_sqlstate; - -use super::gateway_map::GatewayErrorMap; - -/// A named encoder that turns an `Error` into its node-hop wire form. -type HopEncoder = ( - &'static str, - fn(crate::Error) -> nodedb_cluster::rpc_codec::TypedClusterError, -); - -/// An error builder, with the SQLSTATE and public code it must answer. -type ClassCase = ( - fn() -> crate::Error, - &'static str, - nodedb_types::error::ErrorCode, -); - -/// The number of `ErrorCode` variants [`variant_index`] numbers. -const VARIANT_COUNT: usize = 44; - -/// A dense index per variant. Exhaustive, so a new variant fails to compile -/// here until it gets an index, and [`every_variant_has_a_sample`] then fails -/// until [`samples`] carries it. -fn variant_index(code: &ErrorCode) -> usize { - match code { - ErrorCode::DeadlineExceeded => 0, - ErrorCode::RejectedConstraint { .. } => 1, - ErrorCode::RejectedPrevalidation { .. } => 2, - ErrorCode::RetryableRefusal { .. } => 3, - ErrorCode::SyncRejected { .. } => 4, - ErrorCode::SyncNotApplied { .. } => 5, - ErrorCode::NotFound => 6, - ErrorCode::RejectedAuthz { .. } => 7, - ErrorCode::ConflictRetry => 8, - ErrorCode::CrdtFrontierMismatch { .. } => 9, - ErrorCode::FanOutExceeded => 10, - ErrorCode::ResourcesExhausted => 11, - ErrorCode::RejectedDanglingEdge { .. } => 12, - ErrorCode::DuplicateWrite => 13, - ErrorCode::AppendOnlyViolation { .. } => 14, - ErrorCode::BalanceViolation { .. } => 15, - ErrorCode::PeriodLocked { .. } => 16, - ErrorCode::PeriodLockMisconfigured { .. } => 17, - ErrorCode::RetentionViolation { .. } => 18, - ErrorCode::LegalHoldActive { .. } => 19, - ErrorCode::StateTransitionViolation { .. } => 20, - ErrorCode::TransitionCheckViolation { .. } => 21, - ErrorCode::TypeGuardViolation { .. } => 22, - ErrorCode::TypeMismatch { .. } => 23, - ErrorCode::CounterFault { .. } => 24, - ErrorCode::InsufficientBalance { .. } => 25, - ErrorCode::RateExceeded { .. } => 26, - ErrorCode::CollectionDraining { .. } => 27, - ErrorCode::RecursionDepthExceeded { .. } => 28, - ErrorCode::UndefinedColumn { .. } => 29, - ErrorCode::Internal { .. } => 30, - ErrorCode::Unsupported { .. } => 31, - ErrorCode::RollbackFailed { .. } => 32, - ErrorCode::OllpRetryRequired => 33, - ErrorCode::TxnOverlayMemoryExceeded { .. } => 34, - ErrorCode::DivisionByZero => 35, - ErrorCode::UndefinedFunction { .. } => 36, - ErrorCode::DataException { .. } => 37, - ErrorCode::DispatchCapacity { .. } => 38, - ErrorCode::ExpiredBeforeExecution => 39, - ErrorCode::BadRequest { .. } => 40, - ErrorCode::TransactionRollback { .. } => 41, - ErrorCode::ActiveSqlTransaction { .. } => 42, - ErrorCode::DependentObjectsExist { .. } => 43, - } -} - -fn provenance() -> SyncProvenance { - SyncProvenance { - producer_id: 1, - epoch: 1, - stream_id: 1, - seq: 1, - } -} - -/// One sample per variant, plus one per value that picks a different -/// SQLSTATE: each constraint kind and each counter fault. -fn samples() -> Vec { - let text = || "detail".to_owned(); - let collection = || "c".to_owned(); - let mut samples = vec![ - ErrorCode::DeadlineExceeded, - ErrorCode::RejectedPrevalidation { reason: text() }, - ErrorCode::RetryableRefusal { reason: text() }, - ErrorCode::SyncRejected { - violation: ViolationType::PermissionDenied, - applied_seq: 1, - provenance: provenance(), - }, - ErrorCode::SyncRejected { - violation: ViolationType::RateLimited, - applied_seq: 1, - provenance: provenance(), - }, - ErrorCode::SyncNotApplied { - hold: SyncHold::Gap { expected: 2 }, - applied_seq: 1, - }, - ErrorCode::NotFound, - ErrorCode::RejectedAuthz { resource: text() }, - ErrorCode::ConflictRetry, - ErrorCode::CrdtFrontierMismatch { - expected: [0; 32], - actual: [1; 32], - }, - ErrorCode::FanOutExceeded, - ErrorCode::ResourcesExhausted, - ErrorCode::RejectedDanglingEdge { - missing_node: text(), - }, - ErrorCode::DuplicateWrite, - ErrorCode::AppendOnlyViolation { - collection: collection(), - }, - ErrorCode::BalanceViolation { - collection: collection(), - detail: text(), - }, - ErrorCode::PeriodLocked { - collection: collection(), - }, - ErrorCode::PeriodLockMisconfigured { - collection: collection(), - ref_table: "periods".into(), - status_column: "status".into(), - row_identity: "p1".into(), - }, - ErrorCode::RetentionViolation { - collection: collection(), - }, - ErrorCode::LegalHoldActive { - collection: collection(), - }, - ErrorCode::StateTransitionViolation { - collection: collection(), - detail: text(), - }, - ErrorCode::TransitionCheckViolation { - collection: collection(), - detail: text(), - }, - ErrorCode::TypeGuardViolation { - collection: collection(), - detail: text(), - }, - ErrorCode::TypeMismatch { - collection: collection(), - detail: text(), - }, - ErrorCode::InsufficientBalance { - collection: collection(), - detail: text(), - }, - ErrorCode::RateExceeded { - gate: "g".into(), - retry_after_ms: 10, - }, - ErrorCode::CollectionDraining { - collection: collection(), - }, - ErrorCode::RecursionDepthExceeded { - cte_name: "walk".into(), - max_depth: 100, - }, - ErrorCode::UndefinedColumn { column: "x".into() }, - ErrorCode::Internal { detail: text() }, - ErrorCode::Unsupported { detail: text() }, - ErrorCode::RollbackFailed { - entry_index: 0, - detail: text(), - }, - ErrorCode::OllpRetryRequired, - ErrorCode::TxnOverlayMemoryExceeded { limit: 1 << 20 }, - ErrorCode::DivisionByZero, - ErrorCode::UndefinedFunction { name: "f".into() }, - ErrorCode::DataException { detail: text() }, - ErrorCode::DispatchCapacity { reason: text() }, - ErrorCode::ExpiredBeforeExecution, - ErrorCode::BadRequest { detail: text() }, - ErrorCode::TransactionRollback { detail: text() }, - ErrorCode::ActiveSqlTransaction { detail: text() }, - ErrorCode::DependentObjectsExist { - object: "role \"analyst\"".into(), - detail: text(), - }, - ]; - for constraint in [ - "not_null", - "unique", - "generated_always", - "fk_missing", - "rls_policy", - "permission_denied", - "check", - ] { - samples.push(ErrorCode::RejectedConstraint { - constraint: constraint.into(), - detail: text(), - }); - } - for fault in [ - CounterFault::NotAnInteger, - CounterFault::NotAFloat, - CounterFault::IntegerOverflow, - CounterFault::NonFinite, - ] { - samples.push(ErrorCode::CounterFault { - collection: collection(), - fault, - }); - } - samples -} - -fn class(state: &str) -> &str { - state.get(..2).unwrap_or(state) -} - -#[test] -fn every_variant_has_a_sample() { - let mut seen = [false; VARIANT_COUNT]; - for code in samples() { - seen[variant_index(&code)] = true; - } - let missing: Vec = (0..VARIANT_COUNT).filter(|i| !seen[*i]).collect(); - assert!(missing.is_empty(), "variants with no sample: {missing:?}"); -} - -/// The native frame carries pgwire's SQLSTATE, and its numeric code has the -/// same SQLSTATE class, on both native renderings: the typed `Err` and the -/// raw response frame. -#[test] -fn every_data_plane_code_has_one_class_on_native_and_pgwire() { - for code in samples() { - let err = crate::Error::DataPlane(code.clone()); - let (_, pg_state, _) = error_to_sqlstate(&err); - - let native = native_error_fields(&err); - assert_eq!(native.sqlstate, pg_state, "native SQLSTATE for {code:?}"); - let native_state = numeric_code_to_sqlstate(native.code); - assert_eq!( - class(native_state), - class(pg_state), - "{code:?}: pgwire sends {pg_state}, native code {} renders {native_state}", - native.code - ); - - let frame = error_code_to_native(1, Some(&code)); - let payload = frame.error.expect("error frames carry a payload"); - assert_eq!( - payload.code, pg_state, - "response-frame SQLSTATE for {code:?}" - ); - assert_eq!( - payload.ndb_code, native.code.0, - "response-frame code for {code:?}" - ); - } -} - -/// A classified Data-Plane verdict never reads as a server fault over HTTP. -#[test] -fn classified_data_plane_codes_are_not_http_500() { - for code in samples() { - let err = crate::Error::DataPlane(code.clone()); - let (_, pg_state, _) = error_to_sqlstate(&err); - let (status, _) = GatewayErrorMap::to_http(&err); - if pg_state == sqlstate::INTERNAL_ERROR { - assert_eq!(status, 500, "{code:?}"); - } else { - assert_ne!(status, 500, "{code:?} is {pg_state} on pgwire"); - } - } -} - -/// The SQLSTATE status table agrees with the gateway status table for every -/// Data-Plane code. A DDL error and a query error of one class answer one -/// HTTP status. -#[test] -fn sqlstate_status_agrees_with_the_gateway_status() { - for code in samples() { - let err = crate::Error::DataPlane(code.clone()); - let (_, pg_state, _) = error_to_sqlstate(&err); - let (status, _) = GatewayErrorMap::to_http(&err); - assert_eq!( - GatewayErrorMap::sqlstate_to_http(pg_state), - status, - "{code:?} is {pg_state} on pgwire" - ); - } -} - -/// `Unsupported` is feature-not-supported on every surface. -#[test] -fn unsupported_is_feature_not_supported_everywhere() { - let err = crate::Error::DataPlane(ErrorCode::Unsupported { - detail: "not on this engine".into(), - }); - let (_, pg_state, _) = error_to_sqlstate(&err); - assert_eq!(pg_state, sqlstate::FEATURE_NOT_SUPPORTED); - let native = native_error_fields(&err); - assert_eq!(native.code, nodedb_types::error::ErrorCode::SQL_NOT_ENABLED); - assert_eq!(GatewayErrorMap::to_http(&err).0, 501); -} - -/// Control-Plane errors a client acts on. Each has a class of its own. -fn control_plane_samples() -> Vec { - vec![ - crate::Error::RetryableSchemaChanged { - descriptor: "orders".into(), - }, - crate::Error::SessionTokenExpired, - ] -} - -/// A Control-Plane error has one class on native and pgwire, and its native -/// numeric code renders in that class. -#[test] -fn control_plane_errors_have_one_class_on_native_and_pgwire() { - for err in control_plane_samples() { - let (_, pg_state, _) = error_to_sqlstate(&err); - assert_ne!(pg_state, sqlstate::INTERNAL_ERROR, "{err:?} has no class"); - - let native = native_error_fields(&err); - assert_eq!(native.sqlstate, pg_state, "native SQLSTATE for {err:?}"); - let native_state = numeric_code_to_sqlstate(native.code); - assert_eq!( - class(native_state), - class(pg_state), - "{err:?}: pgwire sends {pg_state}, native code {} renders {native_state}", - native.code - ); - } -} - -/// A schema change the server could not absorb is the retryable -/// serialization class on every surface. -#[test] -fn schema_change_is_a_retryable_serialization_failure() { - let err = crate::Error::RetryableSchemaChanged { - descriptor: "orders".into(), - }; - assert_eq!(error_to_sqlstate(&err).1, sqlstate::SERIALIZATION_FAILURE); - let native = native_error_fields(&err); - assert_eq!(native.sqlstate, sqlstate::SERIALIZATION_FAILURE); - assert_eq!(native.code, nodedb_types::error::ErrorCode::WRITE_CONFLICT); - assert!(crate::error_classify::classify(&err).is_retriable()); - let status = GatewayErrorMap::to_http(&err).0; - assert_eq!(status, 409); - assert_eq!( - GatewayErrorMap::sqlstate_to_http(sqlstate::SERIALIZATION_FAILURE), - status - ); -} - -/// An expired session token is invalid authorization on every surface. -#[test] -fn expired_session_token_is_invalid_authorization_everywhere() { - let err = crate::Error::SessionTokenExpired; - assert_eq!(error_to_sqlstate(&err).1, sqlstate::AUTH_TOKEN_EXPIRED.0); - let native = native_error_fields(&err); - assert_eq!(native.sqlstate, sqlstate::AUTH_TOKEN_EXPIRED.0); - assert_eq!(native.code, nodedb_types::error::ErrorCode::AUTH_EXPIRED); - assert_eq!( - numeric_code_to_sqlstate(native.code), - sqlstate::AUTH_TOKEN_EXPIRED.0 - ); - let status = GatewayErrorMap::to_http(&err).0; - assert_eq!(status, 401); - assert_eq!( - GatewayErrorMap::sqlstate_to_http(sqlstate::AUTH_TOKEN_EXPIRED.0), - status - ); -} - -/// The number of `crate::Error` variants [`error_variant_index`] numbers. -const ERROR_VARIANT_COUNT: usize = 110; - -/// A dense index per `crate::Error` variant. Exhaustive, so a new variant -/// fails to compile here until it gets an index, and -/// [`every_error_variant_has_a_sample`] then fails until -/// [`error_samples`] carries it. -pub(crate) fn error_variant_index(err: &crate::Error) -> usize { - use crate::Error as E; - match err { - E::RejectedConstraint { .. } => 0, - E::TxnOverlayMemoryExceeded { .. } => 1, - E::RejectedAuthz { .. } => 2, - E::OffsetRegression { .. } => 3, - E::DeadlineExceeded { .. } => 4, - E::ConflictRetry { .. } => 5, - E::CalvinSerializationConflict => 6, - E::CalvinParticipantError => 7, - E::RejectedPrevalidation { .. } => 8, - E::RetryableRefusal { .. } => 9, - E::AppendOnlyViolation { .. } => 10, - E::BalanceViolation { .. } => 11, - E::MaterializedSumTargetNotFound { .. } => 12, - E::MaterializedSumResolutionMissing { .. } => 13, - E::PeriodLocked { .. } => 14, - E::PeriodLockMisconfigured { .. } => 15, - E::RetentionViolation { .. } => 16, - E::LegalHoldActive { .. } => 17, - E::StateTransitionViolation { .. } => 18, - E::TransitionCheckViolation { .. } => 19, - E::TypeGuardViolation { .. } => 20, - E::TypeMismatch { .. } => 21, - E::InsufficientBalance { .. } => 22, - E::RateExceeded { .. } => 23, - E::CollectionNotFound { .. } => 24, - E::DocumentNotFound { .. } => 25, - E::CollectionDeactivated { .. } => 26, - E::VShardAdmissionCapacityExceeded { .. } => 27, - E::CrdtAdmissionRetriesExhausted { .. } => 28, - E::CrdtAdmissionInvalidPlan { .. } => 29, - E::CrdtAdmissionCallerFence => 30, - E::CrdtApplyRequiresAdmission => 31, - E::CrdtApplyForbiddenInTransaction => 32, - E::NotInTransactionBlock { .. } => 33, - E::CrdtAdmissionTimeout { .. } => 34, - E::NoLeader { .. } => 35, - E::NotLeader { .. } => 36, - E::FanOutExceeded { .. } => 37, - E::CrossCollectionNotColocated { .. } => 38, - E::SourceFrozen { .. } => 39, - E::CloneWriteRequiresMaterialize { .. } => 40, - E::BadRequest { .. } => 41, - E::BackupTenantMismatch { .. } => 42, - E::BackupKeyMismatch => 43, - E::QuotaOvercommit { .. } => 44, - E::PlanError { .. } => 45, - E::FeatureNotSupported { .. } => 46, - E::UndefinedFunction { .. } => 47, - E::UndefinedObject { .. } => 48, - E::ObjectNotInPrerequisiteState { .. } => 49, - E::UndefinedColumn { .. } => 50, - E::AmbiguousColumn { .. } => 51, - E::UnknownStrictField { .. } => 52, - E::DivisionByZero => 53, - E::DataException { .. } => 54, - E::InvalidLimitValue { .. } => 55, - E::RetryableSchemaChanged { .. } => 56, - E::RetryableLeaderChange { .. } => 57, - E::GroupQuorumUnavailable { .. } => 58, - E::GroupMarksUnavailable { .. } => 59, - E::MetadataLeaderUnavailable => 60, - E::AuthorizationStateBehind { .. } => 61, - E::ExecutionLimitExceeded { .. } => 62, - E::LimitExceeded { .. } => 63, - E::Wal(_) => 64, - E::Dispatch { .. } => 65, - E::DispatchCapacity { .. } => 66, - E::Storage { .. } => 67, - E::ColdStorage { .. } => 68, - E::Serialization { .. } => 69, - E::Codec { .. } => 70, - E::SegmentCorrupted { .. } => 71, - E::MemoryExhausted { .. } => 72, - E::Backpressure { .. } => 73, - E::Crdt(_) => 74, - E::Io(_) => 75, - E::Config { .. } => 76, - E::Encryption { .. } => 77, - E::Bridge { .. } => 78, - E::VersionCompat { .. } => 79, - E::Internal { .. } => 80, - E::Shaping(_) => 81, - E::RemoteTyped { .. } => 82, - E::DescriptorVersionAnomaly { .. } => 83, - E::CollectionPurgeRowMissing { .. } => 84, - E::CatalogIntegrityViolation { .. } => 85, - E::DataPlane(_) => 86, - E::Promql(_) => 87, - E::DependentObjectsExist { .. } => 88, - E::CascadeCycle { .. } => 89, - E::CrossShardInExplicitTransaction => 90, - E::SequencerUnavailable => 91, - E::SessionCapExceeded { .. } => 92, - E::SessionIdleTimeout => 93, - E::SessionTokenExpired => 94, - E::SessionKilledByAdmin => 95, - E::SessionUserDropped => 96, - E::OidcProviderTenantUnbound => 97, - E::OidcProviderTenantUnavailable { .. } => 98, - E::ExternalRoleUndefined { .. } => 99, - E::OidcNoDefaultDatabase { .. } => 100, - E::TenantVectorDimExceeded { .. } => 101, - E::TenantGraphDepthExceeded { .. } => 102, - E::RoleInheritanceCycle { .. } => 103, - E::RoleInheritanceDepthExceeded { .. } => 104, - E::OllpExhausted { .. } => 105, - E::MirrorReadOnly { .. } => 106, - E::StaleReadNotLeader { .. } => 107, - E::RoleInUse { .. } => 108, - E::Ddl(_) => 109, - } -} - -/// One sample per `crate::Error` variant. -pub(crate) fn error_samples() -> Vec { - use crate::Error as E; - use crate::types::{DatabaseId, RequestId, TenantId, VShardId}; - - let text = || "detail".to_owned(); - let collection = || "c".to_owned(); - vec![ - E::RejectedConstraint { - collection: collection(), - constraint: "unique".into(), - detail: text(), - }, - E::TxnOverlayMemoryExceeded { limit: 1 << 20 }, - E::RejectedAuthz { - tenant_id: TenantId::new(1), - resource: text(), - }, - E::OffsetRegression { - stream: "s".into(), - group: "g".into(), - partition_id: 0, - current_lsn: 2, - current_sequence: 2, - attempted_lsn: 1, - attempted_sequence: 1, - }, - E::DeadlineExceeded { - request_id: RequestId::new(1), - }, - E::ConflictRetry { - collection: collection(), - document_id: "d".into(), - }, - E::CalvinSerializationConflict, - E::CalvinParticipantError, - E::RejectedPrevalidation { - constraint: "check".into(), - reason: text(), - }, - E::RetryableRefusal { reason: text() }, - E::AppendOnlyViolation { - collection: collection(), - detail: text(), - }, - E::BalanceViolation { - collection: collection(), - detail: text(), - }, - E::MaterializedSumTargetNotFound { - target_collection: "t".into(), - join_column: "k".into(), - join_value: "1".into(), - }, - E::MaterializedSumResolutionMissing { - target_collection: "t".into(), - join_column: "k".into(), - join_value: "1".into(), - }, - E::PeriodLocked { - collection: collection(), - detail: text(), - }, - E::PeriodLockMisconfigured { - collection: collection(), - ref_table: "periods".into(), - status_column: "status".into(), - row_identity: "p1".into(), - }, - E::RetentionViolation { - collection: collection(), - detail: text(), - }, - E::LegalHoldActive { - collection: collection(), - detail: text(), - }, - E::StateTransitionViolation { - collection: collection(), - detail: text(), - }, - E::TransitionCheckViolation { - collection: collection(), - detail: text(), - }, - E::TypeGuardViolation { - collection: collection(), - detail: text(), - }, - E::TypeMismatch { - collection: collection(), - key: "k".into(), - detail: text(), - }, - E::InsufficientBalance { - collection: collection(), - key: "k".into(), - detail: text(), - }, - E::RateExceeded { - gate: "g".into(), - detail: text(), - retry_after_ms: 10, - }, - E::CollectionNotFound { - tenant_id: TenantId::new(1), - collection: collection(), - }, - E::DocumentNotFound { - collection: collection(), - document_id: "d".into(), - }, - E::CollectionDeactivated { - tenant_id: TenantId::new(1), - collection: collection(), - retention_expires_at_ns: 1, - }, - E::VShardAdmissionCapacityExceeded { - vshard_id: VShardId::new(1), - capacity: 4, - }, - E::CrdtAdmissionRetriesExhausted { - vshard_id: VShardId::new(1), - attempts: 3, - }, - E::CrdtAdmissionInvalidPlan { reason: "empty" }, - E::CrdtAdmissionCallerFence, - E::CrdtApplyRequiresAdmission, - E::CrdtApplyForbiddenInTransaction, - E::NotInTransactionBlock { - statement: "VACUUM".into(), - }, - E::CrdtAdmissionTimeout { - vshard_id: VShardId::new(1), - timeout_ms: 10, - }, - E::NoLeader { - vshard_id: VShardId::new(1), - }, - E::NotLeader { - vshard_id: VShardId::new(1), - leader_node: 2, - leader_addr: "10.0.0.1:9000".into(), - }, - E::FanOutExceeded { - shards_touched: 9, - limit: 8, - }, - E::CrossCollectionNotColocated { - op: "insert-select", - source_collection: "a".into(), - target_collection: "b".into(), - }, - E::SourceFrozen { - database_id: DatabaseId::new(7), - }, - E::CloneWriteRequiresMaterialize { - collection: collection(), - engine: "kv".into(), - database: "db".into(), - reason: "shadowed", - }, - E::BadRequest { detail: text() }, - E::BackupTenantMismatch { - expected: 1, - actual: 2, - }, - E::BackupKeyMismatch, - E::QuotaOvercommit { - field: "max_storage".into(), - detail: text(), - }, - E::PlanError { detail: text() }, - E::FeatureNotSupported { detail: text() }, - E::UndefinedFunction { name: "f".into() }, - E::UndefinedObject { - kind: "sequence", - name: "s".into(), - }, - E::ObjectNotInPrerequisiteState { - object: "s".into(), - detail: text(), - }, - E::UndefinedColumn { column: "x".into() }, - E::AmbiguousColumn { - column: "id".into(), - }, - E::UnknownStrictField { - collection: collection(), - column: "x".into(), - }, - E::DivisionByZero, - E::DataException { detail: text() }, - E::InvalidLimitValue { - clause: "LIMIT", - value: "-1".into(), - }, - E::RetryableSchemaChanged { - descriptor: "orders".into(), - }, - E::RetryableLeaderChange { - group_id: 1, - log_index: 2, - }, - E::GroupQuorumUnavailable { - group_id: 1, - voters: vec![1, 2, 3], - unreachable: vec![2, 3], - }, - E::GroupMarksUnavailable { - group_id: 1, - refused_by: vec![2], - }, - E::MetadataLeaderUnavailable, - E::AuthorizationStateBehind { detail: text() }, - E::ExecutionLimitExceeded { detail: text() }, - E::LimitExceeded { - limit_name: "max_rows", - value: 10, - max: 5, - }, - E::Wal(nodedb_wal::WalError::Sealed), - E::Dispatch { detail: text() }, - E::DispatchCapacity { - scope: crate::DispatchCapacityScope::QueueFull { - core_id: 0, - capacity: 4, - }, - }, - E::Storage { - engine: "kv".into(), - detail: text(), - }, - E::ColdStorage { detail: text() }, - E::Serialization { - format: "msgpack".into(), - detail: text(), - }, - E::Codec { detail: text() }, - E::SegmentCorrupted { detail: text() }, - E::MemoryExhausted { - engine: "kv".into(), - }, - E::Backpressure { - engine: nodedb_mem::EngineId::Vector, - }, - E::Crdt(nodedb_crdt::CrdtError::ConstraintViolation { - constraint: "unique".into(), - collection: collection(), - detail: text(), - }), - E::Io(std::io::Error::other("disk")), - E::Config { detail: text() }, - E::Encryption { detail: text() }, - E::Bridge { detail: text() }, - E::VersionCompat { detail: text() }, - E::Internal { detail: text() }, - E::Shaping(Box::new(nodedb_types::NodeDbError::bad_request(text()))), - E::RemoteTyped { - code: nodedb_types::error::ErrorCode::WRITE_CONFLICT, - message: text(), - }, - E::DescriptorVersionAnomaly { - descriptor: "orders".into(), - carried: 5, - prior: 2, - }, - E::CollectionPurgeRowMissing { - database_id: 1, - tenant_id: 1, - name: collection(), - }, - E::CatalogIntegrityViolation { - entry_kind: "PutCollection".into(), - detail: text(), - }, - E::DataPlane(ErrorCode::NotFound), - E::Promql(crate::control::promql::PromqlError::UnexpectedEof), - E::DependentObjectsExist { - tenant_id: 1, - root_kind: "collection", - root_name: collection(), - dependent_count: 1, - dependents: vec![("view".into(), "v".into())], - }, - E::CascadeCycle { - tenant_id: 1, - root: collection(), - depth: 64, - }, - E::CrossShardInExplicitTransaction, - E::SequencerUnavailable, - E::SessionCapExceeded { cap: 8 }, - E::SessionIdleTimeout, - E::SessionTokenExpired, - E::SessionKilledByAdmin, - E::SessionUserDropped, - E::OidcProviderTenantUnbound, - E::OidcProviderTenantUnavailable { tenant_id: 1 }, - E::ExternalRoleUndefined { - subject: "alice".into(), - role: "auditor".into(), - tenant_id: 1, - }, - E::OidcNoDefaultDatabase { - sub: "alice".into(), - }, - E::TenantVectorDimExceeded { - dim: 4096, - limit: 1024, - }, - E::TenantGraphDepthExceeded { - depth: 20, - limit: 10, - }, - E::RoleInheritanceCycle { - child: "a".into(), - parent: "b".into(), - }, - E::RoleInheritanceDepthExceeded { depth: 9, limit: 8 }, - E::OllpExhausted { - retries: 3, - cause: crate::OllpExhaustedCause::PredicateDrift, - }, - E::MirrorReadOnly { - database: "db".into(), - }, - E::StaleReadNotLeader { - database: "db".into(), - source_cluster: "src".into(), - detail: text(), - }, - E::RoleInUse { - role: "analyst".into(), - dependents: crate::control::security::role_assignment::RoleDependents::Users(vec![ - "bob".into(), - ]), - }, - E::Ddl(Box::new( - crate::control::server::shared::ddl::DdlError::new( - sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, - text(), - ), - )), - ] -} - -#[test] -fn every_error_variant_has_a_sample() { - let mut seen = [false; ERROR_VARIANT_COUNT]; - for err in error_samples() { - seen[error_variant_index(&err)] = true; - } - let missing: Vec = (0..ERROR_VARIANT_COUNT).filter(|i| !seen[*i]).collect(); - assert!( - missing.is_empty(), - "error variants with no sample: {missing:?}" - ); -} - -/// Every `crate::Error` variant answers the HTTP status its pgwire SQLSTATE -/// class has. Only an internal or system error reads as a 500. -#[test] -fn every_error_variant_has_the_http_status_of_its_sqlstate() { - for err in error_samples() { - let (_, pg_state, _) = error_to_sqlstate(&err); - let (status, _) = GatewayErrorMap::to_http(&err); - assert_eq!( - status, - GatewayErrorMap::sqlstate_to_http(pg_state), - "{err:?} is {pg_state} on pgwire" - ); - let server_fault = matches!(class(pg_state), "XX" | "58"); - assert_eq!( - status == 500, - server_fault, - "{err:?} is {pg_state} on pgwire but HTTP {status}" - ); - } -} - -/// Every `crate::Error` variant renders the SQLSTATE class it renders locally -/// after it crosses a node hop, through both wire encoders and the decoder. -#[test] -fn every_error_variant_keeps_its_class_across_a_node_hop() { - use nodedb_cluster::rpc_codec::TypedClusterError; - - use crate::control::cluster::data_plane_error_wire::execution_error_to_typed; - - let encoders: [HopEncoder; 2] = [ - ("execution_error_to_typed", execution_error_to_typed), - ("From", TypedClusterError::from), - ]; - for (name, encode) in encoders { - for (err, twin) in error_samples().into_iter().zip(error_samples()) { - let (_, local, _) = error_to_sqlstate(&err); - let rebuilt = crate::Error::from(encode(twin)); - let (_, remote, _) = error_to_sqlstate(&rebuilt); - assert_eq!( - class(remote), - class(local), - "{name}: {err:?} is {local} locally but {remote} after the hop as {rebuilt:?}" - ); - } - } -} - -/// The SQLSTATE each Control-Plane variant renders where it has a class of -/// its own, pinned by variant index. -fn classified_sqlstates() -> Vec<(usize, &'static str)> { - vec![ - (3, sqlstate::INVALID_PARAMETER_VALUE), - (27, sqlstate::TOO_MANY_CONNECTIONS), - (28, sqlstate::SERIALIZATION_FAILURE), - (29, sqlstate::SYNTAX_ERROR), - (30, sqlstate::SYNTAX_ERROR), - (31, sqlstate::SYNTAX_ERROR), - (32, sqlstate::ACTIVE_SQL_TRANSACTION), - (34, sqlstate::QUERY_CANCELED.0), - (44, sqlstate::QUOTA_OVERCOMMIT), - (62, sqlstate::SYNTAX_ERROR), - (63, sqlstate::SYNTAX_ERROR), - (87, sqlstate::SYNTAX_ERROR), - (88, sqlstate::DEPENDENT_OBJECTS_STILL_EXIST), - (90, sqlstate::ACTIVE_SQL_TRANSACTION), - (91, sqlstate::SYNTAX_ERROR), - (92, sqlstate::SYNTAX_ERROR), - (93, sqlstate::SYNTAX_ERROR), - (95, sqlstate::SYNTAX_ERROR), - (96, sqlstate::SYNTAX_ERROR), - (97, sqlstate::SYNTAX_ERROR), - (98, sqlstate::SYNTAX_ERROR), - (99, sqlstate::SYNTAX_ERROR), - (100, sqlstate::SYNTAX_ERROR), - (101, sqlstate::QUOTA_EXCEEDED), - (102, sqlstate::QUOTA_EXCEEDED), - (103, sqlstate::SYNTAX_ERROR), - (104, sqlstate::SYNTAX_ERROR), - (106, sqlstate::READ_ONLY_SQL_TRANSACTION), - (107, sqlstate::STALE_READ_NOT_LEADER), - (108, sqlstate::DEPENDENT_OBJECTS_STILL_EXIST), - (109, sqlstate::DEPENDENT_OBJECTS_STILL_EXIST), - ] -} - -/// A client-facing Control-Plane variant renders its own SQLSTATE, never the -/// internal-error default. -#[test] -fn client_facing_variants_render_their_own_sqlstate() { - let expected = classified_sqlstates(); - let mut seen = 0; - for err in error_samples() { - let index = error_variant_index(&err); - if let Some((_, state)) = expected.iter().find(|(i, _)| *i == index) { - assert_eq!(error_to_sqlstate(&err).1, *state, "{err:?}"); - seen += 1; - } - } - assert_eq!(seen, expected.len(), "a pinned variant has no sample"); -} - -/// A variant with a dedicated public code renders that code's class on the -/// numeric table too, so native and remote renderings agree with pgwire. -#[test] -fn dedicated_codes_render_the_class_of_their_variant() { - use nodedb_types::error::ErrorCode as Ec; - - assert_eq!( - numeric_code_to_sqlstate(Ec::QUOTA_OVERCOMMIT), - sqlstate::QUOTA_OVERCOMMIT - ); - assert_eq!( - numeric_code_to_sqlstate(Ec::TENANT_VECTOR_DIM_EXCEEDED), - sqlstate::QUOTA_EXCEEDED - ); - assert_eq!( - numeric_code_to_sqlstate(Ec::TENANT_GRAPH_DEPTH_EXCEEDED), - sqlstate::QUOTA_EXCEEDED - ); - assert_eq!( - numeric_code_to_sqlstate(Ec::MIRROR_READ_ONLY), - sqlstate::READ_ONLY_SQL_TRANSACTION - ); - assert_eq!( - numeric_code_to_sqlstate(Ec::STALE_READ_NOT_LEADER), - sqlstate::STALE_READ_NOT_LEADER - ); - assert_eq!( - numeric_code_to_sqlstate(Ec::TRANSACTION_ROLLBACK), - sqlstate::TRANSACTION_ROLLBACK - ); - assert_eq!( - numeric_code_to_sqlstate(Ec::ACTIVE_SQL_TRANSACTION), - sqlstate::ACTIVE_SQL_TRANSACTION - ); - assert_eq!( - numeric_code_to_sqlstate(Ec::DEPENDENT_OBJECTS_EXIST), - sqlstate::DEPENDENT_OBJECTS_STILL_EXIST - ); -} - -/// The transaction-state and dependency variants render their exact -/// SQLSTATE after a node hop through both encoders, and carry the public -/// code of that class. -#[test] -fn transaction_and_dependency_variants_keep_their_sqlstate_across_a_hop() { - use nodedb_cluster::rpc_codec::TypedClusterError; - use nodedb_types::error::ErrorCode as Ec; - - use crate::control::cluster::data_plane_error_wire::execution_error_to_typed; - - let cases: [ClassCase; 6] = [ - ( - || crate::Error::CalvinParticipantError, - sqlstate::TRANSACTION_ROLLBACK, - Ec::TRANSACTION_ROLLBACK, - ), - ( - || crate::Error::NotInTransactionBlock { - statement: "VACUUM".into(), - }, - sqlstate::ACTIVE_SQL_TRANSACTION, - Ec::ACTIVE_SQL_TRANSACTION, - ), - ( - || crate::Error::CrdtApplyForbiddenInTransaction, - sqlstate::ACTIVE_SQL_TRANSACTION, - Ec::ACTIVE_SQL_TRANSACTION, - ), - ( - || crate::Error::CrossShardInExplicitTransaction, - sqlstate::ACTIVE_SQL_TRANSACTION, - Ec::ACTIVE_SQL_TRANSACTION, - ), - ( - || crate::Error::DependentObjectsExist { - tenant_id: 1, - root_kind: "collection", - root_name: "c".into(), - dependent_count: 1, - dependents: vec![("view".into(), "v".into())], - }, - sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, - Ec::DEPENDENT_OBJECTS_EXIST, - ), - ( - || crate::Error::RoleInUse { - role: "analyst".into(), - dependents: crate::control::security::role_assignment::RoleDependents::ChildRoles( - vec!["junior".into()], - ), - }, - sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, - Ec::DEPENDENT_OBJECTS_EXIST, - ), - ]; - let encoders: [HopEncoder; 2] = [ - ("execution_error_to_typed", execution_error_to_typed), - ("From", TypedClusterError::from), - ]; - for (make, state, code) in cases { - let err = make(); - assert_eq!(error_to_sqlstate(&err).1, state, "{err:?} locally"); - assert_eq!(native_error_fields(&err).code, code, "{err:?} native code"); - for (name, encode) in encoders { - let rebuilt = crate::Error::from(encode(make())); - assert_eq!( - error_to_sqlstate(&rebuilt).1, - state, - "{name}: {err:?} after the hop as {rebuilt:?}" - ); - } - - // Across the SPSC bridge: the Data-Plane code the variant becomes. - let bridged = ErrorCode::from(make()); - let on_bridge = crate::Error::DataPlane(bridged.clone()); - assert_eq!( - error_to_sqlstate(&on_bridge).1, - state, - "{err:?} across the bridge as {bridged:?}" - ); - assert_eq!( - native_error_fields(&on_bridge).code, - code, - "{err:?} native code across the bridge" - ); - - // Across the cluster Data-Plane wire: the code survives verbatim. - let wire = nodedb_cluster::rpc_codec::DataPlaneErrorCode::from(bridged.clone()); - let back = ErrorCode::from(wire); - assert_eq!(back, bridged, "{err:?} across the cluster wire"); - assert_eq!( - error_to_sqlstate(&crate::Error::DataPlane(back)).1, - state, - "{err:?} after the cluster wire" - ); - } -} - -/// Each transaction-state and dependency Data-Plane code crosses the cluster -/// wire verbatim, and renders one SQLSTATE on both sides. -#[test] -fn transaction_and_dependency_codes_roundtrip_the_cluster_wire() { - use nodedb_cluster::rpc_codec::DataPlaneErrorCode; - - let cases = [ - ( - ErrorCode::TransactionRollback { - detail: "participant aborted".into(), - }, - sqlstate::TRANSACTION_ROLLBACK, - ), - ( - ErrorCode::ActiveSqlTransaction { - detail: "VACUUM cannot run inside a transaction block".into(), - }, - sqlstate::ACTIVE_SQL_TRANSACTION, - ), - ( - ErrorCode::DependentObjectsExist { - object: "collection 'c'".into(), - detail: "cannot drop collection 'c': 1 dependent(s) exist (view:v)".into(), - }, - sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, - ), - ]; - for (code, state) in cases { - let local = crate::Error::DataPlane(code.clone()); - assert_eq!(error_to_sqlstate(&local).1, state, "{code:?} locally"); - let back = ErrorCode::from(DataPlaneErrorCode::from(code.clone())); - assert_eq!(back, code, "{code:?} across the cluster wire"); - let remote = crate::Error::DataPlane(back); - assert_eq!(error_to_sqlstate(&remote).1, state, "{code:?} remotely"); - assert_eq!( - native_error_fields(&remote).code, - native_error_fields(&local).code, - "{code:?} native code" - ); - } -} - -/// A DDL error keeps its exact SQLSTATE and code, including a SQLSTATE no -/// named constant covers. -#[test] -fn a_ddl_error_keeps_its_exact_sqlstate_and_code() { - use crate::control::server::shared::ddl::DdlError; - - for state in ["42710", "42P07", sqlstate::INSUFFICIENT_PRIVILEGE, "57014"] { - let ddl = if state == "57014" { - DdlError::from_error(&crate::Error::DeadlineExceeded { - request_id: crate::types::RequestId::new(1), - }) - } else { - DdlError::new(state, "refused") - }; - let expected_code = ddl.code; - let err = crate::Error::from(ddl); - assert_eq!(error_to_sqlstate(&err).1, state, "{err:?}"); - let native = native_error_fields(&err); - assert_eq!(native.sqlstate, state, "{err:?} native SQLSTATE"); - assert_eq!(native.code, expected_code, "{err:?} native code"); - } -} diff --git a/nodedb/src/control/gateway/error_map/class_parity/code_parity.rs b/nodedb/src/control/gateway/error_map/class_parity/code_parity.rs new file mode 100644 index 000000000..217ad15f7 --- /dev/null +++ b/nodedb/src/control/gateway/error_map/class_parity/code_parity.rs @@ -0,0 +1,103 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Every Data-Plane `ErrorCode` answers one class on native, pgwire and HTTP. + +use nodedb_types::error::sqlstate; + +use crate::bridge::envelope::ErrorCode; +use crate::control::server::native::dispatch::{error_code_to_native, native_error_fields}; +use crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate; +use crate::control::server::pgwire::types::error_to_sqlstate; + +use super::super::GatewayErrorMap; +use super::code_samples::VARIANT_COUNT; +use super::code_samples::samples; +use super::code_samples::variant_index; +use super::support::class; + +#[test] +fn every_variant_has_a_sample() { + let mut seen = [false; VARIANT_COUNT]; + for code in samples() { + seen[variant_index(&code)] = true; + } + let missing: Vec = (0..VARIANT_COUNT).filter(|i| !seen[*i]).collect(); + assert!(missing.is_empty(), "variants with no sample: {missing:?}"); +} + +/// The native frame carries pgwire's SQLSTATE, and its numeric code has the +/// same SQLSTATE class, on both native renderings: the typed `Err` and the +/// raw response frame. +#[test] +fn every_data_plane_code_has_one_class_on_native_and_pgwire() { + for code in samples() { + let err = crate::Error::DataPlane(code.clone()); + let (_, pg_state, _) = error_to_sqlstate(&err); + + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, pg_state, "native SQLSTATE for {code:?}"); + let native_state = numeric_code_to_sqlstate(native.code); + assert_eq!( + class(native_state), + class(pg_state), + "{code:?}: pgwire sends {pg_state}, native code {} renders {native_state}", + native.code + ); + + let frame = error_code_to_native(1, Some(&code)); + let payload = frame.error.expect("error frames carry a payload"); + assert_eq!( + payload.code, pg_state, + "response-frame SQLSTATE for {code:?}" + ); + assert_eq!( + payload.ndb_code, native.code.0, + "response-frame code for {code:?}" + ); + } +} + +/// A classified Data-Plane verdict never reads as a server fault over HTTP. +#[test] +fn classified_data_plane_codes_are_not_http_500() { + for code in samples() { + let err = crate::Error::DataPlane(code.clone()); + let (_, pg_state, _) = error_to_sqlstate(&err); + let (status, _) = GatewayErrorMap::to_http(&err); + if pg_state == sqlstate::INTERNAL_ERROR { + assert_eq!(status, 500, "{code:?}"); + } else { + assert_ne!(status, 500, "{code:?} is {pg_state} on pgwire"); + } + } +} + +/// The SQLSTATE status table agrees with the gateway status table for every +/// Data-Plane code. A DDL error and a query error of one class answer one +/// HTTP status. +#[test] +fn sqlstate_status_agrees_with_the_gateway_status() { + for code in samples() { + let err = crate::Error::DataPlane(code.clone()); + let (_, pg_state, _) = error_to_sqlstate(&err); + let (status, _) = GatewayErrorMap::to_http(&err); + assert_eq!( + GatewayErrorMap::sqlstate_to_http(pg_state), + status, + "{code:?} is {pg_state} on pgwire" + ); + } +} + +/// `Unsupported` is feature-not-supported on every surface. +#[test] +fn unsupported_is_feature_not_supported_everywhere() { + let err = crate::Error::DataPlane(ErrorCode::Unsupported { + detail: "not on this engine".into(), + }); + let (_, pg_state, _) = error_to_sqlstate(&err); + assert_eq!(pg_state, sqlstate::FEATURE_NOT_SUPPORTED); + let native = native_error_fields(&err); + assert_eq!(native.code, nodedb_types::error::ErrorCode::SQL_NOT_ENABLED); + assert_eq!(GatewayErrorMap::to_http(&err).0, 501); +} diff --git a/nodedb/src/control/gateway/error_map/class_parity/code_samples.rs b/nodedb/src/control/gateway/error_map/class_parity/code_samples.rs new file mode 100644 index 000000000..6c677fd60 --- /dev/null +++ b/nodedb/src/control/gateway/error_map/class_parity/code_samples.rs @@ -0,0 +1,209 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One sample per Data-Plane `ErrorCode` variant. + +use nodedb_types::sync::violation::ViolationType; +use nodedb_types::sync::wire::SyncProvenance; + +use crate::bridge::envelope::{CounterFault, ErrorCode, SyncHold}; + +/// The number of `ErrorCode` variants [`variant_index`] numbers. +pub(super) const VARIANT_COUNT: usize = 43; + +/// A dense index per variant. Exhaustive, so a new variant fails to compile +/// here until it gets an index, and [`every_variant_has_a_sample`] then fails +/// until [`samples`] carries it. +pub(super) fn variant_index(code: &ErrorCode) -> usize { + match code { + ErrorCode::DeadlineExceeded => 0, + ErrorCode::RejectedConstraint { .. } => 1, + ErrorCode::RejectedPrevalidation { .. } => 2, + ErrorCode::RetryableRefusal { .. } => 3, + ErrorCode::SyncRejected { .. } => 4, + ErrorCode::SyncNotApplied { .. } => 5, + ErrorCode::NotFound => 6, + ErrorCode::RejectedAuthz { .. } => 7, + ErrorCode::ConflictRetry => 8, + ErrorCode::CrdtFrontierMismatch { .. } => 9, + ErrorCode::ResourcesExhausted => 11, + ErrorCode::RejectedDanglingEdge { .. } => 12, + ErrorCode::DuplicateWrite => 13, + ErrorCode::AppendOnlyViolation { .. } => 14, + ErrorCode::BalanceViolation { .. } => 15, + ErrorCode::PeriodLocked { .. } => 16, + ErrorCode::PeriodLockMisconfigured { .. } => 17, + ErrorCode::RetentionViolation { .. } => 18, + ErrorCode::LegalHoldActive { .. } => 19, + ErrorCode::StateTransitionViolation { .. } => 20, + ErrorCode::TransitionCheckViolation { .. } => 21, + ErrorCode::TypeGuardViolation { .. } => 22, + ErrorCode::TypeMismatch { .. } => 23, + ErrorCode::CounterFault { .. } => 24, + ErrorCode::InsufficientBalance { .. } => 25, + ErrorCode::RateExceeded { .. } => 26, + ErrorCode::CollectionDraining { .. } => 27, + ErrorCode::RecursionDepthExceeded { .. } => 28, + ErrorCode::UndefinedColumn { .. } => 29, + ErrorCode::Internal { .. } => 30, + ErrorCode::Unsupported { .. } => 31, + ErrorCode::RollbackFailed { .. } => 32, + ErrorCode::OllpRetryRequired => 33, + ErrorCode::TxnOverlayMemoryExceeded { .. } => 34, + ErrorCode::DivisionByZero => 35, + ErrorCode::UndefinedFunction { .. } => 36, + ErrorCode::DataException { .. } => 37, + ErrorCode::DispatchCapacity { .. } => 38, + ErrorCode::ExpiredBeforeExecution => 39, + ErrorCode::BadRequest { .. } => 40, + ErrorCode::TransactionRollback { .. } => 41, + ErrorCode::ActiveSqlTransaction { .. } => 42, + ErrorCode::DependentObjectsExist { .. } => 10, + } +} + +fn provenance() -> SyncProvenance { + SyncProvenance { + producer_id: 1, + epoch: 1, + stream_id: 1, + seq: 1, + } +} + +/// One sample per variant, plus one per value that picks a different +/// SQLSTATE: each constraint kind and each counter fault. +pub(super) fn samples() -> Vec { + let text = || "detail".to_owned(); + let collection = || "c".to_owned(); + let mut samples = vec![ + ErrorCode::DeadlineExceeded, + ErrorCode::RejectedPrevalidation { reason: text() }, + ErrorCode::RetryableRefusal { reason: text() }, + ErrorCode::SyncRejected { + violation: ViolationType::PermissionDenied, + applied_seq: 1, + provenance: provenance(), + }, + ErrorCode::SyncRejected { + violation: ViolationType::RateLimited, + applied_seq: 1, + provenance: provenance(), + }, + ErrorCode::SyncNotApplied { + hold: SyncHold::Gap { expected: 2 }, + applied_seq: 1, + }, + ErrorCode::NotFound, + ErrorCode::RejectedAuthz { resource: text() }, + ErrorCode::ConflictRetry, + ErrorCode::CrdtFrontierMismatch { + expected: [0; 32], + actual: [1; 32], + }, + ErrorCode::ResourcesExhausted, + ErrorCode::RejectedDanglingEdge { + missing_node: text(), + }, + ErrorCode::DuplicateWrite, + ErrorCode::AppendOnlyViolation { + collection: collection(), + }, + ErrorCode::BalanceViolation { + collection: collection(), + detail: text(), + }, + ErrorCode::PeriodLocked { + collection: collection(), + }, + ErrorCode::PeriodLockMisconfigured { + collection: collection(), + ref_table: "periods".into(), + status_column: "status".into(), + row_identity: "p1".into(), + }, + ErrorCode::RetentionViolation { + collection: collection(), + }, + ErrorCode::LegalHoldActive { + collection: collection(), + }, + ErrorCode::StateTransitionViolation { + collection: collection(), + detail: text(), + }, + ErrorCode::TransitionCheckViolation { + collection: collection(), + detail: text(), + }, + ErrorCode::TypeGuardViolation { + collection: collection(), + detail: text(), + }, + ErrorCode::TypeMismatch { + collection: collection(), + detail: text(), + }, + ErrorCode::InsufficientBalance { + collection: collection(), + detail: text(), + }, + ErrorCode::RateExceeded { + gate: "g".into(), + retry_after_ms: 10, + }, + ErrorCode::CollectionDraining { + collection: collection(), + }, + ErrorCode::RecursionDepthExceeded { + cte_name: "walk".into(), + max_depth: 100, + }, + ErrorCode::UndefinedColumn { column: "x".into() }, + ErrorCode::Internal { detail: text() }, + ErrorCode::Unsupported { detail: text() }, + ErrorCode::RollbackFailed { + entry_index: 0, + detail: text(), + }, + ErrorCode::OllpRetryRequired, + ErrorCode::TxnOverlayMemoryExceeded { limit: 1 << 20 }, + ErrorCode::DivisionByZero, + ErrorCode::UndefinedFunction { name: "f".into() }, + ErrorCode::DataException { detail: text() }, + ErrorCode::DispatchCapacity { reason: text() }, + ErrorCode::ExpiredBeforeExecution, + ErrorCode::BadRequest { detail: text() }, + ErrorCode::TransactionRollback { detail: text() }, + ErrorCode::ActiveSqlTransaction { detail: text() }, + ErrorCode::DependentObjectsExist { + object: "role \"analyst\"".into(), + detail: text(), + }, + ]; + for constraint in [ + "not_null", + "unique", + "generated_always", + "fk_missing", + "rls_policy", + "permission_denied", + "check", + ] { + samples.push(ErrorCode::RejectedConstraint { + constraint: constraint.into(), + detail: text(), + }); + } + for fault in [ + CounterFault::NotAnInteger, + CounterFault::NotAFloat, + CounterFault::IntegerOverflow, + CounterFault::NonFinite, + ] { + samples.push(ErrorCode::CounterFault { + collection: collection(), + fault, + }); + } + samples +} diff --git a/nodedb/src/control/gateway/error_map/class_parity/control_plane.rs b/nodedb/src/control/gateway/error_map/class_parity/control_plane.rs new file mode 100644 index 000000000..f3e2d2ce1 --- /dev/null +++ b/nodedb/src/control/gateway/error_map/class_parity/control_plane.rs @@ -0,0 +1,124 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Control-Plane errors a client acts on answer one class on every surface. + +use nodedb_types::error::sqlstate; + +use crate::control::server::native::dispatch::native_error_fields; +use crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate; +use crate::control::server::pgwire::types::error_to_sqlstate; + +use super::super::GatewayErrorMap; +use super::support::class; + +/// Control-Plane errors a client acts on. Each has a class of its own. +fn control_plane_samples() -> Vec { + vec![ + crate::Error::RetryableSchemaChanged { + descriptor: "orders".into(), + }, + crate::Error::SessionTokenExpired, + ] +} + +/// A Control-Plane error has one class on native and pgwire, and its native +/// numeric code renders in that class. +#[test] +fn control_plane_errors_have_one_class_on_native_and_pgwire() { + for err in control_plane_samples() { + let (_, pg_state, _) = error_to_sqlstate(&err); + assert_ne!(pg_state, sqlstate::INTERNAL_ERROR, "{err:?} has no class"); + + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, pg_state, "native SQLSTATE for {err:?}"); + let native_state = numeric_code_to_sqlstate(native.code); + assert_eq!( + class(native_state), + class(pg_state), + "{err:?}: pgwire sends {pg_state}, native code {} renders {native_state}", + native.code + ); + } +} + +/// A schema change the server cannot absorb is the retryable +/// serialization class on every surface. +#[test] +fn schema_change_is_a_retryable_serialization_failure() { + let err = crate::Error::RetryableSchemaChanged { + descriptor: "orders".into(), + }; + assert_eq!(error_to_sqlstate(&err).1, sqlstate::SERIALIZATION_FAILURE); + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, sqlstate::SERIALIZATION_FAILURE); + assert_eq!(native.code, nodedb_types::error::ErrorCode::WRITE_CONFLICT); + assert!(crate::error_classify::classify(&err).is_retriable()); + let status = GatewayErrorMap::to_http(&err).0; + assert_eq!(status, 409); + assert_eq!( + GatewayErrorMap::sqlstate_to_http(sqlstate::SERIALIZATION_FAILURE), + status + ); +} + +/// A committed write whose result is gone answers the internal class on every +/// surface, which no client retries, and says the write committed. +#[test] +fn committed_result_unavailable_is_never_retried() { + let err = crate::Error::CommittedResultUnavailable { + group_id: 1, + log_index: 2, + }; + let (_, pg_state, message) = error_to_sqlstate(&err); + assert_eq!(pg_state, sqlstate::INTERNAL_ERROR); + assert!(message.contains("committed"), "{message}"); + let native = native_error_fields(&err); + assert_eq!(native.code, nodedb_types::error::ErrorCode::INTERNAL); + assert_eq!( + class(numeric_code_to_sqlstate(native.code)), + class(pg_state) + ); + assert!(!crate::error_classify::classify(&err).is_retriable()); + assert!(!crate::error_classify::is_unclassified_failure(&err)); +} + +/// A write whose outcome is unknown answers the same internal class, which +/// no client retries, and says the outcome must be checked first. +#[test] +fn proposal_outcome_unknown_is_never_retried() { + let err = crate::Error::ProposalOutcomeUnknown { + group_id: 1, + log_index: 2, + }; + let (_, pg_state, message) = error_to_sqlstate(&err); + assert_eq!(pg_state, sqlstate::INTERNAL_ERROR); + assert!(message.contains("unknown"), "{message}"); + let native = native_error_fields(&err); + assert_eq!(native.code, nodedb_types::error::ErrorCode::INTERNAL); + assert_eq!( + class(numeric_code_to_sqlstate(native.code)), + class(pg_state) + ); + assert!(!crate::error_classify::classify(&err).is_retriable()); + assert!(!crate::error_classify::is_unclassified_failure(&err)); +} + +/// An expired session token is invalid authorization on every surface. +#[test] +fn expired_session_token_is_invalid_authorization_everywhere() { + let err = crate::Error::SessionTokenExpired; + assert_eq!(error_to_sqlstate(&err).1, sqlstate::AUTH_TOKEN_EXPIRED.0); + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, sqlstate::AUTH_TOKEN_EXPIRED.0); + assert_eq!(native.code, nodedb_types::error::ErrorCode::AUTH_EXPIRED); + assert_eq!( + numeric_code_to_sqlstate(native.code), + sqlstate::AUTH_TOKEN_EXPIRED.0 + ); + let status = GatewayErrorMap::to_http(&err).0; + assert_eq!(status, 401); + assert_eq!( + GatewayErrorMap::sqlstate_to_http(sqlstate::AUTH_TOKEN_EXPIRED.0), + status + ); +} diff --git a/nodedb/src/control/gateway/error_map/class_parity/error_index.rs b/nodedb/src/control/gateway/error_map/class_parity/error_index.rs new file mode 100644 index 000000000..9e6da3d79 --- /dev/null +++ b/nodedb/src/control/gateway/error_map/class_parity/error_index.rs @@ -0,0 +1,131 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A dense index per `crate::Error` variant. + +/// The number of `crate::Error` variants [`error_variant_index`] numbers. +pub(super) const ERROR_VARIANT_COUNT: usize = 115; + +/// A dense index per `crate::Error` variant. Exhaustive, so a new variant +/// fails to compile here until it gets an index, and +/// [`every_error_variant_has_a_sample`] then fails until +/// [`error_samples`] carries it. +pub(crate) fn error_variant_index(err: &crate::Error) -> usize { + use crate::Error as E; + match err { + E::RejectedConstraint { .. } => 0, + E::TxnOverlayMemoryExceeded { .. } => 1, + E::RejectedAuthz { .. } => 2, + E::OffsetRegression { .. } => 3, + E::DeadlineExceeded { .. } => 4, + E::ConflictRetry { .. } => 5, + E::CalvinSerializationConflict => 6, + E::CalvinParticipantError => 7, + E::RejectedPrevalidation { .. } => 8, + E::RetryableRefusal { .. } => 9, + E::AppendOnlyViolation { .. } => 10, + E::BalanceViolation { .. } => 11, + E::MaterializedSumTargetNotFound { .. } => 12, + E::MaterializedSumResolutionMissing { .. } => 13, + E::PeriodLocked { .. } => 14, + E::PeriodLockMisconfigured { .. } => 15, + E::RetentionViolation { .. } => 16, + E::LegalHoldActive { .. } => 17, + E::StateTransitionViolation { .. } => 18, + E::TransitionCheckViolation { .. } => 19, + E::TypeGuardViolation { .. } => 20, + E::TypeMismatch { .. } => 21, + E::InsufficientBalance { .. } => 22, + E::RateExceeded { .. } => 23, + E::CollectionNotFound { .. } => 24, + E::DocumentNotFound { .. } => 25, + E::CollectionDeactivated { .. } => 26, + E::VShardAdmissionCapacityExceeded { .. } => 27, + E::CrdtAdmissionRetriesExhausted { .. } => 28, + E::CrdtAdmissionInvalidPlan { .. } => 29, + E::CrdtAdmissionCallerFence => 30, + E::CrdtApplyRequiresAdmission => 31, + E::CrdtApplyForbiddenInTransaction => 32, + E::NotInTransactionBlock { .. } => 33, + E::CrdtAdmissionTimeout { .. } => 34, + E::NoLeader { .. } => 35, + E::NotLeader { .. } => 36, + E::CrossCollectionNotColocated { .. } => 38, + E::CloneWriteRequiresMaterialize { .. } => 40, + E::BadRequest { .. } => 41, + E::BackupTenantMismatch { .. } => 42, + E::BackupKeyMismatch => 43, + E::QuotaOvercommit { .. } => 44, + E::PlanError { .. } => 45, + E::FeatureNotSupported { .. } => 46, + E::UndefinedFunction { .. } => 47, + E::UndefinedObject { .. } => 48, + E::ObjectNotInPrerequisiteState { .. } => 49, + E::UndefinedColumn { .. } => 50, + E::AmbiguousColumn { .. } => 51, + E::UnknownStrictField { .. } => 52, + E::DivisionByZero => 53, + E::DataException { .. } => 54, + E::InvalidLimitValue { .. } => 55, + E::RetryableSchemaChanged { .. } => 56, + E::RetryableLeaderChange { .. } => 57, + E::GroupQuorumUnavailable { .. } => 58, + E::GroupMarksUnavailable { .. } => 59, + E::MetadataLeaderUnavailable => 60, + E::AuthorizationStateBehind { .. } => 61, + E::ExecutionLimitExceeded { .. } => 62, + E::LimitExceeded { .. } => 63, + E::Wal(_) => 64, + E::Dispatch { .. } => 65, + E::DispatchCapacity { .. } => 66, + E::Storage { .. } => 67, + E::ColdStorage { .. } => 68, + E::Serialization { .. } => 69, + E::Codec { .. } => 70, + E::SegmentCorrupted { .. } => 71, + E::MemoryExhausted { .. } => 72, + E::Backpressure { .. } => 73, + E::Crdt(_) => 74, + E::Io(_) => 75, + E::Config { .. } => 76, + E::Encryption { .. } => 77, + E::Bridge { .. } => 78, + E::VersionCompat { .. } => 79, + E::Internal { .. } => 80, + E::Shaping(_) => 81, + E::RemoteTyped { .. } => 82, + E::DescriptorVersionAnomaly { .. } => 83, + E::CollectionPurgeRowMissing { .. } => 84, + E::CatalogIntegrityViolation { .. } => 85, + E::DataPlane(_) => 86, + E::Promql(_) => 87, + E::DependentObjectsExist { .. } => 88, + E::CascadeCycle { .. } => 89, + E::CrossShardInExplicitTransaction => 90, + E::SequencerUnavailable => 91, + E::SessionCapExceeded { .. } => 92, + E::SessionIdleTimeout => 93, + E::SessionTokenExpired => 94, + E::SessionKilledByAdmin => 95, + E::SessionUserDropped => 96, + E::OidcProviderTenantUnbound => 97, + E::OidcProviderTenantUnavailable { .. } => 98, + E::ExternalRoleUndefined { .. } => 99, + E::OidcNoDefaultDatabase { .. } => 100, + E::TenantVectorDimExceeded { .. } => 101, + E::TenantGraphDepthExceeded { .. } => 102, + E::RoleInheritanceCycle { .. } => 103, + E::RoleInheritanceDepthExceeded { .. } => 104, + E::OllpExhausted { .. } => 105, + E::MirrorReadOnly { .. } => 106, + E::StaleReadNotLeader { .. } => 107, + E::RoleInUse { .. } => 108, + E::Ddl(_) => 109, + E::LinearizableReadRefused { .. } => 110, + E::CommittedResultUnavailable { .. } => 111, + E::ProposalOutcomeUnknown { .. } => 39, + E::RestoreTargetNotEmpty { .. } => 112, + E::RestoreVerificationFailed { .. } => 113, + E::BackupCaptureMoved { .. } => 114, + E::CollectionUnstamped { .. } => 37, + } +} diff --git a/nodedb/src/control/gateway/error_map/class_parity/error_parity.rs b/nodedb/src/control/gateway/error_map/class_parity/error_parity.rs new file mode 100644 index 000000000..7c7909189 --- /dev/null +++ b/nodedb/src/control/gateway/error_map/class_parity/error_parity.rs @@ -0,0 +1,169 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Every `crate::Error` variant keeps its class on every surface and across a node hop. + +use nodedb_types::error::sqlstate; + +use crate::control::server::pgwire::types::error_map::numeric_code_to_sqlstate; +use crate::control::server::pgwire::types::error_to_sqlstate; + +use super::super::GatewayErrorMap; +use super::error_index::ERROR_VARIANT_COUNT; +use super::error_index::error_variant_index; +use super::error_samples::error_samples; +use super::support::HopEncoder; +use super::support::class; + +#[test] +fn every_error_variant_has_a_sample() { + let mut seen = [false; ERROR_VARIANT_COUNT]; + for err in error_samples() { + seen[error_variant_index(&err)] = true; + } + let missing: Vec = (0..ERROR_VARIANT_COUNT).filter(|i| !seen[*i]).collect(); + assert!( + missing.is_empty(), + "error variants with no sample: {missing:?}" + ); +} + +/// Every `crate::Error` variant answers the HTTP status its pgwire SQLSTATE +/// class has. Only an internal or system error reads as a 500. +#[test] +fn every_error_variant_has_the_http_status_of_its_sqlstate() { + for err in error_samples() { + let (_, pg_state, _) = error_to_sqlstate(&err); + let (status, _) = GatewayErrorMap::to_http(&err); + assert_eq!( + status, + GatewayErrorMap::sqlstate_to_http(pg_state), + "{err:?} is {pg_state} on pgwire" + ); + let server_fault = matches!(class(pg_state), "XX" | "58"); + assert_eq!( + status == 500, + server_fault, + "{err:?} is {pg_state} on pgwire but HTTP {status}" + ); + } +} + +/// Every `crate::Error` variant renders the SQLSTATE class it renders locally +/// after it crosses a node hop, through both wire encoders and the decoder. +#[test] +fn every_error_variant_keeps_its_class_across_a_node_hop() { + use nodedb_cluster::rpc_codec::TypedClusterError; + + use crate::control::cluster::data_plane_error_wire::execution_error_to_typed; + + let encoders: [HopEncoder; 2] = [ + ("execution_error_to_typed", execution_error_to_typed), + ("From", TypedClusterError::from), + ]; + for (name, encode) in encoders { + for (err, twin) in error_samples().into_iter().zip(error_samples()) { + let (_, local, _) = error_to_sqlstate(&err); + let rebuilt = crate::Error::from(encode(twin)); + let (_, remote, _) = error_to_sqlstate(&rebuilt); + assert_eq!( + class(remote), + class(local), + "{name}: {err:?} is {local} locally but {remote} after the hop as {rebuilt:?}" + ); + } + } +} + +/// The SQLSTATE each Control-Plane variant renders where it has a class of +/// its own, pinned by variant index. +fn classified_sqlstates() -> Vec<(usize, &'static str)> { + vec![ + (3, sqlstate::INVALID_PARAMETER_VALUE), + (27, sqlstate::TOO_MANY_CONNECTIONS), + (28, sqlstate::SERIALIZATION_FAILURE), + (29, sqlstate::SYNTAX_ERROR), + (30, sqlstate::SYNTAX_ERROR), + (31, sqlstate::SYNTAX_ERROR), + (32, sqlstate::ACTIVE_SQL_TRANSACTION), + (34, sqlstate::QUERY_CANCELED.0), + (44, sqlstate::QUOTA_OVERCOMMIT), + (62, sqlstate::SYNTAX_ERROR), + (63, sqlstate::SYNTAX_ERROR), + (87, sqlstate::SYNTAX_ERROR), + (88, sqlstate::DEPENDENT_OBJECTS_STILL_EXIST), + (90, sqlstate::ACTIVE_SQL_TRANSACTION), + (91, sqlstate::SYNTAX_ERROR), + (92, sqlstate::SYNTAX_ERROR), + (93, sqlstate::SYNTAX_ERROR), + (95, sqlstate::SYNTAX_ERROR), + (96, sqlstate::SYNTAX_ERROR), + (97, sqlstate::SYNTAX_ERROR), + (98, sqlstate::SYNTAX_ERROR), + (99, sqlstate::SYNTAX_ERROR), + (100, sqlstate::SYNTAX_ERROR), + (101, sqlstate::QUOTA_EXCEEDED), + (102, sqlstate::QUOTA_EXCEEDED), + (103, sqlstate::SYNTAX_ERROR), + (104, sqlstate::SYNTAX_ERROR), + (106, sqlstate::READ_ONLY_SQL_TRANSACTION), + (107, sqlstate::STALE_READ_NOT_LEADER), + (108, sqlstate::DEPENDENT_OBJECTS_STILL_EXIST), + (109, sqlstate::DEPENDENT_OBJECTS_STILL_EXIST), + ] +} + +/// A client-facing Control-Plane variant renders its own SQLSTATE, never the +/// internal-error default. +#[test] +fn client_facing_variants_render_their_own_sqlstate() { + let expected = classified_sqlstates(); + let mut seen = 0; + for err in error_samples() { + let index = error_variant_index(&err); + if let Some((_, state)) = expected.iter().find(|(i, _)| *i == index) { + assert_eq!(error_to_sqlstate(&err).1, *state, "{err:?}"); + seen += 1; + } + } + assert_eq!(seen, expected.len(), "a pinned variant has no sample"); +} + +/// A variant with a dedicated public code renders that code's class on the +/// numeric table too, so native and remote renderings agree with pgwire. +#[test] +fn dedicated_codes_render_the_class_of_their_variant() { + use nodedb_types::error::ErrorCode as Ec; + + assert_eq!( + numeric_code_to_sqlstate(Ec::QUOTA_OVERCOMMIT), + sqlstate::QUOTA_OVERCOMMIT + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::TENANT_VECTOR_DIM_EXCEEDED), + sqlstate::QUOTA_EXCEEDED + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::TENANT_GRAPH_DEPTH_EXCEEDED), + sqlstate::QUOTA_EXCEEDED + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::MIRROR_READ_ONLY), + sqlstate::READ_ONLY_SQL_TRANSACTION + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::STALE_READ_NOT_LEADER), + sqlstate::STALE_READ_NOT_LEADER + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::TRANSACTION_ROLLBACK), + sqlstate::TRANSACTION_ROLLBACK + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::ACTIVE_SQL_TRANSACTION), + sqlstate::ACTIVE_SQL_TRANSACTION + ); + assert_eq!( + numeric_code_to_sqlstate(Ec::DEPENDENT_OBJECTS_EXIST), + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST + ); +} diff --git a/nodedb/src/control/gateway/error_map/class_parity/error_samples.rs b/nodedb/src/control/gateway/error_map/class_parity/error_samples.rs new file mode 100644 index 000000000..27c125fce --- /dev/null +++ b/nodedb/src/control/gateway/error_map/class_parity/error_samples.rs @@ -0,0 +1,387 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One sample per `crate::Error` variant. + +use nodedb_types::error::sqlstate; + +use crate::bridge::envelope::ErrorCode; + +/// One sample per `crate::Error` variant. +pub(crate) fn error_samples() -> Vec { + use crate::Error as E; + use crate::types::{RequestId, TenantId, VShardId}; + + let text = || "detail".to_owned(); + let collection = || "c".to_owned(); + vec![ + E::RejectedConstraint { + collection: collection(), + constraint: "unique".into(), + detail: text(), + }, + E::TxnOverlayMemoryExceeded { limit: 1 << 20 }, + E::RejectedAuthz { + tenant_id: TenantId::new(1), + resource: text(), + }, + E::OffsetRegression { + stream: "s".into(), + group: "g".into(), + partition_id: 0, + offsets: Box::new(crate::error::RegressedOffsets { + current: crate::event::cdc::CdcOffset { + epoch: 0, + index: 2, + sequence: 2, + }, + attempted: crate::event::cdc::CdcOffset { + epoch: 0, + index: 1, + sequence: 1, + }, + }), + }, + E::DeadlineExceeded { + request_id: RequestId::new(1), + }, + E::ConflictRetry { + collection: collection(), + document_id: "d".into(), + }, + E::CalvinSerializationConflict, + E::CalvinParticipantError, + E::RejectedPrevalidation { + constraint: "check".into(), + reason: text(), + }, + E::RetryableRefusal { reason: text() }, + E::AppendOnlyViolation { + collection: collection(), + detail: text(), + }, + E::BalanceViolation { + collection: collection(), + detail: text(), + }, + E::MaterializedSumTargetNotFound { + target_collection: "t".into(), + join_column: "k".into(), + join_value: "1".into(), + }, + E::MaterializedSumResolutionMissing { + target_collection: "t".into(), + join_column: "k".into(), + join_value: "1".into(), + }, + E::PeriodLocked { + collection: collection(), + detail: text(), + }, + E::PeriodLockMisconfigured { + collection: collection(), + ref_table: "periods".into(), + status_column: "status".into(), + row_identity: "p1".into(), + }, + E::RetentionViolation { + collection: collection(), + detail: text(), + }, + E::LegalHoldActive { + collection: collection(), + detail: text(), + }, + E::StateTransitionViolation { + collection: collection(), + detail: text(), + }, + E::TransitionCheckViolation { + collection: collection(), + detail: text(), + }, + E::TypeGuardViolation { + collection: collection(), + detail: text(), + }, + E::TypeMismatch { + collection: collection(), + key: "k".into(), + detail: text(), + }, + E::InsufficientBalance { + collection: collection(), + key: "k".into(), + detail: text(), + }, + E::RateExceeded { + gate: "g".into(), + detail: text(), + retry_after_ms: 10, + }, + E::CollectionNotFound { + tenant_id: TenantId::new(1), + collection: collection(), + }, + E::DocumentNotFound { + collection: collection(), + document_id: "d".into(), + }, + E::CollectionDeactivated { + tenant_id: TenantId::new(1), + collection: collection(), + retention_expires_at_ns: 1, + }, + E::VShardAdmissionCapacityExceeded { + vshard_id: VShardId::new(1), + capacity: 4, + }, + E::CrdtAdmissionRetriesExhausted { + vshard_id: VShardId::new(1), + attempts: 3, + }, + E::CrdtAdmissionInvalidPlan { reason: "empty" }, + E::CrdtAdmissionCallerFence, + E::CrdtApplyRequiresAdmission, + E::CrdtApplyForbiddenInTransaction, + E::NotInTransactionBlock { + statement: "VACUUM".into(), + }, + E::CrdtAdmissionTimeout { + vshard_id: VShardId::new(1), + timeout_ms: 10, + }, + E::NoLeader { + vshard_id: VShardId::new(1), + }, + E::NotLeader { + vshard_id: VShardId::new(1), + leader_node: 2, + leader_addr: "10.0.0.1:9000".into(), + leader_term: 3, + }, + E::CrossCollectionNotColocated { + op: "insert-select", + source_collection: "a".into(), + target_collection: "b".into(), + }, + E::CloneWriteRequiresMaterialize { + collection: collection(), + engine: "kv".into(), + database: "db".into(), + reason: "shadowed", + }, + E::BadRequest { detail: text() }, + E::BackupTenantMismatch { + expected: 1, + actual: 2, + }, + E::BackupKeyMismatch, + E::QuotaOvercommit { + field: "max_storage".into(), + detail: text(), + }, + E::PlanError { detail: text() }, + E::FeatureNotSupported { detail: text() }, + E::UndefinedFunction { name: "f".into() }, + E::UndefinedObject { + kind: "sequence", + name: "s".into(), + }, + E::ObjectNotInPrerequisiteState { + object: "s".into(), + detail: text(), + }, + E::UndefinedColumn { column: "x".into() }, + E::AmbiguousColumn { + column: "id".into(), + }, + E::UnknownStrictField { + collection: collection(), + column: "x".into(), + }, + E::DivisionByZero, + E::DataException { detail: text() }, + E::InvalidLimitValue { + clause: "LIMIT", + value: "-1".into(), + }, + E::RetryableSchemaChanged { + descriptor: "orders".into(), + }, + E::RetryableLeaderChange { + group_id: 1, + log_index: 2, + }, + E::CommittedResultUnavailable { + group_id: 1, + log_index: 2, + }, + E::ProposalOutcomeUnknown { + group_id: 1, + log_index: 2, + }, + E::GroupQuorumUnavailable { + group_id: 1, + voters: vec![1, 2, 3], + unreachable: vec![2, 3], + }, + E::GroupMarksUnavailable { + group_id: 1, + refused_by: vec![2], + }, + E::BackupCaptureMoved { group_id: 1 }, + E::MetadataLeaderUnavailable, + E::AuthorizationStateBehind { detail: text() }, + E::LinearizableReadRefused { + group_id: 1, + detail: text(), + }, + E::ExecutionLimitExceeded { detail: text() }, + E::LimitExceeded { + limit_name: "max_rows", + value: 10, + max: 5, + }, + E::Wal(nodedb_wal::WalError::Sealed), + E::Dispatch { detail: text() }, + E::DispatchCapacity { + scope: crate::DispatchCapacityScope::QueueFull { + core_id: 0, + capacity: 4, + }, + }, + E::Storage { + engine: "kv".into(), + detail: text(), + }, + E::ColdStorage { detail: text() }, + E::Serialization { + format: "msgpack".into(), + detail: text(), + }, + E::Codec { detail: text() }, + E::SegmentCorrupted { detail: text() }, + E::MemoryExhausted { + engine: "kv".into(), + }, + E::Backpressure { + engine: nodedb_mem::EngineId::Vector, + }, + E::Crdt(nodedb_crdt::CrdtError::ConstraintViolation { + constraint: "unique".into(), + collection: collection(), + detail: text(), + }), + E::Io(std::io::Error::other("disk")), + E::Config { detail: text() }, + E::Encryption { detail: text() }, + E::Bridge { detail: text() }, + E::VersionCompat { detail: text() }, + E::RestoreTargetNotEmpty { + path: std::path::PathBuf::from("/restore"), + }, + E::RestoreVerificationFailed { + phase: nodedb_types::backup_envelope::VerificationPhase::Destination, + mismatches: vec![nodedb_types::backup_envelope::VerificationMismatch { + database_id: 0, + collection: collection(), + part: nodedb_types::backup_envelope::VerifiedPart::Documents, + expected: nodedb_types::backup_envelope::VerifiedTally::default(), + found: nodedb_types::backup_envelope::VerifiedTally::default(), + }], + }, + E::Internal { detail: text() }, + E::Shaping(Box::new(nodedb_types::NodeDbError::bad_request(text()))), + E::RemoteTyped { + code: nodedb_types::error::ErrorCode::WRITE_CONFLICT, + message: text(), + }, + E::DescriptorVersionAnomaly { + descriptor: "orders".into(), + carried: 5, + prior: 2, + }, + E::CollectionPurgeRowMissing { + database_id: 1, + tenant_id: 1, + name: collection(), + }, + E::CollectionUnstamped { + database_id: 1, + tenant_id: 1, + name: collection(), + }, + E::CatalogIntegrityViolation { + entry_kind: "PutCollection".into(), + detail: text(), + }, + E::DataPlane(ErrorCode::NotFound), + E::Promql(crate::control::promql::PromqlError::UnexpectedEof), + E::DependentObjectsExist { + tenant_id: 1, + root_kind: "collection", + root_name: collection(), + dependent_count: 1, + dependents: vec![("view".into(), "v".into())], + }, + E::CascadeCycle { + tenant_id: 1, + root: collection(), + depth: 64, + }, + E::CrossShardInExplicitTransaction, + E::SequencerUnavailable, + E::SessionCapExceeded { cap: 8 }, + E::SessionIdleTimeout, + E::SessionTokenExpired, + E::SessionKilledByAdmin, + E::SessionUserDropped, + E::OidcProviderTenantUnbound, + E::OidcProviderTenantUnavailable { tenant_id: 1 }, + E::ExternalRoleUndefined { + subject: "alice".into(), + role: "auditor".into(), + tenant_id: 1, + }, + E::OidcNoDefaultDatabase { + sub: "alice".into(), + }, + E::TenantVectorDimExceeded { + dim: 4096, + limit: 1024, + }, + E::TenantGraphDepthExceeded { + depth: 20, + limit: 10, + }, + E::RoleInheritanceCycle { + child: "a".into(), + parent: "b".into(), + }, + E::RoleInheritanceDepthExceeded { depth: 9, limit: 8 }, + E::OllpExhausted { + retries: 3, + cause: crate::OllpExhaustedCause::PredicateDrift, + }, + E::MirrorReadOnly { + database: "db".into(), + }, + E::StaleReadNotLeader { + database: "db".into(), + source_cluster: "src".into(), + detail: text(), + }, + E::RoleInUse { + role: "analyst".into(), + dependents: crate::control::security::role_assignment::RoleDependents::Users(vec![ + "bob".into(), + ]), + }, + E::Ddl(Box::new( + crate::control::server::shared::ddl::DdlError::new( + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + text(), + ), + )), + ] +} diff --git a/nodedb/src/control/gateway/error_map/class_parity/hop_parity.rs b/nodedb/src/control/gateway/error_map/class_parity/hop_parity.rs new file mode 100644 index 000000000..77c1f7805 --- /dev/null +++ b/nodedb/src/control/gateway/error_map/class_parity/hop_parity.rs @@ -0,0 +1,181 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Transaction-state, dependency and DDL errors keep their exact SQLSTATE across a node hop. + +use nodedb_types::error::sqlstate; + +use crate::bridge::envelope::ErrorCode; +use crate::control::server::native::dispatch::native_error_fields; +use crate::control::server::pgwire::types::error_to_sqlstate; + +use super::support::HopEncoder; + +/// An error builder, with the SQLSTATE and public code it must answer. +type ClassCase = ( + fn() -> crate::Error, + &'static str, + nodedb_types::error::ErrorCode, +); + +/// The transaction-state and dependency variants render their exact +/// SQLSTATE after a node hop through both encoders, and carry the public +/// code of that class. +#[test] +fn transaction_and_dependency_variants_keep_their_sqlstate_across_a_hop() { + use nodedb_cluster::rpc_codec::TypedClusterError; + use nodedb_types::error::ErrorCode as Ec; + + use crate::control::cluster::data_plane_error_wire::execution_error_to_typed; + + let cases: [ClassCase; 6] = [ + ( + || crate::Error::CalvinParticipantError, + sqlstate::TRANSACTION_ROLLBACK, + Ec::TRANSACTION_ROLLBACK, + ), + ( + || crate::Error::NotInTransactionBlock { + statement: "VACUUM".into(), + }, + sqlstate::ACTIVE_SQL_TRANSACTION, + Ec::ACTIVE_SQL_TRANSACTION, + ), + ( + || crate::Error::CrdtApplyForbiddenInTransaction, + sqlstate::ACTIVE_SQL_TRANSACTION, + Ec::ACTIVE_SQL_TRANSACTION, + ), + ( + || crate::Error::CrossShardInExplicitTransaction, + sqlstate::ACTIVE_SQL_TRANSACTION, + Ec::ACTIVE_SQL_TRANSACTION, + ), + ( + || crate::Error::DependentObjectsExist { + tenant_id: 1, + root_kind: "collection", + root_name: "c".into(), + dependent_count: 1, + dependents: vec![("view".into(), "v".into())], + }, + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + Ec::DEPENDENT_OBJECTS_EXIST, + ), + ( + || crate::Error::RoleInUse { + role: "analyst".into(), + dependents: crate::control::security::role_assignment::RoleDependents::ChildRoles( + vec!["junior".into()], + ), + }, + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + Ec::DEPENDENT_OBJECTS_EXIST, + ), + ]; + let encoders: [HopEncoder; 2] = [ + ("execution_error_to_typed", execution_error_to_typed), + ("From", TypedClusterError::from), + ]; + for (make, state, code) in cases { + let err = make(); + assert_eq!(error_to_sqlstate(&err).1, state, "{err:?} locally"); + assert_eq!(native_error_fields(&err).code, code, "{err:?} native code"); + for (name, encode) in encoders { + let rebuilt = crate::Error::from(encode(make())); + assert_eq!( + error_to_sqlstate(&rebuilt).1, + state, + "{name}: {err:?} after the hop as {rebuilt:?}" + ); + } + + // Across the SPSC bridge: the Data-Plane code the variant becomes. + let bridged = ErrorCode::from(make()); + let on_bridge = crate::Error::DataPlane(bridged.clone()); + assert_eq!( + error_to_sqlstate(&on_bridge).1, + state, + "{err:?} across the bridge as {bridged:?}" + ); + assert_eq!( + native_error_fields(&on_bridge).code, + code, + "{err:?} native code across the bridge" + ); + + // Across the cluster Data-Plane wire: the code survives verbatim. + let wire = nodedb_cluster::rpc_codec::DataPlaneErrorCode::from(bridged.clone()); + let back = ErrorCode::from(wire); + assert_eq!(back, bridged, "{err:?} across the cluster wire"); + assert_eq!( + error_to_sqlstate(&crate::Error::DataPlane(back)).1, + state, + "{err:?} after the cluster wire" + ); + } +} + +/// Each transaction-state and dependency Data-Plane code crosses the cluster +/// wire verbatim, and renders one SQLSTATE on both sides. +#[test] +fn transaction_and_dependency_codes_roundtrip_the_cluster_wire() { + use nodedb_cluster::rpc_codec::DataPlaneErrorCode; + + let cases = [ + ( + ErrorCode::TransactionRollback { + detail: "participant aborted".into(), + }, + sqlstate::TRANSACTION_ROLLBACK, + ), + ( + ErrorCode::ActiveSqlTransaction { + detail: "VACUUM cannot run inside a transaction block".into(), + }, + sqlstate::ACTIVE_SQL_TRANSACTION, + ), + ( + ErrorCode::DependentObjectsExist { + object: "collection 'c'".into(), + detail: "cannot drop collection 'c': 1 dependent(s) exist (view:v)".into(), + }, + sqlstate::DEPENDENT_OBJECTS_STILL_EXIST, + ), + ]; + for (code, state) in cases { + let local = crate::Error::DataPlane(code.clone()); + assert_eq!(error_to_sqlstate(&local).1, state, "{code:?} locally"); + let back = ErrorCode::from(DataPlaneErrorCode::from(code.clone())); + assert_eq!(back, code, "{code:?} across the cluster wire"); + let remote = crate::Error::DataPlane(back); + assert_eq!(error_to_sqlstate(&remote).1, state, "{code:?} remotely"); + assert_eq!( + native_error_fields(&remote).code, + native_error_fields(&local).code, + "{code:?} native code" + ); + } +} + +/// A DDL error keeps its exact SQLSTATE and code, including a SQLSTATE no +/// named constant covers. +#[test] +fn a_ddl_error_keeps_its_exact_sqlstate_and_code() { + use crate::control::server::shared::ddl::DdlError; + + for state in ["42710", "42P07", sqlstate::INSUFFICIENT_PRIVILEGE, "57014"] { + let ddl = if state == "57014" { + DdlError::from_error(&crate::Error::DeadlineExceeded { + request_id: crate::types::RequestId::new(1), + }) + } else { + DdlError::new(state, "refused") + }; + let expected_code = ddl.code; + let err = crate::Error::from(ddl); + assert_eq!(error_to_sqlstate(&err).1, state, "{err:?}"); + let native = native_error_fields(&err); + assert_eq!(native.sqlstate, state, "{err:?} native SQLSTATE"); + assert_eq!(native.code, expected_code, "{err:?} native code"); + } +} diff --git a/nodedb/src/control/gateway/error_map/class_parity/mod.rs b/nodedb/src/control/gateway/error_map/class_parity/mod.rs new file mode 100644 index 000000000..0f489b84b --- /dev/null +++ b/nodedb/src/control/gateway/error_map/class_parity/mod.rs @@ -0,0 +1,23 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Every Data-Plane `ErrorCode`, and each Control-Plane error a client acts +//! on, answers one class on native, pgwire and HTTP. +//! +//! pgwire renders a Data-Plane verdict as its SQLSTATE. A native client reads +//! the numeric `nodedb_types` code on the frame. The two agree when the +//! numeric code renders, through the numeric-code SQLSTATE table, in the same +//! SQLSTATE class (the first two characters) as the pgwire SQLSTATE. That same +//! table renders a code that crossed a node as a bare number, so agreement +//! also keeps a verdict's class across nodes. + +mod code_parity; +mod code_samples; +mod control_plane; +mod error_index; +mod error_parity; +mod error_samples; +mod hop_parity; +mod support; + +pub(crate) use error_index::error_variant_index; +pub(crate) use error_samples::error_samples; diff --git a/nodedb/src/control/gateway/error_map/class_parity/support.rs b/nodedb/src/control/gateway/error_map/class_parity/support.rs new file mode 100644 index 000000000..30b0c2227 --- /dev/null +++ b/nodedb/src/control/gateway/error_map/class_parity/support.rs @@ -0,0 +1,14 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Helpers shared by the class-parity suites. + +/// A named encoder that turns an `Error` into its node-hop wire form. +pub(super) type HopEncoder = ( + &'static str, + fn(crate::Error) -> nodedb_cluster::rpc_codec::TypedClusterError, +); + +/// The SQLSTATE class: the first two characters. +pub(super) fn class(state: &str) -> &str { + state.get(..2).unwrap_or(state) +} diff --git a/nodedb/src/control/gateway/error_map/pgwire.rs b/nodedb/src/control/gateway/error_map/pgwire.rs index 7bfbf67ff..96d088b8a 100644 --- a/nodedb/src/control/gateway/error_map/pgwire.rs +++ b/nodedb/src/control/gateway/error_map/pgwire.rs @@ -75,7 +75,7 @@ mod tests { /// and the direct class is not `XX000`. #[test] fn gateway_and_direct_mapper_agree_on_classified_errors() { - use crate::types::{DatabaseId, TenantId}; + use crate::types::TenantId; let samples = vec![ Error::RejectedConstraint { @@ -123,9 +123,6 @@ mod tests { }, Error::CalvinSerializationConflict, Error::CalvinParticipantError, - Error::SourceFrozen { - database_id: DatabaseId::new(7), - }, Error::CloneWriteRequiresMaterialize { collection: "c".into(), engine: "kv".into(), @@ -140,10 +137,6 @@ mod tests { Error::MemoryExhausted { engine: "kv".into(), }, - Error::FanOutExceeded { - shards_touched: 9, - limit: 8, - }, Error::MaterializedSumTargetNotFound { target_collection: "t".into(), join_column: "k".into(), diff --git a/nodedb/src/control/gateway/error_map/resp.rs b/nodedb/src/control/gateway/error_map/resp.rs index 738c16b72..086a60bdf 100644 --- a/nodedb/src/control/gateway/error_map/resp.rs +++ b/nodedb/src/control/gateway/error_map/resp.rs @@ -79,9 +79,7 @@ impl GatewayErrorMap { | Error::NotInTransactionBlock { .. } | Error::CrdtAdmissionTimeout { .. } | Error::NoLeader { .. } - | Error::FanOutExceeded { .. } | Error::CrossCollectionNotColocated { .. } - | Error::SourceFrozen { .. } | Error::CloneWriteRequiresMaterialize { .. } | Error::BackupTenantMismatch { .. } | Error::BackupKeyMismatch @@ -97,10 +95,14 @@ impl GatewayErrorMap { | Error::DataException { .. } | Error::InvalidLimitValue { .. } | Error::RetryableLeaderChange { .. } + | Error::CommittedResultUnavailable { .. } + | Error::ProposalOutcomeUnknown { .. } | Error::GroupQuorumUnavailable { .. } | Error::GroupMarksUnavailable { .. } + | Error::BackupCaptureMoved { .. } | Error::MetadataLeaderUnavailable | Error::AuthorizationStateBehind { .. } + | Error::LinearizableReadRefused { .. } | Error::ExecutionLimitExceeded { .. } | Error::LimitExceeded { .. } | Error::Wal(_) @@ -118,11 +120,14 @@ impl GatewayErrorMap { | Error::Encryption { .. } | Error::Bridge { .. } | Error::VersionCompat { .. } + | Error::RestoreTargetNotEmpty { .. } + | Error::RestoreVerificationFailed { .. } | Error::Internal { .. } | Error::Shaping(_) | Error::Ddl(_) | Error::DescriptorVersionAnomaly { .. } | Error::CollectionPurgeRowMissing { .. } + | Error::CollectionUnstamped { .. } | Error::CatalogIntegrityViolation { .. } | Error::Promql(_) | Error::DependentObjectsExist { .. } diff --git a/nodedb/src/control/gateway/error_map/test_fixtures.rs b/nodedb/src/control/gateway/error_map/test_fixtures.rs index d7cfa9ddd..e4a833768 100644 --- a/nodedb/src/control/gateway/error_map/test_fixtures.rs +++ b/nodedb/src/control/gateway/error_map/test_fixtures.rs @@ -10,6 +10,7 @@ pub(super) fn not_leader() -> Error { vshard_id: VShardId::new(1), leader_node: 2, leader_addr: "10.0.0.1:9000".into(), + leader_term: 3, } } diff --git a/nodedb/src/control/gateway/live_leaders.rs b/nodedb/src/control/gateway/live_leaders.rs new file mode 100644 index 000000000..929cb8bec --- /dev/null +++ b/nodedb/src/control/gateway/live_leaders.rs @@ -0,0 +1,151 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Live Raft leadership snapshot for routing decisions. +//! +//! Lock order: take the Raft status first, then the routing guard. +//! `raft_status_fn` locks `MultiRaft`, and `MultiRaft` reads the routing table under that lock. +//! A caller that holds a routing guard while it takes Raft status inverts that order. +//! A queued routing writer then blocks the nested routing read, and the node deadlocks. +//! [`LiveLeaders::snapshot`] holds no routing guard, so callers take it before any guard. + +use nodedb_cluster::{GroupStatus, RoutingTable}; + +use crate::control::state::SharedState; + +use super::route::RouteDecision; +use super::router::resolve_decision; + +/// Each hosted group's leader as this node's Raft knows it. +/// +/// `None` means no Raft status source is wired. Routing then falls back to +/// the routing-table hint alone. +pub struct LiveLeaders { + statuses: Option>, +} + +impl LiveLeaders { + /// Snapshot live leadership from `state.raft_status_fn`. + /// + /// The caller must not hold a `cluster_routing` guard. See the module docs. + pub fn snapshot(state: &SharedState) -> Self { + Self { + statuses: state.raft_status_fn.get().map(|status| status()), + } + } + + /// The live leader of `group_id`, or `0` when Raft knows none. + pub fn leader_of(&self, group_id: u64) -> u64 { + self.statuses + .as_deref() + .and_then(|statuses| statuses.iter().find(|s| s.group_id == group_id)) + .map_or(0, |s| s.leader_id) + } + + /// Resolve `vshard_id` against this snapshot, with `routing` as the hint fallback. + pub fn resolve( + &self, + vshard_id: u32, + local_node_id: u64, + routing: Option<&RoutingTable>, + ) -> RouteDecision { + let leader = |group_id: u64| self.leader_of(group_id); + let live_lookup: Option<&dyn Fn(u64) -> u64> = if self.statuses.is_some() { + Some(&leader) + } else { + None + }; + resolve_decision(vshard_id, local_node_id, routing, live_lookup) + } +} + +/// Resolve `vshard_id` to a `RouteDecision` against live Raft leadership. +/// +/// Takes the Raft snapshot first and the routing guard second. The guard drops on return. +pub fn resolve_live_decision(state: &SharedState, vshard_id: u32) -> RouteDecision { + let live = LiveLeaders::snapshot(state); + let routing = state + .cluster_routing + .as_ref() + .map(|lock| lock.read().unwrap_or_else(|p| p.into_inner())); + live.resolve(vshard_id, state.node_id, routing.as_deref()) +} + +#[cfg(test)] +mod tests { + use std::sync::atomic::{AtomicBool, Ordering}; + use std::sync::{Arc, RwLock}; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::wal::WalManager; + + /// A live leader that is neither this node nor the routing hint. + const LIVE_LEADER: u64 = 7; + + fn status(group_id: u64) -> GroupStatus { + GroupStatus { + group_id, + role: "Follower".into(), + leader_id: LIVE_LEADER, + term: 1, + commit_index: 0, + last_applied: 0, + last_log_index: 0, + snapshot_index: 0, + member_count: 2, + learner_count: 0, + vshard_count: 0, + } + } + + /// Resolution takes the Raft status before the routing guard. + /// + /// Production Raft status locks `MultiRaft`, which re-reads routing under that lock. + /// A caller that holds a routing read guard there deadlocks once a writer queues. + /// The fake status source stands in for `MultiRaft`. It probes the routing lock with + /// `try_write`, which fails while any read guard is held. A probe that fails proves + /// the caller held a routing guard across the status call. + #[test] + fn raft_status_is_taken_with_no_routing_guard_held() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = + Arc::new(WalManager::open_for_testing(&dir.path().join("live.wal")).expect("wal")); + let (dispatcher, _sides) = Dispatcher::new(1, 64); + let mut state = SharedState::new(dispatcher, wal).expect("shared state"); + let routing = Arc::new(RwLock::new(RoutingTable::uniform(2, &[1, 2], 2))); + Arc::get_mut(&mut state) + .expect("sole owner of fresh state") + .cluster_routing = Some(Arc::clone(&routing)); + let vshard_id = 0; + let group_id = routing + .read() + .expect("routing lock") + .group_for_vshard(vshard_id) + .expect("the vShard has a group"); + + let held_guard = Arc::new(AtomicBool::new(false)); + let probe_routing = Arc::clone(&routing); + let probe_held = Arc::clone(&held_guard); + let status_fn: Arc Vec + Send + Sync> = Arc::new(move || { + if probe_routing.try_write().is_err() { + probe_held.store(true, Ordering::SeqCst); + } + vec![status(group_id)] + }); + assert!(state.raft_status_fn.set(status_fn).is_ok()); + + let decision = resolve_live_decision(&state, vshard_id); + assert!( + !held_guard.load(Ordering::SeqCst), + "the Raft status was read under a routing guard" + ); + assert!( + matches!(decision, RouteDecision::Remote { node_id, .. } if node_id == LIVE_LEADER), + "live leadership routes the vShard: {decision:?}" + ); + + let live = LiveLeaders::snapshot(&state); + assert!(!held_guard.load(Ordering::SeqCst)); + assert_eq!(live.leader_of(group_id), LIVE_LEADER); + } +} diff --git a/nodedb/src/control/gateway/mod.rs b/nodedb/src/control/gateway/mod.rs index da868d892..873416326 100644 --- a/nodedb/src/control/gateway/mod.rs +++ b/nodedb/src/control/gateway/mod.rs @@ -1,17 +1,21 @@ // SPDX-License-Identifier: BUSL-1.1 pub mod cache_miss; +pub mod cluster_error; pub mod colocation_guard; pub mod core; +pub mod dispatch_local; pub mod dispatch_remote; pub mod dispatcher; pub mod error_map; pub mod fuser; pub mod invalidation; pub mod key_extractor; +pub mod live_leaders; pub mod lowered_plan; pub mod outcome; pub mod plan_cache; +pub mod read_leg; pub mod retry; pub mod route; pub mod router; diff --git a/nodedb/src/control/gateway/outcome.rs b/nodedb/src/control/gateway/outcome.rs index 767b88ef2..1a29714c8 100644 --- a/nodedb/src/control/gateway/outcome.rs +++ b/nodedb/src/control/gateway/outcome.rs @@ -14,6 +14,10 @@ use crate::types::{Lsn, RequestId, VShardId}; use super::core::{Gateway, QueryContext, authorized_plan_for_context}; +/// A gateway execution's payloads, per-shard read watermarks, and +/// read-version LSN, as [`GatewayOutcome::into_parts`] returns them. +pub type GatewayParts = (Vec>, Vec<(VShardId, Lsn)>, Lsn); + /// Everything one gateway execution observed across its routes. pub struct GatewayOutcome { /// One payload, fused when several routes answered. @@ -30,7 +34,7 @@ pub struct GatewayOutcome { impl GatewayOutcome { /// Payloads, per-shard watermarks, and read-version LSN. - pub fn into_parts(self) -> (Vec>, Vec<(VShardId, Lsn)>, Lsn) { + pub fn into_parts(self) -> GatewayParts { (self.payloads, self.shard_watermarks, self.read_version_lsn) } @@ -73,7 +77,7 @@ impl Gateway { ctx: &QueryContext, checked: CloneCheckedTask, ) -> Result { - let plan = authorized_plan_for_context(ctx, checked)?; + let (plan, _lease) = authorized_plan_for_context(ctx, checked)?; match self.execute_plan_outcome(ctx, plan).await { Ok(outcome) => Ok(outcome.into_response()), // A remote leaseholder returns its `NotFound` verdict as a typed diff --git a/nodedb/src/control/gateway/read_leg.rs b/nodedb/src/control/gateway/read_leg.rs new file mode 100644 index 000000000..94b7f29fe --- /dev/null +++ b/nodedb/src/control/gateway/read_leg.rs @@ -0,0 +1,61 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The linearizable half of dispatching one gateway route. +//! +//! A route that is a leg of a linearizable read confirms its group on the node +//! that serves it: here before a local read, on the remote node before a +//! forwarded one (the route's groups ride on the `ExecuteRequest`, and the +//! receiver calls [`confirm_local_read`]). + +use std::time::Duration; + +use crate::Error; +use nodedb_physical::physical_plan::{PhysicalPlan, plan_contains_cluster_partitioned_leaf}; + +use crate::control::cluster::linearizable_read::{ + confirm_linearizable_read, groups_of_vshards, linearizable_read_deadline, +}; +use crate::control::server::graph_dispatch::graph_read_groups; +use crate::control::server::shared::write_admission::plan_is_write; +use crate::control::state::SharedState; +use crate::types::DatabaseId; + +use super::route::TaskRoute; + +/// The groups a route's read confirms where it is served: the group of the +/// route's vShard for a linearizable read, none for a write or a weaker read. +pub(super) fn linearizable_read_groups( + shared: &SharedState, + route: &TaskRoute, + linearizable: bool, +) -> Result, Error> { + if !linearizable || plan_is_write(&route.plan) { + return Ok(Vec::new()); + } + groups_of_vshards(shared, [route.vshard_id]) +} + +/// Confirm a linearizable read on this node before `plan` reads locally, +/// within the statement's remaining `deadline_ms`. Nothing to confirm when +/// `read_groups` is empty. +/// +/// A graph or array plan fans across every local core, so it confirms the +/// groups the plan reads there (`graph_dispatch::read_groups`) rather than the +/// route's one group. +pub(crate) async fn confirm_local_read( + shared: &SharedState, + database_id: DatabaseId, + plan: &PhysicalPlan, + read_groups: &[u64], + deadline_ms: u64, +) -> Result<(), Error> { + if read_groups.is_empty() { + return Ok(()); + } + let deadline = linearizable_read_deadline(Duration::from_millis(deadline_ms)); + if plan_contains_cluster_partitioned_leaf(plan) { + let groups = graph_read_groups(shared, database_id, plan)?; + return confirm_linearizable_read(shared, &groups, deadline).await; + } + confirm_linearizable_read(shared, read_groups, deadline).await +} diff --git a/nodedb/src/control/gateway/retry.rs b/nodedb/src/control/gateway/retry.rs index 4a18dc7cc..3b5880e91 100644 --- a/nodedb/src/control/gateway/retry.rs +++ b/nodedb/src/control/gateway/retry.rs @@ -3,8 +3,10 @@ //! Typed `NotLeader` retry with 3-attempt budget + 50/100/200 ms backoff. //! //! When a remote dispatch returns `Error::NotLeader`, the retry helper: -//! 1. Extracts the hinted new leader from the error. -//! 2. Updates the routing table entry for the affected group. +//! 1. Extracts the hinted new leader and its term from the error. +//! 2. Offers them to the routing table entry for the affected group. The +//! hint moves only when the redirect's term is above the hint's term, so +//! a stale redirect never undoes a newer leader. //! 3. Sleeps for the appropriate backoff duration. //! 4. Re-invokes the closure. //! @@ -52,25 +54,39 @@ where Err(Error::NotLeader { vshard_id, leader_node, + leader_term, .. }) => { debug!( attempt, vshard_id = vshard_id.as_u32(), leader_node, + leader_term, "gateway: NotLeader — will retry with new leader hint" ); - // Update routing table with the new leader hint: - // • `leader_node != 0` → a redirect hint; set the group leader. + // Update routing table with the redirect: + // • `leader_node != 0` with a term → offer the leader at its + // term. It applies above the hint's term, and fills a hint + // cleared at that term. + // • `leader_node != 0` with term 0 → the source named an + // owner without a term. It fills only a hint that holds + // no term. // • `leader_node == 0` → transport failure or no hint; clear - // the group leader to 0 so the next attempt falls back to - // local dispatch rather than retrying the same dead node. + // the group leader and keep its term, so the next attempt + // falls back to local dispatch rather than retrying the + // same dead node. if let Some(rt) = routing && let Ok(mut table) = rt.write() && let Ok(group_id) = table.group_for_vshard(vshard_id.as_u32()) { - table.set_leader(group_id, leader_node); + if leader_node == 0 { + table.clear_leader(group_id); + } else if leader_term == 0 { + table.set_leader(group_id, leader_node); + } else { + table.confirm_leader(group_id, leader_node, leader_term); + } } if attempt + 1 < MAX_RETRIES { @@ -81,6 +97,7 @@ where vshard_id, leader_node, leader_addr: String::new(), + leader_term, }); } Err(Error::RetryableLeaderChange { @@ -145,6 +162,7 @@ mod tests { vshard_id: VShardId::new(0), leader_node: 2, leader_addr: "10.0.0.2:9400".into(), + leader_term: 1, }) } else { Ok::(99) @@ -163,6 +181,7 @@ mod tests { vshard_id: VShardId::new(1), leader_node: 0, leader_addr: String::new(), + leader_term: 0, }) }) .await; @@ -205,6 +224,7 @@ mod tests { vshard_id: VShardId::new(0), leader_node: 2, leader_addr: "addr".into(), + leader_term: 4, }) } else { Ok::<(), Error>(()) @@ -214,6 +234,68 @@ mod tests { .await; let table = rt.read().unwrap(); - assert_eq!(table.leader_for_vshard(0).unwrap(), 2); + assert_eq!(table.leader_at_term_for_vshard(0).unwrap(), (2, 4)); + } + + /// A redirect at a lower term than the hint never moves it back. + #[tokio::test] + async fn a_stale_redirect_leaves_a_newer_hint() { + let mut table = RoutingTable::uniform(1, &[1, 2, 3], 3); + let group_id = table.group_for_vshard(0).unwrap(); + assert!(table.observe_leader(group_id, 3, 7)); + let rt = RwLock::new(table); + + let _ = retry_not_leader(Some(&rt), |attempt| async move { + if attempt == 0 { + Err(Error::NotLeader { + vshard_id: VShardId::new(0), + leader_node: 2, + leader_addr: "addr".into(), + leader_term: 5, + }) + } else { + Ok::<(), Error>(()) + } + }) + .await; + + let table = rt.read().unwrap(); + assert_eq!(table.leader_at_term_for_vshard(0).unwrap(), (3, 7)); + } + + /// A redirect without a term fills a hint that holds no term, and never + /// replaces a termed one. + #[tokio::test] + async fn a_term_less_redirect_never_replaces_a_termed_hint() { + async fn redirect_to(rt: &RwLock, leader_node: u64) { + let _ = retry_not_leader(Some(rt), |attempt| async move { + if attempt == 0 { + Err(Error::NotLeader { + vshard_id: VShardId::new(0), + leader_node, + leader_addr: String::new(), + leader_term: 0, + }) + } else { + Ok::<(), Error>(()) + } + }) + .await; + } + + let rt = RwLock::new(RoutingTable::uniform(1, &[1, 2, 3], 3)); + redirect_to(&rt, 2).await; + assert_eq!( + rt.read().unwrap().leader_at_term_for_vshard(0).unwrap(), + (2, 0) + ); + + let group_id = rt.read().unwrap().group_for_vshard(0).unwrap(); + assert!(rt.write().unwrap().observe_leader(group_id, 3, 7)); + redirect_to(&rt, 2).await; + assert_eq!( + rt.read().unwrap().leader_at_term_for_vshard(0).unwrap(), + (3, 7) + ); } } diff --git a/nodedb/src/control/gateway/router.rs b/nodedb/src/control/gateway/router.rs index d1bf99e84..4e25f199d 100644 --- a/nodedb/src/control/gateway/router.rs +++ b/nodedb/src/control/gateway/router.rs @@ -21,10 +21,11 @@ //! cluster-partitioned child (graph traversal by node-id, array by tile) is //! broadcast to every vShard; a single-vShard-homed child (document / kv / //! columnar / timeseries / spatial / vector / text, and joins/aggregates over -//! them) is routed to its ONE owning vShard — broadcasting it would duplicate +//! them) is routed to its ONE owning vShard — broadcasting it duplicates //! rows, since the data-plane scan is not vshard-scoped. //! -//! In single-node mode (routing table = `None`), all plans route locally. +//! A state with no routing table routes every plan locally. Every booted +//! node has one. A pure-logic fixture builds a state without it. use nodedb_cluster::routing::{RoutingTable, vshard_for_collection}; use nodedb_types::PartitionStrategy; @@ -72,7 +73,7 @@ pub fn route_plan( }); } - // In single-node mode every plan runs locally. + // With no routing table every plan runs locally. let Some(routing) = routing else { let vshard_id = primary_vshard(&plan, database_id)?; return Ok(vec![TaskRoute { @@ -86,7 +87,7 @@ pub fn route_plan( // The coordinator strips the Exchange here: its child is the plan that runs // on each vShard, and the per-vShard payloads are fused on return (see // `fuse_payloads` in the gateway core). Shipping the Exchange wrapper itself - // would let it reach a Data-Plane core, which rejects unresolved Exchange + // lets it reach a Data-Plane core, which rejects unresolved Exchange // nodes ("Exchange must be resolved by the coordinator before dispatch"). use nodedb_physical::physical_plan::{ ExchangeMode, ExchangeOp, QueryOp, plan_contains_cluster_partitioned_leaf, @@ -105,8 +106,8 @@ pub fn route_plan( // payloads. // - A single-vShard-homed source (document/kv/columnar/timeseries/ // spatial/vector/text, and joins/aggregates over them) lives on - // exactly ONE vShard. Broadcasting it to all 1024 vShards would - // return the full result from the owning node once per route that + // exactly ONE vShard. Broadcasting it to all 1024 vShards + // returns the full result from the owning node once per route that // lands there (the data-plane scan is NOT vshard-scoped) → N-fold // duplication. Route it to its single owning vShard instead. Any // nested build-side data movement is resolved at the dispatch site @@ -208,7 +209,7 @@ fn route_single_collection( /// `LeaderUnknown` (surfaced as `Error::NotLeader` by dispatch so the /// gateway retry loop sleeps and re-resolves). /// -/// Single-node mode (`routing == None`) always routes locally. +/// With no routing table (`routing == None`) it always routes locally. pub fn resolve_decision( vshard_id: u32, local_node_id: u64, @@ -237,13 +238,21 @@ pub fn resolve_decision( }; } // Live state has no leader for this group yet — fall through to - // routing-table hint (it may have a stale-but-usable forwarding + // routing-table hint (it can have a stale-but-usable forwarding // target from the last term). } match routing.leader_for_vshard(vshard_id) { Ok(0) => unknown, - Ok(leader) if leader == local_node_id => RouteDecision::Local, + // A hint naming this node is stale once this node holds no replica + // of the group: it left the group, so it cannot serve it. + Ok(leader) if leader == local_node_id => { + if routing.is_replica_of_vshard(vshard_id, local_node_id) { + RouteDecision::Local + } else { + unknown + } + } Ok(leader) => RouteDecision::Remote { node_id: leader, vshard_id: vshard_id as u64, @@ -278,8 +287,8 @@ fn route_broadcast( /// task's own `vshard_id`. /// /// These ops name no collection, so the router cannot derive their vShard. The -/// `primary_vshard` fallback would send them to vShard 0: a staged write would -/// land in an overlay the commit never reads, and a commit would apply on the +/// `primary_vshard` fallback sends them to vShard 0: a staged write +/// lands in an overlay the commit never reads, and a commit applies on the /// wrong core. Callers dispatch them with the task's `vshard_id`, never /// through the gateway. pub fn is_task_vshard_scoped(plan: &PhysicalPlan) -> bool { @@ -359,7 +368,7 @@ mod tests { key: vec![], value: vec![], ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), returning: None, rls_filters: Vec::new(), provenance: None, @@ -429,8 +438,8 @@ mod tests { // A single-vShard-homed read reaches the router wrapped in // Exchange{Gather} (the shape `convert()` produces). The router must // strip the Exchange and route the child to its ONE owning vShard — NOT - // broadcast to every vShard. Broadcasting a single-homed source would - // return the full collection from the owning node once per route that + // broadcast to every vShard. Broadcasting a single-homed source + // returns the full collection from the owning node once per route that // lands there (the data-plane scan is not vshard-scoped) → N-fold // duplication. let plan = PhysicalPlan::Query(nodedb_physical::physical_plan::QueryOp::Exchange( @@ -467,7 +476,7 @@ mod tests { routes[0].plan ); // It routes to the same single vShard a bare scan of the same collection - // would (the collection's owner). + // routes to (the collection's owner). assert_eq!( routes[0].vshard_id, vshard_for_collection(CollectionKey::from_bare(DatabaseId::DEFAULT, "events")) diff --git a/nodedb/src/control/gateway/sql_execute.rs b/nodedb/src/control/gateway/sql_execute.rs index c3f06d3c9..533885582 100644 --- a/nodedb/src/control/gateway/sql_execute.rs +++ b/nodedb/src/control/gateway/sql_execute.rs @@ -84,7 +84,7 @@ impl Gateway { if let Some(cached_plan) = self.plan_cache.get(&full_key) { debug!(sql = %sql, "gateway: plan cache hit (two-phase)"); let checked = authorize_fn(cached_plan.as_ref().clone()).await?; - let plan = authorized_plan_for_context(ctx, checked)?; + let (plan, _lease) = authorized_plan_for_context(ctx, checked)?; return self .execute_with_version_set(ctx, plan, stored_vs) .await @@ -101,7 +101,7 @@ impl Gateway { .await?; // A volatile plan froze its `nextval` / `now()` / UUID values while it - // was built. Admitting it would replay this execution's values into + // was built. Admitting it replays this execution's values into // every later one, so it skips both caches and re-plans next time. // The side cache is skipped too: its entry is pruned only alongside the // plan entry it maps to, so storing one without a plan leaves an orphan @@ -118,7 +118,7 @@ impl Gateway { } let checked = authorize_fn(plan.clone()).await?; - let plan = authorized_plan_for_context(ctx, checked)?; + let (plan, _lease) = authorized_plan_for_context(ctx, checked)?; self.execute_with_version_set(ctx, plan, actual_vs) .await .map(|outcome| outcome.payloads) diff --git a/nodedb/src/control/gateway/stream.rs b/nodedb/src/control/gateway/stream.rs index f26f34af5..ea90ad3cd 100644 --- a/nodedb/src/control/gateway/stream.rs +++ b/nodedb/src/control/gateway/stream.rs @@ -16,9 +16,9 @@ use nodedb_physical::physical_plan::PhysicalPlan; use super::core::{Gateway, QueryContext, authorized_plan_for_context}; use super::dispatcher::{DispatchRouteStreamParams, dispatch_route_stream, statement_deadline_ms}; +use super::live_leaders::resolve_live_decision; use super::retry::retry_not_leader; use super::route::TaskRoute; -use super::router::resolve_decision; impl Gateway { /// Streaming sibling of [`execute`](Gateway::execute). @@ -45,7 +45,9 @@ impl Gateway { ctx: &QueryContext, checked: CloneCheckedTask, ) -> Result { - let plan = authorized_plan_for_context(ctx, checked)?; + // Every streamed plan is an unordered scan (`streamable_gather_child`), + // so it carries no write lease to hold for the stream's lifetime. + let (plan, _lease) = authorized_plan_for_context(ctx, checked)?; self.execute_stream_internal(ctx, plan).await } @@ -85,35 +87,10 @@ impl Gateway { let tenant_id = ctx.tenant_id; let database_id = ctx.database_id; let trace_id = ctx.trace_id; + let linearizable = ctx.linearizable; let version_set = version_set_for_route.clone(); async move { - let decision = { - let routing_guard = shared - .cluster_routing - .as_ref() - .map(|rw| rw.read().unwrap_or_else(|p| p.into_inner())); - let raft_snapshot: Vec = - shared.raft_status_fn.get().map(|f| f()).unwrap_or_default(); - let live_leader = move |group_id: u64| -> u64 { - raft_snapshot - .iter() - .find(|gs| gs.group_id == group_id) - .map(|gs| gs.leader_id) - .unwrap_or(0) - }; - let live_lookup: Option<&dyn Fn(u64) -> u64> = - if shared.raft_status_fn.get().is_some() { - Some(&live_leader) - } else { - None - }; - resolve_decision( - vshard_id_u32, - shared.node_id, - routing_guard.as_deref(), - live_lookup, - ) - }; + let decision = resolve_live_decision(&shared, vshard_id_u32); let route = TaskRoute { plan, decision, @@ -127,6 +104,7 @@ impl Gateway { trace_id, deadline_ms, version_set: &version_set, + linearizable, }) .await } diff --git a/nodedb/src/control/gateway/version_check.rs b/nodedb/src/control/gateway/version_check.rs index d00bdbd51..b1c965681 100644 --- a/nodedb/src/control/gateway/version_check.rs +++ b/nodedb/src/control/gateway/version_check.rs @@ -5,9 +5,9 @@ //! A plan is stamped at plan time with the descriptor versions it was built //! against (`GatewayVersionSet`). Holding a descriptor lease does not make //! that stamp current: a lease grant never compares the requested version -//! against the catalog, so a plan stamped just before a DDL committed still -//! acquires its lease afterwards, at the superseded version. A mixed-version -//! cluster skips the lease drain outright. Every dispatch path therefore +//! against the catalog, so a plan stamped shortly before a DDL committed still +//! acquires its lease afterwards, at the superseded version. Every dispatch +//! path therefore //! re-compares the stamped versions against the executing node's own catalog //! before the plan runs. @@ -31,7 +31,7 @@ pub enum DescriptorCheckError { actual_version: u64, }, - /// The catalog read itself failed, so the versions could not be compared. + /// The catalog read itself failed, so the versions were not compared. #[error("catalog lookup failed for {collection}: {detail}")] CatalogLookup { collection: String, detail: String }, } @@ -147,6 +147,27 @@ pub fn check_descriptor_holds( Ok(()) } +/// Re-compare a plan's stamped descriptor versions against this node's own +/// catalog before a local dispatch. A mismatch surfaces as +/// [`crate::Error::RetryableSchemaChanged`], which the gateway's cache-miss +/// retry absorbs by re-planning against fresh state. +pub(super) fn check_local_descriptor_versions( + shared: &crate::control::state::SharedState, + tenant_id: crate::types::TenantId, + database_id: DatabaseId, + version_set: &super::version_set::GatewayVersionSet, +) -> Result<(), crate::Error> { + check_descriptor_versions( + shared.credentials.catalog(), + database_id, + tenant_id.as_u64(), + version_set + .iter() + .map(|(collection, version)| (collection.as_str(), *version)), + )?; + Ok(()) +} + #[cfg(test)] mod tests { use super::*; @@ -157,7 +178,7 @@ mod tests { fn catalog_with(collections: &[(&str, u64)]) -> SystemCatalog { let catalog = SystemCatalog::open_in_memory().expect("in-memory catalog"); for (name, version) in collections { - let mut stored = StoredCollection::new(TENANT, name, "owner"); + let mut stored = StoredCollection::stamped_for_test(TENANT, name, "owner"); stored.descriptor_version = *version; catalog .put_collection(DatabaseId::DEFAULT, &stored) diff --git a/nodedb/src/control/gateway/version_set/plan_keys.rs b/nodedb/src/control/gateway/version_set/plan_keys.rs index 1a61717db..9a9180486 100644 --- a/nodedb/src/control/gateway/version_set/plan_keys.rs +++ b/nodedb/src/control/gateway/version_set/plan_keys.rs @@ -135,7 +135,7 @@ fn document_touched_collections( /// Extract every collection name touched by a `PhysicalPlan`. /// -/// Returns a `Vec` that may contain duplicates; callers are +/// Returns a `Vec` that can contain duplicates; callers are /// responsible for de-duplication (e.g., `GatewayVersionSet::from_plan`). pub fn touched_collections(plan: &PhysicalPlan) -> Vec { use nodedb_physical::physical_plan::*; @@ -234,8 +234,15 @@ pub fn touched_collections(plan: &PhysicalPlan) -> Vec { } | SetNodeLabels { .. } | RemoveNodeLabels { .. } + // A node delete's guard is keyed on the node's key home. + | NodeEdgeGuard { .. } // The wrapped delete is structural too — node IDs, no collection. | ResolveEdgeDelete(_) => {} + TruncateEdges { collection, .. } + | NodePresenceGuard { collection, .. } + | NodePresenceRead { collection, .. } => { + out.push(collection.as_str().to_owned()) + } } } @@ -265,7 +272,7 @@ pub fn touched_collections(plan: &PhysicalPlan) -> Vec { // The wrapped ingest is the intercepted write verbatim. ResolveIngest(inner) => { - if let Ingest { collection, .. } = inner.as_ref() { + if let Ingest { collection, .. } = &inner.ingest { out.push(collection.as_str().to_owned()); } } @@ -430,7 +437,7 @@ mod tests { ), item_key: vec![], dest_key: vec![], - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), source_rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), dest_rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), }); diff --git a/nodedb/src/control/insert_select/copy_rows.rs b/nodedb/src/control/insert_select/copy_rows.rs index 8f61f34e7..6b66c0dbe 100644 --- a/nodedb/src/control/insert_select/copy_rows.rs +++ b/nodedb/src/control/insert_select/copy_rows.rs @@ -8,7 +8,7 @@ //! Every step here assumes standard msgpack bodies carrying an `id` field — //! `MaterializeScan` already normalized a strict source's Binary Tuple and //! injected the row's storage-key identity on the Data Plane. Never re-add -//! a Control-Plane decode here; it would silently corrupt the filter, PK +//! a Control-Plane decode here; it silently corrupts the filter, PK //! extraction, and target write. use nodedb_types::{DatabaseId, Surrogate, TenantId}; @@ -17,7 +17,7 @@ use crate::bridge::expr_eval::ComputedColumn; use crate::bridge::scan_filter::ScanFilter; use crate::control::state::SharedState; use crate::control::target_identity::{ - TargetPk, assign_target_surrogate, bare_collection_name, resolve_target_pk, + TargetPk, assign_target_surrogates, bare_collection_name, resolve_target_pk, }; use crate::engine::document::store::StorageKey; @@ -91,7 +91,7 @@ pub(crate) fn resolve_copy_spec( /// bounds the total copied-row count across pages (the SELECT `LIMIT`) and is /// decremented per emitted row. Returns the concrete /// `(target_doc_id, msgpack_value, fresh_surrogate)` to write. -pub(crate) fn assign_page_rows( +pub(crate) async fn assign_page_rows( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, @@ -100,7 +100,7 @@ pub(crate) fn assign_page_rows( entries: Vec<(String, u32, Vec)>, remaining: &mut usize, ) -> crate::Result, Surrogate)>> { - let mut out = Vec::with_capacity(entries.len()); + let mut shaped: Vec> = Vec::with_capacity(entries.len()); for (_source_doc_id, _source_surrogate, value) in entries { if *remaining == 0 { break; @@ -117,21 +117,30 @@ pub(crate) fn assign_page_rows( } else { shape_row(&value, &spec.column_map)? }; - let surrogate = assign_target_surrogate( - state, - nodedb_types::CollectionKey::from_qualified_str(database_id, target_collection)?, - tenant_id, - &spec.target_pk, - &value, - )?; - out.push(( - StorageKey::for_surrogate(surrogate).to_string(), - value, - surrogate, - )); + shaped.push(value); *remaining -= 1; } - Ok(out) + // Every copied row's surrogate in one batch at the target's home. + let bodies: Vec<&[u8]> = shaped.iter().map(Vec::as_slice).collect(); + let surrogates = assign_target_surrogates( + state, + nodedb_types::CollectionKey::from_qualified_str(database_id, target_collection)?, + tenant_id, + &spec.target_pk, + &bodies, + ) + .await?; + Ok(shaped + .into_iter() + .zip(surrogates) + .map(|(value, surrogate)| { + ( + StorageKey::for_surrogate(surrogate).to_string(), + value, + surrogate, + ) + }) + .collect()) } /// Evaluate each `(target_column, expression)` pair against the source row and diff --git a/nodedb/src/control/insert_select/expand_staged.rs b/nodedb/src/control/insert_select/expand_staged.rs index ff8ce7398..3d1c33613 100644 --- a/nodedb/src/control/insert_select/expand_staged.rs +++ b/nodedb/src/control/insert_select/expand_staged.rs @@ -13,7 +13,7 @@ //! //! Emits `PointInsert`, not `BatchInsert`: only `PointPut`/`PointInsert`/ //! `PointDelete` have an undo-tracked arm in transactional replay: a -//! `BatchInsert` here would survive an atomic rollback (partial commit). +//! `BatchInsert` here survives an atomic rollback (partial commit). use nodedb_types::{DatabaseId, Surrogate, TenantId}; @@ -78,7 +78,7 @@ pub(crate) async fn resolve_and_emit_insert_select_ops( // Resolve materialized-sum targets: these ops stage directly, bypassing // statement-level resolution, so without this a bound target collection - // would fold against an empty resolution. + // folds against an empty resolution. let sum_bodies: Vec<&[u8]> = rows.iter().map(|(_, value, _)| value.as_slice()).collect(); let mut resolved_sum_targets = crate::control::planner::materialized_sum::resolve_sum_targets_for_bodies( @@ -196,7 +196,8 @@ async fn materialize_copy( &spec, entries, &mut remaining, - )?; + ) + .await?; rows.extend(page); if next_cursor.is_empty() { diff --git a/nodedb/src/control/insert_select/orchestrator.rs b/nodedb/src/control/insert_select/orchestrator.rs index 6eb173b0e..0ea9ca185 100644 --- a/nodedb/src/control/insert_select/orchestrator.rs +++ b/nodedb/src/control/insert_select/orchestrator.rs @@ -121,7 +121,8 @@ pub(crate) async fn run_insert_select( &spec, entries, &mut remaining, - )?; + ) + .await?; // Phase 3: one atomic batch write for this page. if !rows.is_empty() { diff --git a/nodedb/src/control/lease/admission.rs b/nodedb/src/control/lease/admission.rs new file mode 100644 index 000000000..d69da9eae --- /dev/null +++ b/nodedb/src/control/lease/admission.rs @@ -0,0 +1,103 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The refcount units a plan admission reserved and has not yet handed to a +//! [`QueryLeaseScope`]. +//! +//! Admission reserves every descriptor under the admission gate, then grants +//! each one. A grant awaits the metadata group, so the admission future can be +//! cancelled at any point in between. [`PendingAdmission`] owns the units +//! meanwhile: +//! +//! - on success, [`PendingAdmission::into_scope`] hands them to the scope; +//! - on error, [`PendingAdmission::rollback`] gives them back and awaits the +//! release of every descriptor no statement holds any more; +//! - on cancellation, `Drop` gives them back and hands that release to the +//! background releaser, so nothing blocks. + +use nodedb_cluster::DescriptorId; + +use super::QueryLeaseScope; +use super::releaser::ReleaseRequest; +use crate::control::state::SharedState; +use crate::error::Error; + +/// Refcount units one admission reserved. +pub(crate) struct PendingAdmission<'a> { + shared: &'a SharedState, + held: Vec<(DescriptorId, u64)>, +} + +impl<'a> PendingAdmission<'a> { + pub(crate) fn new(shared: &'a SharedState) -> Self { + Self { + shared, + held: Vec::new(), + } + } + + /// Reserve one unit of `(id, version)`. The caller holds the admission + /// gate. + pub(crate) fn reserve(&mut self, id: &DescriptorId, version: u64) { + self.shared.lease_refcount.increment(id, version); + self.held.push((id.clone(), version)); + } + + /// The units reserved so far, in admission order. + pub(crate) fn held(&self) -> &[(DescriptorId, u64)] { + &self.held + } + + /// Hand the units to a new scope. A full holder table rolls them back. + pub(crate) async fn into_scope(mut self) -> Result { + match QueryLeaseScope::new(self.held.clone(), self.shared) { + Ok(scope) => { + self.held.clear(); + Ok(scope) + } + Err(error) => { + self.rollback().await; + Err(error) + } + } + } + + /// Give every unit back and await the release of each descriptor no + /// statement holds any more, so a failed first-holder grant leaves no + /// local lease behind. A failed release is logged: the lease expires. + pub(crate) async fn rollback(mut self) { + let unheld = self.give_back(); + if let Err(error) = super::release::release_unheld_leases(self.shared, unheld).await { + tracing::warn!(%error, "plan lease admission rollback: lease release failed"); + } + } + + /// Decrement every unit. Returns the descriptors left with no holder. + fn give_back(&mut self) -> Vec { + let mut unheld = Vec::new(); + let mut drained_hold_ended = false; + for (id, version) in self.held.drain(..) { + self.shared.lease_refcount.decrement(&id, version); + drained_hold_ended |= self.shared.lease_drain.is_draining(&id, version); + if self.shared.lease_refcount.current(&id) == 0 && !unheld.contains(&id) { + unheld.push(id); + } + } + // A drain counts this node's holds directly, so it re-counts now. + if drained_hold_ended { + self.shared.lease_drain.wake_drain_waiters(); + } + unheld + } +} + +impl Drop for PendingAdmission<'_> { + fn drop(&mut self) { + let unheld = self.give_back(); + if !unheld.is_empty() { + self.shared + .lease_runtime + .releaser + .submit(ReleaseRequest::UnheldDescriptors(unheld)); + } + } +} diff --git a/nodedb/src/control/lease/descriptor_lookup.rs b/nodedb/src/control/lease/descriptor_lookup.rs index bd9d3c7ac..1a2148414 100644 --- a/nodedb/src/control/lease/descriptor_lookup.rs +++ b/nodedb/src/control/lease/descriptor_lookup.rs @@ -6,20 +6,20 @@ use nodedb_types::DatabaseId; -use nodedb_cluster::{DescriptorId, DescriptorKind}; +use nodedb_cluster::{DescriptorId, DescriptorKind, DrainOwner}; use crate::control::catalog_entry::CatalogEntry; use crate::control::state::SharedState; /// For a `Put*` entry that carries `descriptor_version`, return -/// the `DescriptorId` whose drain should be implicitly cleared +/// the `DescriptorId` whose drain ends implicitly /// after the entry applies. Returns `None` for variants without /// descriptor versioning (auth, schedules, change streams, etc.). /// /// Called from `MetadataCommitApplier::apply_host_side_effects` /// on every node — after the `apply_to` succeeds, the applier -/// looks up the drained id via this helper and calls -/// `shared.lease_drain.install_end` on it. This is how drain +/// looks up the drained id via this helper and ends the +/// [`DrainOwner::Ddl`] drain on it. This is how drain /// clears implicitly on the happy path without a second raft /// round-trip. pub fn descriptor_id_for_implicit_clear(entry: &CatalogEntry) -> Option { @@ -85,13 +85,88 @@ pub fn descriptor_id_for_implicit_clear(entry: &CatalogEntry) -> Option Vec<(DescriptorId, DrainOwner)> { + match entry { + CatalogEntry::MoveTenantCutover { + tenant_id, + source_db_id, + collections, + .. + } => { + let owner = move_tenant_drain_owner(*tenant_id, *source_db_id); + collections + .iter() + .map(|coll| { + ( + move_source_descriptor(*source_db_id, coll.tenant_id, &coll.name), + owner.clone(), + ) + }) + .collect() + } + // The source side of a moved array ends the move's drain on it. + CatalogEntry::DeleteArray { + database_id, + tenant_id, + name, + moved_to: Some(moved), + .. + } => vec![( + DescriptorId::new( + *database_id, + *tenant_id, + DescriptorKind::Array, + name.clone(), + ), + move_tenant_drain_owner(moved.mover_tenant_id, *database_id), + )], + other => descriptor_id_for_implicit_clear(other) + .map(|id| (id, DrainOwner::Ddl)) + .into_iter() + .collect(), + } +} + +/// End on this node every drain [`drains_for_implicit_clear`] names for +/// `entry`, rows first. +pub fn clear_implicit_drains(shared: &SharedState, entry: &CatalogEntry) -> crate::Result<()> { + super::drain_apply::apply_drain_ends(shared, &drains_for_implicit_clear(entry)) +} + +/// The owner of a `MOVE TENANT`'s drain. One tenant moves at most once at a +/// time out of one database, so the pair names the move. +pub fn move_tenant_drain_owner(tenant_id: u64, source_db_id: u64) -> DrainOwner { + DrainOwner::MoveTenant { + tenant_id, + source_db_id, + } +} + +/// The descriptor a `MOVE TENANT` drains for one collection of its source +/// database. +pub fn move_source_descriptor(source_db_id: u64, tenant_id: u64, name: &str) -> DescriptorId { + DescriptorId::new( + source_db_id, + tenant_id, + DescriptorKind::Collection, + name.to_string(), + ) +} + /// For a `Put*` entry that carries `descriptor_version`, return /// `(descriptor_id, prior_persisted_version)` so the proposer can /// decide whether to run drain. `prior_persisted_version` is `0` -/// on create (no prior record) and causes `drain_for_ddl` to +/// on create (no prior record) and causes `drain_for_ddl_async` to /// return immediately. /// -/// Called from `metadata_proposer::propose_catalog_entry_with_timeout` +/// Called from `metadata_proposer::propose_catalog_entry_async` /// BEFORE the raft propose path. Reads from `SystemCatalog` under /// a short read txn — the read is consistent with the subsequent /// propose because the stamp logic in the applier increments diff --git a/nodedb/src/control/lease/drain.rs b/nodedb/src/control/lease/drain.rs index dad5f5fcd..2ab209fd2 100644 --- a/nodedb/src/control/lease/drain.rs +++ b/nodedb/src/control/lease/drain.rs @@ -16,8 +16,7 @@ //! `is_draining` check in `force_refresh_lease`) and during the //! proposer's drain wait loop. This file owns the in-memory //! state only; the propose-side orchestration (including the -//! rolling-upgrade gate and the wait-for-leases-to-release loop) -//! lives in `drain_propose.rs`. +//! wait-for-leases-to-release loop) lives in `drain_propose.rs`. //! //! **TTL semantics**: every drain entry carries an `expires_at` //! HLC, but `is_draining` never compares it against a local @@ -29,22 +28,31 @@ //! `wait_for_lease_drain` bounds its own wait with a same-node //! `Instant` deadline and, on timeout, proposes //! `DescriptorDrainEnd` explicitly — that replicated entry, not -//! `expires_at`, is what clears a stale drain everywhere. We do -//! NOT run a periodic GC task. If nothing ever re-writes the -//! key, the entry sits in the map until the next `install_end` -//! on the same id or until process restart (drain state is not -//! persisted to redb — it's raft-log-derived and rebuilds on -//! replay). +//! `expires_at`, is what clears a stale drain everywhere. +//! +//! A proposer that crashes holding a drain ends it itself on restart, +//! through its own recovery. A proposer that leaves the topology never +//! restarts: the `TopologyChange::Leave` apply hook ends every drain that +//! node proposed (`lease::gc::end_drains_for_node`). A node that is only +//! suspected Dead keeps its drains, because it can still be running the DDL +//! the drain protects. +//! +//! **Durability**: `drain_apply` writes one `SystemCatalog` row per +//! (descriptor, owner) before it changes the tracker, for the metadata +//! applier and the single-node fallback alike. Boot seeds the tracker from +//! those rows. use std::collections::HashMap; use std::sync::RwLock; +use std::sync::atomic::{AtomicBool, Ordering}; -use nodedb_cluster::DescriptorId; +use nodedb_cluster::{DescriptorId, DrainOwner}; use nodedb_types::Hlc; +use tokio::sync::{Notify, futures::Notified}; -/// One drain entry: "this descriptor is draining leases at +/// One owner's drain: "this descriptor is draining leases at /// versions <= `up_to_version`, active until an explicit -/// `DescriptorDrainEnd`". +/// `DescriptorDrainEnd` for this owner". #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub struct DrainEntry { pub up_to_version: u64, @@ -54,17 +62,34 @@ pub struct DrainEntry { /// the module doc for why a wall-clock comparison is unsafe /// here. pub expires_at: Hlc, + /// Node that proposed the drain. + pub proposer_node_id: u64, } /// In-memory drain state for descriptors being altered. /// +/// Each descriptor keeps one entry per active owner. It is draining while +/// any owner's entry covers the requested version. +/// /// All public mutations (`install_start`, `install_end`) are /// called by the metadata applier's decode path. All public /// reads (`is_draining`, `snapshot`, `count`) are called by the /// lease acquire path and the drain wait loop. +/// +/// Two wake-ups replace fixed polling on both sides of a drain: +/// +/// - `holds_changed` wakes a drain waiting for this node's holds and leases. +/// A hold given back and a lease release applied both fire it. +/// - `ended` wakes a statement waiting out a drain. It fires once the +/// metadata entry that ended the drain has applied all its effects, so the +/// retried statement plans against the new descriptor. #[derive(Debug, Default)] pub struct DescriptorDrainTracker { - active: RwLock>, + active: RwLock>>, + holds_changed: Notify, + ended: Notify, + /// A drain ended since the last [`Self::settle`]. + ended_unsettled: AtomicBool, } impl DescriptorDrainTracker { @@ -72,67 +97,156 @@ impl DescriptorDrainTracker { Self::default() } - /// Record the start of a drain for `id` at `up_to_version`, - /// stamped with `expires_at` for observability (see - /// [`DrainEntry::expires_at`]). Overwrites any prior entry - /// for the same key — a subsequent start with a higher - /// `up_to_version` extends the drain rather than creating a - /// conflicting record. + /// Wake every drain waiting for holds and leases to go. + pub fn wake_drain_waiters(&self) { + self.holds_changed.notify_waiters(); + } + + /// Resolves at the next [`Self::wake_drain_waiters`]. The caller enables + /// it before it counts, so a wake during the count is not lost. + pub fn holds_changed(&self) -> Notified<'_> { + self.holds_changed.notified() + } + + /// Wake every statement waiting out a drain, when a drain ended since the + /// last call. The metadata applier calls it after each entry's effects. + pub fn settle(&self) { + if self.ended_unsettled.swap(false, Ordering::AcqRel) { + self.ended.notify_waiters(); + } + } + + /// Resolves at the next [`Self::settle`] that follows a drain end. The + /// caller enables it before its attempt, so an end during the attempt is + /// not lost. + pub fn drain_ended(&self) -> Notified<'_> { + self.ended.notified() + } + + /// Drop every drain. Boot and a metadata snapshot install call this + /// before loading the persisted drains. + pub fn clear(&self) { + let mut map = self.active.write().unwrap_or_else(|p| p.into_inner()); + if !map.is_empty() { + self.ended_unsettled.store(true, Ordering::Release); + } + map.clear(); + } + + /// Record `owner`'s drain of `id` at `up_to_version`, stamped with + /// `expires_at` for observability (see [`DrainEntry::expires_at`]). + /// Overwrites that owner's prior entry only: a restart by the same owner + /// replaces its range, and other owners' entries stay. /// /// Called by the metadata applier on every node when a /// `DescriptorDrainStart` raft entry commits. - pub fn install_start(&self, id: DescriptorId, up_to_version: u64, expires_at: Hlc) { + pub fn install_start( + &self, + id: DescriptorId, + owner: DrainOwner, + up_to_version: u64, + expires_at: Hlc, + proposer_node_id: u64, + ) { tracing::debug!( ?id, + ?owner, up_to_version, expires_wall_ns = expires_at.wall_ns, + proposer_node_id, "drain: install_start" ); let mut map = self.active.write().unwrap_or_else(|p| p.into_inner()); - map.insert( - id, + map.entry(id).or_default().insert( + owner, DrainEntry { up_to_version, expires_at, + proposer_node_id, }, ); } - /// Remove the drain entry for `id`, if any. Called by the - /// metadata applier both on explicit `DescriptorDrainEnd` + /// Every `(descriptor, owner)` drain `node_id` proposed. + pub fn proposed_by(&self, node_id: u64) -> Vec<(DescriptorId, DrainOwner)> { + let map = self.active.read().unwrap_or_else(|p| p.into_inner()); + map.iter() + .flat_map(|(id, owners)| { + owners + .iter() + .filter(|(_, entry)| entry.proposer_node_id == node_id) + .map(|(owner, _)| (id.clone(), owner.clone())) + }) + .collect() + } + + /// Remove `owner`'s drain of `id`, if any. Other owners' drains stay. + /// Called by the metadata applier both on explicit `DescriptorDrainEnd` /// raft entries AND on the implicit clear path that runs /// after a successful `Put*` apply. - pub fn install_end(&self, id: &DescriptorId) { - tracing::debug!(?id, "drain: install_end"); + pub fn install_end(&self, id: &DescriptorId, owner: &DrainOwner) { + tracing::debug!(?id, ?owner, "drain: install_end"); let mut map = self.active.write().unwrap_or_else(|p| p.into_inner()); - map.remove(id); + if let Some(owners) = map.get_mut(id) { + if owners.remove(owner).is_some() { + self.ended_unsettled.store(true, Ordering::Release); + } + if owners.is_empty() { + map.remove(id); + } + } } /// Whether an acquire on `(id, requested_version)` must be /// rejected because a drain is active that covers this /// version. /// - /// Returns `true` iff an entry exists for `id` with + /// Returns `true` iff any owner's entry for `id` has /// `requested_version <= entry.up_to_version` (i.e. the - /// requested version is inside the drain range). Drain state + /// requested version is inside that drain's range). Drain state /// is raft-replicated, so presence of an entry is authoritative /// on every node — a node never judges another node's deadline /// (`expires_at`, stamped by whichever node proposed the drain) /// against its own wall clock, since nothing bounds clock skew /// between nodes. An entry stays active until an explicit - /// `DescriptorDrainEnd` clears it via `install_end`. + /// `DescriptorDrainEnd` for its owner clears it via `install_end`. pub fn is_draining(&self, id: &DescriptorId, requested_version: u64) -> bool { let map = self.active.read().unwrap_or_else(|p| p.into_inner()); - match map.get(id) { - Some(entry) => requested_version <= entry.up_to_version, - None => false, - } + map.get(id).is_some_and(|owners| { + owners + .values() + .any(|entry| requested_version <= entry.up_to_version) + }) + } + + /// Every owner whose drain on `id` covers `requested_version`, named in a + /// stable order. Empty when `is_draining` is false. + pub fn draining_owners(&self, id: &DescriptorId, requested_version: u64) -> Vec { + let map = self.active.read().unwrap_or_else(|p| p.into_inner()); + let mut owners: Vec = map + .get(id) + .map(|owners| { + owners + .iter() + .filter(|(_, entry)| requested_version <= entry.up_to_version) + .map(|(owner, _)| owner.clone()) + .collect() + }) + .unwrap_or_default(); + owners.sort_by_cached_key(ToString::to_string); + owners } - /// Snapshot the full (id, entry) set for diagnostics and tests. - pub fn snapshot(&self) -> Vec<(DescriptorId, DrainEntry)> { + /// Snapshot every `(id, owner, entry)` for diagnostics and tests. + pub fn snapshot(&self) -> Vec<(DescriptorId, DrainOwner, DrainEntry)> { let map = self.active.read().unwrap_or_else(|p| p.into_inner()); - map.iter().map(|(id, e)| (id.clone(), *e)).collect() + map.iter() + .flat_map(|(id, owners)| { + owners + .iter() + .map(|(owner, entry)| (id.clone(), owner.clone(), *entry)) + }) + .collect() } /// Count of active drain entries. Every installed entry is @@ -143,11 +257,11 @@ impl DescriptorDrainTracker { self.total_count() } - /// Total count of installed drain entries. Mainly for - /// debugging. + /// Total count of installed `(descriptor, owner)` drain entries. + /// Mainly for debugging. pub fn total_count(&self) -> usize { let map = self.active.read().unwrap_or_else(|p| p.into_inner()); - map.len() + map.values().map(HashMap::len).sum() } } @@ -164,11 +278,22 @@ mod tests { Hlc::new(wall_ns, 0) } + const PROPOSER: u64 = 1; + const DDL: DrainOwner = DrainOwner::Ddl; + + fn materializer(clone_collection: &str) -> DrainOwner { + DrainOwner::CloneMaterialize { + clone_database: 1025, + tenant_id: 1, + clone_collection: clone_collection.to_string(), + } + } + #[test] fn install_then_is_draining_true_for_versions_in_range() { let tracker = DescriptorDrainTracker::new(); let d = id("orders"); - tracker.install_start(d.clone(), 5, hlc(1_000_000)); + tracker.install_start(d.clone(), DDL, 5, hlc(1_000_000), PROPOSER); // Versions 1..=5 are inside the drain range; version 6 is // outside. assert!(tracker.is_draining(&d, 1)); @@ -182,31 +307,31 @@ mod tests { fn install_end_clears_entry() { let tracker = DescriptorDrainTracker::new(); let d = id("orders"); - tracker.install_start(d.clone(), 5, hlc(1_000_000)); + tracker.install_start(d.clone(), DDL, 5, hlc(1_000_000), PROPOSER); assert!(tracker.is_draining(&d, 5)); - tracker.install_end(&d); + tracker.install_end(&d, &DDL); assert!(!tracker.is_draining(&d, 5)); assert_eq!(tracker.total_count(), 0); } - /// Pins the fix for the cross-node clock-skew bug: a node must - /// never judge another node's drain deadline by its own wall - /// clock. An entry whose `expires_at` is far in the local past - /// stays active — only an explicit `install_end` clears it. + /// A node never judges another node's drain deadline by its own wall + /// clock, so cross-node clock skew cannot end a drain. An entry whose + /// `expires_at` is far in the local past stays active. Only an explicit + /// `install_end` clears it. #[test] fn is_draining_stays_active_past_local_wall_clock_expiry() { let tracker = DescriptorDrainTracker::new(); let d = id("stale-clock"); // expires_at is stamped far in the past relative to any - // wall clock a checking node could plausibly read. - tracker.install_start(d.clone(), 5, hlc(1_000)); + // wall clock a checking node can plausibly read. + tracker.install_start(d.clone(), DDL, 5, hlc(1_000), PROPOSER); assert!(tracker.is_draining(&d, 1)); assert!(tracker.is_draining(&d, 5)); assert!(!tracker.is_draining(&d, 6)); // Only an explicit end clears it. - tracker.install_end(&d); + tracker.install_end(&d, &DDL); assert!(!tracker.is_draining(&d, 1)); } @@ -215,8 +340,8 @@ mod tests { let tracker = DescriptorDrainTracker::new(); let a = id("a"); let b = id("b"); - tracker.install_start(a.clone(), 1, hlc(1_000_000)); - tracker.install_start(b.clone(), 10, hlc(1_000_000)); + tracker.install_start(a.clone(), DDL, 1, hlc(1_000_000), PROPOSER); + tracker.install_start(b.clone(), DDL, 10, hlc(1_000_000), PROPOSER); assert!(tracker.is_draining(&a, 1)); assert!(!tracker.is_draining(&a, 2)); @@ -229,34 +354,77 @@ mod tests { fn install_start_overwrites_prior_entry() { let tracker = DescriptorDrainTracker::new(); let d = id("orders"); - tracker.install_start(d.clone(), 5, hlc(1_000_000)); + tracker.install_start(d.clone(), DDL, 5, hlc(1_000_000), PROPOSER); // Start again with a higher up_to_version — the new // entry extends the drain range. - tracker.install_start(d.clone(), 10, hlc(2_000_000)); + tracker.install_start(d.clone(), DDL, 10, hlc(2_000_000), PROPOSER); assert!(tracker.is_draining(&d, 10)); assert_eq!(tracker.total_count(), 1); let snap = tracker.snapshot(); - assert_eq!(snap[0].1.up_to_version, 10); - assert_eq!(snap[0].1.expires_at.wall_ns, 2_000_000); + assert_eq!(snap[0].2.up_to_version, 10); + assert_eq!(snap[0].2.expires_at.wall_ns, 2_000_000); } - /// `count_active` no longer filters by wall-clock expiry, so it - /// pins to the same value as `total_count` for any installed - /// set, regardless of how far in the past `expires_at` is. + /// `count_active` does not filter by wall-clock expiry, so it equals + /// `total_count` for any installed set, however far in the past + /// `expires_at` is. #[test] fn count_active_matches_total_count_regardless_of_expiry() { let tracker = DescriptorDrainTracker::new(); let a = id("live"); let b = id("expired-by-wall-clock"); - tracker.install_start(a, 1, hlc(10_000_000)); - tracker.install_start(b, 1, hlc(100)); + tracker.install_start(a, DDL, 1, hlc(10_000_000), PROPOSER); + tracker.install_start(b, DDL, 1, hlc(100), PROPOSER); assert_eq!(tracker.total_count(), 2); assert_eq!(tracker.count_active(), 2); - tracker.install_end(&id("live")); + tracker.install_end(&id("live"), &DDL); assert_eq!(tracker.count_active(), 1); assert_eq!(tracker.count_active(), tracker.total_count()); } + + #[test] + fn proposed_by_names_only_that_nodes_drains() { + let tracker = DescriptorDrainTracker::new(); + tracker.install_start(id("a"), DDL, 1, hlc(1), 7); + tracker.install_start(id("b"), DDL, 1, hlc(1), 8); + tracker.install_start(id("c"), DDL, 1, hlc(1), 7); + + let mut mine = tracker.proposed_by(7); + mine.sort_by(|x, y| format!("{x:?}").cmp(&format!("{y:?}"))); + assert_eq!(mine, vec![(id("a"), DDL), (id("c"), DDL)]); + assert!(tracker.proposed_by(9).is_empty()); + } + + /// Ending one owner's drain leaves the descriptor drained while another + /// owner's drain remains. + #[test] + fn descriptor_stays_drained_while_any_owner_remains() { + let tracker = DescriptorDrainTracker::new(); + let d = id("orders"); + tracker.install_start(d.clone(), materializer("a"), u64::MAX, hlc(1), PROPOSER); + tracker.install_start(d.clone(), DDL, 3, hlc(1), PROPOSER); + tracker.install_start(d.clone(), materializer("b"), u64::MAX, hlc(1), PROPOSER); + assert_eq!(tracker.total_count(), 3); + + assert_eq!( + tracker.draining_owners(&d, 4), + vec![materializer("a"), materializer("b")], + "the DDL's version-bounded drain does not cover version 4" + ); + tracker.install_end(&d, &DDL); + assert!( + tracker.is_draining(&d, 4), + "the DDL's end leaves both holders" + ); + tracker.install_end(&d, &materializer("a")); + assert!(tracker.is_draining(&d, 4), "one holder is left"); + tracker.install_end(&d, &materializer("a")); + assert!(tracker.is_draining(&d, 4), "a repeated end is a no-op"); + tracker.install_end(&d, &materializer("b")); + assert!(!tracker.is_draining(&d, 4)); + assert_eq!(tracker.total_count(), 0); + } } diff --git a/nodedb/src/control/lease/drain_apply.rs b/nodedb/src/control/lease/drain_apply.rs new file mode 100644 index 000000000..28ef60d72 --- /dev/null +++ b/nodedb/src/control/lease/drain_apply.rs @@ -0,0 +1,114 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Durable drain start and end on this node. +//! +//! The metadata applier and the single-node fallback both go through these +//! functions. Each writes or deletes the drain's `SystemCatalog` row before +//! it changes the tracker, so boot seeds the same drains a restart dropped. + +use nodedb_cluster::{DescriptorId, DrainOwner}; +use nodedb_types::Hlc; + +use crate::control::security::catalog::StoredDrain; +use crate::control::state::SharedState; + +/// Persist and install `owner`'s drain of `descriptor_id`, then release an +/// idle lease on it: no new statement can take the lease now. +pub fn apply_drain_start( + shared: &SharedState, + descriptor_id: &DescriptorId, + owner: &DrainOwner, + up_to_version: u64, + expires_at: Hlc, + proposer_node_id: u64, +) -> crate::Result<()> { + shared + .credentials + .catalog() + .put_descriptor_drain(&StoredDrain { + descriptor_id: descriptor_id.clone(), + owner: owner.clone(), + up_to_version, + expires_at, + proposer_node_id, + })?; + { + // Shares plan admission's gate: an admission completes before this + // start installs, or this drain wins and admission fails closed. + let _admission_gate = shared + .lease_admission_gate + .lock() + .unwrap_or_else(|poison| poison.into_inner()); + shared.lease_drain.install_start( + descriptor_id.clone(), + owner.clone(), + up_to_version, + expires_at, + proposer_node_id, + ); + } + super::release::release_idle_on_drain(shared, descriptor_id); + Ok(()) +} + +/// Delete the rows of `drains`, then end them in the tracker. Other owners' +/// drains of the same descriptors stay. +pub fn apply_drain_ends( + shared: &SharedState, + drains: &[(DescriptorId, DrainOwner)], +) -> crate::Result<()> { + if drains.is_empty() { + return Ok(()); + } + shared + .credentials + .catalog() + .remove_descriptor_drains(drains)?; + for (id, owner) in drains { + shared.lease_drain.install_end(id, owner); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_cluster::DescriptorKind; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::wal::WalManager; + + /// A single-node drain writes its row on start and deletes only its own + /// owner's row on end. + #[tokio::test] + async fn local_drain_rows_follow_start_and_end() { + let dir = tempfile::tempdir().unwrap(); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("drain.wal")).unwrap()); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let state = SharedState::new(dispatcher, wal).unwrap(); + let id = DescriptorId::new(0, 1, DescriptorKind::Collection, "orders"); + let moving = DrainOwner::MoveTenant { + tenant_id: 1, + source_db_id: 0, + }; + + apply_drain_start(&state, &id, &DrainOwner::Ddl, 4, Hlc::new(9, 0), 1).unwrap(); + apply_drain_start(&state, &id, &moving, 5, Hlc::new(9, 0), 1).unwrap(); + apply_drain_ends(&state, &[(id.clone(), DrainOwner::Ddl)]).unwrap(); + + let rows = state + .credentials + .catalog() + .load_descriptor_drains() + .unwrap(); + assert_eq!(rows.len(), 1); + assert_eq!(rows[0].owner, moving); + assert!(state.lease_drain.is_draining(&id, 5)); + assert!( + state.lease_drain.is_draining(&id, 4), + "the move owner still covers 4" + ); + } +} diff --git a/nodedb/src/control/lease/drain_propose.rs b/nodedb/src/control/lease/drain_propose.rs index bf87d865c..4b9c39b71 100644 --- a/nodedb/src/control/lease/drain_propose.rs +++ b/nodedb/src/control/lease/drain_propose.rs @@ -5,192 +5,111 @@ //! 1. Propose `DescriptorDrainStart`. Every node's applier installs it into //! `shared.lease_drain`, so `force_refresh_lease` then rejects new acquires //! at the drained version. -//! 2. Poll `metadata_cache.leases` for entries on the same descriptor at -//! `version <= up_to_version`. Return once none remain. +//! 2. Count `metadata_cache.leases` entries and local holds on the same +//! descriptor at `version <= up_to_version`, again at each hold or lease +//! change. Return once none remain. //! 3. On deadline, propose `DescriptorDrainEnd` so the cluster can progress, //! then return the timeout error. //! //! The happy path emits no `DescriptorDrainEnd`: the following `Put*` carries -//! the new version and the applier's post-apply hook calls `install_end` on -//! every node. That saves a raft round-trip per DDL. -//! -//! The drain variants are wire-format v4, so mixed clusters gate on -//! `DESCRIPTOR_DRAIN_VERSION` and run without drain safety until every node is -//! upgraded. +//! the new version and the applier's post-apply hook ends the +//! [`DrainOwner::Ddl`] drain on every node. That saves a raft round-trip per +//! DDL. Every drain has an owner, and each end removes only its own owner's +//! drain, so a DDL never ends a `MOVE TENANT` or materializer drain. use std::time::{Duration, Instant}; -use tokio::runtime::RuntimeFlavor; -use nodedb_cluster::{DescriptorId, MetadataEntry, encode_entry}; +use nodedb_cluster::{DescriptorId, DrainOwner, MetadataEntry, encode_entry}; use nodedb_types::Hlc; -use crate::control::rolling_upgrade::DESCRIPTOR_DRAIN_VERSION; use crate::control::state::SharedState; use crate::error::Error; -/// Re-poll interval for the drain wait loop. -const POLL_INTERVAL: Duration = Duration::from_millis(50); +/// The longest the drain wait sleeps between counts with no hold or lease +/// change. Expiry and holder death end a lease with no event to wake it. +pub(super) const POLL_INTERVAL: Duration = Duration::from_millis(50); /// Grace added to the lease duration for the `expires_at` stamped on a drain /// entry. `is_draining` never reads it, so it affects observability only. const DRAIN_TTL_GRACE: Duration = Duration::from_secs(30); -/// Drain every lease on `id` at `version <= up_to_version` for a `Put*` DDL. -/// -/// Returns `Ok(())` once they have drained, or immediately when the -/// rolling-upgrade gate is closed. Errors on timeout or propose failure. -/// -/// `own_holds` is how many of those refcount units the requesting transaction -/// holds itself — `0` for a caller with no lease scope of its own. A -/// transaction altering a descriptor it also holds a statement-time lease on -/// cannot wait for its own hold: it cannot release that lease until this -/// call returns. -pub fn drain_for_ddl( +/// The `DescriptorDrainStart` entry of a drain of `id` under `owner`. +pub(super) fn drain_start_entry( shared: &SharedState, - id: DescriptorId, + id: &DescriptorId, + owner: &DrainOwner, up_to_version: u64, max_wait: Duration, - own_holds: u32, -) -> Result<(), Error> { - // Rolling upgrade gate: no drain in mixed-version clusters. - { - let vs = shared.cluster_version_view(); - if !vs.can_activate_feature(DESCRIPTOR_DRAIN_VERSION) { - tracing::warn!( - min_version = vs.min_version, - required = DESCRIPTOR_DRAIN_VERSION, - "descriptor lease drain: cluster in compat mode, skipping drain" - ); - return Ok(()); - } - } - - // No prior version means no lease can exist. Callers skip this case - // already; the guard is cheap. - if up_to_version == 0 { - return Ok(()); - } - +) -> MetadataEntry { let now_hlc = shared.hlc_clock.now(); let ttl_ns: u64 = (max_wait + DRAIN_TTL_GRACE) .as_nanos() .try_into() .unwrap_or(u64::MAX); - let expires_at = Hlc::new(now_hlc.wall_ns.saturating_add(ttl_ns), 0); - - propose_drain( - shared, - MetadataEntry::DescriptorDrainStart { - descriptor_id: id.clone(), - up_to_version, - expires_at, - }, - "drain_start", - )?; - - match poll_leases_drained(shared, &id, up_to_version, max_wait, own_holds) { - Ok(()) => Ok(()), - Err(e) => { - // `is_draining` has no expiry backstop, so this explicit propose - // is the only thing that clears the drain after a timeout. Its own - // errors are logged and dropped. - if let Err(cleanup_err) = propose_drain( - shared, - MetadataEntry::DescriptorDrainEnd { - descriptor_id: id.clone(), - }, - "drain_end", - ) { - tracing::warn!( - error = %cleanup_err, - "descriptor lease drain: cleanup propose failed after timeout" - ); - } - Err(e) - } + MetadataEntry::DescriptorDrainStart { + descriptor_id: id.clone(), + up_to_version, + expires_at: Hlc::new(now_hlc.wall_ns.saturating_add(ttl_ns), 0), + proposer_node_id: shared.node_id, + owner: owner.clone(), } } -/// Wait until no lease or admission reservation remains on `id` at -/// `version <= up_to_version`, polling every [`POLL_INTERVAL`]. -/// -/// Sync on purpose: `metadata_proposer` beneath it is sync because pgwire DDL -/// handlers are, so an `async fn` here would ripple through every catalog-DDL -/// call site and strand the sync callers (GC sweeper, clone materializer, -/// backup restore). -/// -/// Async tasks still reach it, so on a multi-thread runtime the wait goes back -/// to tokio — parking a worker for the whole drain can delay the very -/// lease-release and raft-apply work it is waiting on. -pub(crate) fn poll_leases_drained( +/// One poll of a drain wait: `true` once no matching lease remains, an error +/// once `deadline` passed with some still held, `false` otherwise. +pub(super) fn drained_or_timed_out( shared: &SharedState, id: &DescriptorId, up_to_version: u64, max_wait: Duration, own_holds: u32, -) -> Result<(), Error> { - // `block_in_place` panics on the current-thread runtime and has no worker - // pool to hand the parked work to, so it is used only where it is legal. - match tokio::runtime::Handle::try_current() { - Ok(handle) if handle.runtime_flavor() == RuntimeFlavor::MultiThread => { - tokio::task::block_in_place(|| { - wait_for_lease_drain(shared, id, up_to_version, max_wait, own_holds) - }) - } - _ => wait_for_lease_drain(shared, id, up_to_version, max_wait, own_holds), + deadline: Instant, +) -> Result { + let remaining = count_matching_leases(shared, id, up_to_version, own_holds); + if remaining == 0 { + return Ok(true); } -} - -/// The wait loop itself, split out so the convergence condition and deadline -/// handling are identical on both paths above. -fn wait_for_lease_drain( - shared: &SharedState, - id: &DescriptorId, - up_to_version: u64, - max_wait: Duration, - own_holds: u32, -) -> Result<(), Error> { - let deadline = Instant::now() + max_wait; - loop { - let remaining = count_matching_leases(shared, id, up_to_version, own_holds); - if remaining == 0 { - return Ok(()); - } - if Instant::now() >= deadline { - return Err(Error::Config { - detail: format!( - "descriptor lease drain timed out after {max_wait:?} \ - waiting for {id:?} up to version {up_to_version} \ - (still held: {remaining})" - ), - }); - } - std::thread::sleep(POLL_INTERVAL); + if Instant::now() >= deadline { + return Err(Error::Config { + detail: format!( + "descriptor lease drain timed out after {max_wait:?} \ + waiting for {id:?} up to version {up_to_version} \ + (still held: {remaining})" + ), + }); } + Ok(false) } /// Count leases and admission reservations on `id` at `version <= /// up_to_version`. `0` means the drain has cleared; a nonzero value is /// diagnostic only, so it saturates rather than overflowing. /// -/// Leases from non-member nodes and leases past `expires_at` are both ignored. -/// A crashed node never releases its leases (no SIGTERM path runs), so without -/// these filters every DDL on those descriptors wedges forever. Missing -/// topology treats every holder as a member — the filter only drops holds it -/// is certain about. +/// Three filters drop leases whose holder can no longer use them. A crashed +/// node never releases its leases, so without them every DDL on those +/// descriptors waits out the full lease duration. /// -/// Dropping an expired lease is safe because a live holder never has one: the -/// renewal loop re-acquires before expiry, so an expired record means that -/// node's renewal stopped. A live hold on THIS node is counted through -/// `lease_refcount` and is unaffected by expiry. +/// - Non-member holders are dropped. Missing topology treats every holder as a +/// member, so the filter only drops holds it is certain about. +/// - Expired leases are dropped. A live holder never has one: the renewal loop +/// re-acquires before expiry. +/// - A SWIM-Dead holder's leases are dropped only on the metadata leader, once +/// [`nodedb_cluster::DEAD_HOLDER_LEASE_GRACE`] has passed since it went Dead +/// and the leader has seen no Raft response from it for +/// [`nodedb_cluster::DEAD_HOLDER_RAFT_SILENCE`]. By then the holder has +/// self-fenced, or it still hears the leader and applies the release. Any +/// other drainer waits for the leader's lease GC to release them. +/// +/// `expires_at.wall_ns` is stamped on the holder's own wall clock. A lease +/// held by another node therefore stays live until +/// [`nodedb_types::MAX_CLOCK_SKEW_NS`] past it. This node's own leases get no +/// margin. A live hold on THIS node is also counted through `lease_refcount`, +/// which expiry never touches. /// /// Expiry compares against wall time, not [`HlcClock::peek`]: `peek` stays -/// frozen on a quiet cluster, which would find every lease unexpired and -/// reinstate the wedge — and an idle cluster is exactly when a crashed node's -/// leases are the only ones left. `expires_at.wall_ns` is stamped from -/// `HlcClock::now()`, local wall time held monotonic; a peer's HLC is never -/// merged in, so for a lease granted elsewhere the comparison carries that -/// node's clock offset, unbounded. +/// frozen on a quiet cluster, which finds every lease unexpired and +/// reinstates the wedge — and an idle cluster is exactly when a crashed node's +/// leases are the only ones left. /// /// `own_holds` excludes that many local refcount units — the requester's own — /// from both the refcount safety net and this node's replicated cache entry, @@ -201,7 +120,22 @@ fn count_matching_leases( up_to_version: u64, own_holds: u32, ) -> usize { - let now_wall_ns = super::wall_now_ns(); + let instant = Instant::now(); + let now = nodedb_cluster::LeaseNow { + wall_ns: super::wall_now_ns(), + instant, + // Reading the leader term takes the raft coordinator lock, so skip + // it unless some holder can qualify for early release. + metadata_leader_term: if shared + .lease_runtime + .holder_liveness + .any_dead_grace_elapsed(instant) + { + metadata_leader_term(shared) + } else { + None + }, + }; let other_local_holds = shared .lease_refcount .current_at_or_below(id, up_to_version) @@ -220,7 +154,12 @@ fn count_matching_leases( .filter(|((lid, holder), l)| { lid == id && l.version <= up_to_version - && l.expires_at.wall_ns > now_wall_ns + && shared.lease_runtime.holder_liveness.lease_is_live( + *holder, + shared.node_id, + l.expires_at.wall_ns, + &now, + ) && lease_holder_is_member(shared, *holder) && !(self_only && *holder == shared.node_id) }) @@ -234,6 +173,15 @@ fn count_matching_leases( } } +/// This node's metadata-group term while it leads the group, else `None`. +fn metadata_leader_term(shared: &SharedState) -> Option { + shared + .lease_runtime + .metadata_leader_term + .get() + .and_then(|term| term()) +} + /// Whether `node_id` is a current cluster member. Missing topology treats /// every holder as a member. fn lease_holder_is_member(shared: &SharedState, node_id: u64) -> bool { @@ -246,64 +194,32 @@ fn lease_holder_is_member(shared: &SharedState, node_id: u64) -> bool { } } -/// Encode and propose a drain variant, blocking until the applied-index -/// watcher confirms it applied locally. Separate from `lease::propose_and_wait` -/// because drain variants are not `CatalogDdl` and encode differently. -fn propose_drain( - shared: &SharedState, - entry: MetadataEntry, - operation: &'static str, -) -> Result<(), Error> { - let Some(handle) = shared.metadata_raft.get() else { - // Single-node fallback: apply through the same path the applier uses, - // so drain state is exercised without a raft loop. - apply_drain_locally(shared, &entry); - return Ok(()); - }; - let raw = encode_entry(&entry).map_err(|e| Error::Config { +/// How long a drain variant's propose waits for its local apply. +pub(super) const DRAIN_PROPOSE_TIMEOUT: Duration = Duration::from_secs(5); + +/// Encode a drain variant for the metadata group. +pub(super) fn encode_drain(entry: &MetadataEntry, operation: &str) -> Result, Error> { + encode_entry(entry).map_err(|e| Error::Config { detail: format!("descriptor drain {operation} encode: {e}"), - })?; - let log_index = handle.propose(raw)?; - let watcher = shared.applied_index_watcher(nodedb_cluster::METADATA_GROUP_ID); - const DRAIN_PROPOSE_TIMEOUT: Duration = Duration::from_secs(5); - let outcome = - tokio::task::block_in_place(|| watcher.wait_for(log_index, DRAIN_PROPOSE_TIMEOUT)); - if !outcome.is_reached() { - return Err(Error::Config { - detail: format!( - "descriptor drain {operation} did not apply within {DRAIN_PROPOSE_TIMEOUT:?} \ - (log index {log_index}, current: {}, outcome: {outcome:?})", - watcher.current() - ), - }); - } - Ok(()) + }) } -/// Apply a drain variant to the local tracker without raft, so `drain_for_ddl` -/// has the same semantics in every deployment mode. -fn apply_drain_locally(shared: &SharedState, entry: &MetadataEntry) { - match entry { - MetadataEntry::DescriptorDrainStart { - descriptor_id, - up_to_version, - expires_at, - } => { - // Shares plan admission's gate: an admission completes before this - // start installs, or this drain wins and admission fails closed. - let _admission_gate = shared - .lease_admission_gate - .lock() - .unwrap_or_else(|poison| poison.into_inner()); - shared - .lease_drain - .install_start(descriptor_id.clone(), *up_to_version, *expires_at); - } - MetadataEntry::DescriptorDrainEnd { descriptor_id } => { - shared.lease_drain.install_end(descriptor_id); - } - _ => {} +/// The result of waiting for a drain variant's apply at `log_index`. +pub(super) fn drain_applied_or_error( + outcome: nodedb_cluster::WaitOutcome, + operation: &str, + log_index: u64, + current: u64, +) -> Result<(), Error> { + if outcome.is_reached() { + return Ok(()); } + Err(Error::Config { + detail: format!( + "descriptor drain {operation} did not apply within {DRAIN_PROPOSE_TIMEOUT:?} \ + (log index {log_index}, current: {current}, outcome: {outcome:?})" + ), + }) } #[cfg(test)] @@ -388,8 +304,8 @@ mod tests { } /// A minute into the future in REAL wall time — the frame the grant path - /// stamps in. Deriving it from `hlc_clock.peek()` would put fixture and - /// code under test in one frozen frame, and the assertion would prove + /// stamps in. Deriving it from `hlc_clock.peek()` puts fixture and + /// code under test in one frozen frame, and the assertion proves /// nothing. fn unexpired() -> nodedb_types::Hlc { nodedb_types::Hlc::new( @@ -458,9 +374,12 @@ mod tests { ); let (dispatcher, _data_sides) = Dispatcher::new(1, 64); let mut state = SharedState::new(dispatcher, wal).expect("construct drain count state"); - Arc::get_mut(&mut state) - .expect("single owner in test") - .cluster_topology = Some(Arc::new(std::sync::RwLock::new(topo_with(&[1])))); + let fixture = Arc::get_mut(&mut state).expect("single owner in test"); + fixture.cluster_topology = Some(Arc::new(std::sync::RwLock::new(topo_with(&[1])))); + // The WAL open stamps its empty-log time anchor from the node HLC, + // which advances it to the open's wall time. A fresh clock stands for + // an HLC that never advanced. + fixture.hlc_clock = Arc::new(nodedb_types::HlcClock::new()); let descriptor = DescriptorId::new(0, 1, DescriptorKind::Collection, "orders".to_string()); // Untouched HLC: `peek()` is at zero while wall time is decades ahead. @@ -586,4 +505,189 @@ mod tests { "a different session's hold on the same descriptor must still block the drain" ); } + + /// State whose topology holds this node and one remote holder. + fn state_with_remote_member() -> (Arc, u64, tempfile::TempDir) { + let directory = tempfile::tempdir().expect("create drain count test directory"); + let wal = Arc::new( + WalManager::open_for_testing(&directory.path().join("drain-count.wal")) + .expect("open drain count test WAL"), + ); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let mut state = SharedState::new(dispatcher, wal).expect("construct drain count state"); + let remote = state.node_id + 1; + let local = state.node_id; + Arc::get_mut(&mut state) + .expect("single owner in test") + .cluster_topology = Some(Arc::new(std::sync::RwLock::new(topo_with(&[ + local, remote, + ])))); + (state, remote, directory) + } + + fn orders() -> DescriptorId { + DescriptorId::new(0, 1, DescriptorKind::Collection, "orders".to_string()) + } + + /// An instant `ago` in the past on the monotonic clock. + fn instant_ago(ago: Duration) -> Instant { + Instant::now() + .checked_sub(ago) + .expect("monotonic clock far enough from its origin") + } + + /// A remote holder stamps `expires_at` on its own clock. A drainer whose + /// clock runs ahead must not treat that lease as expired inside the skew + /// margin. This node's own lease gets no margin. + #[tokio::test] + async fn remote_lease_inside_the_skew_margin_blocks_drain() { + let (state, remote, _directory) = state_with_remote_member(); + let descriptor = orders(); + let just_expired = + nodedb_types::Hlc::new(super::super::wall_now_ns().saturating_sub(1_000_000_000), 0); + + insert_lease(&state, &descriptor, remote, 1, just_expired); + assert_eq!( + count_matching_leases(&state, &descriptor, 1, 0), + 1, + "a remote lease one second past expiry is inside the skew margin" + ); + + state + .metadata_cache + .write() + .unwrap_or_else(|p| p.into_inner()) + .leases + .clear(); + insert_lease(&state, &descriptor, state.node_id, 1, just_expired); + assert_eq!( + count_matching_leases(&state, &descriptor, 1, 0), + 0, + "this node's own expired lease gets no skew margin" + ); + } + + const LEADER_TERM: u64 = 7; + + /// Make `state` report itself as metadata-group leader in [`LEADER_TERM`]. + fn make_metadata_leader(state: &SharedState) { + let term: Arc Option + Send + Sync> = Arc::new(|| Some(LEADER_TERM)); + if state.lease_runtime.metadata_leader_term.set(term).is_err() { + panic!("metadata leader term already set in a fresh test state"); + } + } + + /// Leader samples of `holder`'s Raft responses: `first` acks long enough + /// ago to cover the silence window, then `latest` acks now. + fn sample_raft_acks(state: &SharedState, holder: u64, first: u64, latest: u64) { + let sample = |acks| nodedb_cluster::multi_raft::PeerAckSample { + term: LEADER_TERM, + acks: vec![(holder, acks)], + }; + let window_ago = + instant_ago(nodedb_cluster::DEAD_HOLDER_RAFT_SILENCE + Duration::from_secs(1)); + state + .lease_runtime + .holder_liveness + .observe_raft_contact(&sample(first), window_ago); + state + .lease_runtime + .holder_liveness + .observe_raft_contact(&sample(latest), Instant::now()); + } + + fn dead_past_grace(state: &SharedState, holder: u64) { + state.lease_runtime.holder_liveness.record_dead_at( + holder, + instant_ago(nodedb_cluster::DEAD_HOLDER_LEASE_GRACE + Duration::from_secs(1)), + ); + } + + /// A SWIM-Dead, Raft-silent holder stays in topology. On the metadata + /// leader its unexpired lease blocks the drain until the dead grace + /// passes, then stops blocking. + #[tokio::test] + async fn dead_holder_lease_is_released_only_after_the_clamp() { + let (state, remote, _directory) = state_with_remote_member(); + make_metadata_leader(&state); + sample_raft_acks(&state, remote, 3, 3); + let descriptor = orders(); + insert_lease(&state, &descriptor, remote, 1, unexpired()); + + state + .lease_runtime + .holder_liveness + .record_dead_at(remote, Instant::now()); + assert_eq!( + count_matching_leases(&state, &descriptor, 1, 0), + 1, + "a holder that just went Dead may still be serving until it self-fences" + ); + + dead_past_grace(&state, remote); + assert_eq!( + count_matching_leases(&state, &descriptor, 1, 0), + 0, + "past the dead grace the holder has fenced, so its lease must not block" + ); + } + + /// SWIM says Dead, but the leader heard the holder over Raft recently. + /// The holder has not fenced, so its lease keeps blocking. + #[tokio::test] + async fn dead_by_swim_but_recently_acked_by_raft_keeps_the_lease() { + let (state, remote, _directory) = state_with_remote_member(); + make_metadata_leader(&state); + sample_raft_acks(&state, remote, 3, 4); + let descriptor = orders(); + insert_lease(&state, &descriptor, remote, 1, unexpired()); + dead_past_grace(&state, remote); + + assert_eq!( + count_matching_leases(&state, &descriptor, 1, 0), + 1, + "a holder that still answers Raft must keep its lease" + ); + } + + /// A drainer that does not lead the metadata group keeps a Dead holder's + /// lease live. Only the leader's lease GC releases it. + #[tokio::test] + async fn non_leader_drainer_keeps_a_dead_holders_lease() { + let (state, remote, _directory) = state_with_remote_member(); + sample_raft_acks(&state, remote, 3, 3); + let descriptor = orders(); + insert_lease(&state, &descriptor, remote, 1, unexpired()); + dead_past_grace(&state, remote); + + assert_eq!(count_matching_leases(&state, &descriptor, 1, 0), 1); + } + + /// SWIM refuting a Dead verdict with Alive clears the clamp: the holder's + /// lease blocks the drain again until its own expiry. + #[tokio::test] + async fn alive_refutation_clears_the_clamp() { + use nodedb_cluster::{MemberState, MembershipSubscriber}; + + let (state, remote, _directory) = state_with_remote_member(); + make_metadata_leader(&state); + sample_raft_acks(&state, remote, 3, 3); + let descriptor = orders(); + insert_lease(&state, &descriptor, remote, 1, unexpired()); + dead_past_grace(&state, remote); + assert_eq!(count_matching_leases(&state, &descriptor, 1, 0), 0); + + let swim_id = + nodedb_types::NodeId::try_new(remote.to_string()).expect("numeric SWIM node id"); + state.lease_runtime.holder_liveness.on_state_change( + &swim_id, + Some(MemberState::Dead), + MemberState::Alive, + ); + assert_eq!( + count_matching_leases(&state, &descriptor, 1, 0), + 1, + "an Alive refutation must restore the lease as a blocking hold" + ); + } } diff --git a/nodedb/src/control/lease/drain_propose_async.rs b/nodedb/src/control/lease/drain_propose_async.rs new file mode 100644 index 000000000..3246ccaf3 --- /dev/null +++ b/nodedb/src/control/lease/drain_propose_async.rs @@ -0,0 +1,161 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The descriptor lease drain, proposed and awaited. +//! +//! The flow is the one [`super::drain_propose`] describes. Every wait is +//! awaited: the drain variants' metadata commits apply while the proposer's +//! task yields, and the lease wait sleeps until a hold or lease changes or +//! the timer fires. No runtime flavor loses a worker to it. + +use std::time::{Duration, Instant}; + +use nodedb_cluster::{DescriptorId, DrainOwner, MetadataEntry}; + +use crate::control::metadata_proposer::wait::wait_applied; +use crate::control::state::SharedState; +use crate::error::Error; + +use super::drain_propose::{ + DRAIN_PROPOSE_TIMEOUT, POLL_INTERVAL, drain_applied_or_error, drain_start_entry, + drained_or_timed_out, encode_drain, +}; + +/// Drain every lease on `id` at `version <= up_to_version` for a `Put*` DDL. +/// +/// The drain is owned by [`DrainOwner::Ddl`], so the apply of the DDL's own +/// catalog entry ends it and no other owner's drain. +/// +/// Returns `Ok(())` once they have drained. Errors on timeout or propose +/// failure. +/// +/// `own_holds` is how many of those refcount units the requesting transaction +/// holds itself — `0` for a caller with no lease scope of its own. A +/// transaction altering a descriptor it also holds a statement-time lease on +/// cannot wait for its own hold: it cannot release that lease until this +/// call returns. +pub async fn drain_for_ddl_async( + shared: &SharedState, + id: DescriptorId, + up_to_version: u64, + max_wait: Duration, + own_holds: u32, +) -> Result<(), Error> { + start_drain( + shared, + id, + DrainOwner::Ddl, + up_to_version, + max_wait, + own_holds, + ) + .await +} + +/// Drain every lease on `id`, at any version, under `owner`. +/// +/// For a holder that stops all writes for a span of work rather than for one +/// version bump. Covering every version keeps the drain in force when a DDL +/// bumps the descriptor's version meanwhile. Only [`end_drain_async`] with the +/// same `owner`, or the owner's own implicit clear, ends it. +pub async fn drain_for_owner_async( + shared: &SharedState, + id: DescriptorId, + owner: DrainOwner, + max_wait: Duration, +) -> Result<(), Error> { + start_drain(shared, id, owner, u64::MAX, max_wait, 0).await +} + +async fn start_drain( + shared: &SharedState, + id: DescriptorId, + owner: DrainOwner, + up_to_version: u64, + max_wait: Duration, + own_holds: u32, +) -> Result<(), Error> { + // No prior version means no lease can exist. + if up_to_version == 0 { + return Ok(()); + } + propose_drain_async( + shared, + drain_start_entry(shared, &id, &owner, up_to_version, max_wait), + "drain_start", + ) + .await?; + + let deadline = Instant::now() + max_wait; + let drained = loop { + // A hold given back or a lease release applied wakes the wait. The + // timer covers what no event marks: a lease expiry, a holder's death. + let changed = shared.lease_drain.holds_changed(); + tokio::pin!(changed); + changed.as_mut().enable(); + match drained_or_timed_out(shared, &id, up_to_version, max_wait, own_holds, deadline) { + Ok(true) => break Ok(()), + Ok(false) => { + tokio::select! { + _ = changed.as_mut() => {} + _ = tokio::time::sleep(POLL_INTERVAL) => {} + } + } + Err(error) => break Err(error), + } + }; + if let Err(error) = drained { + // `is_draining` has no expiry backstop, so this explicit propose is + // the only thing that clears the drain after a timeout. Its own + // errors are logged and dropped. + if let Err(cleanup_err) = end_drain_async(shared, id, owner).await { + tracing::warn!( + error = %cleanup_err, + "descriptor lease drain: cleanup propose failed after timeout" + ); + } + return Err(error); + } + Ok(()) +} + +/// End `owner`'s drain on `id` on every node, and wait until the end applies +/// on this node. Other owners' drains on `id` stay. +/// +/// For a DDL whose own catalog entry never commits, so no implicit clear +/// runs, and for every [`drain_for_owner_async`] holder. Ending a drain that +/// is not active is a no-op on every node. +pub async fn end_drain_async( + shared: &SharedState, + id: DescriptorId, + owner: DrainOwner, +) -> Result<(), Error> { + propose_drain_async( + shared, + MetadataEntry::DescriptorDrainEnd { + descriptor_id: id, + owner, + }, + "drain_end", + ) + .await +} + +/// Propose a drain variant and await its apply on this node. +async fn propose_drain_async( + shared: &SharedState, + entry: MetadataEntry, + operation: &'static str, +) -> Result<(), Error> { + let handle = shared.metadata_raft_handle()?; + let log_index = handle + .propose_async(encode_drain(&entry, operation)?) + .await?; + let watcher = shared.applied_index_watcher(nodedb_cluster::METADATA_GROUP_ID); + let outcome = wait_applied( + std::sync::Arc::clone(&watcher), + log_index, + DRAIN_PROPOSE_TIMEOUT, + ) + .await?; + drain_applied_or_error(outcome, operation, log_index, watcher.current()) +} diff --git a/nodedb/src/control/lease/gc.rs b/nodedb/src/control/lease/gc.rs index e207e88e2..8ffb5cc59 100644 --- a/nodedb/src/control/lease/gc.rs +++ b/nodedb/src/control/lease/gc.rs @@ -4,51 +4,23 @@ //! //! A crashed node's leases are never TTL-pruned from `MetadataCache.leases` //! (only a `DescriptorLeaseRelease` entry removes them), so every DDL drain -//! on those descriptors times out forever. Two triggers run this module: -//! the `TopologyChange::Leave` apply hook (immediate) and the metadata -//! leader's periodic sweep (safety net). +//! on those descriptors times out forever. The durable leave cleanup +//! (`lease::leave_cleanup`) runs this module. The metadata leader's periodic sweep in +//! `nodedb-cluster` is the safety net for non-member and SWIM-Dead holders. use nodedb_cluster::DescriptorId; use crate::control::lease::release::LeaseReleaseHandle; use crate::control::state::SharedState; -/// Collect `(node_id, descriptor_ids)` for every lease holder that is no -/// longer a cluster member. Missing topology → empty (never GC on guesswork). -/// -/// Host-side sibling of the cluster-side sweep's pure collector -/// (`nodedb-cluster::raft_loop::lease_gc::collect_non_member_lease_releases`); -/// kept here for the GC API surface and exercised by unit tests — the -/// production sweep path lives in the cluster crate, which cannot depend on -/// this one. -#[allow(dead_code)] -pub(crate) fn collect_non_member_leases(shared: &SharedState) -> Vec<(u64, Vec)> { - let Some(topo) = &shared.cluster_topology else { - return Vec::new(); - }; - let cache = shared - .metadata_cache - .read() - .unwrap_or_else(|p| p.into_inner()); - let topo = topo.read().unwrap_or_else(|p| p.into_inner()); - - let mut by_holder: std::collections::HashMap> = - std::collections::HashMap::new(); - for (id, holder) in cache.leases.keys() { - if !topo.contains(*holder) { - by_holder.entry(*holder).or_default().push(id.clone()); - } - } - let mut out: Vec<(u64, Vec)> = by_holder.into_iter().collect(); - out.sort_by_key(|(node_id, _)| *node_id); - out -} - /// Propose `DescriptorLeaseRelease` for every lease held by `node_id`. /// No-op if the cache has no entries for that node (idempotent vs. the -/// periodic sweep). Blocks on the local applied watermark like the -/// normal release path; callers on hot paths must spawn this. -pub(crate) fn gc_leases_for_node(shared: &SharedState, node_id: u64) -> Result<(), crate::Error> { +/// periodic sweep). Awaits the local applied watermark like the normal +/// release path. +pub(crate) async fn gc_leases_for_node( + shared: &SharedState, + node_id: u64, +) -> Result<(), crate::Error> { let ids: Vec = { let cache = shared .metadata_cache @@ -64,7 +36,57 @@ pub(crate) fn gc_leases_for_node(shared: &SharedState, node_id: u64) -> Result<( if ids.is_empty() { return Ok(()); } - LeaseReleaseHandle::from_shared(shared).release_for_node(node_id, ids) + LeaseReleaseHandle::from_shared(shared)? + .release_for_node(node_id, ids) + .await +} + +/// End every active drain whose proposer is no longer in the topology. +/// +/// The periodic backstop for the `TopologyChange::Leave` hook: a drain end +/// that hook failed to apply is retried here. A node that is only suspected +/// Dead stays in the topology, so its drains stay: it can still be running +/// the DDL they protect. A state no boot wired has no topology and no foreign +/// proposer. +pub(crate) async fn end_orphaned_drains(shared: &SharedState) { + let Some(topology) = shared.cluster_topology.as_ref() else { + return; + }; + let orphaned: Vec = { + let topology = topology.read().unwrap_or_else(|p| p.into_inner()); + let mut proposers: Vec = shared + .lease_drain + .snapshot() + .into_iter() + .map(|(_, _, entry)| entry.proposer_node_id) + .filter(|node_id| !topology.contains(*node_id)) + .collect(); + proposers.sort_unstable(); + proposers.dedup(); + proposers + }; + for node_id in orphaned { + end_drains_for_node(shared, node_id).await; + } +} + +/// Propose `DescriptorDrainEnd` for every active drain `node_id` proposed. +/// +/// Runs only for a node that left the topology: it can never end its own +/// drains, and it runs no DDL any more. A drain whose end does not apply +/// stays active and is logged. +pub(crate) async fn end_drains_for_node(shared: &SharedState, node_id: u64) { + for (id, owner) in shared.lease_drain.proposed_by(node_id) { + if let Err(error) = super::end_drain_async(shared, id.clone(), owner.clone()).await { + tracing::warn!( + node_id, + descriptor = ?id, + ?owner, + %error, + "drain end for a node that left did not apply" + ); + } + } } #[cfg(test)] @@ -95,19 +117,6 @@ mod tests { DescriptorId::new(0, 1, DescriptorKind::Collection, name.to_string()) } - fn topo_with(ids: &[u64]) -> nodedb_cluster::ClusterTopology { - let mut t = nodedb_cluster::ClusterTopology::new(); - for (i, id) in ids.iter().enumerate() { - let addr: std::net::SocketAddr = format!("127.0.0.1:{}", 9000 + i).parse().unwrap(); - t.add_node(nodedb_cluster::NodeInfo::new( - *id, - addr, - nodedb_cluster::NodeState::Active, - )); - } - t - } - fn insert_lease(state: &SharedState, descriptor: &DescriptorId, holder: u64) { let now = state.hlc_clock.peek(); state @@ -129,54 +138,29 @@ mod tests { ); } - #[test] - fn collect_non_member_leases_returns_only_foreign_holders() { - let (state, _directory) = test_state(); - let mut state = state; - Arc::get_mut(&mut state) - .expect("single owner in test") - .cluster_topology = Some(Arc::new(std::sync::RwLock::new(topo_with(&[1])))); - let descriptor = id("orders"); - - insert_lease(&state, &descriptor, 1); // member — must be kept - insert_lease(&state, &descriptor, 2); // not in topology — GC target - - let collected = collect_non_member_leases(&state); - assert_eq!(collected.len(), 1); - assert_eq!(collected[0].0, 2); - assert_eq!(collected[0].1, vec![descriptor]); - } - - #[test] - fn collect_non_member_leases_empty_without_topology() { - let (state, _directory) = test_state(); - // `cluster_topology` is None in single-node mode: never GC on guesswork. - let descriptor = id("orders"); - insert_lease(&state, &descriptor, 99); - - assert!(collect_non_member_leases(&state).is_empty()); - } - /// Fake metadata raft handle: records proposed entries and bumps the - /// applied watcher so the release path's `wait_for` returns immediately. + /// applied watcher so the release path's apply wait returns at once. struct RecordingProposer { proposed: std::sync::Mutex>>, watcher: Arc, } impl crate::control::metadata_proposer::MetadataRaftHandle for RecordingProposer { - fn propose(&self, bytes: Vec) -> Result { + fn propose_async<'a>( + &'a self, + bytes: Vec, + ) -> crate::control::metadata_proposer::ProposeFuture<'a> { self.proposed .lock() .unwrap_or_else(|p| p.into_inner()) .push(bytes); self.watcher.bump(1); - Ok(1) + Box::pin(std::future::ready(Ok(1))) } } - #[test] - fn gc_leases_for_node_proposes_descriptor_lease_release() { + #[tokio::test] + async fn gc_leases_for_node_proposes_descriptor_lease_release() { let (state, _directory) = test_state(); let watcher = state.applied_index_watcher(nodedb_cluster::METADATA_GROUP_ID); let proposer = Arc::new(RecordingProposer { @@ -191,7 +175,9 @@ mod tests { let descriptor = id("orders"); insert_lease(&state, &descriptor, 2); - gc_leases_for_node(&state, 2).expect("gc release for node 2"); + gc_leases_for_node(&state, 2) + .await + .expect("gc release for node 2"); let proposed = proposer.proposed.lock().unwrap_or_else(|p| p.into_inner()); assert_eq!(proposed.len(), 1); @@ -205,8 +191,8 @@ mod tests { )); } - #[test] - fn gc_leases_for_node_noop_when_no_entries() { + #[tokio::test] + async fn gc_leases_for_node_noop_when_no_entries() { let (state, _directory) = test_state(); let watcher = state.applied_index_watcher(nodedb_cluster::METADATA_GROUP_ID); let proposer = Arc::new(RecordingProposer { @@ -219,7 +205,7 @@ mod tests { .unwrap_or_else(|_| panic!("metadata raft handle already set in test")); // No leases at all for node 2 (or anyone): must not propose. - gc_leases_for_node(&state, 2).expect("gc noop"); + gc_leases_for_node(&state, 2).await.expect("gc noop"); assert!( proposer .proposed @@ -230,7 +216,9 @@ mod tests { // Leases held by OTHER nodes are also not this node's GC target. insert_lease(&state, &id("other"), 3); - gc_leases_for_node(&state, 2).expect("gc noop for foreign-only leases"); + gc_leases_for_node(&state, 2) + .await + .expect("gc noop for foreign-only leases"); assert!( proposer .proposed diff --git a/nodedb/src/control/lease/holders.rs b/nodedb/src/control/lease/holders.rs new file mode 100644 index 000000000..6988bad7e --- /dev/null +++ b/nodedb/src/control/lease/holders.rs @@ -0,0 +1,419 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! In-flight query holds per descriptor, so a lease this node loses can end +//! the queries still running under it. +//! +//! A query registers here when its [`super::QueryLeaseScope`] is built and +//! deregisters when the scope drops. Two events revoke holds: +//! +//! - A `DescriptorLeaseRelease` for this node applies here. Other nodes then +//! treat the lease as gone and a DDL drain can pass. +//! - This node self-fences: it has lost metadata-leader contact long enough +//! that other nodes can release its leases. +//! +//! A revoked query ends with [`Error::RetryableSchemaChanged`], so the client +//! retries it against the current descriptor version. + +use std::collections::HashMap; +use std::sync::{Arc, Mutex}; + +use nodedb_cluster::DescriptorId; +use tokio::sync::watch; + +use crate::control::state::SharedState; +use crate::error::Error; + +/// Cap on query holds tracked at once on one node. An untracked hold +/// cannot be revoked, so admission past the cap fails with a retryable error. +const MAX_TRACKED_LEASE_HOLDS: usize = 65_536; + +/// The revocation signal of one query's lease scope. +#[derive(Debug)] +pub struct LeaseRevocation { + /// `Some(detail)` once revoked. Set once, never cleared. + state: watch::Sender>, +} + +impl LeaseRevocation { + fn new() -> Self { + let (state, _) = watch::channel(None); + Self { state } + } + + /// Mark the scope revoked because this node lost its lease on `descriptor`. + /// A second revocation keeps the first reason. + pub fn revoke(&self, descriptor: &DescriptorId) { + self.state.send_if_modified(|state| { + if state.is_some() { + return false; + } + *state = Some(format!( + "{descriptor:?} (this node lost its descriptor lease while the statement ran)" + )); + true + }); + } + + /// The retryable error for a revoked scope, or `None` while it is live. + pub fn revoked_error(&self) -> Option { + self.state + .borrow() + .clone() + .map(|descriptor| Error::RetryableSchemaChanged { descriptor }) + } + + /// Resolve with the retryable error once the scope is revoked. + pub async fn revoked(&self) -> Error { + let mut rx = self.state.subscribe(); + loop { + if let Some(descriptor) = rx.borrow_and_update().clone() { + return Error::RetryableSchemaChanged { descriptor }; + } + if rx.changed().await.is_err() { + // The sender lives as long as `self`, so this never resolves. + std::future::pending::<()>().await; + } + } + } +} + +#[derive(Debug, Default)] +struct HoldersInner { + next_key: u64, + total: usize, + by_descriptor: HashMap>>, +} + +/// Every in-flight query hold on this node, keyed by descriptor. +#[derive(Debug, Default)] +pub struct LeaseHolders { + inner: Mutex, +} + +/// One registered query hold. Pass it back to [`LeaseHolders::deregister`]. +#[derive(Debug)] +pub struct HolderTicket { + key: u64, + descriptors: Vec, + revocation: Arc, +} + +impl HolderTicket { + pub fn revocation(&self) -> &Arc { + &self.revocation + } +} + +impl LeaseHolders { + pub fn new() -> Self { + Self::default() + } + + /// Register one query holding `descriptors`. Fails with a retryable + /// error when the node already tracks [`MAX_TRACKED_LEASE_HOLDS`] holds. + pub fn register(&self, descriptors: Vec) -> Result { + let mut inner = self.inner.lock().unwrap_or_else(|p| p.into_inner()); + if inner.total.saturating_add(descriptors.len()) > MAX_TRACKED_LEASE_HOLDS { + return Err(Error::RetryableSchemaChanged { + descriptor: format!( + "{descriptors:?} (descriptor lease holder table full at \ + {MAX_TRACKED_LEASE_HOLDS} holds; retry once statements finish)" + ), + }); + } + let key = inner.next_key; + inner.next_key = inner.next_key.wrapping_add(1); + let revocation = Arc::new(LeaseRevocation::new()); + for descriptor in &descriptors { + inner + .by_descriptor + .entry(descriptor.clone()) + .or_default() + .insert(key, Arc::clone(&revocation)); + } + inner.total = inner.total.saturating_add(descriptors.len()); + Ok(HolderTicket { + key, + descriptors, + revocation, + }) + } + + /// Remove a hold at query end. + pub fn deregister(&self, ticket: &HolderTicket) { + let mut inner = self.inner.lock().unwrap_or_else(|p| p.into_inner()); + let mut removed = 0usize; + for descriptor in &ticket.descriptors { + if let Some(holders) = inner.by_descriptor.get_mut(descriptor) { + if holders.remove(&ticket.key).is_some() { + removed += 1; + } + if holders.is_empty() { + inner.by_descriptor.remove(descriptor); + } + } + } + inner.total = inner.total.saturating_sub(removed); + } + + /// Revoke every hold on any of `descriptors`. Returns how many queries + /// were revoked. + pub fn revoke(&self, descriptors: &[DescriptorId]) -> usize { + let inner = self.inner.lock().unwrap_or_else(|p| p.into_inner()); + let mut revoked = 0usize; + for descriptor in descriptors { + if let Some(holders) = inner.by_descriptor.get(descriptor) { + for revocation in holders.values() { + revocation.revoke(descriptor); + revoked += 1; + } + } + } + revoked + } + + /// Revoke every hold on this node. Returns how many holds were revoked. + pub fn revoke_all(&self) -> usize { + let inner = self.inner.lock().unwrap_or_else(|p| p.into_inner()); + let mut revoked = 0usize; + for (descriptor, holders) in &inner.by_descriptor { + for revocation in holders.values() { + revocation.revoke(descriptor); + revoked += 1; + } + } + revoked + } + + /// Number of tracked holds. + pub fn tracked(&self) -> usize { + self.inner.lock().unwrap_or_else(|p| p.into_inner()).total + } +} + +/// Revoke this node's queries on `descriptor_ids` when a committed +/// `DescriptorLeaseRelease` for `released_node` applies here. +pub fn revoke_on_release( + shared: &SharedState, + released_node: u64, + descriptor_ids: &[DescriptorId], +) { + if released_node != shared.node_id { + return; + } + let revoked = shared.lease_runtime.holders.revoke(descriptor_ids); + if revoked > 0 { + tracing::info!( + revoked, + descriptors = descriptor_ids.len(), + "descriptor lease released while in use; revoked in-flight statements" + ); + } +} + +/// Revoke every in-flight query on this node once it has gone +/// [`nodedb_cluster::LEASE_SELF_FENCE_WINDOW`] without metadata-leader +/// contact. Called periodically by the lease renewal loop. +/// +/// Only lost contact counts here. A replica merely behind on apply refuses +/// its cached lease for new statements, but its running ones stay valid: +/// other nodes release a lease only after Raft silence, not apply lag. Before +/// `start_raft` installs the contact check, nothing is revoked. +pub fn revoke_if_fenced(shared: &SharedState) { + let Some(in_contact) = shared.lease_runtime.metadata_contact.get() else { + return; + }; + if in_contact(nodedb_cluster::LEASE_SELF_FENCE_WINDOW) { + return; + } + let revoked = shared.lease_runtime.holders.revoke_all(); + if revoked > 0 { + tracing::warn!( + revoked, + "metadata-leader contact lost past the lease self-fence window; \ + revoked in-flight statements" + ); + } +} + +#[cfg(test)] +mod tests { + use std::time::Duration; + + use nodedb_cluster::DescriptorKind; + + use super::*; + + fn id(name: &str) -> DescriptorId { + DescriptorId::new(0, 1, DescriptorKind::Collection, name.to_string()) + } + + #[test] + fn revoke_reaches_only_holders_of_that_descriptor() { + let holders = LeaseHolders::new(); + let orders = holders.register(vec![id("orders")]).expect("register"); + let users = holders.register(vec![id("users")]).expect("register"); + + assert_eq!(holders.revoke(&[id("orders")]), 1); + assert!(matches!( + orders.revocation().revoked_error(), + Some(Error::RetryableSchemaChanged { .. }) + )); + assert!(users.revocation().revoked_error().is_none()); + } + + #[test] + fn deregistered_holds_are_not_revoked() { + let holders = LeaseHolders::new(); + let ticket = holders + .register(vec![id("orders"), id("users")]) + .expect("register"); + assert_eq!(holders.tracked(), 2); + holders.deregister(&ticket); + assert_eq!(holders.tracked(), 0); + assert_eq!(holders.revoke_all(), 0); + } + + #[test] + fn registration_is_bounded() { + let holders = LeaseHolders::new(); + let many: Vec = (0..MAX_TRACKED_LEASE_HOLDS) + .map(|i| id(&format!("c{i}"))) + .collect(); + let _full = holders.register(many).expect("fill to the cap"); + assert!(matches!( + holders.register(vec![id("one_more")]), + Err(Error::RetryableSchemaChanged { .. }) + )); + } + + #[tokio::test] + async fn revoked_future_resolves_on_revoke() { + let holders = Arc::new(LeaseHolders::new()); + let ticket = holders.register(vec![id("orders")]).expect("register"); + let revocation = Arc::clone(ticket.revocation()); + let waiter = tokio::spawn(async move { revocation.revoked().await }); + tokio::task::yield_now().await; + holders.revoke(&[id("orders")]); + let error = tokio::time::timeout(Duration::from_secs(5), waiter) + .await + .expect("revocation observed promptly") + .expect("waiter did not panic"); + assert!(matches!(error, Error::RetryableSchemaChanged { .. })); + } + + /// A long-running statement under a lease is cancelled with a retryable + /// error once a release of that lease for this node applies. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn long_running_query_is_cancelled_when_its_lease_is_released() { + use crate::control::planner::descriptor_set::DescriptorVersionSet; + + let cluster = crate::control::cluster::test_one_node::boot().await; + let state = Arc::clone(&cluster.state); + let descriptor = id("orders"); + let mut versions = DescriptorVersionSet::new(); + versions.record(descriptor.clone(), 1); + let scope = state + .acquire_plan_lease_scope(&versions) + .await + .expect("admit the statement"); + assert_eq!(state.lease_runtime.holders.tracked(), 1); + + let running = tokio::spawn(async move { + scope + .guard(tokio::time::sleep(Duration::from_secs(600))) + .await + }); + tokio::task::yield_now().await; + + revoke_on_release(&state, state.node_id + 1, std::slice::from_ref(&descriptor)); + assert!( + !running.is_finished(), + "another node's release must not cancel" + ); + + revoke_on_release(&state, state.node_id, std::slice::from_ref(&descriptor)); + let outcome = tokio::time::timeout(Duration::from_secs(5), running) + .await + .expect("the statement ends promptly") + .expect("the statement task did not panic"); + assert!(matches!(outcome, Err(Error::RetryableSchemaChanged { .. }))); + assert_eq!( + state.lease_runtime.holders.tracked(), + 0, + "the scope deregistered on drop" + ); + drop(state); + cluster.shutdown().await; + } + + /// Lost metadata-leader contact revokes every running statement; contact + /// within the window revokes none. + #[tokio::test] + async fn lost_leader_contact_revokes_running_statements() { + use std::sync::atomic::{AtomicBool, Ordering}; + + use crate::bridge::dispatch::Dispatcher; + use crate::wal::WalManager; + + let directory = tempfile::tempdir().expect("create holder test directory"); + let wal = Arc::new( + WalManager::open_for_testing(&directory.path().join("fence.wal")) + .expect("open holder test WAL"), + ); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let state = SharedState::new(dispatcher, wal).expect("construct holder test state"); + let in_contact = Arc::new(AtomicBool::new(true)); + let probe = Arc::clone(&in_contact); + let contact: Arc bool + Send + Sync> = + Arc::new(move |_window| probe.load(Ordering::SeqCst)); + if state.lease_runtime.metadata_contact.set(contact).is_err() { + panic!("metadata contact fn already set in a fresh test state"); + } + let ticket = state + .lease_runtime + .holders + .register(vec![id("orders")]) + .expect("register"); + + revoke_if_fenced(&state); + assert!(ticket.revocation().revoked_error().is_none()); + + in_contact.store(false, Ordering::SeqCst); + revoke_if_fenced(&state); + assert!(ticket.revocation().revoked_error().is_some()); + state.lease_runtime.holders.deregister(&ticket); + } + + /// A statement boundary after revocation fails, and a guard refuses even + /// work that is already complete: nothing runs on a revoked scope. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn revoked_scope_fails_boundary_checks_and_guards() { + use crate::control::planner::descriptor_set::DescriptorVersionSet; + + let cluster = crate::control::cluster::test_one_node::boot().await; + let state = Arc::clone(&cluster.state); + let descriptor = id("orders"); + let mut versions = DescriptorVersionSet::new(); + versions.record(descriptor.clone(), 1); + let scope = state + .acquire_plan_lease_scope(&versions) + .await + .expect("admit the statement"); + assert!(scope.check_not_revoked().is_ok()); + assert!(matches!(scope.guard(async { 7 }).await, Ok(7))); + + revoke_on_release(&state, state.node_id, std::slice::from_ref(&descriptor)); + assert!(matches!( + scope.check_not_revoked(), + Err(Error::RetryableSchemaChanged { .. }) + )); + assert!(matches!( + scope.guard(async { 7 }).await, + Err(Error::RetryableSchemaChanged { .. }) + )); + drop(scope); + drop(state); + cluster.shutdown().await; + } +} diff --git a/nodedb/src/control/lease/leader_wait.rs b/nodedb/src/control/lease/leader_wait.rs index 6679376d9..52c7aca70 100644 --- a/nodedb/src/control/lease/leader_wait.rs +++ b/nodedb/src/control/lease/leader_wait.rs @@ -1,8 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 //! Propose-and-wait: encode a metadata entry, propose it through raft, -//! and block until it applies locally, retrying past a transient -//! leader election. +//! and await its local apply, retrying past a transient leader election. use std::time::Duration; @@ -30,13 +29,13 @@ const LEADER_ELECTION_BACKOFF_MS: [u64; 7] = [10, 25, 50, 100, 200, 400, 800]; /// Every error other than [`Error::MetadataLeaderUnavailable`] is returned /// immediately — only the transient no-leader case is retried, and only for a /// bounded number of attempts. -fn propose_once_leader_is_elected( +async fn propose_once_leader_is_elected( handle: &dyn crate::control::metadata_proposer::MetadataRaftHandle, raw: Vec, operation: &'static str, ) -> Result { for (attempt, backoff_ms) in LEADER_ELECTION_BACKOFF_MS.iter().enumerate() { - match handle.propose(raw.clone()) { + match handle.propose_async(raw.clone()).await { Ok(log_index) => return Ok(log_index), Err(Error::MetadataLeaderUnavailable) => { tracing::debug!( @@ -44,53 +43,40 @@ fn propose_once_leader_is_elected( operation, "descriptor lease: metadata election in progress; re-proposing" ); - tokio::task::block_in_place(|| { - std::thread::sleep(Duration::from_millis(*backoff_ms)); - }); + tokio::time::sleep(Duration::from_millis(*backoff_ms)).await; } Err(other) => return Err(other), } } // One final attempt so the caller sees a live verdict rather than a stale // one from before the last backoff. - handle.propose(raw) + handle.propose_async(raw).await } -/// Encode `entry`, propose through the metadata raft handle, and -/// block on the local applied watermark until the proposed log -/// index is applied (or the timeout fires). +/// Encode `entry`, propose it through the metadata raft handle, and await +/// the local applied watermark until the proposed log index applies or the +/// timeout fires. /// -/// Shared by `acquire_lease` and `release_leases`. `operation` is a -/// short label used for diagnostic error messages — it appears in -/// both the encode-failure and timeout paths. -/// -/// Caller must already have checked `shared.metadata_raft.get()` -/// and decided to take the cluster path; this helper does NOT -/// implement the single-node fallback, because the two callers -/// have different fallback semantics (acquire writes the lease -/// into the cache, release removes entries) and inlining the -/// fallback here would couple them artificially. -pub(in crate::control) fn propose_and_wait( +/// Shared by the lease grant and release paths. `operation` is a short label +/// for the encode-failure and timeout errors. +pub(in crate::control) async fn propose_and_wait( shared: &SharedState, entry: &MetadataEntry, operation: &'static str, ) -> Result { - let Some(handle) = shared.metadata_raft.get() else { - // Programmer error — callers must check this themselves. - return Err(Error::Config { - detail: format!("descriptor lease {operation}: no metadata raft handle"), - }); - }; + let handle = shared.metadata_raft_handle()?; let raw = encode_entry(entry).map_err(|e| Error::Config { detail: format!("descriptor lease {operation} encode: {e}"), })?; - let log_index = propose_once_leader_is_elected(handle.as_ref(), raw, operation)?; + let log_index = propose_once_leader_is_elected(handle.as_ref(), raw, operation).await?; - // `wait_for` parks the calling thread on a Condvar — wrap in - // `block_in_place` so tokio reassigns a fresh worker and the - // raft tick that bumps the watcher is not starved. let watcher = shared.applied_index_watcher(nodedb_cluster::METADATA_GROUP_ID); - let outcome = tokio::task::block_in_place(|| watcher.wait_for(log_index, PROPOSE_TIMEOUT)); + let outcome = crate::control::metadata_proposer::wait::wait_applied( + std::sync::Arc::clone(&watcher), + log_index, + PROPOSE_TIMEOUT, + ) + .await?; if !outcome.is_reached() { return Err(Error::Config { detail: format!( diff --git a/nodedb/src/control/lease/leave_cleanup.rs b/nodedb/src/control/lease/leave_cleanup.rs new file mode 100644 index 000000000..9d13af3e6 --- /dev/null +++ b/nodedb/src/control/lease/leave_cleanup.rs @@ -0,0 +1,126 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Durable lease and drain cleanup for a node that left the cluster. +//! +//! The `TopologyChange::Leave` apply writes a `_system.pending_leave_cleanup` +//! row before it returns. [`drive_leave_cleanup`] releases the node's leases +//! and ends the drains it proposed, from the singleton worker, and removes +//! the row once the metadata cache holds neither. The Leave post-apply, the +//! boot drain, and the retry worker all call it, so a cleanup a crash or a +//! failed proposal interrupted is driven again. + +use crate::control::state::SharedState; + +/// Whether this node's metadata state holds no lease of `node_id` and no +/// drain it proposed. +fn cleanup_done(shared: &SharedState, node_id: u64) -> bool { + let holds_lease = shared + .metadata_cache + .read() + .unwrap_or_else(|p| p.into_inner()) + .leases + .keys() + .any(|(_, holder)| *holder == node_id); + !holds_lease && shared.lease_drain.proposed_by(node_id).is_empty() +} + +/// Drive the cleanup `node_id` owes. Returns whether its row is gone. +/// +/// Only the singleton worker proposes. Every node removes its row once the +/// release and drain-end entries applied here. Awaits the local applied +/// watermark while it proposes. +pub async fn drive_leave_cleanup(shared: &SharedState, node_id: u64) -> crate::Result { + if shared.is_singleton_worker() && !cleanup_done(shared, node_id) { + if let Err(error) = super::gc::gc_leases_for_node(shared, node_id).await { + tracing::warn!( + node_id, + %error, + "lease release for a node that left did not apply; the retry worker re-drives it" + ); + } + super::gc::end_drains_for_node(shared, node_id).await; + } + if !cleanup_done(shared, node_id) { + return Ok(false); + } + shared + .credentials + .catalog() + .remove_pending_leave_cleanup(node_id)?; + Ok(true) +} + +/// Drive every owed leave cleanup. Returns how many rows remain. +/// +/// `Err` only when the rows cannot be read. +pub async fn drain_pending_leave_cleanups(shared: &SharedState) -> crate::Result { + let mut remaining = 0usize; + for node_id in shared.credentials.catalog().load_pending_leave_cleanups()? { + match drive_leave_cleanup(shared, node_id).await { + Ok(true) => {} + Ok(false) => remaining += 1, + Err(error) => { + remaining += 1; + tracing::warn!(node_id, %error, "leave cleanup row could not be removed"); + } + } + } + Ok(remaining) +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use nodedb_cluster::{DescriptorId, DescriptorKind, DescriptorLease, DrainOwner}; + use nodedb_types::Hlc; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::wal::WalManager; + + fn state() -> (Arc, tempfile::TempDir) { + let dir = tempfile::tempdir().unwrap(); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("leave.wal")).unwrap()); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + (SharedState::new(dispatcher, wal).unwrap(), dir) + } + + /// A row stays while the left node still holds a lease or a drain, and + /// goes once neither remains. + #[tokio::test] + async fn row_stays_until_leases_and_drains_are_gone() { + let (state, _dir) = state(); + let id = DescriptorId::new(0, 1, DescriptorKind::Collection, "orders"); + state + .credentials + .catalog() + .enqueue_pending_leave_cleanup(7, 3) + .unwrap(); + state.metadata_cache.write().unwrap().leases.insert( + (id.clone(), 7), + DescriptorLease { + descriptor_id: id.clone(), + version: 1, + node_id: 7, + expires_at: Hlc::new(1, 0), + }, + ); + state + .lease_drain + .install_start(id.clone(), DrainOwner::Ddl, 1, Hlc::new(1, 0), 7); + assert!(!cleanup_done(&state, 7)); + + state.metadata_cache.write().unwrap().leases.clear(); + state.lease_drain.install_end(&id, &DrainOwner::Ddl); + assert_eq!(drain_pending_leave_cleanups(&state).await.unwrap(), 0); + assert!( + state + .credentials + .catalog() + .load_pending_leave_cleanups() + .unwrap() + .is_empty() + ); + } +} diff --git a/nodedb/src/control/lease/mod.rs b/nodedb/src/control/lease/mod.rs index a0877731c..4b398ee51 100644 --- a/nodedb/src/control/lease/mod.rs +++ b/nodedb/src/control/lease/mod.rs @@ -11,35 +11,50 @@ //! entry commits on the metadata raft group. //! //! This module provides the host-side API surface — `acquire_lease` -//! and `release_leases` — that proposes those entries and blocks -//! on the local applied watermark, mirroring the -//! `metadata_proposer::propose_catalog_entry` pattern. +//! and `release_leases` — that proposes those entries and awaits +//! the local applied watermark. Synchronous code hands a release to +//! the background `releaser` instead of waiting. //! //! The planner acquires a lease before reading a descriptor to prevent //! stale reads across DDL. DDL drain consumes the `MetadataCache.leases` //! view before committing a new descriptor version. On `SIGTERM`, leases //! are released explicitly so they drain faster than expiry. +pub mod admission; pub mod descriptor_lookup; pub mod drain; +pub mod drain_apply; pub mod drain_propose; +pub mod drain_propose_async; pub mod gc; +pub mod holders; mod leader_wait; +pub mod leave_cleanup; pub mod propose; pub mod refcount; pub mod release; +pub mod releaser; pub mod renewal; +pub mod runtime; +mod self_fence; pub mod shutdown_release; mod wall_time; +pub(crate) use self_fence::lease_use_is_fenced; pub(super) use wall_time::wall_now_ns; -pub use descriptor_lookup::{descriptor_id_and_prior_version, descriptor_id_for_implicit_clear}; +pub use descriptor_lookup::{ + clear_implicit_drains, descriptor_id_and_prior_version, descriptor_id_for_implicit_clear, + drains_for_implicit_clear, move_source_descriptor, move_tenant_drain_owner, +}; pub use drain::{DescriptorDrainTracker, DrainEntry}; -pub use drain_propose::drain_for_ddl; +pub use drain_apply::{apply_drain_ends, apply_drain_start}; +pub use drain_propose_async::{drain_for_ddl_async, drain_for_owner_async, end_drain_async}; +pub use holders::{LeaseHolders, LeaseRevocation, revoke_if_fenced, revoke_on_release}; pub(super) use leader_wait::{PROPOSE_TIMEOUT, propose_and_wait}; pub use propose::{DEFAULT_LEASE_DURATION, acquire_lease, compute_expires_at, force_refresh_lease}; -pub(crate) use propose::{acquire_lease_after_admission, ensure_not_draining}; +pub(crate) use propose::{acquire_lease_after_admission, drain_owner_list, ensure_not_draining}; pub use refcount::{LeaseRefCount, QueryLeaseScope}; pub use release::release_leases; pub use renewal::{LeaseRenewalConfig, LeaseRenewalLoop}; +pub use runtime::LeaseRuntime; diff --git a/nodedb/src/control/lease/propose.rs b/nodedb/src/control/lease/propose.rs index b37f8a419..46e7bf764 100644 --- a/nodedb/src/control/lease/propose.rs +++ b/nodedb/src/control/lease/propose.rs @@ -1,7 +1,16 @@ // SPDX-License-Identifier: BUSL-1.1 -//! `acquire_lease` — synchronous propose-and-wait helper for -//! descriptor leases. Mirrors `metadata_proposer::propose_catalog_entry`. +//! `acquire_lease` — the descriptor lease grant, proposed through the +//! metadata group and awaited. +//! +//! Two gates order it: +//! +//! - `lease_admission_gate`, a std mutex held only while drain state is +//! checked and a refcount unit is reserved. It is never held across an +//! await. +//! - `lease_grant_gate`, an async mutex held from the grant's cache check +//! through its metadata apply. A release of this node's leases holds it +//! too, so a grant and a release never interleave. use std::time::Duration; @@ -16,13 +25,13 @@ use crate::error::Error; pub const DEFAULT_LEASE_DURATION: Duration = Duration::from_secs(300); /// Compute the HLC at which a lease granted at `now` for the given -/// duration should expire. Pure function so it can be unit-tested +/// duration expires. Pure function so it can be unit-tested /// without spinning up a cluster. /// /// HLC arithmetic: we only advance the wall-clock component. The /// logical counter resets to 0 on the synthetic future timestamp /// because it represents a "this is the earliest moment a real HLC -/// could observe past expiry" sentinel, not a real causal event. +/// can observe past expiry" sentinel, not a real causal event. pub fn compute_expires_at(now: Hlc, duration: Duration) -> Hlc { let delta_ns: u64 = duration.as_nanos().try_into().unwrap_or(u64::MAX); Hlc::new(now.wall_ns.saturating_add(delta_ns), 0) @@ -35,13 +44,17 @@ pub fn compute_expires_at(now: Hlc, duration: Duration) -> Hlc { /// held while proposing or waiting for raft. A slow-path caller reserves its /// exact descriptor version under the gate before releasing it; that reservation /// is visible to a drain that applies while the grant is in flight. -pub fn acquire_lease( +pub async fn acquire_lease( shared: &SharedState, descriptor_id: DescriptorId, version: u64, duration: Duration, ) -> Result { - { + // Read before any lease lock: the fence takes the raft coordinator lock. + let fenced = super::lease_use_is_fenced(shared); + // Keep this reservation live until the grant path has either installed a + // metadata lease or returned an error. A cancelled grant drops it too. + let _reservation = { let _admission_gate = shared .lease_admission_gate .lock() @@ -55,22 +68,19 @@ pub fn acquire_lease( .read() .unwrap_or_else(|p| p.into_inner()); // A cached metadata lease itself keeps a drain from clearing, so it - // needs no temporary reservation after the gate is released. - if let Some(existing) = cache.leases.get(&cache_key) + // needs no temporary reservation after the gate is released. A fenced + // holder skips it and re-acquires through raft. + if !fenced + && let Some(existing) = cache.leases.get(&cache_key) && existing.version >= version && existing.expires_at > now { return Ok(existing.clone()); } + super::refcount::RefcountReservation::reserve(shared, descriptor_id.clone(), version) + }; - // Keep this reservation live until the grant path has either installed - // a metadata lease or returned an error. - shared.lease_refcount.increment(&descriptor_id, version); - } - - let result = acquire_lease_after_admission(shared, descriptor_id.clone(), version, duration); - shared.lease_refcount.decrement(&descriptor_id, version); - result + acquire_lease_after_admission(shared, descriptor_id, version, duration).await } /// Acquire a descriptor lease after plan admission has already checked drain @@ -78,18 +88,17 @@ pub fn acquire_lease( /// takes neither the admission gate nor another drain snapshot: the existing /// reservation is the linearized admission record while its raft grant is in /// flight. -pub(crate) fn acquire_lease_after_admission( +pub(crate) async fn acquire_lease_after_admission( shared: &SharedState, descriptor_id: DescriptorId, version: u64, duration: Duration, ) -> Result { + // Read before any lease lock: the fence takes the raft coordinator lock. + let fenced = super::lease_use_is_fenced(shared); // This gate is intentionally independent from admission: it serializes // first-holder and version-upgrade grants while raft applies metadata. - let _grant_gate = shared - .lease_grant_gate - .lock() - .unwrap_or_else(|poison| poison.into_inner()); + let _grant_gate = shared.lease_grant_gate.lock().await; let now = shared.hlc_clock.now(); let cache_key = (descriptor_id.clone(), shared.node_id); { @@ -97,7 +106,8 @@ pub(crate) fn acquire_lease_after_admission( .metadata_cache .read() .unwrap_or_else(|p| p.into_inner()); - if let Some(existing) = cache.leases.get(&cache_key) + if !fenced + && let Some(existing) = cache.leases.get(&cache_key) && existing.version >= version && existing.expires_at > now { @@ -105,7 +115,7 @@ pub(crate) fn acquire_lease_after_admission( } } - refresh_lease_after_admission(shared, descriptor_id, version, duration) + refresh_lease_after_admission(shared, descriptor_id, version, duration).await } /// Reject an acquisition covered by an active descriptor drain. @@ -122,68 +132,77 @@ pub(crate) fn ensure_not_draining( descriptor_id: &DescriptorId, version: u64, ) -> Result<(), Error> { - if shared.lease_drain.is_draining(descriptor_id, version) { - return Err(drain_in_progress_error(descriptor_id, version)); + let owners = shared.lease_drain.draining_owners(descriptor_id, version); + if !owners.is_empty() { + return Err(drain_in_progress_error(descriptor_id, version, &owners)); } Ok(()) } /// Build the retryable error for an acquisition covered by an active drain. /// -/// The full descriptor identity and the requested version stay in the message -/// so a retry-budget exhaustion is still diagnosable from the client error. -fn drain_in_progress_error(descriptor_id: &DescriptorId, version: u64) -> Error { +/// The full descriptor identity, the requested version, and every drain owner +/// stay in the message, so a retry-budget exhaustion names what holds the +/// descriptor. +fn drain_in_progress_error( + descriptor_id: &DescriptorId, + version: u64, + owners: &[nodedb_cluster::DrainOwner], +) -> Error { Error::RetryableSchemaChanged { descriptor: format!( - "{descriptor_id:?} at version {version} (descriptor lease drain in progress)" + "{descriptor_id:?} at version {version} (descriptor lease drain in progress: {})", + drain_owner_list(owners) ), } } +/// Every drain owner, joined for an error message. +pub(crate) fn drain_owner_list(owners: &[nodedb_cluster::DrainOwner]) -> String { + owners + .iter() + .map(ToString::to_string) + .collect::>() + .join("; ") +} + /// Unconditionally propose a fresh lease grant, skipping the /// "existing lease still valid" fast path. Used by the renewal /// loop, which has already decided the current lease is near /// expiry and must be refreshed even though it hasn't technically /// expired yet. /// -/// The single-node fallback and the cluster propose path are -/// identical to [`acquire_lease`]; the only difference is that -/// this function always stamps a new `expires_at = now + duration`. -pub fn force_refresh_lease( +/// The propose path is identical to [`acquire_lease`]; the only +/// difference is that this function always stamps a new +/// `expires_at = now + duration`. +pub async fn force_refresh_lease( shared: &SharedState, descriptor_id: DescriptorId, version: u64, duration: Duration, ) -> Result { - { + // The existing metadata lease plus this reservation keeps drain safe + // until the renewal's raft grant has completed. + let _reservation = { let _admission_gate = shared .lease_admission_gate .lock() .unwrap_or_else(|poison| poison.into_inner()); ensure_not_draining(shared, &descriptor_id, version)?; - // The existing metadata lease plus this reservation keeps drain safe - // until the renewal's raft grant has completed. - shared.lease_refcount.increment(&descriptor_id, version); - } + super::refcount::RefcountReservation::reserve(shared, descriptor_id.clone(), version) + }; // Force refresh bypasses the cache fast path, but serializes its raw // proposal with all first-holder and version-upgrade grants. - let result = { - let _grant_gate = shared - .lease_grant_gate - .lock() - .unwrap_or_else(|poison| poison.into_inner()); - refresh_lease_after_admission(shared, descriptor_id.clone(), version, duration) - }; - shared.lease_refcount.decrement(&descriptor_id, version); - result + let _grant_gate = shared.lease_grant_gate.lock().await; + refresh_lease_after_admission(shared, descriptor_id, version, duration).await } /// Unconditionally refresh after the caller has already linearized admission. /// This raw helper must not lock the admission gate or re-check drain state: -/// doing either while its raft operation is in flight would reintroduce the +/// doing either while its raft operation is in flight reintroduces the /// drain/applier deadlock. -fn refresh_lease_after_admission( +async fn refresh_lease_after_admission( shared: &SharedState, descriptor_id: DescriptorId, version: u64, @@ -199,22 +218,13 @@ fn refresh_lease_after_admission( expires_at, }; - // Single-node / no-cluster fallback: write straight into the - // local cache. The cache is shared with the rest of the process - // via `Arc>` so subsequent reads see it immediately. - if shared.metadata_raft.get().is_none() { - install_into_local_cache(shared, &lease); - return Ok(lease); - } - - // Cluster path: encode + propose + block on apply via the - // shared `propose_and_wait` helper. + // Encode, propose, and await the local apply. let entry = MetadataEntry::DescriptorLeaseGrant(lease.clone()); - super::propose_and_wait(shared, &entry, "grant")?; + super::propose_and_wait(shared, &entry, "grant").await?; // Re-read the cache. Under normal conditions the apply path - // already installed the lease before `wait_for` returned, so - // this read is just confirmation. If for some reason the lease + // already installed the lease before the apply wait returned, so + // this read is only confirmation. If for some reason the lease // is missing (race with cluster shutdown, lost commit), return // the in-memory copy we proposed — every committed lease at the // applied index is by definition durable. @@ -230,22 +240,6 @@ fn refresh_lease_after_admission( Ok(lease) } -/// Install a lease directly into the in-memory cache. Used by the -/// single-node fallback only — the cluster path goes through the -/// raft applier, which calls `MetadataCache::apply` on every node. -fn install_into_local_cache(shared: &SharedState, lease: &DescriptorLease) { - let mut cache = shared - .metadata_cache - .write() - .unwrap_or_else(|p| p.into_inner()); - cache - .leases - .insert((lease.descriptor_id.clone(), lease.node_id), lease.clone()); - if lease.expires_at > cache.last_applied_hlc { - cache.last_applied_hlc = lease.expires_at; - } -} - #[cfg(test)] mod tests { use super::*; @@ -271,13 +265,18 @@ mod tests { use nodedb_cluster::DescriptorKind; let descriptor = DescriptorId::new(0, 1, DescriptorKind::Collection, "orders".to_string()); - match drain_in_progress_error(&descriptor, 7) { + let owners = [nodedb_cluster::DrainOwner::MoveTenant { + tenant_id: 1, + source_db_id: 1024, + }]; + match drain_in_progress_error(&descriptor, 7, &owners) { Error::RetryableSchemaChanged { descriptor: detail } => { assert!( detail.contains("orders"), "descriptor identity lost: {detail}" ); assert!(detail.contains("version 7"), "version lost: {detail}"); + assert!(detail.contains("being moved"), "drain owner lost: {detail}"); } other => panic!("expected RetryableSchemaChanged, got {other:?}"), } diff --git a/nodedb/src/control/lease/refcount.rs b/nodedb/src/control/lease/refcount.rs index f3311cca6..07bb191b7 100644 --- a/nodedb/src/control/lease/refcount.rs +++ b/nodedb/src/control/lease/refcount.rs @@ -2,47 +2,36 @@ //! Per-query descriptor lease refcount + scope guard. //! -//! Descriptor leases are acquired at plan time and held -//! through execute. Two concurrent queries touching the same -//! descriptor version share a single underlying raft lease — we track -//! per-node exact-version refcounts so only a missing or lower-version lease -//! pays an acquire round-trip and only the last query across every version to -//! finish pays the release round-trip. Intermediate queries hit the fast-path -//! increment / decrement with no raft traffic. +//! Descriptor leases are acquired at plan time, or at authorization for a +//! write that runs no planner, and held through execute. Queries touching the +//! same descriptor version share one underlying raft lease: per-node +//! exact-version refcounts mean only a missing or lower-version lease pays an +//! acquire round trip. //! -//! The DDL drain path relies on this: when every in-flight -//! query using a descriptor finishes, the refcount hits zero, -//! the lease is actually released via a -//! `DescriptorLeaseRelease` raft entry, and drain's poll loop -//! observes the lease clear. Long-running queries naturally -//! bound the drain window — if a query exceeds -//! `DEFAULT_DRAIN_TIMEOUT` the ALTER fails with a drain-timeout -//! error and the operator retries. +//! A lease whose refcount returns to 0 stays granted. The next statement on +//! the descriptor reuses it with no raft traffic. It ends one of three ways: //! -//! ## Guard semantics +//! - a `DescriptorDrainStart` applies on this node, which releases this +//! node's unheld leases on that descriptor at once, so the drain waits only +//! for statements still running; +//! - the last statement still running under an active drain ends, which +//! hands the release to the background releaser; +//! - it reaches expiry: the renewal loop renews a held lease and releases an +//! idle one. //! -//! `QueryLeaseScope` is the owned collection of leases a -//! single query accumulated during planning. The scope drops -//! when the query's pgwire handler finishes executing (after -//! every response has been returned). Drop walks the scope, -//! decrements each exact-version refcount, and — when no version of a -//! descriptor remains held — spawns a background task to propose the release -//! entry. The spawn is mandatory because `Drop` cannot -//! be async; the drop handler itself returns immediately. +//! ## Guard semantics //! -//! A dropped `QueryLeaseScope` therefore schedules (but does -//! not await) the release. Drain's poll loop observes the -//! release after the raft round-trip lands on the leader — -//! sub-10ms in a healthy cluster. +//! `QueryLeaseScope` is the owned collection of leases a single statement +//! holds. Drop decrements each exact-version refcount and never waits. use std::collections::HashMap; use std::sync::{Arc, Mutex}; use nodedb_cluster::DescriptorId; -use tracing::warn; -use super::release::LeaseReleaseHandle; +use super::holders::{HolderTicket, LeaseHolders}; use crate::control::state::SharedState; +use crate::error::Error; /// Host-side lease reference counts. One entry per descriptor id and /// descriptor version this node currently holds; the value is the number of @@ -107,6 +96,85 @@ impl LeaseRefCount { } } +/// Gives back refcount units and releases a lease whose last hold ends while +/// a drain covers it. +/// +/// A drain start releases this node's idle leases on its descriptor once, as +/// it installs. A lease still held then is released here, when its last hold +/// ends: otherwise it stays granted until expiry and the drain waits for it. +/// The drain installs before its start checks the refcount, and a hold +/// decrements before it checks the drain, so one of the two always releases. +#[derive(Clone)] +pub(crate) struct HoldRelease { + refcounts: Arc, + drains: Arc, + queue: super::releaser::ReleaseQueue, +} + +impl HoldRelease { + pub(crate) fn for_state(shared: &SharedState) -> Self { + Self { + refcounts: Arc::clone(&shared.lease_refcount), + drains: Arc::clone(&shared.lease_drain), + queue: shared.lease_runtime.releaser.queue(), + } + } + + /// Give back one unit of each hold. A descriptor left with no hold while + /// a drain covers the version it held is handed to the background + /// releaser. + fn give_back(&self, holds: impl IntoIterator) { + let mut drained_idle: Vec = Vec::new(); + let mut drained_hold_ended = false; + for (id, version) in holds { + self.refcounts.decrement(&id, version); + if !self.drains.is_draining(&id, version) { + continue; + } + drained_hold_ended = true; + if self.refcounts.current(&id) == 0 && !drained_idle.contains(&id) { + drained_idle.push(id); + } + } + // A drain counts this node's holds directly, so it re-counts now. + if drained_hold_ended { + self.drains.wake_drain_waiters(); + } + if !drained_idle.is_empty() { + self.queue + .submit(super::releaser::ReleaseRequest::UnheldDescriptors( + drained_idle, + )); + } + } +} + +/// One exact-version refcount unit a grant in flight holds. Dropping it, +/// on return or when its future is cancelled, gives the unit back. +pub(crate) struct RefcountReservation { + release: HoldRelease, + id: DescriptorId, + version: u64, +} + +impl RefcountReservation { + /// Take one unit of `(id, version)`. The caller holds the admission gate. + pub(crate) fn reserve(shared: &SharedState, id: DescriptorId, version: u64) -> Self { + shared.lease_refcount.increment(&id, version); + Self { + release: HoldRelease::for_state(shared), + id, + version, + } + } +} + +impl Drop for RefcountReservation { + fn drop(&mut self) { + self.release.give_back([(self.id.clone(), self.version)]); + } +} + /// Owned collection of lease holds for one query. /// /// Created by `OriginCatalog::take_lease_scope()` after @@ -115,10 +183,11 @@ impl LeaseRefCount { pub struct QueryLeaseScope { /// Exact descriptor-version refcounts this query holds. descriptor_versions: Vec<(DescriptorId, u64)>, - /// Refcount state shared independently of the process-wide state. - refcounts: Option>, - /// Minimal owned capability needed to release the underlying lease. - releaser: Option, + /// Gives the holds back on drop, independently of the process-wide state. + release: Option, + /// This query's entry in the node's holder table, which lets a lost + /// lease revoke it. `None` for an empty scope. + holder: Option<(Arc, HolderTicket)>, } impl QueryLeaseScope { @@ -128,19 +197,61 @@ impl QueryLeaseScope { pub fn empty() -> Self { Self { descriptor_versions: Vec::new(), - refcounts: None, - releaser: None, + release: None, + holder: None, } } /// Build a scope from exact descriptor-version holds already incremented - /// on the node's `lease_refcount`. Only cloneable release capabilities are - /// retained, so the scope neither owns nor weak-references `SharedState`. - pub fn new(descriptor_versions: Vec<(DescriptorId, u64)>, shared: &SharedState) -> Self { - Self { + /// on the node's `lease_refcount`, and register it in the node's holder + /// table. Only cloneable capabilities are retained, so the scope neither + /// owns nor weak-references `SharedState`. + /// + /// Fails with a retryable error when the holder table is full. The + /// caller still owns the refcounts and rolls them back. + pub fn new( + descriptor_versions: Vec<(DescriptorId, u64)>, + shared: &SharedState, + ) -> Result { + let mut seen = std::collections::HashSet::new(); + let descriptors: Vec = descriptor_versions + .iter() + .filter(|(id, _)| seen.insert(id.clone())) + .map(|(id, _)| id.clone()) + .collect(); + let holders = Arc::clone(&shared.lease_runtime.holders); + let ticket = holders.register(descriptors)?; + Ok(Self { descriptor_versions, - refcounts: Some(Arc::clone(&shared.lease_refcount)), - releaser: Some(LeaseReleaseHandle::from_shared(shared)), + release: Some(HoldRelease::for_state(shared)), + holder: Some((holders, ticket)), + }) + } + + /// The retryable error if this node lost a lease the scope holds. + pub fn check_not_revoked(&self) -> Result<(), Error> { + match self + .holder + .as_ref() + .and_then(|(_, t)| t.revocation().revoked_error()) + { + Some(error) => Err(error), + None => Ok(()), + } + } + + /// Run `fut` while the scope holds its leases. Ends early with + /// [`Error::RetryableSchemaChanged`] once this node loses one of them; + /// dropping `fut` drops its pending Data Plane requests. + pub async fn guard(&self, fut: F) -> Result { + let Some((_, ticket)) = self.holder.as_ref() else { + return Ok(fut.await); + }; + let revocation = Arc::clone(ticket.revocation()); + tokio::select! { + biased; + error = revocation.revoked() => Err(error), + output = fut => Ok(output), } } @@ -166,57 +277,20 @@ impl QueryLeaseScope { impl Drop for QueryLeaseScope { fn drop(&mut self) { + if let Some((holders, ticket)) = self.holder.take() { + holders.deregister(&ticket); + } if self.descriptor_versions.is_empty() { return; } - let Some(refcounts) = self.refcounts.take() else { + let Some(release) = self.release.take() else { return; }; - let Some(releaser) = self.releaser.take() else { - return; - }; - // Decrement exact-version refcounts and collect ids whose total across - // every version just hit zero — only those need metadata release. - let mut to_release = Vec::new(); - for (id, version) in self.descriptor_versions.drain(..) { - refcounts.decrement(&id, version); - if refcounts.current(&id) == 0 { - to_release.push(id); - } - } - if to_release.is_empty() { - return; - } - // Release is synchronous, so run it on Tokio's blocking pool when a - // runtime owns this drop. Drops can also occur on non-Tokio threads - // (notably teardown paths); then use an independent OS thread rather - // than silently retaining the metadata lease. Both paths call the same - // conditional release, which serializes with admissions on grant_gate. - if let Ok(handle) = tokio::runtime::Handle::try_current() { - handle.spawn(async move { - let result = - tokio::task::spawn_blocking(move || releaser.release_if_unheld(to_release)) - .await; - match result { - Ok(Ok(())) => {} - Ok(Err(error)) => { - warn!(error = %error, "QueryLeaseScope drop: background release failed"); - } - Err(error) => { - warn!(error = %error, "QueryLeaseScope drop: spawn_blocking panicked"); - } - } - }); - } else if let Err(error) = std::thread::Builder::new() - .name("nodedb-lease-release".into()) - .spawn(move || { - if let Err(error) = releaser.release_if_unheld(to_release) { - warn!(error = %error, "QueryLeaseScope drop: fallback release failed"); - } - }) - { - warn!(error = %error, "QueryLeaseScope drop: failed to spawn fallback release"); - } + // The lease stays granted at refcount 0, so the next statement on the + // descriptor reuses it with no Raft round trip. A drain start releases + // it on this node, a drain that covers it releases it when this last + // hold ends, and the renewal loop lets it lapse at expiry. + release.give_back(self.descriptor_versions.drain(..)); } } @@ -229,7 +303,7 @@ mod tests { use nodedb_cluster::DescriptorKind; use crate::bridge::dispatch::Dispatcher; - use crate::control::lease::{DEFAULT_LEASE_DURATION, acquire_lease_after_admission}; + use crate::control::lease::DEFAULT_LEASE_DURATION; use crate::wal::WalManager; fn id(name: &str) -> DescriptorId { @@ -308,11 +382,11 @@ mod tests { #[test] fn empty_scope_drops_cleanly() { let scope = QueryLeaseScope::empty(); - drop(scope); // should not panic even without a runtime + drop(scope); // does not panic even without a runtime } #[test] - fn no_runtime_drop_releases_last_unheld_lease() { + fn dropping_the_last_holder_keeps_the_lease_granted() { let (state, descriptor, scope, _directory) = { let runtime = tokio::runtime::Builder::new_current_thread() .enable_all() @@ -329,14 +403,28 @@ mod tests { .expect("construct lease release state"); let descriptor = id("no-runtime-drop"); state.lease_refcount.increment(&descriptor, 1); - acquire_lease_after_admission( - &state, - descriptor.clone(), - 1, - DEFAULT_LEASE_DURATION, - ) - .expect("install single-node lease"); - let scope = QueryLeaseScope::new(vec![(descriptor.clone(), 1)], &state); + // The grant as the metadata applier installs it. This test + // covers the holder count, not the proposal. + let expires_at = nodedb_types::Hlc::new( + state.hlc_clock.peek().wall_ns + DEFAULT_LEASE_DURATION.as_nanos() as u64, + 0, + ); + state + .metadata_cache + .write() + .unwrap_or_else(|p| p.into_inner()) + .leases + .insert( + (descriptor.clone(), state.node_id), + nodedb_cluster::DescriptorLease { + descriptor_id: descriptor.clone(), + version: 1, + node_id: state.node_id, + expires_at, + }, + ); + let scope = QueryLeaseScope::new(vec![(descriptor.clone(), 1)], &state) + .expect("register the scope's holder"); (state, descriptor, scope, directory) }); @@ -347,12 +435,11 @@ mod tests { drop(scope); - for _ in 0..100 { - if state.lookup_lease_for_self(&descriptor).is_none() { - return; - } - std::thread::sleep(Duration::from_millis(10)); - } - panic!("no-runtime fallback did not release the unheld descriptor lease"); + assert_eq!(state.lease_refcount.current(&descriptor), 0); + std::thread::sleep(Duration::from_millis(50)); + assert!( + state.lookup_lease_for_self(&descriptor).is_some(), + "an idle lease stays granted until a drain or its expiry" + ); } } diff --git a/nodedb/src/control/lease/release.rs b/nodedb/src/control/lease/release.rs index 90d0e5f19..805e958a4 100644 --- a/nodedb/src/control/lease/release.rs +++ b/nodedb/src/control/lease/release.rs @@ -1,93 +1,79 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Batched descriptor lease release for explicit shutdown and query-scope drop. +//! Batched descriptor lease release for explicit shutdown, admission +//! rollback, lease GC, and the background releaser. -use std::sync::{Arc, Mutex, RwLock}; +use std::sync::Arc; -use nodedb_cluster::{AppliedIndexWatcher, DescriptorId, MetadataCache, MetadataEntry}; +use nodedb_cluster::{AppliedIndexWatcher, DescriptorId, MetadataEntry}; use crate::control::lease::LeaseRefCount; use crate::control::state::SharedState; use crate::error::Error; -/// Owned release capability captured when a query scope is admitted. -/// -/// It deliberately contains only the immutable node identity and cloneable -/// metadata handles needed for release. Query scopes therefore do not keep the -/// whole `SharedState` alive and do not require a `Weak` (which -/// would prevent bootstrap-time `Arc::get_mut` wiring). +/// Owned release capability: the node identity and the cloneable metadata +/// handles a release needs, without the whole `SharedState`. pub(crate) struct LeaseReleaseHandle { node_id: u64, - metadata_cache: Arc>, - metadata_raft: Option>, + metadata_raft: Arc, applied_watcher: Arc, - grant_gate: Arc>, + grant_gate: Arc>, refcounts: Arc, } impl LeaseReleaseHandle { - pub(crate) fn from_shared(shared: &SharedState) -> Self { - Self { + /// Capture the release capability of `shared`. Fails before `start_raft` + /// installed the metadata raft handle. + pub(crate) fn from_shared(shared: &SharedState) -> Result { + Ok(Self { node_id: shared.node_id, - metadata_cache: Arc::clone(&shared.metadata_cache), - metadata_raft: shared.metadata_raft.get().cloned(), + metadata_raft: Arc::clone(shared.metadata_raft_handle()?), applied_watcher: shared.applied_index_watcher(nodedb_cluster::METADATA_GROUP_ID), grant_gate: Arc::clone(&shared.lease_grant_gate), refcounts: Arc::clone(&shared.lease_refcount), - } + }) } /// Explicit release for shutdown, the public API, and tests. It is /// unconditional, but cannot race a grant because both operations hold the /// same gate through metadata apply. - pub(crate) fn release(&self, descriptor_ids: Vec) -> Result<(), Error> { - let _grant_gate = self - .grant_gate - .lock() - .unwrap_or_else(|poison| poison.into_inner()); - self.release_raw(descriptor_ids) + pub(crate) async fn release(&self, descriptor_ids: Vec) -> Result<(), Error> { + let _grant_gate = self.grant_gate.lock().await; + self.release_raw_for_node(self.node_id, descriptor_ids) + .await } /// Release only descriptors that remain unheld when the grant gate is /// acquired. A new admission reserves its refcount before taking this gate, /// so a queued release skips that descriptor. Conversely, an admission that /// arrives after release waits for the gate, cache-rechecks, and re-grants. - /// - /// This is called from `QueryLeaseScope`'s blocking worker, so its applied - /// watcher wait is intentionally direct rather than wrapped in - /// `tokio::task::block_in_place`. - pub(crate) fn release_if_unheld(&self, descriptor_ids: Vec) -> Result<(), Error> { - let _grant_gate = self - .grant_gate - .lock() - .unwrap_or_else(|poison| poison.into_inner()); + pub(crate) async fn release_if_unheld( + &self, + descriptor_ids: Vec, + ) -> Result<(), Error> { + let _grant_gate = self.grant_gate.lock().await; let unheld = descriptor_ids .into_iter() .filter(|id| self.refcounts.current(id) == 0) .collect(); - self.release_raw(unheld) - } - - /// Raw metadata release for this node. The caller must hold `grant_gate`. - fn release_raw(&self, descriptor_ids: Vec) -> Result<(), Error> { - self.release_raw_for_node(self.node_id, descriptor_ids) + self.release_raw_for_node(self.node_id, unheld).await } /// Release leases held by an ARBITRARY node. Used by lease GC for /// nodes that left the topology (crashed/decommissioned). Does NOT take /// `grant_gate` (no contention with local grants — the foreign holder /// cannot grant anymore). - pub(crate) fn release_for_node( + pub(crate) async fn release_for_node( &self, node_id: u64, descriptor_ids: Vec, ) -> Result<(), Error> { - self.release_raw_for_node(node_id, descriptor_ids) + self.release_raw_for_node(node_id, descriptor_ids).await } - /// Raw metadata release for `node_id`. `release_raw` keeps its gate-taking - /// wrapper for the self path; this is the ungated core. - fn release_raw_for_node( + /// Raw metadata release for `node_id`. The self path holds `grant_gate` + /// around it. + async fn release_raw_for_node( &self, node_id: u64, descriptor_ids: Vec, @@ -96,20 +82,6 @@ impl LeaseReleaseHandle { return Ok(()); } - let Some(metadata_raft) = &self.metadata_raft else { - // Single-node fallback: hanya meaningful untuk self. - if node_id == self.node_id { - let mut cache = self - .metadata_cache - .write() - .unwrap_or_else(|poison| poison.into_inner()); - for id in descriptor_ids { - cache.leases.remove(&(id, node_id)); - } - } - return Ok(()); - }; - let entry = MetadataEntry::DescriptorLeaseRelease { node_id, descriptor_ids, @@ -117,10 +89,13 @@ impl LeaseReleaseHandle { let raw = nodedb_cluster::encode_entry(&entry).map_err(|error| Error::Config { detail: format!("descriptor lease release encode: {error}"), })?; - let log_index = metadata_raft.propose(raw)?; - let outcome = self - .applied_watcher - .wait_for(log_index, super::PROPOSE_TIMEOUT); + let log_index = self.metadata_raft.propose_async(raw).await?; + let outcome = crate::control::metadata_proposer::wait::wait_applied( + Arc::clone(&self.applied_watcher), + log_index, + super::PROPOSE_TIMEOUT, + ) + .await?; if !outcome.is_reached() { return Err(Error::Config { detail: format!( @@ -138,34 +113,60 @@ impl LeaseReleaseHandle { /// Release every lease this node currently holds against any of /// `descriptor_ids`. Empty input is a no-op. /// -/// Cluster mode proposes one `DescriptorLeaseRelease` entry and waits for the -/// local applied watermark. Single-node mode removes the local cache entries. -pub fn release_leases( +/// Proposes one `DescriptorLeaseRelease` entry and awaits the local applied +/// watermark. +pub async fn release_leases( shared: &SharedState, descriptor_ids: Vec, ) -> Result<(), Error> { - let releaser = LeaseReleaseHandle::from_shared(shared); - if shared.metadata_raft.get().is_none() { - return releaser.release(descriptor_ids); + LeaseReleaseHandle::from_shared(shared)? + .release(descriptor_ids) + .await +} + +/// Release this node's lease on `id` when no statement holds it, because a +/// drain on `id` has started. +/// +/// An idle lease stays granted after its last statement, so without this the +/// drain waits for it to expire. A held lease stays: the drain waits for +/// its statement. Called from the drain-start apply, which must not block, so +/// the background releaser proposes the release. +/// +/// The check reads the raw cache entry. The self-fence and the expiry filter +/// of `lookup_lease_for_self` gate the reuse of a lease, never its release. +/// A follower applying this entry has not yet advanced its applied index past +/// it, so the fence always reports it behind the leader here. +pub(crate) fn release_idle_on_drain(shared: &SharedState, id: &DescriptorId) { + let granted_here = shared + .metadata_cache + .read() + .unwrap_or_else(|p| p.into_inner()) + .leases + .contains_key(&(id.clone(), shared.node_id)); + if !granted_here || shared.lease_refcount.current(id) > 0 { + return; } - // `AppliedIndexWatcher::wait_for` parks on a Condvar. Preserve the prior - // cluster-path behavior by yielding this Tokio worker while it waits. - tokio::task::block_in_place(|| releaser.release(descriptor_ids)) + shared + .lease_runtime + .releaser + .submit(super::releaser::ReleaseRequest::UnheldDescriptors(vec![ + id.clone(), + ])); } /// Conditionally release descriptors that have no remaining query admission. -/// This synchronous rollback path preserves the public release wrapper's Tokio -/// behavior; query-scope drop invokes `LeaseReleaseHandle::release_if_unheld` -/// from its own blocking worker instead. -pub(crate) fn release_unheld_leases( +/// Used by admission rollback, by the background releaser, and by the renewal +/// loop's release of an idle lease at expiry. +pub(crate) async fn release_unheld_leases( shared: &SharedState, descriptor_ids: Vec, ) -> Result<(), Error> { - let releaser = LeaseReleaseHandle::from_shared(shared); - if shared.metadata_raft.get().is_none() { - return releaser.release_if_unheld(descriptor_ids); + if descriptor_ids.is_empty() { + return Ok(()); } - tokio::task::block_in_place(|| releaser.release_if_unheld(descriptor_ids)) + LeaseReleaseHandle::from_shared(shared)? + .release_if_unheld(descriptor_ids) + .await } #[cfg(test)] @@ -175,59 +176,51 @@ mod tests { use nodedb_cluster::{DescriptorId, DescriptorKind}; use super::*; - use crate::bridge::dispatch::Dispatcher; + use crate::control::cluster::test_one_node; use crate::control::lease::{DEFAULT_LEASE_DURATION, acquire_lease_after_admission}; - use crate::control::state::SharedState; - use crate::wal::WalManager; - - fn test_state() -> (Arc, tempfile::TempDir) { - let directory = tempfile::tempdir().expect("create lease release test directory"); - let wal = Arc::new( - WalManager::open_for_testing(&directory.path().join("lease-release.wal")) - .expect("open lease release test WAL"), - ); - let (dispatcher, _data_sides) = Dispatcher::new(1, 64); - let state = SharedState::new(dispatcher, wal).expect("construct lease release state"); - (state, directory) - } fn id(name: &str) -> DescriptorId { DescriptorId::new(0, 1, DescriptorKind::Collection, name.to_string()) } - #[tokio::test] + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn last_scope_release_removes_unheld_lease() { - let (state, _directory) = test_state(); + let cluster = test_one_node::boot().await; + let state = &cluster.state; let descriptor = id("last-scope"); state.lease_refcount.increment(&descriptor, 1); - acquire_lease_after_admission(&state, descriptor.clone(), 1, DEFAULT_LEASE_DURATION) - .expect("install single-node lease"); + acquire_lease_after_admission(state, descriptor.clone(), 1, DEFAULT_LEASE_DURATION) + .await + .expect("grant the lease through the metadata group"); assert_eq!(state.lease_refcount.decrement(&descriptor, 1), 0); - LeaseReleaseHandle::from_shared(&state) + LeaseReleaseHandle::from_shared(state) + .expect("release handle") .release_if_unheld(vec![descriptor.clone()]) + .await .expect("release last scope lease"); assert!(state.lookup_lease_for_self(&descriptor).is_none()); + cluster.shutdown().await; } - #[tokio::test] + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn readmission_before_release_gate_check_preserves_lease() { - let (state, _directory) = test_state(); + let cluster = test_one_node::boot().await; + let state = &cluster.state; let descriptor = id("readmitted"); state .acquire_descriptor_lease(descriptor.clone(), 1, DEFAULT_LEASE_DURATION) - .expect("install single-node lease"); + .await + .expect("grant the lease through the metadata group"); - let gate = state - .lease_grant_gate - .lock() - .unwrap_or_else(|poison| poison.into_inner()); - let release_state = Arc::clone(&state); + let gate = state.lease_grant_gate.lock().await; + let release_state = Arc::clone(state); let release_descriptor = descriptor.clone(); - let release = std::thread::spawn(move || { - LeaseReleaseHandle::from_shared(&release_state) + let release = tokio::spawn(async move { + LeaseReleaseHandle::from_shared(&release_state)? .release_if_unheld(vec![release_descriptor]) + .await }); // The release cannot inspect refcounts until the held grant gate is @@ -235,33 +228,40 @@ mod tests { state.lease_refcount.increment(&descriptor, 1); drop(gate); release - .join() - .expect("release thread panicked") + .await + .expect("release task panicked") .expect("conditional release failed"); assert!(state.lookup_lease_for_self(&descriptor).is_some()); state.lease_refcount.decrement(&descriptor, 1); + cluster.shutdown().await; } - #[tokio::test] + #[tokio::test(flavor = "multi_thread", worker_threads = 4)] async fn release_first_requires_later_admission_to_regrant() { - let (state, _directory) = test_state(); + let cluster = test_one_node::boot().await; + let state = &cluster.state; let descriptor = id("regrant"); state .acquire_descriptor_lease(descriptor.clone(), 1, DEFAULT_LEASE_DURATION) - .expect("install single-node lease"); - LeaseReleaseHandle::from_shared(&state) + .await + .expect("grant the lease through the metadata group"); + LeaseReleaseHandle::from_shared(state) + .expect("release handle") .release_if_unheld(vec![descriptor.clone()]) + .await .expect("release unheld lease"); assert!(state.lookup_lease_for_self(&descriptor).is_none()); // This mirrors an admission that follows release: it reserves before // the grant path, which cache-rechecks under the same gate and grants. state.lease_refcount.increment(&descriptor, 1); - acquire_lease_after_admission(&state, descriptor.clone(), 1, DEFAULT_LEASE_DURATION) + acquire_lease_after_admission(state, descriptor.clone(), 1, DEFAULT_LEASE_DURATION) + .await .expect("regrant after release"); assert!(state.lookup_lease_for_self(&descriptor).is_some()); state.lease_refcount.decrement(&descriptor, 1); + cluster.shutdown().await; } } diff --git a/nodedb/src/control/lease/releaser.rs b/nodedb/src/control/lease/releaser.rs new file mode 100644 index 000000000..615d70e82 --- /dev/null +++ b/nodedb/src/control/lease/releaser.rs @@ -0,0 +1,194 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Background release of leases that synchronous code gives up. +//! +//! A `Drop`, a metadata applier step, or any other synchronous site never +//! waits for a metadata proposal. It hands the release to [`LeaseReleaser`] +//! instead: a bounded queue owned by `SharedState`. One Control-Plane task +//! drains the queue, proposes each release, and retries a failed one with +//! backoff. +//! +//! A release that is never proposed, because the queue is full, every +//! attempt failed, or the node shut down, costs latency only. A descriptor +//! lease expires at its `expires_at`, and the renewal loop releases an idle +//! one at expiry. A DDL preparation lease is reclaimed by the metadata leader +//! once its lease window passed. + +use std::sync::{Arc, Mutex, Weak}; +use std::time::Duration; + +use nodedb_cluster::DescriptorId; +use tokio::sync::mpsc; + +use crate::control::shutdown::{ShutdownPhase, spawn_loop}; +use crate::control::state::SharedState; +use crate::error::Error; + +/// Requests the queue holds before a new one is dropped to its expiry path. +const QUEUE_CAPACITY: usize = 1024; + +/// Proposals one request makes before it is left to its expiry path. +const MAX_ATTEMPTS: u32 = 6; + +/// Wait before the second attempt. Each later wait doubles it. +const FIRST_BACKOFF: Duration = Duration::from_millis(50); + +/// One release the background task proposes. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) enum ReleaseRequest { + /// Release this node's leases on these descriptors that no statement + /// holds when the release runs. + UnheldDescriptors(Vec), + /// Release the DDL preparation lease `token`. + DdlPrepare { token: u64 }, +} + +/// The bounded queue of releases synchronous code hands off. +pub struct LeaseReleaser { + tx: mpsc::Sender, + rx: Mutex>>, +} + +impl Default for LeaseReleaser { + fn default() -> Self { + let (tx, rx) = mpsc::channel(QUEUE_CAPACITY); + Self { + tx, + rx: Mutex::new(Some(rx)), + } + } +} + +impl LeaseReleaser { + /// Queue `request` without waiting. A full or closed queue drops it and + /// logs: the lease then ends through its expiry path. + pub(crate) fn submit(&self, request: ReleaseRequest) { + self.queue().submit(request); + } + + /// A handle that queues releases into this releaser, for an owner that + /// outlives no `SharedState` borrow. + pub(crate) fn queue(&self) -> ReleaseQueue { + ReleaseQueue { + tx: self.tx.clone(), + } + } + + /// The queue's receiving end. The first call takes it, later calls get + /// `None`. + pub(super) fn take_receiver(&self) -> Option> { + self.rx.lock().unwrap_or_else(|p| p.into_inner()).take() + } +} + +/// The sending end of a [`LeaseReleaser`]. +#[derive(Clone)] +pub(crate) struct ReleaseQueue { + tx: mpsc::Sender, +} + +impl ReleaseQueue { + /// Queue `request` without waiting. A full or closed queue drops it and + /// logs: the lease then ends through its expiry path. + pub(crate) fn submit(&self, request: ReleaseRequest) { + match self.tx.try_send(request) { + Ok(()) => {} + Err(mpsc::error::TrySendError::Full(request)) => tracing::warn!( + ?request, + "lease release queue full; the lease ends at its expiry" + ), + Err(mpsc::error::TrySendError::Closed(request)) => tracing::warn!( + ?request, + "lease release queue closed; the lease ends at its expiry" + ), + } + } +} + +/// Spawn the task that drains `shared`'s release queue. `start_raft` calls it +/// once the metadata raft handle is installed. A second call spawns nothing. +pub(crate) fn spawn_lease_releaser(shared: &Arc) { + let Some(mut rx) = shared.lease_runtime.releaser.take_receiver() else { + tracing::warn!("lease releaser already running; start_raft appears to have run twice"); + return; + }; + let weak = Arc::downgrade(shared); + spawn_loop( + &shared.loop_registry, + &shared.shutdown, + "lease_releaser", + ShutdownPhase::DrainingControlPlane, + move |mut shutdown| async move { + loop { + tokio::select! { + biased; + _ = shutdown.wait_cancelled() => break, + request = rx.recv() => { + let Some(request) = request else { break }; + release_with_retry(&weak, &request).await; + } + } + } + }, + ); +} + +/// Propose `request` until it applies or [`MAX_ATTEMPTS`] failed. +async fn release_with_retry(shared: &Weak, request: &ReleaseRequest) { + let mut backoff = FIRST_BACKOFF; + for attempt in 1..=MAX_ATTEMPTS { + let Some(state) = shared.upgrade() else { + return; + }; + match release_once(&state, request).await { + Ok(()) => return, + Err(error) if attempt < MAX_ATTEMPTS => { + tracing::debug!(?request, attempt, %error, "lease release failed; retrying"); + } + Err(error) => { + tracing::warn!( + ?request, + %error, + "lease release failed on every attempt; the lease ends at its expiry" + ); + return; + } + } + drop(state); + tokio::time::sleep(backoff).await; + backoff = backoff.saturating_mul(2); + } +} + +async fn release_once(shared: &SharedState, request: &ReleaseRequest) -> Result<(), Error> { + match request { + ReleaseRequest::UnheldDescriptors(ids) => { + super::release::release_unheld_leases(shared, ids.clone()).await + } + ReleaseRequest::DdlPrepare { token } => { + crate::control::metadata_proposer::release_ddl_prepare_token(shared, *token).await + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_full_queue_drops_the_request_without_blocking() { + let releaser = LeaseReleaser::default(); + for token in 0..QUEUE_CAPACITY as u64 { + releaser.submit(ReleaseRequest::DdlPrepare { token }); + } + releaser.submit(ReleaseRequest::DdlPrepare { token: u64::MAX }); + let mut rx = releaser.take_receiver().expect("receiver"); + let mut queued = 0; + while let Ok(request) = rx.try_recv() { + assert_ne!(request, ReleaseRequest::DdlPrepare { token: u64::MAX }); + queued += 1; + } + assert_eq!(queued, QUEUE_CAPACITY); + assert!(releaser.take_receiver().is_none()); + } +} diff --git a/nodedb/src/control/lease/renewal.rs b/nodedb/src/control/lease/renewal.rs index 7274b6872..07fecec82 100644 --- a/nodedb/src/control/lease/renewal.rs +++ b/nodedb/src/control/lease/renewal.rs @@ -4,9 +4,10 @@ //! //! Spawned once per cluster node at startup. Wakes every //! `check_interval`, walks the local node's leases in -//! `metadata_cache.leases`, and re-acquires every lease whose +//! `metadata_cache.leases`, and re-acquires every held lease whose //! remaining time is below `threshold_pct` of the original -//! duration. Re-acquire goes through the standard +//! duration. A lease no statement holds is released instead, so an +//! idle lease lapses at its expiry. Re-acquire goes through the standard //! `SharedState::acquire_descriptor_lease` slow path, which //! transparently forwards to the metadata-group leader if this //! node isn't it. @@ -15,15 +16,11 @@ //! blocked on its tokio task. On every tick it upgrades to a //! strong reference, doing nothing if the upgrade fails. //! -//! **Single-node clusters skip this loop entirely.** In single-node -//! mode there is no metadata raft handle, every `acquire_lease` -//! call writes straight into the local cache, and there is no -//! concurrent writer that could expire a lease behind the loop's -//! back. The `spawn` constructor returns `None` in that case so -//! the embedded usage path doesn't carry an idle tokio task. +//! Every node runs the loop, a one-node cluster included: every lease +//! is granted and released through the metadata raft group. //! //! The DDL drain gate reads `metadata_cache.leases` to decide -//! when a `Put*` of a new descriptor version may commit. The +//! when a `Put*` of a new descriptor version can commit. The //! renewal loop is what keeps that map populated past initial //! acquisition. @@ -40,11 +37,16 @@ use tokio::sync::watch; use tokio::task::JoinHandle; use tracing::{debug, error, info}; +/// How often the loop checks the self-fence and revokes in-flight statements. +/// Revocation then trails the fence by at most this, well inside the 5 s +/// clock-skew margin other nodes wait on top of the fence window. +const LEASE_FENCE_CHECK_INTERVAL: Duration = Duration::from_secs(1); + use crate::control::state::SharedState; /// Configuration extracted from `ClusterTransportTuning` at spawn /// time. Captured so the loop has stable values for the duration -/// of its life — tuning hot-reload (if it ever lands) would need +/// of its life — tuning hot-reload (if it ever lands) needs /// to restart the loop. #[derive(Debug, Clone, Copy)] pub struct LeaseRenewalConfig { @@ -90,9 +92,9 @@ pub struct LeaseRenewalLoop { } impl LeaseRenewalLoop { - /// Spawn the renewal loop on the current tokio runtime. Returns - /// `None` (and does not spawn anything) on single-node clusters - /// where `metadata_raft` is not wired — see the module docstring. + /// Spawn the renewal loop on the current tokio runtime. Fails, and + /// spawns nothing, before `start_raft` installed the metadata raft + /// handle. /// /// The returned handle is `(JoinHandle, LoopMetrics)`; the caller /// registers the metrics with the cluster's loop-metrics registry @@ -102,11 +104,8 @@ impl LeaseRenewalLoop { shared: Arc, tuning: &ClusterTransportTuning, shutdown_rx: watch::Receiver, - ) -> Option<(JoinHandle<()>, Arc)> { - if shared.metadata_raft.get().is_none() { - debug!("descriptor lease renewal: skipping spawn (no metadata raft handle)"); - return None; - } + ) -> crate::Result<(JoinHandle<()>, Arc)> { + shared.metadata_raft_handle()?; let config = LeaseRenewalConfig::from_tuning(tuning); info!( check_interval_secs = config.check_interval.as_secs(), @@ -126,7 +125,7 @@ impl LeaseRenewalLoop { loop_handle.run().await; metrics_for_task.set_up(false); }); - Some((join, loop_metrics)) + Ok((join, loop_metrics)) } async fn run(mut self) { @@ -136,6 +135,8 @@ impl LeaseRenewalLoop { // acquires won't be near expiry. interval.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); interval.tick().await; + let mut fence_check = tokio::time::interval(LEASE_FENCE_CHECK_INTERVAL); + fence_check.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); self.loop_metrics.set_up(true); loop { tokio::select! { @@ -146,9 +147,14 @@ impl LeaseRenewalLoop { return; } } + _ = fence_check.tick() => { + if let Some(shared) = self.shared.upgrade() { + super::revoke_if_fenced(&shared); + } + } _ = interval.tick() => { let started = Instant::now(); - self.tick(); + self.tick().await; self.loop_metrics.observe(started.elapsed()); } } @@ -161,7 +167,7 @@ impl LeaseRenewalLoop { /// next tick retries it. /// /// **Why we use wall-clock nanoseconds, not `hlc_clock.peek()`**: - /// `peek` returns the last HLC the clock observed, which may + /// `peek` returns the last HLC the clock observed, which can /// be frozen at the moment the lease was stamped if nothing /// else has advanced the clock since. We want "real time now" /// to compute `remaining = expires_at - now`, and the lease's @@ -169,8 +175,8 @@ impl LeaseRenewalLoop { /// it was stamped. Comparing against `SystemTime::now()` keeps /// both sides of the subtraction in the same reference frame /// and avoids spuriously classifying leases as "not near - /// expiry" just because the HLC hasn't ticked. - fn tick(&self) { + /// expiry" only because the HLC hasn't ticked. + async fn tick(&self) { let Some(shared) = self.shared.upgrade() else { return; }; @@ -184,16 +190,45 @@ impl LeaseRenewalLoop { "descriptor lease renewal: re-acquiring near-expiry leases" ); for (id, held_version) in near_expiry { - let current_version = lookup_current_version(&shared, &id); - match current_version { - Some(v) => { - let version = v.max(held_version); + // An idle lease lapses at expiry instead of renewing: no statement + // holds it, and the next one re-acquires it. + if shared.lease_refcount.current(&id) == 0 { + if let Err(e) = + super::release::release_unheld_leases(&shared, vec![id.clone()]).await + { + error!( + descriptor = ?id, + held_version, + error = %e, + "descriptor lease renewal: idle lease release failed; it \ + lapses at its expiry" + ); + self.loop_metrics.record_error("release"); + } + continue; + } + let lookup = lookup_current_version(&shared, &id); + if let Err(e) = &lookup { + error!( + descriptor = ?id, + held_version, + error = %e, + "descriptor lease renewal: catalog read failed; the lease is left \ + alone this tick" + ); + self.loop_metrics.record_error("lookup"); + } + match renewal_step(held_version, &lookup) { + RenewalStep::Skip => {} + RenewalStep::Renew(version) => { if let Err(e) = super::propose::force_refresh_lease( &shared, id.clone(), version, self.config.full_duration, - ) { + ) + .await + { error!( descriptor = ?id, version, @@ -211,8 +246,9 @@ impl LeaseRenewalLoop { self.loop_metrics.record_error("renew"); } } - None => { - if let Err(e) = super::release::release_leases(&shared, vec![id.clone()]) { + RenewalStep::ReleaseDropped => { + if let Err(e) = super::release::release_leases(&shared, vec![id.clone()]).await + { error!( descriptor = ?id, held_version, @@ -235,47 +271,69 @@ impl LeaseRenewalLoop { } } +/// What one renewal tick does with one near-expiry lease. +#[derive(Debug, Clone, PartialEq, Eq)] +enum RenewalStep { + /// Re-acquire the lease at this version. + Renew(u64), + /// The descriptor is gone. Release the lease. + ReleaseDropped, + /// The catalog read failed. Leave the lease alone this tick. + Skip, +} + +/// Decide the step for a held lease from its catalog lookup. +/// +/// A catalog read error says nothing about the descriptor. Releasing +/// drops a lease a statement still uses, so the lease is skipped +/// and the next tick retries the lookup. +fn renewal_step(held_version: u64, lookup: &crate::Result>) -> RenewalStep { + match lookup { + Ok(Some(v)) => RenewalStep::Renew((*v).max(held_version)), + Ok(None) => RenewalStep::ReleaseDropped, + Err(_) => RenewalStep::Skip, + } +} + /// Look up the current persisted version for a descriptor. -/// Returns `None` if the descriptor has been dropped, the -/// catalog is unavailable, or the descriptor kind is not one -/// the planner / renewal path tracks. -fn lookup_current_version(shared: &SharedState, id: &DescriptorId) -> Option { +/// +/// Returns `Ok(None)` when the descriptor is dropped or its kind is not +/// one the planner and renewal path track. A catalog read error returns +/// `Err`. +fn lookup_current_version(shared: &SharedState, id: &DescriptorId) -> crate::Result> { use nodedb_cluster::DescriptorKind; let catalog = shared.credentials.catalog(); - match id.kind { + let database_id = DatabaseId::new(id.database_id); + let version = match id.kind { DescriptorKind::Collection => catalog - .get_collection(DatabaseId::new(id.database_id), id.tenant_id, &id.name) - .ok() - .flatten() + .get_collection(database_id, id.tenant_id, &id.name)? .filter(|c| c.is_active) .map(|c| c.descriptor_version.max(1)), DescriptorKind::Function => catalog - .get_function_in_database(DatabaseId::new(id.database_id), id.tenant_id, &id.name) - .ok() - .flatten() + .get_function_in_database(database_id, id.tenant_id, &id.name)? .map(|f| f.descriptor_version.max(1)), DescriptorKind::Procedure => catalog - .get_procedure_in_database(DatabaseId::new(id.database_id), id.tenant_id, &id.name) - .ok() - .flatten() + .get_procedure_in_database(database_id, id.tenant_id, &id.name)? .map(|p| p.descriptor_version.max(1)), DescriptorKind::Trigger => catalog - .get_trigger_in_database(DatabaseId::new(id.database_id), id.tenant_id, &id.name) - .ok() - .flatten() + .get_trigger_in_database(database_id, id.tenant_id, &id.name)? .map(|t| t.descriptor_version.max(1)), DescriptorKind::Sequence => catalog - .get_sequence(id.database_id, id.tenant_id, &id.name) - .ok() - .flatten() + .get_sequence(id.database_id, id.tenant_id, &id.name)? .map(|s| s.descriptor_version.max(1)), DescriptorKind::MaterializedView => catalog - .get_materialized_view(id.database_id, id.tenant_id, &id.name) - .ok() - .flatten() + .get_materialized_view(id.database_id, id.tenant_id, &id.name)? .map(|v| v.descriptor_version.max(1)), + DescriptorKind::Array => catalog + .get_array_in_database( + nodedb_types::TenantId::new(id.tenant_id), + database_id, + &id.name, + )? + .map(|_| crate::control::server::shared::clone_write::ARRAY_DESCRIPTOR_VERSION), _ => None, - } + }; + Ok(version) } /// Snapshot every lease in `(_, this_node_id)` whose remaining @@ -422,6 +480,25 @@ mod tests { assert!(!names.contains(&"c".to_string())); } + /// A catalog read error skips the lease: no renew, no release. + #[test] + fn catalog_read_error_skips_the_lease() { + let lookup: crate::Result> = Err(crate::Error::Storage { + engine: "catalog".into(), + detail: "read failed".into(), + }); + assert_eq!(renewal_step(3, &lookup), RenewalStep::Skip); + } + + /// A dropped descriptor releases the lease. A live one renews at the + /// higher of the held and catalog versions. + #[test] + fn lookup_outcome_picks_the_step() { + assert_eq!(renewal_step(3, &Ok(None)), RenewalStep::ReleaseDropped); + assert_eq!(renewal_step(3, &Ok(Some(5))), RenewalStep::Renew(5)); + assert_eq!(renewal_step(7, &Ok(Some(5))), RenewalStep::Renew(7)); + } + #[test] fn empty_cache_returns_empty_vec() { let leases = HashMap::new(); diff --git a/nodedb/src/control/lease/runtime.rs b/nodedb/src/control/lease/runtime.rs new file mode 100644 index 000000000..3a32ac9eb --- /dev/null +++ b/nodedb/src/control/lease/runtime.rs @@ -0,0 +1,41 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-node lease liveness and revocation state, owned by `SharedState`. + +use std::sync::{Arc, OnceLock}; +use std::time::Duration; + +use super::holders::LeaseHolders; + +/// This node's metadata-group term while it leads the group, else `None`. +pub type LeaderTermFn = Arc Option + Send + Sync>; + +/// Whether this node heard from the metadata leader within the window. The +/// leader always has. +pub type LeaderContactFn = Arc bool + Send + Sync>; + +/// Everything the lease module tracks per node beyond the lease cache itself. +#[derive(Default)] +pub struct LeaseRuntime { + /// SWIM Dead records and the leader's Raft contact samples. Shared with + /// the SWIM detector and the metadata leader's lease-GC sweep. + pub holder_liveness: Arc, + /// In-flight statement holds per descriptor. A lease this node loses + /// revokes the statements still running under it. + pub holders: Arc, + /// Reads one Raft group, so the drainer can poll it cheaply. `start_raft` + /// installs it. + pub metadata_leader_term: OnceLock, + /// Contact only: a replica behind on apply still counts as in contact. + /// `start_raft` installs it. + pub metadata_contact: OnceLock, + /// Releases synchronous code hands off. `start_raft` spawns the task + /// that proposes them. + pub releaser: super::releaser::LeaseReleaser, +} + +impl LeaseRuntime { + pub fn new() -> Self { + Self::default() + } +} diff --git a/nodedb/src/control/lease/self_fence.rs b/nodedb/src/control/lease/self_fence.rs new file mode 100644 index 000000000..a4a16eb3c --- /dev/null +++ b/nodedb/src/control/lease/self_fence.rs @@ -0,0 +1,213 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Holder-side lease fence. +//! +//! A holder refuses its cached lease once its last metadata-leader contact is +//! older than [`LEASE_SELF_FENCE_WINDOW`]. Other nodes release a SWIM-Dead +//! holder's leases only after that window, plus the clock-skew margin, has +//! passed. So a live holder that SWIM misjudges has stopped using its lease +//! before anyone else treats the lease as gone. +//! +//! A replica behind the leader's commit index also fails the check. It can lack +//! a release or drain the leader already committed. +//! +//! The fence covers only the cached fast path. A refused holder re-acquires +//! through a fresh raft grant, which proves leader contact again. + +use nodedb_cluster::{LEASE_SELF_FENCE_WINDOW, METADATA_GROUP_ID}; + +use crate::control::state::SharedState; + +/// Whether this node must not reuse a cached descriptor lease now. +/// +/// `start_raft` installs the raft read gate. Before it runs no metadata group +/// exists to drain or release this node's leases, so the node never fences. +pub(crate) fn lease_use_is_fenced(shared: &SharedState) -> bool { + match shared.raft_read_gate.get() { + Some(gate) => !gate.within_staleness_bound(METADATA_GROUP_ID, LEASE_SELF_FENCE_WINDOW), + None => false, + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + use std::sync::atomic::{AtomicU64, Ordering}; + use std::time::Duration; + + use async_trait::async_trait; + use nodedb_cluster::{DescriptorId, DescriptorKind}; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::control::cluster::read_index::{RaftReadGate, ReadIndexRefusal}; + use crate::control::lease::DEFAULT_LEASE_DURATION; + use crate::wal::WalManager; + + /// Gate whose metadata-leader contact is `contact_age_ms` old. + struct ContactAgeGate { + contact_age_ms: AtomicU64, + } + + impl ContactAgeGate { + fn set_contact_age(&self, age: Duration) { + let ms = u64::try_from(age.as_millis()).unwrap_or(u64::MAX); + self.contact_age_ms.store(ms, Ordering::SeqCst); + } + } + + #[async_trait] + impl RaftReadGate for ContactAgeGate { + async fn confirm_leader( + &self, + _group_id: u64, + _timeout: Duration, + ) -> Result { + Err(ReadIndexRefusal::NotLeader) + } + + fn within_staleness_bound(&self, group_id: u64, max_staleness: Duration) -> bool { + group_id == METADATA_GROUP_ID + && Duration::from_millis(self.contact_age_ms.load(Ordering::SeqCst)) + <= max_staleness + } + + fn holds_leader_lease(&self, _group_id: u64) -> bool { + false + } + + fn leader_lease_term(&self, _group_id: u64) -> Option { + None + } + + fn lease_read_index(&self, _group_id: u64) -> Option { + None + } + } + + fn fenced_state() -> (Arc, Arc, tempfile::TempDir) { + let directory = tempfile::tempdir().expect("create self-fence test directory"); + let wal = Arc::new( + WalManager::open_for_testing(&directory.path().join("self-fence.wal")) + .expect("open self-fence test WAL"), + ); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let state = SharedState::new(dispatcher, wal).expect("construct self-fence state"); + let gate = Arc::new(ContactAgeGate { + contact_age_ms: AtomicU64::new(0), + }); + if state + .raft_read_gate + .set(Arc::clone(&gate) as Arc) + .is_err() + { + panic!("raft read gate already set in a fresh test state"); + } + (state, gate, directory) + } + + fn orders() -> DescriptorId { + DescriptorId::new(0, 1, DescriptorKind::Collection, "orders".to_string()) + } + + /// Install a granted lease on `orders` the way the metadata applier + /// installs a committed grant. This state runs no metadata group. + fn grant_orders(state: &SharedState) -> nodedb_cluster::DescriptorLease { + let lease = nodedb_cluster::DescriptorLease { + descriptor_id: orders(), + version: 1, + node_id: state.node_id, + expires_at: nodedb_types::Hlc::new( + state.hlc_clock.peek().wall_ns + DEFAULT_LEASE_DURATION.as_nanos() as u64, + 0, + ), + }; + state + .metadata_cache + .write() + .unwrap_or_else(|p| p.into_inner()) + .leases + .insert((orders(), state.node_id), lease.clone()); + lease + } + + #[tokio::test] + async fn no_gate_never_fences() { + let directory = tempfile::tempdir().expect("create self-fence test directory"); + let wal = Arc::new( + WalManager::open_for_testing(&directory.path().join("self-fence.wal")) + .expect("open self-fence test WAL"), + ); + let (dispatcher, _data_sides) = Dispatcher::new(1, 64); + let state = SharedState::new(dispatcher, wal).expect("construct self-fence state"); + assert!(!lease_use_is_fenced(&state)); + } + + #[tokio::test] + async fn cached_lease_is_used_inside_the_window() { + let (state, gate, _directory) = fenced_state(); + gate.set_contact_age(LEASE_SELF_FENCE_WINDOW - Duration::from_secs(1)); + + let first = grant_orders(&state); + assert!(state.lookup_lease_for_self(&orders()).is_some()); + let reused = state + .acquire_descriptor_lease(orders(), 1, DEFAULT_LEASE_DURATION) + .await + .expect("fast-path acquire"); + assert_eq!( + reused.expires_at, first.expires_at, + "fast path reuses the lease" + ); + } + + #[tokio::test] + async fn cached_lease_is_refused_past_the_window() { + let (state, gate, _directory) = fenced_state(); + grant_orders(&state); + + gate.set_contact_age(LEASE_SELF_FENCE_WINDOW + Duration::from_secs(1)); + assert!(lease_use_is_fenced(&state)); + assert!( + state.lookup_lease_for_self(&orders()).is_none(), + "a fenced holder must not report its cached lease as usable" + ); + + // A fenced holder proposes a fresh grant instead of reusing the + // cached lease. This state runs no metadata group, so the proposal + // refuses. + let reacquired = state + .acquire_descriptor_lease(orders(), 1, DEFAULT_LEASE_DURATION) + .await; + assert!( + reacquired.is_err(), + "a fenced holder must re-acquire instead of reusing the cached lease: {reacquired:?}" + ); + } + + /// A follower applying a drain start has not advanced its applied index + /// past the entry, so it always reads as fenced there. The fence gates + /// reuse only: the drain start still hands the idle lease to the + /// releaser. + #[tokio::test] + async fn a_fenced_holder_still_releases_its_idle_lease_on_drain_start() { + let (state, gate, _directory) = fenced_state(); + grant_orders(&state); + gate.set_contact_age(LEASE_SELF_FENCE_WINDOW + Duration::from_secs(1)); + assert!(lease_use_is_fenced(&state)); + + crate::control::lease::release::release_idle_on_drain(&state, &orders()); + + let mut rx = state + .lease_runtime + .releaser + .take_receiver() + .expect("no releaser task runs in this state"); + assert_eq!( + rx.try_recv().ok(), + Some( + crate::control::lease::releaser::ReleaseRequest::UnheldDescriptors(vec![orders()]) + ), + "the drain start must hand the idle lease to the releaser" + ); + } +} diff --git a/nodedb/src/control/lease/shutdown_release.rs b/nodedb/src/control/lease/shutdown_release.rs index d799288b4..188d4fdb6 100644 --- a/nodedb/src/control/lease/shutdown_release.rs +++ b/nodedb/src/control/lease/shutdown_release.rs @@ -33,8 +33,7 @@ use crate::control::state::SharedState; pub const DEFAULT_SHUTDOWN_RELEASE_TIMEOUT: Duration = Duration::from_secs(2); /// Release every lease this node currently holds in the -/// metadata cache. No-op in single-node mode (no metadata raft -/// handle) and on empty lease sets. +/// metadata cache. No-op on an empty lease set. /// /// Returns once the release raft entry has been applied locally /// (via `release_descriptor_leases`'s internal wait), or once @@ -49,31 +48,18 @@ pub async fn release_all_local_leases(shared: Arc, deadline: Durati } let count = descriptor_ids.len(); - // Bound the release call by `deadline`. `release_descriptor_leases` - // is sync and uses `block_in_place + wait_for` internally, so - // we run it via `spawn_blocking` to release the async runtime. - let release_shared = Arc::clone(&shared); - let release_task = tokio::task::spawn_blocking(move || { - release_shared.release_descriptor_leases(descriptor_ids) - }); - - match timeout(deadline, release_task).await { - Ok(Ok(Ok(()))) => { + // Bound the release call by `deadline`. + match timeout(deadline, shared.release_descriptor_leases(descriptor_ids)).await { + Ok(Ok(())) => { tracing::info!(count, "shutdown release: released {count} local leases"); } - Ok(Ok(Err(e))) => { + Ok(Err(e)) => { tracing::warn!( error = %e, count, "shutdown release: propose failed, leases will drain via TTL" ); } - Ok(Err(join_err)) => { - tracing::warn!( - error = %join_err, - "shutdown release: spawn_blocking task panicked" - ); - } Err(_) => { tracing::warn!( count, diff --git a/nodedb/src/control/local_dispatch/local_read.rs b/nodedb/src/control/local_dispatch/local_read.rs index b751f1d38..ee1f232df 100644 --- a/nodedb/src/control/local_dispatch/local_read.rs +++ b/nodedb/src/control/local_dispatch/local_read.rs @@ -81,6 +81,7 @@ pub(crate) async fn dispatch_local_read( txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: Admission::Exempt(ExemptReason::Read), }; diff --git a/nodedb/src/control/maintenance/clone_materializer/chained.rs b/nodedb/src/control/maintenance/clone_materializer/chained.rs new file mode 100644 index 000000000..30758a4ca --- /dev/null +++ b/nodedb/src/control/maintenance/clone_materializer/chained.rs @@ -0,0 +1,270 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Clone materialization of a `HASH_CHAIN` document collection. +//! +//! Each row's link covers its stored contents and its position, so the clone +//! copies the rows as stored, in source position order, as relink requests: +//! redo records of origin `Restore`, committed through the target vShard's +//! apply log. Every replica links each row from the target head. The target is +//! empty, so the links it derives equal the source's. +//! +//! A re-run after a crash sends the same rows in the same order. A row already +//! in the target is kept as installed, and the rest link after it. + +use std::collections::HashSet; + +use nodedb_physical::physical_plan::RedoOrigin; +use nodedb_types::sync::wire::SyncProvenance; +use nodedb_types::{CollectionKey, DatabaseId, Surrogate, TenantId}; +use nodedb_wal::record::RecordType; + +use crate::control::planner::sql_plan_convert::convert::db_qualified; +use crate::control::security::catalog::{StoredCollection, SystemCatalog}; +use crate::control::state::SharedState; +use crate::control::surrogate::CarriedIdentity; +use crate::control::wal_replication::encode::transaction_redo_entry; +use crate::control::wal_replication::propose_replicated_entry; +use crate::control::wal_replication::transaction_redo::{RedoTarget, TransactionRedoPayload}; +use crate::event::EventSource; +use crate::types::hash_chain::CHAIN_SEQ_FIELD; +use crate::types::{HomedRecord, RecordHomes}; +use crate::wal::{RedoRecord, RedoRowChange, RedoRowKind, RedoSubRecord}; + +use super::document::{SourcePage, scan_page}; + +/// Most rows one relink record carries. +const ROWS_PER_RECORD: usize = 512; + +/// One scanned source row, before its target surrogate is bound. +struct ScannedRow { + seq: u64, + document_id: String, + pk_bytes: Vec, + body: Vec, +} + +/// One source row, ready to copy under its bound target surrogate. +struct ChainedRow { + seq: u64, + document_id: String, + pk_bytes: Vec, + target_surrogate: Surrogate, + body: Vec, +} + +fn clone_error(detail: String) -> crate::Error { + crate::Error::Storage { + engine: "clone_materializer".into(), + detail, + } +} + +/// Copy every live source row of `coll` into the target in source position +/// order. Returns the rows copied. +pub(super) async fn materialize_chained_collection( + state: &SharedState, + catalog: &SystemCatalog, + db_id: DatabaseId, + coll: &StoredCollection, + tombstoned: &HashSet, + system_as_of_ms: Option, +) -> crate::Result { + let Some(ref origin) = coll.cloned_from else { + return Ok(0); + }; + let tenant_id = TenantId::new(coll.tenant_id); + let source_qualified = db_qualified(origin.source_database, &origin.source_collection); + let target_qualified = db_qualified(db_id, &coll.name); + let source_key = CollectionKey::from_bare(origin.source_database, &origin.source_collection); + let target_key = CollectionKey::from_bare(db_id, &coll.name); + + let mut scanned: Vec = Vec::new(); + let mut cursor: Vec = Vec::new(); + loop { + let page = SourcePage { + tenant_id, + source_db_id: origin.source_database, + source_qualified: &source_qualified, + cursor: &cursor, + system_as_of_ms, + raw_bodies: true, + }; + let (entries, next_cursor) = scan_page(state, page, None).await?; + for (doc_id_hex, source_surrogate, body) in entries { + if tombstoned.contains(&source_surrogate) + || catalog + .get_clone_copyup(&target_qualified, source_surrogate)? + .is_some() + { + continue; + } + let doc = nodedb_types::json_from_msgpack(&body).map_err(|e| { + clone_error(format!( + "row {doc_id_hex} of hash-chained '{source_qualified}' does not decode: {e}" + )) + })?; + let seq = doc + .get(CHAIN_SEQ_FIELD) + .and_then(|seq| seq.as_u64()) + .ok_or_else(|| { + clone_error(format!( + "row {doc_id_hex} of hash-chained '{source_qualified}' has no unsigned \ + integer '{CHAIN_SEQ_FIELD}' field" + )) + })?; + let pk_bytes = catalog + .get_pk_for_surrogate(source_key, tenant_id, Surrogate::new(source_surrogate)) + .map_err(|e| { + clone_error(format!( + "get_pk_for_surrogate failed for surrogate {source_surrogate} in \ + '{source_qualified}': {e}" + )) + })? + .unwrap_or_else(|| doc_id_hex.as_bytes().to_vec()); + // A chained schemaless row already carries `id` from its first + // write, which the source link covers, so this leaves its bytes + // unchanged and the relinked target link equals the source link. + let body = match nodedb_types::StorageKey::parse(&doc_id_hex) { + Some(storage_key) => { + let identity = nodedb_types::RowIdentity::of_stored_row( + &body, + coll.declared_primary_key.as_deref(), + storage_key, + ); + crate::control::clone::identity::carry_identity(coll, body, identity.as_str()) + } + None => body, + }; + scanned.push(ScannedRow { + seq, + document_id: String::from_utf8_lossy(&pk_bytes).into_owned(), + pk_bytes, + body, + }); + } + if next_cursor.is_empty() { + break; + } + cursor = next_cursor; + } + // Every row's target surrogate in one batch at the target collection's + // home, under the same (collection, pk_bytes) keys the INSERT path uses. + let pks: Vec<&[u8]> = scanned.iter().map(|row| row.pk_bytes.as_slice()).collect(); + let bound = crate::control::server::surrogate_exchange::assign_surrogates_routed( + state, + target_key, + tenant_id, + &pks, + crate::types::TraceId::ZERO, + ) + .await + .map_err(|e| { + clone_error(format!( + "surrogate assign failed for the rows of '{target_qualified}': {e}" + )) + })?; + super::status::check_bound_surrogates(&target_qualified, bound.len(), scanned.len())?; + let mut rows: Vec = scanned + .into_iter() + .zip(bound) + .map(|(row, target_surrogate)| ChainedRow { + seq: row.seq, + document_id: row.document_id, + pk_bytes: row.pk_bytes, + target_surrogate, + body: row.body, + }) + .collect(); + rows.sort_by_key(|row| row.seq); + + let target = RedoTarget { + tenant_id, + database_id: db_id, + vshard_id: RecordHomes::of(HomedRecord::Row(target_key)).owner(), + }; + let stored = nodedb_types::QualifiedCollection::new(db_id, &coll.name); + let copied = rows.len() as u64; + let mut pending = rows.into_iter().peekable(); + while pending.peek().is_some() { + let batch: Vec = pending.by_ref().take(ROWS_PER_RECORD).collect(); + let payload = relink_payload(stored.as_str(), &coll.name, batch)?; + commit_relink(state, target, &payload).await?; + } + Ok(copied) +} + +/// One batch as a relink record: each row a document put carrying its +/// source link, in position order. +fn relink_payload( + stored: &str, + bare: &str, + batch: Vec, +) -> crate::Result { + let mut ops = Vec::with_capacity(batch.len()); + let mut identities = Vec::with_capacity(batch.len()); + // Each relinked row installs in the clone as a new row. + let mut row_changes = Vec::with_capacity(batch.len()); + for row in batch { + row_changes.push(RedoRowChange { + collection: stored.to_string(), + row: row.document_id.as_str().to_owned(), + kind: RedoRowKind::Insert, + }); + let prov: Option = None; + let payload = zerompk::to_msgpack_vec(&( + stored, + row.document_id.as_str(), + row.body, + prov, + row.target_surrogate.as_u32(), + )) + .map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("clone relink: encode document put: {e}"), + })?; + ops.push(RedoSubRecord { + record_type: RecordType::Put as u32, + payload, + }); + identities.push(CarriedIdentity { + collection: bare.to_string(), + pk_bytes: row.pk_bytes, + surrogate: row.target_surrogate, + }); + } + Ok(TransactionRedoPayload { + redo: RedoRecord { + version: 1, + ops, + calvin_stamp: None, + cross_shard_applied: None, + row_sources: Vec::new(), + publishes: Vec::new(), + row_changes, + }, + collections: vec![stored.to_string()], + sum_targets: Vec::new(), + identities, + // The rows passed their rules when the source first wrote them, and + // a copy fires no AFTER trigger. + event_source: EventSource::Restore, + origin: RedoOrigin::Restore, + }) +} + +/// Commit one relink record and wait until it is durable and installed here. +async fn commit_relink( + state: &SharedState, + target: RedoTarget, + payload: &TransactionRedoPayload, +) -> crate::Result<()> { + let proposer = state.async_raft_proposer()?; + let entry = transaction_redo_entry( + target.tenant_id, + target.database_id, + target.vshard_id, + payload, + ); + propose_replicated_entry(state, proposer, entry).await?; + Ok(()) +} diff --git a/nodedb/src/control/maintenance/clone_materializer/columnar.rs b/nodedb/src/control/maintenance/clone_materializer/columnar.rs index da1303f48..901d130c9 100644 --- a/nodedb/src/control/maintenance/clone_materializer/columnar.rs +++ b/nodedb/src/control/maintenance/clone_materializer/columnar.rs @@ -24,13 +24,13 @@ //! storage layer. A single `ColumnarOp::MaterializeScan` handler (and this //! Control Plane loop) serves all three profiles. -use nodedb_types::{CloneStatus, DatabaseId, Lsn, RlsWriteCheck, TenantId}; +use std::collections::HashSet; -use super::dispatch::dispatch_local; +use nodedb_types::{DatabaseId, Lsn, RlsWriteCheck, Surrogate, TenantId}; + +use super::dispatch::dispatch_to_owner; use super::reaper::{ReapParams, reap_materialized_collection}; -use crate::bridge::envelope::Status; -use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use super::status::{check_bound_surrogates, checkpoint_progress, mark_materializing}; use crate::control::planner::sql_plan_convert::convert::db_qualified; use crate::control::security::catalog::{StoredCollection, SystemCatalog}; use crate::control::state::SharedState; @@ -52,217 +52,214 @@ pub(super) async fn materialize_columnar_collection( let Some(ref origin) = coll.cloned_from else { return Ok(()); }; + mark_materializing(state, coll).await?; let target_qualified = db_qualified(db_id, &coll.name); - let source_qualified = db_qualified(origin.source_database, &origin.source_collection); - let tenant_id = TenantId::new(coll.tenant_id); - - // Flip status to `Materializing` if still `Shadowed`. - if matches!(coll.clone_status, CloneStatus::Shadowed) { - let mut updated = coll.clone(); - updated.clone_status = CloneStatus::Materializing { - progress_lsn: Lsn::new(0), - bytes_done: 0, - bytes_total: 0, - }; - let outcome = propose_catalog_entry( - state, - &CatalogEntry::PutCollection(Box::new(updated.clone())), - )?; - if outcome.needs_local_apply() { - catalog.put_collection(db_id, &updated)?; - } - } - // Tombstones: synthetic source surrogates deleted from the clone before // materialization. The Data Plane scan encodes a unique u32 per row as the // surrogate (segment_id in upper 16 bits, row_idx in lower 16 bits). let tombstoned = catalog.list_clone_tombstones(&target_qualified)?; - // Convert as_of_lsn to milliseconds for the source-side scan. - let system_as_of_ms = state.ms_to_lsn_inverse(origin.as_of_lsn); - - // Detect target engine profile so INSERT dispatches to the right handler. - // Timeseries collections use `TimeseriesOp::Ingest` (msgpack array format) - // so rows land in `columnar_memtables` — not `columnar_engines` (plain - // columnar). Plain / Spatial use `ColumnarOp::Insert`. - let target_is_timeseries = coll.collection_type.is_timeseries(); + let system_as_of_ms = crate::control::clone::lsn_resolve::source_as_of_ms( + state, + coll.bitemporal, + origin.as_of_lsn, + ); - let mut cursor: Vec = Vec::new(); - let mut copied: u64 = 0; - let mut total_seen: u64 = 0; + ColumnarCopy { + state, + catalog, + db_id, + coll, + tenant_id: TenantId::new(coll.tenant_id), + source_db_id: origin.source_database, + source_qualified: db_qualified(origin.source_database, &origin.source_collection), + target_qualified: &target_qualified, + tombstoned: &tombstoned, + system_as_of_ms, + as_of_lsn: origin.as_of_lsn, + } + .run() + .await?; - loop { - let (entries, next_cursor) = scan_source_page( - state, - tenant_id, - origin.source_database, - &source_qualified, - &cursor, - system_as_of_ms, - ) - .await?; + reap_materialized_collection(ReapParams { + db_id, + tenant_id: coll.tenant_id, + name: &coll.name, + state, + catalog, + }) + .await?; + Ok(()) +} - for (source_surrogate_u32, value_bytes) in entries { - total_seen += 1; +/// The row copy of one columnar clone collection. +struct ColumnarCopy<'a> { + state: &'a SharedState, + catalog: &'a SystemCatalog, + db_id: DatabaseId, + coll: &'a StoredCollection, + tenant_id: TenantId, + source_db_id: DatabaseId, + source_qualified: String, + target_qualified: &'a str, + tombstoned: &'a HashSet, + system_as_of_ms: Option, + as_of_lsn: Lsn, +} - // Skip rows deleted from the clone (CoW tombstone). - if tombstoned.contains(&source_surrogate_u32) { - continue; +impl ColumnarCopy<'_> { + /// Copy every source page, with a progress checkpoint after each one. + async fn run(&self) -> crate::Result<()> { + let mut cursor: Vec = Vec::new(); + let mut copied: u64 = 0; + let mut total_seen: u64 = 0; + loop { + let (entries, next_cursor) = scan_source_page( + self.state, + self.tenant_id, + self.source_db_id, + &self.source_qualified, + &cursor, + self.system_as_of_ms, + ) + .await?; + total_seen += entries.len() as u64; + let pending = self.pending_rows(entries)?; + copied += self.copy_rows(pending).await?; + checkpoint_progress(self.state, self.coll, self.as_of_lsn, copied, total_seen).await?; + if next_cursor.is_empty() { + break; } + cursor = next_cursor; + } + tracing::info!( + db_id = self.db_id.as_u64(), + collection = %self.coll.name, + copied, + skipped_tombstoned = self.tombstoned.len(), + source_total = total_seen, + "columnar materialize: source rows copied to target", + ); + Ok(()) + } - // Skip rows already copy-up'd into target by the CoW write path. - if catalog - .get_clone_copyup(&target_qualified, source_surrogate_u32)? - .is_some() + /// The rows of one page still to copy. A row deleted from the clone (CoW + /// tombstone) or already copied up by the CoW write path is skipped. + fn pending_rows(&self, entries: Vec<(u32, Vec)>) -> crate::Result)>> { + let mut pending = Vec::with_capacity(entries.len()); + for (source_surrogate, value_bytes) in entries { + if self.tombstoned.contains(&source_surrogate) + || self + .catalog + .get_clone_copyup(self.target_qualified, source_surrogate)? + .is_some() { continue; } - - // Allocate target surrogate using a synthetic key derived from the - // source surrogate bytes so allocation is deterministic across - // retries. The source surrogate encodes (segment_id, row_idx) and - // is unique per source row within the collection. - let target_surrogate = state - .surrogate_assigner - .assign( - nodedb_types::CollectionKey::from_bare(db_id, &coll.name), - tenant_id, - &source_surrogate_u32.to_be_bytes(), - ) - .map_err(|e| crate::Error::Storage { - engine: "clone_materializer".into(), - detail: format!( - "surrogate assign failed for source surrogate {source_surrogate_u32} in \ - '{target_qualified}': {e}" - ), - })?; - - // Wrap the value_bytes (msgpack Value::Object) in a msgpack array - // so the Insert / Ingest handler can decode it as a row sequence. - let payload = wrap_in_array(value_bytes)?; - - let plan = if target_is_timeseries { - // Timeseries target: use TimeseriesOp::Ingest so rows land in - // `columnar_memtables` (the timeseries scan path reads from - // there, not from `columnar_engines`). - // Format "msgpack" = msgpack array-of-maps (same layout as - // SQL VALUES ingest produced by the planner). - PhysicalPlan::Timeseries(TimeseriesOp::Ingest { - collection: nodedb_types::QualifiedCollection::new(db_id, &coll.name), - payload, - format: "msgpack".into(), - wal_lsn: None, - surrogates: vec![target_surrogate], - provenance: None, - // `materialize_one` (walker.rs) already refused this - // materialization if either side carried an RLS policy, - // so no policy applies to source or target here — this - // reflects a check that ran, not an assumption. - rls_write_check: RlsWriteCheck::NoPolicyApplies, - returning: None, - rls_filters: Vec::new(), - }) - } else { - PhysicalPlan::Columnar(ColumnarOp::Insert { - collection: nodedb_types::QualifiedCollection::new(db_id, &coll.name), - payload, - format: "msgpack".into(), - intent: ColumnarInsertIntent::InsertIfAbsent, - on_conflict_updates: Vec::<(String, UpdateValue)>::new(), - surrogates: vec![target_surrogate], - schema_bytes: Vec::new(), - provenance: None, - wal_lsn: None, - // `materialize_one` (walker.rs) already refused this - // materialization if either side carried an RLS policy, - // so no policy applies to source or target here — this - // reflects a check that ran, not an assumption. - rls_write_check: RlsWriteCheck::NoPolicyApplies, - // Internal row copy — nothing is projected back and no - // caller identity's reads are being gated. - returning: None, - rls_filters: Vec::new(), - }) - }; - - let resp = - dispatch_local(state, tenant_id, db_id, &target_qualified, plan, None).await?; - if resp.status != Status::Ok { - return Err(crate::Error::Storage { - engine: "clone_materializer".into(), - detail: format!( - "columnar insert on target '{target_qualified}' for source surrogate \ - {source_surrogate_u32} returned status {:?}", - resp.status - ), - }); - } - copied += 1; + pending.push((source_surrogate, value_bytes)); } + Ok(pending) + } - // Per-page progress checkpoint. - checkpoint_progress( - state, - catalog, - db_id, - coll, - origin.as_of_lsn, - copied, - total_seen, + /// Bind target surrogates for `pending` and insert each row into the + /// target. Returns the rows copied. + async fn copy_rows(&self, pending: Vec<(u32, Vec)>) -> crate::Result { + // Target surrogates for the page in one batch at the target + // collection's home. Each keys on the source surrogate's bytes, so the + // allocation is deterministic across retries: the source surrogate + // encodes (segment_id, row_idx) and is unique per source row. + let keys: Vec<[u8; 4]> = pending.iter().map(|(s, _)| s.to_be_bytes()).collect(); + let key_refs: Vec<&[u8]> = keys.iter().map(|k| k.as_slice()).collect(); + let target_surrogates = + crate::control::server::surrogate_exchange::assign_surrogates_routed( + self.state, + nodedb_types::CollectionKey::from_bare(self.db_id, &self.coll.name), + self.tenant_id, + &key_refs, + crate::types::TraceId::ZERO, + ) + .await + .map_err(|e| crate::Error::Storage { + engine: "clone_materializer".into(), + detail: format!( + "surrogate assign failed for a page of '{}': {e}", + self.target_qualified + ), + })?; + check_bound_surrogates( + self.target_qualified, + target_surrogates.len(), + pending.len(), )?; - - if next_cursor.is_empty() { - break; + let mut copied = 0; + for ((_, value_bytes), target_surrogate) in pending.into_iter().zip(target_surrogates) { + let plan = self.insert_plan(value_bytes, target_surrogate); + dispatch_to_owner( + self.state, + self.tenant_id, + self.db_id, + self.target_qualified, + plan, + ) + .await?; + copied += 1; } - cursor = next_cursor; + Ok(copied) } - tracing::info!( - db_id = db_id.as_u64(), - collection = %coll.name, - copied, - skipped_tombstoned = tombstoned.len(), - source_total = total_seen, - "columnar materialize: source rows copied to target", - ); - - reap_materialized_collection(ReapParams { - target_collection_qualified: &target_qualified, - db_id, - tenant_id: coll.tenant_id, - name: &coll.name, - state, - catalog, - })?; - - Ok(()) -} - -/// Persist a `Materializing { progress_lsn, .. }` checkpoint between scan pages. -fn checkpoint_progress( - state: &SharedState, - catalog: &SystemCatalog, - db_id: DatabaseId, - coll: &StoredCollection, - as_of_lsn: Lsn, - copied: u64, - total_seen: u64, -) -> crate::Result<()> { - let mut updated = coll.clone(); - updated.clone_status = CloneStatus::Materializing { - progress_lsn: as_of_lsn, - bytes_done: copied, - bytes_total: total_seen, - }; - let outcome = propose_catalog_entry( - state, - &CatalogEntry::PutCollection(Box::new(updated.clone())), - )?; - if outcome.needs_local_apply() { - catalog.put_collection(db_id, &updated)?; + /// The plan that inserts one row into the target unless it is present. + /// + /// A timeseries target uses `TimeseriesOp::Ingest` (msgpack array format) + /// so rows land in `columnar_memtables`, which the timeseries scan path + /// reads. Plain / Spatial use `ColumnarOp::Insert` into + /// `columnar_engines`. + fn insert_plan(&self, value_bytes: Vec, target_surrogate: Surrogate) -> PhysicalPlan { + // Wrap the value_bytes (msgpack Value::Object) in a msgpack array + // so the Insert / Ingest handler can decode it as a row sequence. + let payload = wrap_in_array(value_bytes); + let collection = nodedb_types::QualifiedCollection::new(self.db_id, &self.coll.name); + if self.coll.collection_type.is_timeseries() { + // Format "msgpack" = msgpack array-of-maps (same layout as SQL + // VALUES ingest produced by the planner). + PhysicalPlan::Timeseries(TimeseriesOp::Ingest { + collection, + payload, + format: "msgpack".into(), + wal_lsn: None, + surrogates: vec![target_surrogate], + provenance: None, + // `materialize_one` (walker.rs) already refused this + // materialization if either side carried an RLS policy, + // so no policy applies to source or target here — this + // reflects a check that ran, not an assumption. + rls_write_check: RlsWriteCheck::NoPolicyApplies, + returning: None, + rls_filters: Vec::new(), + }) + } else { + PhysicalPlan::Columnar(ColumnarOp::Insert { + collection, + payload, + format: "msgpack".into(), + intent: ColumnarInsertIntent::InsertIfAbsent, + on_conflict_updates: Vec::<(String, UpdateValue)>::new(), + surrogates: vec![target_surrogate], + schema_bytes: Vec::new(), + provenance: None, + wal_lsn: None, + // `materialize_one` (walker.rs) already refused this + // materialization if either side carried an RLS policy, + // so no policy applies to source or target here — this + // reflects a check that ran, not an assumption. + rls_write_check: RlsWriteCheck::NoPolicyApplies, + // Internal row copy — nothing is projected back and no + // caller identity's reads are being gated. + returning: None, + rls_filters: Vec::new(), + }) + } } - Ok(()) } /// `(source_surrogate_u32, value_bytes)` returned by one scan page. @@ -283,17 +280,8 @@ async fn scan_source_page( count: SCAN_PAGE, system_as_of_ms, }); - let resp = dispatch_local(state, tenant_id, source_db_id, source_qualified, plan, None).await?; - if resp.status != Status::Ok { - return Err(crate::Error::Storage { - engine: "clone_materializer".into(), - detail: format!( - "columnar materialize-scan on source '{source_qualified}' returned status {:?}", - resp.status - ), - }); - } - parse_materialize_scan_payload(resp.payload.as_ref()) + let payload = dispatch_to_owner(state, tenant_id, source_db_id, source_qualified, plan).await?; + parse_materialize_scan_payload(&payload) } /// Parse the msgpack payload emitted by `execute_columnar_materialize_scan`: @@ -343,10 +331,10 @@ fn parse_materialize_scan_payload(payload: &[u8]) -> crate::Result { /// Wrap a single msgpack Value::Object blob in a msgpack fixarray of length 1 /// so the columnar insert handler can decode it as `Vec`. -fn wrap_in_array(value_bytes: Vec) -> crate::Result> { +fn wrap_in_array(value_bytes: Vec) -> Vec { // fixarray header for 1 element: 0x91 let mut out = Vec::with_capacity(1 + value_bytes.len()); out.push(0x91); out.extend_from_slice(&value_bytes); - Ok(out) + out } diff --git a/nodedb/src/control/maintenance/clone_materializer/dispatch.rs b/nodedb/src/control/maintenance/clone_materializer/dispatch.rs index 1d46d7211..c5a4b634d 100644 --- a/nodedb/src/control/maintenance/clone_materializer/dispatch.rs +++ b/nodedb/src/control/maintenance/clone_materializer/dispatch.rs @@ -1,26 +1,70 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Local-data-plane dispatch helper used by the clone materializer. +//! Data Plane dispatch helpers used by the clone materializer. //! //! Lets the walker issue scans and writes against source/target collections -//! without pgwire. The materializer runs on a Tokio blocking thread and uses -//! [`tokio::runtime::Handle::block_on`] to drive these futures synchronously. +//! without pgwire. The materializer awaits these futures on the caller's +//! runtime. use std::sync::atomic::Ordering; use std::time::{Duration, Instant}; use nodedb_types::{DatabaseId, TenantId}; -use crate::bridge::envelope::{Priority, Request, Response}; +use crate::bridge::envelope::{ErrorCode, Priority, Request, Response}; +use crate::control::gateway::core::QueryContext; +use crate::control::server::dispatch_utils::{OwnedRead, OwnedReadScope, route_owned_read}; +use crate::control::server::shared::write_admission::plan_is_write; use crate::control::state::SharedState; use crate::types::{ReadConsistency, RequestId, TraceId, TxnId, VShardId}; use nodedb_physical::physical_plan::PhysicalPlan; -/// Dispatch a `PhysicalPlan` to the local Data Plane and await the response. +/// Run a plan on the node that owns its vShard and return its payload. /// -/// Bypasses WAL replication coordination (the engine handler still appends -/// the WAL on mutation). Used for both source scans and target writes; every -/// shard the materializer touches is owned locally. +/// The plan routes through the gateway like any statement: a scan reads the +/// owner's shard, and a write goes through the owner's replicated write +/// path, so every replica of the shard applies it. +/// +/// A `NotFound` verdict returns an empty payload. A plan that fans out to +/// several shards is refused: a materializer cursor walks one shard. +pub(crate) async fn dispatch_to_owner( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection_qualified: &str, + plan: PhysicalPlan, +) -> crate::Result> { + let gateway = state.installed_gateway()?; + let ctx = QueryContext { + tenant_id, + trace_id: TraceId::generate(), + database_id, + txn_id: None, + linearizable: true, + }; + let mut payloads = match gateway.execute_internal(&ctx, plan).await { + Ok(payloads) => payloads, + Err(crate::Error::DataPlane(ErrorCode::NotFound)) => return Ok(Vec::new()), + Err(error) => return Err(error), + }; + if payloads.len() > 1 { + return Err(crate::Error::Storage { + engine: "clone_materializer".into(), + detail: format!( + "plan on '{collection_qualified}' answered from {} shards; \ + a materializer page walks one shard", + payloads.len() + ), + }); + } + Ok(payloads.pop().unwrap_or_default()) +} + +/// Dispatch a `PhysicalPlan` to the Data Plane and await the response. +/// +/// A write bypasses routing and replication: it reaches only this node's copy +/// of the shard. A read runs where the owning group serves it. Materializer +/// scans and writes use [`dispatch_to_owner`]. /// /// `txn_id` stamps the request with the transaction whose staging overlay /// the handler must fold in. Autocommit callers pass `None`; COMMIT-time @@ -40,12 +84,41 @@ pub(crate) async fn dispatch_local( dispatch_local_on_vshard(state, tenant_id, database_id, vshard_id, plan, txn_id).await } +/// Run a read-only resolve pass that carries a `Write` grant (columnar DML's +/// `ResolveDml`) on this node's cores. `MERGE` and `UPDATE ... FROM` resolve +/// through `DocumentOp::ResolveWrite`, a read, which [`dispatch_local`] routes +/// to the owner. +/// +/// The gateway carries such a plan as a write, so the pass cannot be sent +/// to the owner. It runs here only when this node replicates the collection's +/// home group, confirmed, and is refused with `NotLeader` otherwise +/// (`dispatch_utils::prepare_local_pass`). +pub(crate) async fn dispatch_resolve_pass( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection_qualified: &str, + plan: PhysicalPlan, + txn_id: Option, +) -> crate::Result { + let vshard_id = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection_qualified)? + .vshard(); + crate::control::server::dispatch_utils::prepare_local_pass(state, vshard_id, txn_id).await?; + dispatch_local_on_vshard(state, tenant_id, database_id, vshard_id, plan, txn_id).await +} + /// [`dispatch_local`] for a plan whose home vShard is not derived from a /// collection name. /// /// A graph edge plan is key-homed on its endpoints, so the vShard that holds /// the edge is `VShardId::from_key(src_id)` and not the collection's hash. /// Routing such a plan by collection reads a shard that never held it. +/// +/// A read of this node's copy feeds a write (MERGE and `UPDATE ... FROM` +/// resolve passes, predicate write resolution). A read runs where the group +/// that owns `vshard_id` serves it, confirmed (`dispatch_utils::owner_read`): +/// a stale or absent replica aims the write at the wrong rows. pub(crate) async fn dispatch_local_on_vshard( state: &SharedState, tenant_id: TenantId, @@ -53,6 +126,35 @@ pub(crate) async fn dispatch_local_on_vshard( vshard_id: VShardId, plan: PhysicalPlan, txn_id: Option, +) -> crate::Result { + let plan = if plan_is_write(&plan) { + plan + } else { + let scope = OwnedReadScope { + tenant_id, + database_id, + vshard_id, + trace_id: TraceId::ZERO, + txn_id, + linearizable: true, + }; + match route_owned_read(state, scope, plan).await? { + OwnedRead::Local(plan) => *plan, + OwnedRead::Served(response) => return Ok(response), + } + }; + dispatch_on_this_node(state, tenant_id, database_id, vshard_id, plan, txn_id).await +} + +/// Dispatch `plan` to this node's core for `vshard_id` and await the +/// response, with no routing: the plan runs against this node's own state. +pub(crate) async fn dispatch_on_this_node( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + plan: PhysicalPlan, + txn_id: Option, ) -> crate::Result { let req_id = RequestId::new(state.request_id_counter.fetch_add(1, Ordering::Relaxed)); let deadline_secs = state.tuning.network.default_deadline_secs; @@ -75,6 +177,7 @@ pub(crate) async fn dispatch_local_on_vshard( txn_id, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: crate::bridge::envelope::Admission::Exempt( crate::bridge::envelope::ExemptReason::AlreadyOrdered, ), diff --git a/nodedb/src/control/maintenance/clone_materializer/document.rs b/nodedb/src/control/maintenance/clone_materializer/document.rs index 5ff243b4c..c6c4acb2f 100644 --- a/nodedb/src/control/maintenance/clone_materializer/document.rs +++ b/nodedb/src/control/maintenance/clone_materializer/document.rs @@ -11,15 +11,15 @@ //! filters CoW-copied rows, and `if_absent` skips already-written rows — the //! per-page checkpoint is best-effort but the per-key probes cover it. -use nodedb_types::{CloneStatus, DatabaseId, Lsn, Surrogate, TenantId}; +use nodedb_types::{DatabaseId, TenantId}; use crate::types::TxnId; -use super::dispatch::dispatch_local; +use super::dispatch::{dispatch_local, dispatch_to_owner}; +use super::document_copy::RowCopy; use super::reaper::{ReapParams, reap_materialized_collection}; +use super::status::{checkpoint_progress, mark_materializing}; use crate::bridge::envelope::Status; -use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; use crate::control::planner::sql_plan_convert::convert::db_qualified; use crate::control::security::catalog::{StoredCollection, SystemCatalog}; use crate::control::state::SharedState; @@ -38,203 +38,56 @@ pub(super) async fn materialize_document_collection( let Some(ref origin) = coll.cloned_from else { return Ok(()); }; + mark_materializing(state, coll).await?; let target_qualified = db_qualified(db_id, &coll.name); - let source_qualified = db_qualified(origin.source_database, &origin.source_collection); - let tenant_id = TenantId::new(coll.tenant_id); - - // Flip status to `Materializing` if still `Shadowed` so concurrent - // readers see in-progress state. - if matches!(coll.clone_status, CloneStatus::Shadowed) { - let mut updated = coll.clone(); - updated.clone_status = CloneStatus::Materializing { - progress_lsn: Lsn::new(0), - bytes_done: 0, - bytes_total: 0, - }; - let outcome = propose_catalog_entry( - state, - &CatalogEntry::PutCollection(Box::new(updated.clone())), - )?; - if outcome.needs_local_apply() { - catalog.put_collection(db_id, &updated)?; - } - } - // Tombstones: source surrogates deleted from the clone before materialization. let tombstoned = catalog.list_clone_tombstones(&target_qualified)?; - // Convert as_of_lsn to milliseconds for the source-side scan. - let system_as_of_ms = state.ms_to_lsn_inverse(origin.as_of_lsn); - - let mut cursor: Vec = Vec::new(); - let mut copied: u64 = 0; - let mut total_seen: u64 = 0; + let system_as_of_ms = crate::control::clone::lsn_resolve::source_as_of_ms( + state, + coll.bitemporal, + origin.as_of_lsn, + ); - loop { - let (entries, next_cursor) = scan_source_page( + if coll.hash_chain { + let copied = super::chained::materialize_chained_collection( state, - tenant_id, - origin.source_database, - &source_qualified, - &cursor, + catalog, + db_id, + coll, + &tombstoned, system_as_of_ms, - None, ) .await?; - - for (doc_id_hex, source_surrogate_u32, value_bytes) in entries { - total_seen += 1; - - let source_surrogate = Surrogate::new(source_surrogate_u32); - - // Skip rows deleted from the clone (CoW tombstone). - if tombstoned.contains(&source_surrogate_u32) { - continue; - } - - // Skip rows already copy-up'd into target by the CoW write path. - if catalog - .get_clone_copyup(&target_qualified, source_surrogate_u32)? - .is_some() - { - continue; - } - - // Recover the user-visible PK bytes from the catalog so the - // surrogate assigner produces the same surrogate the write path - // would allocate for this (collection, pk) pair. - let pk_bytes = catalog - .get_pk_for_surrogate( - nodedb_types::CollectionKey::from_bare( - origin.source_database, - &origin.source_collection, - ), - tenant_id, - source_surrogate, - ) - .map_err(|e| crate::Error::Storage { - engine: "clone_materializer".into(), - detail: format!( - "get_pk_for_surrogate failed for surrogate {source_surrogate_u32} \ - in '{source_qualified}': {e}" - ), - })? - .unwrap_or_else(|| { - // No PK binding (e.g. very old row): fall back to the hex doc_id - // as key bytes — deterministic but may differ from the write path. - doc_id_hex.as_bytes().to_vec() - }); - - let document_id = String::from_utf8_lossy(&pk_bytes).into_owned(); - - // Allocate target surrogate using the same (collection, pk_bytes) key - // the normal INSERT path would use. - let target_surrogate = state - .surrogate_assigner - .assign( - nodedb_types::CollectionKey::from_bare(db_id, &coll.name), - tenant_id, - &pk_bytes, - ) - .map_err(|e| crate::Error::Storage { - engine: "clone_materializer".into(), - detail: format!( - "surrogate assign failed for doc '{doc_id_hex}' in \ - '{target_qualified}': {e}" - ), - })?; - - let plan = PhysicalPlan::Document(DocumentOp::PointInsert { - collection: nodedb_types::QualifiedCollection::new(db_id, &coll.name), - document_id: document_id.clone(), - value: value_bytes, - if_absent: true, - surrogate: target_surrogate, - // A materializer copy answers no client, so it projects - // nothing and needs no read gate. - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - deferred_sum_targets: Vec::new(), - }); - - let resp = - dispatch_local(state, tenant_id, db_id, &target_qualified, plan, None).await?; - if resp.status != Status::Ok { - return Err(crate::Error::Storage { - engine: "clone_materializer".into(), - detail: format!( - "document insert on target '{target_qualified}' for doc \ - '{document_id}' returned status {:?}", - resp.status - ), - }); - } - copied += 1; - } - - // Per-page progress checkpoint. - checkpoint_progress( + checkpoint_progress(state, coll, origin.as_of_lsn, copied, copied).await?; + } else { + RowCopy { state, catalog, db_id, coll, - origin.as_of_lsn, - copied, - total_seen, - )?; - - if next_cursor.is_empty() { - break; + tenant_id: TenantId::new(coll.tenant_id), + source_db_id: origin.source_database, + source_collection: &origin.source_collection, + source_qualified: db_qualified(origin.source_database, &origin.source_collection), + target_qualified: &target_qualified, + tombstoned: &tombstoned, + system_as_of_ms, + as_of_lsn: origin.as_of_lsn, } - cursor = next_cursor; + .run() + .await?; } - tracing::info!( - db_id = db_id.as_u64(), - collection = %coll.name, - copied, - skipped_tombstoned = tombstoned.len(), - source_total = total_seen, - "document materialize: source rows copied to target", - ); - reap_materialized_collection(ReapParams { - target_collection_qualified: &target_qualified, db_id, tenant_id: coll.tenant_id, name: &coll.name, state, catalog, - })?; - - Ok(()) -} - -/// Persist a `Materializing { progress_lsn, .. }` checkpoint between scan pages. -fn checkpoint_progress( - state: &SharedState, - catalog: &SystemCatalog, - db_id: DatabaseId, - coll: &StoredCollection, - as_of_lsn: Lsn, - copied: u64, - total_seen: u64, -) -> crate::Result<()> { - let mut updated = coll.clone(); - updated.clone_status = CloneStatus::Materializing { - progress_lsn: as_of_lsn, - bytes_done: copied, - bytes_total: total_seen, - }; - let outcome = propose_catalog_entry( - state, - &CatalogEntry::PutCollection(Box::new(updated.clone())), - )?; - if outcome.needs_local_apply() { - catalog.put_collection(db_id, &updated)?; - } + }) + .await?; Ok(()) } @@ -255,12 +108,57 @@ pub(crate) async fn scan_source_page( system_as_of_ms: Option, txn_id: Option, ) -> crate::Result { + let page = SourcePage { + tenant_id, + source_db_id, + source_qualified, + cursor, + system_as_of_ms, + raw_bodies: false, + }; + scan_page(state, page, txn_id).await +} + +/// One source-side scan page request. +pub(super) struct SourcePage<'a> { + pub tenant_id: TenantId, + pub source_db_id: DatabaseId, + pub source_qualified: &'a str, + pub cursor: &'a [u8], + pub system_as_of_ms: Option, + /// Bodies as stored, with no `id` added. + pub raw_bodies: bool, +} + +/// Run one `MaterializeScan` round-trip for `page`. +pub(super) async fn scan_page( + state: &SharedState, + page: SourcePage<'_>, + txn_id: Option, +) -> crate::Result { + let SourcePage { + tenant_id, + source_db_id, + source_qualified, + cursor, + system_as_of_ms, + raw_bodies, + } = page; let plan = PhysicalPlan::Document(DocumentOp::MaterializeScan { collection: nodedb_types::QualifiedCollection::from_stored(source_qualified.to_string()), cursor: cursor.to_vec(), count: SCAN_PAGE, system_as_of_ms, + raw_bodies, }); + // A transaction's staging overlay lives on the source shard's leader, so + // a staged read runs there (`dispatch_local` routes it). Every other read + // goes to the source shard's owner. + if txn_id.is_none() { + let payload = + dispatch_to_owner(state, tenant_id, source_db_id, source_qualified, plan).await?; + return parse_materialize_scan_payload(&payload); + } let resp = dispatch_local( state, tenant_id, diff --git a/nodedb/src/control/maintenance/clone_materializer/document_copy.rs b/nodedb/src/control/maintenance/clone_materializer/document_copy.rs new file mode 100644 index 000000000..57bcde928 --- /dev/null +++ b/nodedb/src/control/maintenance/clone_materializer/document_copy.rs @@ -0,0 +1,214 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Row copy of one non-chained Document clone collection. +//! +//! Each source page is filtered, bound to target surrogates in one batch, and +//! inserted row by row with `PointInsert { if_absent: true }`. A progress +//! checkpoint follows each page. + +use std::collections::HashSet; + +use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; +use nodedb_types::{DatabaseId, Lsn, Surrogate, TenantId}; + +use super::dispatch::dispatch_to_owner; +use super::document::{SourcePage, scan_page}; +use super::status::{check_bound_surrogates, checkpoint_progress}; +use crate::control::security::catalog::{StoredCollection, SystemCatalog}; +use crate::control::state::SharedState; + +/// One source row to copy, with the pk bytes it keys under. +struct PendingRow { + doc_id_hex: String, + pk_bytes: Vec, + value_bytes: Vec, +} + +/// The row copy of one non-chained Document clone collection. +pub(super) struct RowCopy<'a> { + pub state: &'a SharedState, + pub catalog: &'a SystemCatalog, + pub db_id: DatabaseId, + pub coll: &'a StoredCollection, + pub tenant_id: TenantId, + pub source_db_id: DatabaseId, + pub source_collection: &'a str, + pub source_qualified: String, + pub target_qualified: &'a str, + pub tombstoned: &'a HashSet, + pub system_as_of_ms: Option, + pub as_of_lsn: Lsn, +} + +impl RowCopy<'_> { + /// Copy every source page, with a progress checkpoint after each one. + pub(super) async fn run(&self) -> crate::Result<()> { + let mut cursor: Vec = Vec::new(); + let mut copied: u64 = 0; + let mut total_seen: u64 = 0; + loop { + // The copy reads bodies as stored. A scan that adds `id` + // hands a strict target a field its schema lacks, and a + // declared-key target a second identity. `insert_row` writes the + // source identity back only where the target keeps it in `id`. + let page = SourcePage { + tenant_id: self.tenant_id, + source_db_id: self.source_db_id, + source_qualified: &self.source_qualified, + cursor: &cursor, + system_as_of_ms: self.system_as_of_ms, + raw_bodies: true, + }; + let (entries, next_cursor) = scan_page(self.state, page, None).await?; + total_seen += entries.len() as u64; + let pending = self.pending_rows(entries)?; + copied += self.copy_rows(pending).await?; + checkpoint_progress(self.state, self.coll, self.as_of_lsn, copied, total_seen).await?; + if next_cursor.is_empty() { + break; + } + cursor = next_cursor; + } + tracing::info!( + db_id = self.db_id.as_u64(), + collection = %self.coll.name, + copied, + skipped_tombstoned = self.tombstoned.len(), + source_total = total_seen, + "document materialize: source rows copied to target", + ); + Ok(()) + } + + /// The rows of one page still to copy. A row deleted from the clone (CoW + /// tombstone) or already copied up by the CoW write path is skipped. + fn pending_rows(&self, entries: Vec<(String, u32, Vec)>) -> crate::Result> { + let mut pending = Vec::with_capacity(entries.len()); + for (doc_id_hex, source_surrogate, value_bytes) in entries { + if self.tombstoned.contains(&source_surrogate) + || self + .catalog + .get_clone_copyup(self.target_qualified, source_surrogate)? + .is_some() + { + continue; + } + let pk_bytes = self.source_pk_bytes(&doc_id_hex, source_surrogate)?; + pending.push(PendingRow { + doc_id_hex, + pk_bytes, + value_bytes, + }); + } + Ok(pending) + } + + /// Recover the user-visible PK bytes from the catalog, so the surrogate + /// assigner produces the same surrogate the write path allocates for this + /// (collection, pk) pair. + fn source_pk_bytes(&self, doc_id_hex: &str, source_surrogate: u32) -> crate::Result> { + let pk_bytes = self + .catalog + .get_pk_for_surrogate( + nodedb_types::CollectionKey::from_bare(self.source_db_id, self.source_collection), + self.tenant_id, + Surrogate::new(source_surrogate), + ) + .map_err(|e| crate::Error::Storage { + engine: "clone_materializer".into(), + detail: format!( + "get_pk_for_surrogate failed for surrogate {source_surrogate} \ + in '{}': {e}", + self.source_qualified + ), + })?; + // No PK binding (e.g. very old row): the hex doc_id is the key bytes. + // It is deterministic but can differ from the write path. + Ok(pk_bytes.unwrap_or_else(|| doc_id_hex.as_bytes().to_vec())) + } + + /// Bind target surrogates for `pending` and insert each row into the + /// target. Returns the rows copied. + async fn copy_rows(&self, pending: Vec) -> crate::Result { + // Target surrogates for the whole page in one batch at the target + // collection's home, under the same (collection, pk_bytes) keys the + // normal INSERT path uses. + let pks: Vec<&[u8]> = pending.iter().map(|row| row.pk_bytes.as_slice()).collect(); + let target_surrogates = + crate::control::server::surrogate_exchange::assign_surrogates_routed( + self.state, + nodedb_types::CollectionKey::from_bare(self.db_id, &self.coll.name), + self.tenant_id, + &pks, + crate::types::TraceId::ZERO, + ) + .await + .map_err(|e| crate::Error::Storage { + engine: "clone_materializer".into(), + detail: format!( + "surrogate assign failed for a page of '{}': {e}", + self.target_qualified + ), + })?; + check_bound_surrogates( + self.target_qualified, + target_surrogates.len(), + pending.len(), + )?; + let mut copied = 0; + for (row, target_surrogate) in pending.into_iter().zip(target_surrogates) { + self.insert_row(row, target_surrogate).await?; + copied += 1; + } + Ok(copied) + } + + /// Insert one row into the target under `target_surrogate`, unless the + /// target already holds it. + async fn insert_row(&self, row: PendingRow, target_surrogate: Surrogate) -> crate::Result<()> { + let PendingRow { + doc_id_hex, + pk_bytes, + value_bytes, + } = row; + // The copy keeps the row's client identity under its new surrogate, + // so a clone read matches it to its source row by primary key. + let value_bytes = match nodedb_types::StorageKey::parse(&doc_id_hex) { + Some(storage_key) => { + let identity = nodedb_types::RowIdentity::of_stored_row( + &value_bytes, + self.coll.declared_primary_key.as_deref(), + storage_key, + ); + crate::control::clone::identity::carry_identity( + self.coll, + value_bytes, + identity.as_str(), + ) + } + None => value_bytes, + }; + let plan = PhysicalPlan::Document(DocumentOp::PointInsert { + collection: nodedb_types::QualifiedCollection::new(self.db_id, &self.coll.name), + document_id: String::from_utf8_lossy(&pk_bytes).into_owned(), + value: value_bytes, + if_absent: true, + surrogate: target_surrogate, + // A materializer copy answers no client, so it projects + // nothing and needs no read gate. + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }); + dispatch_to_owner( + self.state, + self.tenant_id, + self.db_id, + self.target_qualified, + plan, + ) + .await?; + Ok(()) + } +} diff --git a/nodedb/src/control/maintenance/clone_materializer/kv.rs b/nodedb/src/control/maintenance/clone_materializer/kv.rs index 7c73cdb18..af0fdc52b 100644 --- a/nodedb/src/control/maintenance/clone_materializer/kv.rs +++ b/nodedb/src/control/maintenance/clone_materializer/kv.rs @@ -2,9 +2,10 @@ //! KV engine source-to-target row copy. //! -//! Drives the source `KvOp::MaterializeScan` cursor to completion, and for -//! each non-tombstoned key not yet present in target dispatches a `KvOp::Put` -//! against target with a fresh surrogate. Calls the reaper at the end to flip +//! Drives the source `KvOp::MaterializeScan` cursor to completion on the +//! source shard's owner, and for each non-tombstoned key not yet present in +//! target dispatches a replicated `KvOp::Put` against target with a fresh +//! surrogate. Calls the reaper at the end to flip //! status to `Materialized` and clear `cloned_from`. //! //! ## Idempotency / restart-safety @@ -15,18 +16,16 @@ //! is the atomic Raft proposal at the end of the per-collection pass. The //! walker re-runs this function on the next sweep until the reaper succeeds. -use nodedb_types::{CloneStatus, DatabaseId, Lsn, TenantId}; +use nodedb_types::{DatabaseId, TenantId}; -use crate::bridge::envelope::Status; -use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; use crate::control::planner::sql_plan_convert::convert::db_qualified; use crate::control::security::catalog::{StoredCollection, SystemCatalog}; use crate::control::state::SharedState; use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; -use super::dispatch::dispatch_local; +use super::dispatch::dispatch_to_owner; use super::reaper::{ReapParams, reap_materialized_collection}; +use super::status::{check_bound_surrogates, checkpoint_progress, mark_materializing}; /// Rows fetched per scan round-trip. Larger = fewer round-trips, more memory /// per response. 4096 is a balance for typical clone sizes; very large @@ -47,24 +46,7 @@ pub(super) async fn materialize_kv_collection( let target_qualified = db_qualified(db_id, &coll.name); let source_qualified = db_qualified(origin.source_database, &origin.source_collection); let tenant_id = TenantId::new(coll.tenant_id); - - // Flip status to `Materializing` if still `Shadowed` so concurrent - // readers see in-progress state and a crash here resumes from `progress_lsn = 0`. - if matches!(coll.clone_status, CloneStatus::Shadowed) { - let mut updated = coll.clone(); - updated.clone_status = CloneStatus::Materializing { - progress_lsn: Lsn::new(0), - bytes_done: 0, - bytes_total: 0, - }; - let outcome = propose_catalog_entry( - state, - &CatalogEntry::PutCollection(Box::new(updated.clone())), - )?; - if outcome.needs_local_apply() { - catalog.put_collection(db_id, &updated)?; - } - } + mark_materializing(state, coll).await?; let tombstoned = catalog.list_kv_clone_tombstones(&target_qualified)?; @@ -82,6 +64,7 @@ pub(super) async fn materialize_kv_collection( ) .await?; + let mut pending: Vec<(Vec, Vec)> = Vec::with_capacity(entries.len()); for (key, value) in entries { total_seen += 1; let key_str = String::from_utf8_lossy(&key).into_owned(); @@ -92,12 +75,21 @@ pub(super) async fn materialize_kv_collection( if probe_target_key(state, tenant_id, db_id, &target_qualified, &key).await? { continue; } + pending.push((key, value)); + } - let surrogate = state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(db_id, &coll.name), - tenant_id, - &key, - )?; + // The page's surrogates in one batch at the target collection's home. + let keys: Vec<&[u8]> = pending.iter().map(|(key, _)| key.as_slice()).collect(); + let surrogates = crate::control::server::surrogate_exchange::assign_surrogates_routed( + state, + nodedb_types::CollectionKey::from_bare(db_id, &coll.name), + tenant_id, + &keys, + crate::types::TraceId::ZERO, + ) + .await?; + check_bound_surrogates(&target_qualified, surrogates.len(), pending.len())?; + for ((key, value), surrogate) in pending.into_iter().zip(surrogates) { let plan = PhysicalPlan::Kv(KvOp::Put { collection: nodedb_types::QualifiedCollection::new(db_id, &coll.name), key: key.clone(), @@ -108,32 +100,14 @@ pub(super) async fn materialize_kv_collection( rls_filters: Vec::new(), provenance: None, }); - let resp = - dispatch_local(state, tenant_id, db_id, &target_qualified, plan, None).await?; - if resp.status != Status::Ok { - return Err(crate::Error::Storage { - engine: "clone_materializer".into(), - detail: format!( - "kv put on target '{target_qualified}' for key {key_str} returned status {:?}", - resp.status - ), - }); - } + dispatch_to_owner(state, tenant_id, db_id, &target_qualified, plan).await?; copied += 1; } // Persist a chunk-level progress checkpoint so a crash after this // batch resumes without re-walking already-copied keys (the per-key // probe makes that safe regardless, but cheaper to skip the round-trip). - checkpoint_progress( - state, - catalog, - db_id, - coll, - origin.as_of_lsn, - copied, - total_seen, - )?; + checkpoint_progress(state, coll, origin.as_of_lsn, copied, total_seen).await?; if next_cursor.is_empty() { break; @@ -151,43 +125,17 @@ pub(super) async fn materialize_kv_collection( ); reap_materialized_collection(ReapParams { - target_collection_qualified: &target_qualified, db_id, tenant_id: coll.tenant_id, name: &coll.name, state, catalog, - })?; + }) + .await?; Ok(()) } -/// Persist a `Materializing { progress_lsn, .. }` checkpoint between scan pages. -fn checkpoint_progress( - state: &SharedState, - catalog: &SystemCatalog, - db_id: DatabaseId, - coll: &StoredCollection, - as_of_lsn: Lsn, - copied: u64, - total_seen: u64, -) -> crate::Result<()> { - let mut updated = coll.clone(); - updated.clone_status = CloneStatus::Materializing { - progress_lsn: as_of_lsn, - bytes_done: copied, - bytes_total: total_seen, - }; - let outcome = propose_catalog_entry( - state, - &CatalogEntry::PutCollection(Box::new(updated.clone())), - )?; - if outcome.needs_local_apply() { - catalog.put_collection(db_id, &updated)?; - } - Ok(()) -} - /// Run one source-side `MaterializeScan` round-trip. Returns the entries in /// this page (raw `(key, value)` byte pairs) plus the next-cursor; the /// cursor is empty when the scan is complete. @@ -203,17 +151,8 @@ async fn scan_source_page( cursor: cursor.to_vec(), count: SCAN_PAGE, }); - let resp = dispatch_local(state, tenant_id, source_db_id, source_qualified, plan, None).await?; - if resp.status != Status::Ok { - return Err(crate::Error::Storage { - engine: "clone_materializer".into(), - detail: format!( - "kv materialize-scan on source '{source_qualified}' returned status {:?}", - resp.status - ), - }); - } - parse_materialize_scan_payload(resp.payload.as_ref()) + let payload = dispatch_to_owner(state, tenant_id, source_db_id, source_qualified, plan).await?; + parse_materialize_scan_payload(&payload) } /// `(key, value)` pairs returned by one materialize-scan page. @@ -283,6 +222,9 @@ async fn probe_target_key( // ceiling, so reads against the target stay unbounded here. surrogate_ceiling: None, }); - let resp = dispatch_local(state, tenant_id, db_id, target_qualified, plan, None).await?; - Ok(resp.status == Status::Ok && !resp.payload.is_empty()) + Ok( + !dispatch_to_owner(state, tenant_id, db_id, target_qualified, plan) + .await? + .is_empty(), + ) } diff --git a/nodedb/src/control/maintenance/clone_materializer/mod.rs b/nodedb/src/control/maintenance/clone_materializer/mod.rs index 23f0ef7ca..4ce7d83fd 100644 --- a/nodedb/src/control/maintenance/clone_materializer/mod.rs +++ b/nodedb/src/control/maintenance/clone_materializer/mod.rs @@ -1,21 +1,26 @@ // SPDX-License-Identifier: BUSL-1.1 +mod chained; mod columnar; mod dispatch; mod document; +mod document_copy; mod kv; mod reaper; mod rls_gate; +mod source_drain; +mod status; pub mod progress; pub mod walker; pub use progress::CloneMaterializerHandle; -pub use walker::{ - MaterializeParams, force_materialize_blocking, materialize_database, run_scheduled_sweep, -}; +pub use walker::{MaterializeParams, force_materialize, materialize_database, run_scheduled_sweep}; -// Shared with the `INSERT ... SELECT` orchestrator, which reuses the same -// local-dispatch primitive and source-scan cursor decode. -pub(crate) use dispatch::{dispatch_local, dispatch_local_on_vshard}; +// Shared with the `INSERT ... SELECT` orchestrator (local dispatch, source +// scan) and the clone copy-up (owner-routed replicated write). +pub(crate) use dispatch::{ + dispatch_local, dispatch_local_on_vshard, dispatch_on_this_node, dispatch_resolve_pass, + dispatch_to_owner, +}; pub(crate) use document::{read_all_source_rows, scan_source_page}; diff --git a/nodedb/src/control/maintenance/clone_materializer/progress.rs b/nodedb/src/control/maintenance/clone_materializer/progress.rs index 908a5e65a..7c6cea7cd 100644 --- a/nodedb/src/control/maintenance/clone_materializer/progress.rs +++ b/nodedb/src/control/maintenance/clone_materializer/progress.rs @@ -28,8 +28,9 @@ struct Slot { /// Intended use-cases: /// 1. Background scheduler: calls `notify_start(n)` once it begins N collections, /// then calls `notify_collection_done()` for each that completes. -/// 2. `ALTER DATABASE … MATERIALIZE` / `DROP DATABASE … FORCE`: calls `wait_until_done()` -/// and blocks (with the Tokio `spawn_blocking` wrapper the DDL handler uses). +/// 2. `ALTER DATABASE … MATERIALIZE` / `DROP DATABASE … FORCE`: pass the handle +/// to the awaited `force_materialize`, which drives it. `wait_until_done()` +/// blocks its thread, so only a thread outside the async runtime calls it. #[derive(Debug)] pub struct CloneMaterializerHandle { db_id: DatabaseId, diff --git a/nodedb/src/control/maintenance/clone_materializer/reaper.rs b/nodedb/src/control/maintenance/clone_materializer/reaper.rs index 6e8df36ae..876d57322 100644 --- a/nodedb/src/control/maintenance/clone_materializer/reaper.rs +++ b/nodedb/src/control/maintenance/clone_materializer/reaper.rs @@ -1,24 +1,23 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Post-copy cleanup. Reaps `clone_copyups` and `clone_tombstones` (both -//! surrogate-keyed and KV-key-keyed) catalog rows for a collection that has -//! just been fully materialized, then flips the collection's `clone_status` -//! to `Materialized` and clears `cloned_from`. +//! Post-copy cleanup: flip a fully copied clone collection to `Materialized` +//! and clear `cloned_from`. +//! +//! The flip is one replicated `PutCollection`. Its apply drops each node's +//! copy-up and tombstone rows of the collection before it writes the row, so +//! every node reaps its own copy-on-write state at the same log position. //! //! Idempotent: calling on a collection already in `Materialized` is a no-op. use nodedb_types::{CloneStatus, DatabaseId}; use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::catalog::SystemCatalog; use crate::control::state::SharedState; /// Parameters for reaping a single fully-copied clone collection. pub struct ReapParams<'a> { - /// `db_qualified(target_db_id, name)` — same key the CoW write helpers - /// use when stamping copyups/tombstones (see `control::clone::tombstone`). - pub target_collection_qualified: &'a str, pub db_id: DatabaseId, pub tenant_id: u64, pub name: &'a str, @@ -26,17 +25,12 @@ pub struct ReapParams<'a> { pub catalog: &'a SystemCatalog, } -/// Reap CoW auxiliary rows and flip the collection to `Materialized`. +/// Flip the collection to `Materialized` and reap its CoW rows on every node. /// -/// Order matters for crash safety: aux rows are deleted first (idempotent). -/// A crash after aux delete but before status flip is recovered on the -/// next sweep — re-deletion is a no-op, and the status flip then proceeds. -/// A crash before aux delete leaves both the rows and the pre-flip status -/// in place — the next sweep redoes the whole step. The status-flip + clear -/// `cloned_from` is one Raft proposal, so it commits atomically. -pub fn reap_materialized_collection(params: ReapParams<'_>) -> crate::Result<()> { +/// A crash before the flip commits leaves the pre-flip status, and the next +/// sweep redoes the step. +pub async fn reap_materialized_collection(params: ReapParams<'_>) -> crate::Result<()> { let ReapParams { - target_collection_qualified, db_id, tenant_id, name, @@ -44,9 +38,6 @@ pub fn reap_materialized_collection(params: ReapParams<'_>) -> crate::Result<()> catalog, } = params; - catalog.delete_all_clone_copyups_for_collection(target_collection_qualified)?; - catalog.delete_all_clone_tombstones_for_collection(target_collection_qualified)?; - let Some(mut desc) = catalog.get_collection(db_id, tenant_id, name)? else { // Collection was concurrently dropped — nothing to reap. return Ok(()); @@ -62,11 +53,6 @@ pub fn reap_materialized_collection(params: ReapParams<'_>) -> crate::Result<()> // fast path skip the `cloned_from` lookup entirely. desc.cloned_from = None; - let outcome = - propose_catalog_entry(state, &CatalogEntry::PutCollection(Box::new(desc.clone())))?; - if outcome.needs_local_apply() { - catalog.put_collection(db_id, &desc)?; - } - + propose_catalog_entry_async(state, &CatalogEntry::PutCollection(Box::new(desc))).await?; Ok(()) } diff --git a/nodedb/src/control/maintenance/clone_materializer/source_drain.rs b/nodedb/src/control/maintenance/clone_materializer/source_drain.rs new file mode 100644 index 000000000..6277b0216 --- /dev/null +++ b/nodedb/src/control/maintenance/clone_materializer/source_drain.rs @@ -0,0 +1,366 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Cluster-wide write stop on a KV clone's source for the copy. +//! +//! KV keeps no row versions. An overwrite keeps the key's surrogate, and a +//! delete removes the row, so neither a surrogate ceiling nor an as-of bound +//! can hide a source write made after the clone. The copy is exact only if no +//! source write lands while it runs. +//! +//! The replicated descriptor drain gives that on every node: its start entry +//! applies everywhere, every node's planner then refuses a new plan on the +//! source collection, and the start returns only once no node holds a lease +//! on it, so no write admitted earlier is still in flight. The materializer's +//! own source scans run as pre-built gateway plans, which take no lease, so +//! the drain does not block them. +//! +//! The drain is owned by the clone collection +//! ([`nodedb_cluster::DrainOwner::CloneMaterialize`]) and covers every +//! descriptor version. A DDL on the source neither ends it nor escapes it by +//! bumping the version, and two copies from one source each hold their own. +//! +//! ## Claims and recovery +//! +//! A replicated claim row (`_system.clone_source_drains`) is written before +//! the drain starts and removed after it ends. A crash leaves the claim, and +//! [`recover_orphaned_source_drains`], run by the singleton worker before each +//! sweep, then settles it: a claim whose clone collection still needs its +//! copy stays, and the sweep re-drives that copy under the drain. Any other +//! claim ends its drain and is removed. The drain always ends before its claim +//! goes, so no crash leaves a drain with no claim to find it. + +use std::future::Future; + +use crate::control::catalog_entry::CatalogEntry; +use crate::control::clone::cow_entry::replicate_async; +use crate::control::lease::{drain_for_owner_async, end_drain_async, move_source_descriptor}; +use crate::control::metadata_proposer::DEFAULT_DRAIN_TIMEOUT; +use crate::control::security::catalog::StoredCollection; +use crate::control::security::catalog::clone_source_drains::CloneSourceDrain; +use crate::control::state::SharedState; + +/// Await `copy` with the source collection of `coll` drained cluster-wide. +/// +/// A collection whose source no longer exists has no writer to stop. +pub(super) async fn with_source_drain( + state: &SharedState, + coll: &StoredCollection, + copy: impl Future>, +) -> crate::Result { + let Some(origin) = &coll.cloned_from else { + return copy.await; + }; + if state + .credentials + .catalog() + .get_collection( + origin.source_database, + coll.tenant_id, + &origin.source_collection, + )? + .is_none() + { + return copy.await; + } + let claim = CloneSourceDrain { + clone_database: coll.database_id.as_u64(), + tenant_id: coll.tenant_id, + clone_collection: coll.name.clone(), + source_database: origin.source_database.as_u64(), + source_collection: origin.source_collection.clone(), + }; + replicate_async( + state, + &CatalogEntry::PutCloneSourceDrain(Box::new(claim.clone())), + ) + .await?; + if let Err(error) = drain_for_owner_async( + state, + source_descriptor(&claim), + drain_owner(&claim), + DEFAULT_DRAIN_TIMEOUT, + ) + .await + { + // `drain_for_owner_async` ended its own drain on failure. + release(state, &claim).await?; + return Err(error); + } + let copied = copy.await; + let released = release(state, &claim).await; + let value = copied?; + released?; + Ok(value) +} + +/// Settle every claim no running copy needs. Run by the singleton worker. +/// +/// A claim whose clone collection still delegates to its source stays: the +/// sweep re-drives that copy, which ends the drain when it finishes. Every +/// other claim ends its own drain and is removed. +pub(super) async fn recover_orphaned_source_drains(state: &SharedState) -> crate::Result<()> { + let claims = state.credentials.catalog().list_clone_source_drains()?; + for claim in claims { + let still_cloning = state + .credentials + .catalog() + .get_collection( + claim.clone_database_id(), + claim.tenant_id, + &claim.clone_collection, + )? + .is_some_and(|coll| coll.cloned_from.is_some()); + if !still_cloning { + release(state, &claim).await?; + } + } + Ok(()) +} + +/// End `claim`'s own drain, then remove the claim. Other owners' drains on +/// the source stay. The drain ends first: a crash in between leaves a claim +/// that recovery settles again. +async fn release(state: &SharedState, claim: &CloneSourceDrain) -> crate::Result<()> { + end_drain_async(state, source_descriptor(claim), drain_owner(claim)).await?; + replicate_async( + state, + &CatalogEntry::DeleteCloneSourceDrain { + clone_database: claim.clone_database, + tenant_id: claim.tenant_id, + clone_collection: claim.clone_collection.clone(), + }, + ) + .await +} + +fn drain_owner(claim: &CloneSourceDrain) -> nodedb_cluster::DrainOwner { + nodedb_cluster::DrainOwner::CloneMaterialize { + clone_database: claim.clone_database, + tenant_id: claim.tenant_id, + clone_collection: claim.clone_collection.clone(), + } +} + +fn source_descriptor(claim: &CloneSourceDrain) -> nodedb_cluster::DescriptorId { + move_source_descriptor( + claim.source_database, + claim.tenant_id, + &claim.source_collection, + ) +} + +#[cfg(test)] +mod tests { + use nodedb_types::{CloneOrigin, DatabaseId, Lsn}; + + use super::*; + + /// Write `coll` the way a committed `PutCollection` applies it: the + /// collection row plus its `StoredOwner` row. A later replicated entry's + /// apply checks catalog integrity, and a collection row without its + /// owner row fails that check. + fn put_collection(state: &SharedState, coll: &StoredCollection) { + crate::control::catalog_entry::apply::collection::put(coll, state.credentials.catalog()) + .unwrap(); + } + + /// The source is drained for exactly the copy, on success and on error. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn source_is_drained_for_the_copy_only() { + let cluster = crate::control::cluster::test_one_node::boot().await; + let state = &cluster.state; + + let source_db = DatabaseId::new(1024); + let mut source = StoredCollection::stamped_for_test(1, "kv_src", "admin"); + source.database_id = source_db; + source.descriptor_version = 3; + put_collection(state, &source); + let mut clone = StoredCollection::new(1, "kv_src", "admin"); + clone.database_id = DatabaseId::new(1025); + clone.cloned_from = Some(CloneOrigin { + source_database: source_db, + source_collection: "kv_src".into(), + as_of_lsn: Lsn::new(1), + clone_created_at: Lsn::new(2), + kv_surrogate_ceiling: None, + }); + let id = move_source_descriptor(source_db.as_u64(), 1, "kv_src"); + + let seen = with_source_drain(state, &clone, async { + Ok(state.lease_drain.is_draining(&id, 3)) + }) + .await + .unwrap(); + assert!(seen, "the source is drained while the copy runs"); + assert!( + !state.lease_drain.is_draining(&id, 3), + "the drain ends after" + ); + + let failed: crate::Result<()> = with_source_drain(state, &clone, async { + Err(crate::Error::Internal { + detail: "copy failed".into(), + }) + }) + .await; + assert!(failed.is_err()); + assert!( + !state.lease_drain.is_draining(&id, 3), + "a failed copy ends the drain too" + ); + assert!( + state + .credentials + .catalog() + .list_clone_source_drains() + .unwrap() + .is_empty(), + "every claim is released" + ); + cluster.shutdown().await; + } + + /// A drain left by a crashed copy whose clone collection is gone ends at + /// recovery. One whose clone still needs its copy stays for the sweep. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn recovery_ends_orphaned_drains_only() { + let cluster = crate::control::cluster::test_one_node::boot().await; + let state = &cluster.state; + let catalog = state.credentials.catalog(); + + let source_db = DatabaseId::new(1024); + let mut source = StoredCollection::stamped_for_test(1, "kv_src", "admin"); + source.database_id = source_db; + source.descriptor_version = 2; + put_collection(state, &source); + + // A live clone: its claim and drain must survive recovery. + let live_db = DatabaseId::new(1025); + let mut live = StoredCollection::stamped_for_test(1, "kv_src", "admin"); + live.database_id = live_db; + live.cloned_from = Some(CloneOrigin { + source_database: source_db, + source_collection: "kv_src".into(), + as_of_lsn: Lsn::new(1), + clone_created_at: Lsn::new(2), + kv_surrogate_ceiling: None, + }); + put_collection(state, &live); + + let claim = |clone_database: u64| CloneSourceDrain { + clone_database, + tenant_id: 1, + clone_collection: "kv_src".into(), + source_database: source_db.as_u64(), + source_collection: "kv_src".into(), + }; + let id = move_source_descriptor(source_db.as_u64(), 1, "kv_src"); + // Both copies crashed holding their drains. The dropped clone's claim + // shares the source with the live one. + for clone_database in [1025, 1026] { + let row = claim(clone_database); + drain_for_owner_async(state, id.clone(), drain_owner(&row), DEFAULT_DRAIN_TIMEOUT) + .await + .unwrap(); + catalog.put_clone_source_drain(&row).unwrap(); + } + assert_eq!(state.lease_drain.total_count(), 2); + + recover_orphaned_source_drains(state).await.unwrap(); + assert_eq!( + catalog.list_clone_source_drains().unwrap(), + vec![claim(1025)] + ); + assert_eq!( + state.lease_drain.snapshot()[0].1, + drain_owner(&claim(1025)), + "only the dropped clone's drain ends" + ); + assert!( + state.lease_drain.is_draining(&id, 2), + "the live clone's copy still needs the drain" + ); + + // The live clone finishes materializing; recovery now ends the drain. + live.cloned_from = None; + live.clone_status = nodedb_types::CloneStatus::Materialized; + put_collection(state, &live); + recover_orphaned_source_drains(state).await.unwrap(); + assert!(catalog.list_clone_source_drains().unwrap().is_empty()); + assert!(!state.lease_drain.is_draining(&id, 2)); + cluster.shutdown().await; + } + + /// A DDL on the source during the copy ends only its own drain. The source + /// stays drained, at the bumped version too, until the copy releases it. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn ddl_during_copy_leaves_source_drained_until_release() { + let cluster = crate::control::cluster::test_one_node::boot().await; + let state = &cluster.state; + + let source_db = DatabaseId::new(1024); + let mut source = StoredCollection::stamped_for_test(1, "kv_src", "admin"); + source.database_id = source_db; + source.descriptor_version = 3; + put_collection(state, &source); + let mut clone = StoredCollection::new(1, "kv_src", "admin"); + clone.database_id = DatabaseId::new(1025); + clone.cloned_from = Some(CloneOrigin { + source_database: source_db, + source_collection: "kv_src".into(), + as_of_lsn: Lsn::new(1), + clone_created_at: Lsn::new(2), + kv_surrogate_ceiling: None, + }); + let id = move_source_descriptor(source_db.as_u64(), 1, "kv_src"); + + with_source_drain(state, &clone, async { + // The DDL drains the source, then its bumped `PutCollection` + // applies and runs the implicit clear. + crate::control::lease::drain_for_ddl_async( + state, + id.clone(), + 3, + DEFAULT_DRAIN_TIMEOUT, + 0, + ) + .await?; + let mut altered = source.clone(); + altered.descriptor_version = 4; + crate::control::lease::clear_implicit_drains( + state, + &CatalogEntry::PutCollection(Box::new(altered)), + )?; + assert!( + state.lease_drain.is_draining(&id, 3), + "the old version stays drained" + ); + assert!( + state.lease_drain.is_draining(&id, 4), + "the bumped version is drained too" + ); + assert_eq!( + state.lease_drain.total_count(), + 1, + "only the DDL's drain ended" + ); + // A statement refused meanwhile is retryable and names the owner. + match crate::control::lease::ensure_not_draining(state, &id, 4) { + Err(error @ crate::Error::RetryableSchemaChanged { .. }) => assert!( + error.to_string().contains("clone materializing"), + "the refusal names the drain owner: {error}" + ), + other => panic!("expected a retryable drain refusal, got {other:?}"), + } + Ok(()) + }) + .await + .unwrap(); + assert!( + !state.lease_drain.is_draining(&id, 4), + "the copy's release ends it" + ); + assert_eq!(state.lease_drain.total_count(), 0); + cluster.shutdown().await; + } +} diff --git a/nodedb/src/control/maintenance/clone_materializer/status.rs b/nodedb/src/control/maintenance/clone_materializer/status.rs new file mode 100644 index 000000000..1a4d382da --- /dev/null +++ b/nodedb/src/control/maintenance/clone_materializer/status.rs @@ -0,0 +1,68 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Clone status writes and checks shared by every engine's row copy. + +use nodedb_types::{CloneStatus, Lsn}; + +use crate::control::catalog_entry::entry::CatalogEntry; +use crate::control::metadata_proposer::propose_catalog_entry_async; +use crate::control::security::catalog::StoredCollection; +use crate::control::state::SharedState; + +/// Flip the status to `Materializing` if still `Shadowed`, so concurrent +/// readers see in-progress state. A crash after this resumes from +/// `progress_lsn = 0`. +pub(super) async fn mark_materializing( + state: &SharedState, + coll: &StoredCollection, +) -> crate::Result<()> { + if !matches!(coll.clone_status, CloneStatus::Shadowed) { + return Ok(()); + } + let mut updated = coll.clone(); + updated.clone_status = CloneStatus::Materializing { + progress_lsn: Lsn::new(0), + bytes_done: 0, + bytes_total: 0, + }; + propose_catalog_entry_async(state, &CatalogEntry::PutCollection(Box::new(updated))).await?; + Ok(()) +} + +/// Persist a `Materializing { progress_lsn, .. }` checkpoint between scan pages. +pub(super) async fn checkpoint_progress( + state: &SharedState, + coll: &StoredCollection, + as_of_lsn: Lsn, + copied: u64, + total_seen: u64, +) -> crate::Result<()> { + let mut updated = coll.clone(); + updated.clone_status = CloneStatus::Materializing { + progress_lsn: as_of_lsn, + bytes_done: copied, + bytes_total: total_seen, + }; + propose_catalog_entry_async(state, &CatalogEntry::PutCollection(Box::new(updated))).await?; + Ok(()) +} + +/// Check that the home of `target_qualified` bound one surrogate per row. +/// +/// `zip` drops the rows past a short answer. Every row is copied under its own +/// bound surrogate, or the copy fails here. +pub(super) fn check_bound_surrogates( + target_qualified: &str, + bound: usize, + rows: usize, +) -> crate::Result<()> { + if bound == rows { + return Ok(()); + } + Err(crate::Error::Storage { + engine: "clone_materializer".into(), + detail: format!( + "the home of '{target_qualified}' answered {bound} surrogates for {rows} rows" + ), + }) +} diff --git a/nodedb/src/control/maintenance/clone_materializer/walker.rs b/nodedb/src/control/maintenance/clone_materializer/walker.rs index 8fc285854..7dc384c4e 100644 --- a/nodedb/src/control/maintenance/clone_materializer/walker.rs +++ b/nodedb/src/control/maintenance/clone_materializer/walker.rs @@ -14,19 +14,25 @@ //! - **Columnar / Timeseries / Spatial** — implemented (all three share the //! same columnar materializer path via `ColumnarOp::MaterializeScan`). //! -//! ## Sync wrapper +//! ## Cluster //! -//! The public API is sync because it is invoked from `spawn_blocking` on -//! both the DDL hot path and the maintenance scheduler. Internally we call -//! [`tokio::runtime::Handle::block_on`] so the per-engine async helpers can -//! use the SPSC bridge. +//! Source scans run on the source shard's owner, and target writes go +//! through the target shard's replicated write path, so every replica holds +//! the copied rows. Status flips are replicated `PutCollection` entries. The +//! scheduled sweep runs on one node cluster-wide. +//! +//! ## Async end to end +//! +//! Every entry point is async and awaits the per-engine copies, which +//! dispatch through the SPSC bridge. Nothing blocks a runtime thread on a +//! future, so the DDL handlers that await it and the maintenance sweep run +//! on any runtime flavor. -use std::collections::HashSet; use std::sync::atomic::{AtomicBool, Ordering}; use nodedb_types::{CloneStatus, CollectionType, DatabaseId}; -use crate::control::maintenance::wrapper::{MaintenanceOutcome, with_budget}; +use crate::control::maintenance::wrapper::{MaintenanceOutcome, with_budget_async}; use crate::control::security::catalog::{StoredCollection, SystemCatalog}; use crate::control::state::SharedState; @@ -35,15 +41,16 @@ use super::document::materialize_document_collection; use super::kv::materialize_kv_collection; use super::progress::CloneMaterializerHandle; use super::rls_gate::refuse_if_rls_policy_applies; +use super::source_drain::with_source_drain; /// Result of a single `materialize_database` call. #[derive(Debug)] pub enum MaterializeOutcome { /// Every clone collection in the database is `Materialized`. AllComplete, - /// `n` collections still need work and the caller should reschedule. + /// `n` collections still need work and the caller reschedules. Incomplete { collections_remaining: usize }, - /// The maintenance budget was exhausted before any work could be done. + /// The maintenance budget was exhausted before any work was done. BudgetDeferred, /// Cooperative shutdown signal received. Cancelled, @@ -66,17 +73,20 @@ pub struct MaterializeParams<'a> { } /// Drive one materialization sweep for `db_id`. -pub fn materialize_database(params: MaterializeParams<'_>) -> crate::Result { +pub async fn materialize_database( + params: MaterializeParams<'_>, +) -> crate::Result { if params.cancel.load(Ordering::Relaxed) { return Ok(MaterializeOutcome::Cancelled); } - let outcome = with_budget( + let outcome = with_budget_async( ¶ms.state.maintenance_budget, params.db_id, params.estimated_secs, - || do_materialize_database(¶ms), - ); + do_materialize_database(¶ms), + ) + .await; match outcome { MaintenanceOutcome::Deferred => Ok(MaterializeOutcome::BudgetDeferred), @@ -85,7 +95,13 @@ pub fn materialize_database(params: MaterializeParams<'_>) -> crate::Result) -> crate::Result { +/// +/// Document and columnar sources are read as of the clone point through +/// `system_as_of_ms`, and a KV source is drained cluster-wide in +/// `materialize_one`. +async fn do_materialize_database( + params: &MaterializeParams<'_>, +) -> crate::Result { if params.cancel.load(Ordering::Relaxed) { return Ok(MaterializeOutcome::Cancelled); } @@ -100,35 +116,12 @@ fn do_materialize_database(params: &MaterializeParams<'_>) -> crate::Result = pending - .iter() - .filter_map(|c| c.cloned_from.as_ref().map(|o| o.source_database)) - .collect(); - let _freeze_guards: Vec = source_db_ids - .iter() - .map(|db_id| params.state.materialize_freeze.freeze(*db_id)) - .collect(); - - let runtime_handle = tokio::runtime::Handle::try_current().map_err(|_| { - // Materializer must run inside a Tokio runtime so SPSC dispatch - // futures can drive. The DDL hot path runs sync handlers on runtime - // worker threads; the background sweep runs sync handlers on - // `spawn_blocking` threads. Both share the runtime. - crate::Error::Dispatch { - detail: "clone materializer requires a Tokio runtime context".into(), - } - })?; - let mut remaining = 0usize; for coll in &pending { if params.cancel.load(Ordering::Relaxed) { return Ok(MaterializeOutcome::Cancelled); } - match materialize_one(&runtime_handle, params, coll) { + match materialize_one(params, coll).await { Ok(()) => { if let Some(h) = params.handle { h.notify_collection_done(); @@ -181,62 +174,45 @@ fn pending_clone_collections( .collect()) } -/// Route one collection to its per-engine materializer. -/// -/// The per-engine implementations are async (they dispatch through the SPSC -/// bridge). We bridge sync↔async with [`tokio::task::block_in_place`] so the -/// runtime worker stays usable while we drive the future on this thread. This -/// requires a multi-threaded runtime — the production server uses -/// `#[tokio::main]` (multi-thread by default) and tests must annotate with -/// `#[tokio::test(flavor = "multi_thread")]`. +/// Route one collection to its per-engine materializer and await it. /// /// Every engine's materialization flows through this function, so the RLS /// policy-existence gate is checked once here, before any scan or write plan /// is built for any of the four write sites (KV `Put`, Document /// `PointInsert`, Columnar `Insert`, Timeseries `Ingest`) or their matching /// source-side `MaterializeScan`s. -fn materialize_one( - runtime: &tokio::runtime::Handle, +async fn materialize_one( params: &MaterializeParams<'_>, coll: &StoredCollection, ) -> crate::Result<()> { refuse_if_rls_policy_applies(params.state, params.db_id, coll)?; match &coll.collection_type { - CollectionType::KeyValue(_) => tokio::task::block_in_place(|| { - runtime.block_on(materialize_kv_collection( + // KV keeps no row versions, so its source takes no write for the copy. + CollectionType::KeyValue(_) => { + with_source_drain( params.state, - params.catalog, - params.db_id, coll, - )) - }), - CollectionType::Document(_) => tokio::task::block_in_place(|| { - runtime.block_on(materialize_document_collection( - params.state, - params.catalog, - params.db_id, - coll, - )) - }), - CollectionType::Columnar(_) => tokio::task::block_in_place(|| { - runtime.block_on(materialize_columnar_collection( - params.state, - params.catalog, - params.db_id, - coll, - )) - }), + materialize_kv_collection(params.state, params.catalog, params.db_id, coll), + ) + .await + } + CollectionType::Document(_) => { + materialize_document_collection(params.state, params.catalog, params.db_id, coll).await + } + CollectionType::Columnar(_) => { + materialize_columnar_collection(params.state, params.catalog, params.db_id, coll).await + } } } -/// Drive materialization to completion synchronously. +/// Drive materialization of `db_id` to completion. /// /// Used by `ALTER DATABASE … MATERIALIZE` and `DROP DATABASE … FORCE`. Returns /// `Err(Error::BadRequest)` for unsupported engines (mapped to SQLSTATE /// `0A000` by the DDL handlers); returns `Ok(())` on success or after the /// budget defers. -pub fn force_materialize_blocking( +pub async fn force_materialize( db_id: DatabaseId, state: &SharedState, catalog: &SystemCatalog, @@ -252,7 +228,7 @@ pub fn force_materialize_blocking( estimated_secs: 0.0, }; - match do_materialize_database(¶ms)? { + match do_materialize_database(¶ms).await? { MaterializeOutcome::AllComplete => Ok(()), MaterializeOutcome::Incomplete { collections_remaining, @@ -270,11 +246,19 @@ pub fn force_materialize_blocking( } /// Entry point called by the maintenance scheduler on each tick. -pub fn run_scheduled_sweep( +pub async fn run_scheduled_sweep( state: &SharedState, catalog: &SystemCatalog, cancel: &AtomicBool, ) -> crate::Result<()> { + // Materializer writes route to each shard's owner and replicate, so one + // node sweeps for the whole cluster. + if !state.is_singleton_worker() { + return Ok(()); + } + // A copy that crashed on any node, or whose clone went away, leaves its + // source drain to this node. Settled before the sweep re-drives copies. + super::source_drain::recover_orphaned_source_drains(state).await?; let database_ids: Vec = catalog .list_databases()? .into_iter() @@ -295,7 +279,7 @@ pub fn run_scheduled_sweep( estimated_secs: 5.0, }; - match materialize_database(params) { + match materialize_database(params).await { Ok(MaterializeOutcome::AllComplete) => {} Ok(MaterializeOutcome::Incomplete { collections_remaining, diff --git a/nodedb/src/control/maintenance/wrapper.rs b/nodedb/src/control/maintenance/wrapper.rs index ce17c5ded..3b087e5a2 100644 --- a/nodedb/src/control/maintenance/wrapper.rs +++ b/nodedb/src/control/maintenance/wrapper.rs @@ -63,10 +63,37 @@ where } } +/// [`with_budget`] for async work: awaits `work` only if `db` has CPU budget +/// remaining. The lease covers the whole await. +pub async fn with_budget_async( + tracker: &Arc, + db: DatabaseId, + estimated_secs: f64, + work: Fut, +) -> MaintenanceOutcome +where + Fut: std::future::Future, +{ + match tracker.try_acquire(db, estimated_secs) { + None => MaintenanceOutcome::Deferred, + Some(_lease) => MaintenanceOutcome::Ran(work.await), + } +} + #[cfg(test)] mod tests { use super::*; + #[tokio::test(flavor = "current_thread")] + async fn with_budget_async_awaits_within_cap() { + let tracker = Arc::new(MaintenanceBudgetTracker::new()); + let db = DatabaseId::new(3); + tracker.set_cap(db, 25); + + let outcome = with_budget_async(&tracker, db, 1.0, async { 7u32 }).await; + assert!(matches!(outcome, MaintenanceOutcome::Ran(7))); + } + #[test] fn with_budget_runs_within_cap() { let tracker = Arc::new(MaintenanceBudgetTracker::new()); @@ -101,8 +128,8 @@ mod tests { } } - // At this point the budget may be near exhaustion. The exact behavior - // depends on timing; we just verify the API compiles and returns a + // At this point the budget can be near exhaustion. The exact behavior + // depends on timing; we only verify the API compiles and returns a // valid variant. let outcome = with_budget(&tracker, db, 0.0, || 99u32); let _ = outcome.ran() || outcome.deferred(); diff --git a/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs b/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs index c83443852..820370b45 100644 --- a/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs +++ b/nodedb/src/control/merge_orchestrator/expand_staged_merge.rs @@ -27,8 +27,8 @@ use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use super::resolve_arms::{ResolvedMergeArms, decode_resolve}; use crate::control::target_identity::{ - TargetPk, assign_target_surrogate, bare_collection_name, derive_document_id, require_surrogate, - resolve_target_pk, + TargetPk, assign_target_surrogates, bare_collection_name, derive_document_id, + require_surrogate, resolve_target_pk, }; /// Resolve one in-transaction `DocumentOp::Merge` task into the concrete, @@ -63,7 +63,7 @@ pub(crate) async fn resolve_and_emit_merge_ops( // Gate every resolved arm on the target's write policy (post-image for // UPDATE/INSERT, pre-image for DELETE): this expansion rewrites the // statement past RLS injection, so without this check a governed MERGE - // would launder into ungoverned point writes. + // launders into ungoverned point writes. if let nodedb_types::WriteGateDecision::Evaluate(predicate) = rls_write_check.decision() { let bodies = arms .updates @@ -105,7 +105,8 @@ pub(crate) async fn resolve_and_emit_merge_ops( vshard_id, arms, &mut out, - )?; + ) + .await?; Ok(out) } @@ -189,9 +190,9 @@ async fn resolve_merge_arms( /// Rewrite the three resolved arms into concrete point-write tasks appended /// to `out`. An UPDATE/DELETE arm with no registered surrogate is a hard -/// error — emitting a degraded raw op would reproduce the indexing / +/// error — emitting a degraded raw op reproduces the indexing / /// durability defect this expansion fixes. -fn emit_arms( +async fn emit_arms( state: &SharedState, task: &PhysicalTask, target_collection: &str, @@ -200,14 +201,21 @@ fn emit_arms( arms: ResolvedMergeArms, out: &mut Vec, ) -> crate::Result<()> { - for (_join_key, body) in arms.inserts { - let surrogate = assign_target_surrogate( - state, - nodedb_types::CollectionKey::from_qualified_str(task.database_id, target_collection)?, - task.tenant_id, - target_pk, - &body, - )?; + // Every inserted row's surrogate in one batch at the target's home. + let insert_bodies: Vec<&[u8]> = arms + .inserts + .iter() + .map(|(_, body)| body.as_slice()) + .collect(); + let insert_surrogates = assign_target_surrogates( + state, + nodedb_types::CollectionKey::from_qualified_str(task.database_id, target_collection)?, + task.tenant_id, + target_pk, + &insert_bodies, + ) + .await?; + for ((_join_key, body), surrogate) in arms.inserts.into_iter().zip(insert_surrogates) { let document_id = derive_document_id(target_pk, &body, surrogate); out.push(point_task( task, @@ -267,13 +275,13 @@ fn emit_arms( target_collection.to_string(), ), document_id, - surrogate, + surrogate: Some(surrogate), pk_bytes, returning: None, rls_filters: Vec::new(), // Already decided against the merge's write predicate by // `admit_compiled_write_image` above; this op removes that - // same row, so re-checking would re-run the same test. + // same row, so re-checking re-runs the same test. rls_write_check: nodedb_types::RlsWriteCheck::decided_earlier_in_request(), resolved_sum_targets: Vec::new(), }), diff --git a/nodedb/src/control/merge_orchestrator/orchestrator.rs b/nodedb/src/control/merge_orchestrator/orchestrator.rs index 1d8cb442f..4ff1ed673 100644 --- a/nodedb/src/control/merge_orchestrator/orchestrator.rs +++ b/nodedb/src/control/merge_orchestrator/orchestrator.rs @@ -5,7 +5,7 @@ //! A NOT-MATCHED insert row needs its OWN registered surrogate, and surrogate //! registration is Control-Plane-only, so autocommit MERGE runs as a //! TOCTOU-safe round trip: (0) ship the source rows (scanned on its own -//! core, since it may differ from the target's) into `source_rows`; (1) +//! core, since it can differ from the target's) into `source_rows`; (1) //! resolve — the Data Plane classifies the merge read-only and returns //! NOT-MATCHED rows; (2) assign a fresh registered surrogate per insert row //! and decide the target's write policy over every resolved arm; (3) apply — @@ -33,7 +33,7 @@ use nodedb_physical::physical_plan::{DocumentOp, ReturningSpec}; use super::resolve_arms::decode_resolve; use crate::control::planner::materialized_sum::resolve_sum_targets_for_bodies; use crate::control::target_identity::{ - assign_target_surrogate, bare_collection_name, derive_document_id, resolve_target_pk, + assign_target_surrogates, bare_collection_name, derive_document_id, resolve_target_pk, }; /// Upper bound on resolve→apply retries under concurrent source/target drift. @@ -57,7 +57,7 @@ pub struct MergeArgs<'a> { /// pre-processor. `None` selects the affected-count response. pub returning: Option<&'a ReturningSpec>, /// RLS read filters, carried onto the apply pass so `RETURNING` rows are - /// gated as a `SELECT` by the same principal would be. + /// gated as a `SELECT` by the same principal is. pub rls_filters: &'a [u8], /// RLS write predicate, carried onto the apply pass which decides every /// arm's image against it. Separate from `rls_filters`: read vs write gate. @@ -133,7 +133,7 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate let mut attempt: u32 = 0; loop { - // Phase 0: read SOURCE on its own core (may differ from target's) + // Phase 0: read SOURCE on its own core (can differ from target's) // and ship raw rows into the plan. A fresh read per attempt keeps // resolve/apply on one consistent snapshot. let source_rows = read_all_source_rows( @@ -166,7 +166,7 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate } let arms = decode_resolve(&resolve_resp.payload)?; - // Phase 2a: resolve materialized-sum targets from the arms just + // Phase 2a: resolve materialized-sum targets from the arms // classified (INSERT credits, DELETE debits, UPDATE the difference, // both sides on a join-key rewrite). Lookup-only: an unmatched join // value fails the statement. Drift is caught by apply's own @@ -227,17 +227,22 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate by_join_key: Vec::with_capacity(insert_rows.len()), identities: Vec::with_capacity(insert_rows.len()), }; - for (join_key, body) in &insert_rows { - let surrogate = assign_target_surrogate( - state, - nodedb_types::CollectionKey::from_qualified_str( - args.database_id, - args.target_collection, - )?, - args.tenant_id, - &target_pk, - body, - )?; + let insert_bodies: Vec<&[u8]> = insert_rows + .iter() + .map(|(_, body)| body.as_slice()) + .collect(); + let insert_surrogates = assign_target_surrogates( + state, + nodedb_types::CollectionKey::from_qualified_str( + args.database_id, + args.target_collection, + )?, + args.tenant_id, + &target_pk, + &insert_bodies, + ) + .await?; + for ((join_key, body), surrogate) in insert_rows.iter().zip(insert_surrogates) { let document_id = derive_document_id(&target_pk, body, surrogate); inserts .by_join_key @@ -273,7 +278,7 @@ pub(crate) async fn run_merge(state: &SharedState, args: MergeArgs<'_>) -> crate }); } // Concurrent drift: re-resolve (fresh phase 1) and retry. The - // surrogates assigned this round are simply unused (harmless — + // surrogates assigned this round are unused (harmless — // the counter is monotonic and gap-tolerant). continue; } diff --git a/nodedb/src/control/metadata_proposer/catalog.rs b/nodedb/src/control/metadata_proposer/catalog.rs index 8f2d7d10d..f964a878a 100644 --- a/nodedb/src/control/metadata_proposer/catalog.rs +++ b/nodedb/src/control/metadata_proposer/catalog.rs @@ -12,52 +12,49 @@ use crate::control::propose_outcome::ProposeOutcome; use crate::control::state::SharedState; use crate::error::Error; -use super::ddl_prepare::{acquire_ddl_prepare_lease, lock_ddl_preparation}; +use super::ddl_prepare::{acquire_ddl_prepare_lease_async, lock_ddl_preparation_async}; +use super::handle::MetadataRaftHandle; use super::timeouts::{DEFAULT_DRAIN_TIMEOUT, DEFAULT_PROPOSE_TIMEOUT}; +use super::wait::wait_tracking_progress_async; -/// Propose a `CatalogEntry` and block until the local applied-index -/// watcher confirms the entry has been applied on this node. +/// Propose a `CatalogEntry` and wait until this node applied it, on any +/// runtime flavor. /// -/// The returned [`ProposeOutcome`] tells the caller whether to write the -/// catalog itself, leave it to the applier, or do nothing because the entry -/// is held for COMMIT. -pub fn propose_catalog_entry( - shared: &SharedState, - entry: &CatalogEntry, -) -> Result { - propose_catalog_entry_with_timeout(shared, entry, DEFAULT_PROPOSE_TIMEOUT) -} - -/// Same as [`propose_catalog_entry`] but with an explicit timeout. +/// The returned [`ProposeOutcome`] tells the caller whether the entry applied +/// here through the metadata group, or is held for COMMIT. The caller never +/// writes the catalog itself. +/// +/// Every wait is awaited: the DDL preparation lock and lease, the descriptor +/// drain, the metadata commit's apply, and the authorization barrier. /// /// An entry that changes authorization state returns only once it binds /// every node: the authorization barrier runs after the local apply, with the -/// DDL preparation lock already released. -pub fn propose_catalog_entry_with_timeout( +/// DDL preparation lock and lease already released. +pub async fn propose_catalog_entry_async( shared: &SharedState, entry: &CatalogEntry, - timeout: Duration, ) -> Result { - let outcome = propose_and_apply_locally(shared, entry, timeout)?; + let outcome = propose_and_apply_locally(shared, entry).await?; if let ProposeOutcome::Replicated { log_index } = outcome && entry.bears_authorization() { - crate::control::security::auth_lease::block_on_barrier( + crate::control::security::auth_lease::authorization_barrier( shared, vec![nodedb_cluster::GroupCoverage { group_id: METADATA_GROUP_ID, through: log_index, }], - )?; + ) + .await?; } Ok(outcome) } -/// Propose `entry` and wait until this node applied it. -fn propose_and_apply_locally( +/// Propose `entry` and wait until this node applied it. The preparation lock +/// and lease are released before it returns. +async fn propose_and_apply_locally( shared: &SharedState, entry: &CatalogEntry, - timeout: Duration, ) -> Result { // Buffering is decided first, ahead of every replication-mode gate: an open // transaction owns the entry regardless of whether this deployment @@ -68,95 +65,135 @@ fn propose_and_apply_locally( return Ok(ProposeOutcome::Buffered); } - let Some(handle) = shared.metadata_raft.get() else { - return Ok(ProposeOutcome::LocalOnly); - }; - - // Rolling-upgrade gate: until every node in the cluster reports - // at least `DISTRIBUTED_CATALOG_VERSION`, fall back to the legacy - // direct-write path on the originating node. Mixing the - // replicated and direct paths during a partial upgrade would - // diverge catalog state across nodes — see - // `control/rolling_upgrade.rs`. - if !shared - .cluster_version_view() - .can_activate_feature(crate::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION) - { - tracing::warn!( - min_version = shared.cluster_version_view().min_version, - required = crate::control::rolling_upgrade::DISTRIBUTED_CATALOG_VERSION, - "metadata propose: cluster in compat mode (mixed-version), \ - falling back to legacy direct-write path" - ); - return Ok(ProposeOutcome::LocalOnly); - } + let handle = shared.metadata_raft_handle()?; // Serialize preparation through local apply confirmation. Without this, // concurrent proposers can both observe persisted version N and emit N+1. - let _local_ddl_guard = lock_ddl_preparation(shared)?; - let distributed_ddl_guard = acquire_ddl_prepare_lease(shared, handle.as_ref())?; - - // Drain for Put* variants that carry descriptor_version. - // Leases acquired at plan time are refcounted and held - // through execute; when the last in-flight query using a - // descriptor completes, its `QueryLeaseScope` drops and the - // refcount hits zero, releasing the lease. Drain is what - // makes this an actual barrier: the proposer waits for all - // prior-version leases to release before committing the new - // `Put*`, giving long-running in-flight queries a bounded - // window (DEFAULT_DRAIN_TIMEOUT) to finish. - if let Some((descriptor_id, prior_version)) = - crate::control::lease::descriptor_id_and_prior_version(entry, shared) - && prior_version > 0 - { - crate::control::lease::drain_for_ddl( + let _local_ddl_guard = lock_ddl_preparation_async(shared).await; + + let lease = acquire_ddl_prepare_lease_async(shared, handle.as_ref()).await?; + let proposed = async { + let drained = drain_prior_version(shared, entry).await?; + let stamped = stamp_or_end_drain(shared, entry, drained).await?; + propose_prepared( shared, - descriptor_id, - prior_version, - DEFAULT_DRAIN_TIMEOUT, - // No transactional lease scope of its own: this is a bare, - // unbuffered DDL statement, not a COMMIT finalizing buffered DDL - // alongside a buffered write to the same descriptor. - 0, - )?; + handle.as_ref(), + lease.token(), + catalog_ddl_entry(&stamped)?, + DEFAULT_PROPOSE_TIMEOUT, + ) + .await } + .await; + lease.release().await; + Ok(ProposeOutcome::Replicated { + log_index: proposed?, + }) +} + +/// Drain the prior version of the descriptor `entry` changes. Returns the +/// drained descriptor, whose drain the entry's apply ends. +/// +/// Leases acquired at plan time are refcounted and held through execute; when +/// the last in-flight query using a descriptor completes, its +/// `QueryLeaseScope` drops and the refcount hits zero, releasing the lease. +/// The drain is the barrier: the proposer waits for every prior-version lease +/// to release before committing the new `Put*`, giving long-running in-flight +/// queries a bounded window (`DEFAULT_DRAIN_TIMEOUT`) to finish. +async fn drain_prior_version( + shared: &SharedState, + entry: &CatalogEntry, +) -> Result, Error> { + let Some((descriptor_id, prior_version)) = + crate::control::lease::descriptor_id_and_prior_version(entry, shared) + else { + return Ok(None); + }; + if prior_version == 0 { + return Ok(None); + } + crate::control::lease::drain_for_ddl_async( + shared, + descriptor_id.clone(), + prior_version, + DEFAULT_DRAIN_TIMEOUT, + // No transactional lease scope of its own: this is a bare, + // unbuffered DDL statement, not a COMMIT finalizing buffered DDL + // alongside a buffered write to the same descriptor. + 0, + ) + .await?; + Ok(Some(descriptor_id)) +} - // Freeze the descriptor_version / constraint_version / - // modification_hlc HERE, at propose time, so the value is computed - // exactly once from this node's local catalog (`prior + 1`) and - // then replicated verbatim inside the entry. Every node applies the - // frozen value without re-deriving it, which makes replay-from-log - // on restart and re-delivery during learner catch-up idempotent — - // the divergence that a per-node apply-time stamp produced is gone. - // - // Gated on the same rolling-upgrade flag the apply path used to - // gate on: only stamp once every node can activate descriptor - // versioning; otherwise leave the entry's sentinel version `0` - // (downstream resolvers treat `0` as `1`). Older nodes in a - // mixed-version cluster lack the stamp logic, so a stamped value - // would not be reproduced symmetrically there. - let stamped_owned; - let entry: &CatalogEntry = if shared - .cluster_version_view() - .can_activate_feature(crate::control::rolling_upgrade::DESCRIPTOR_VERSIONING_VERSION) +/// Stamp `entry`. A failed stamp ends the drain of `drained` and returns the +/// stamp error. +/// +/// Freezes the descriptor_version / constraint_version / modification_hlc +/// HERE, at propose time, so the value is computed exactly once from this +/// node's local catalog (`prior + 1`) and then replicated verbatim inside the +/// entry. Every node applies the frozen value without re-deriving it, which +/// makes replay-from-log on restart and re-delivery during learner catch-up +/// idempotent. +async fn stamp_or_end_drain( + shared: &SharedState, + entry: &CatalogEntry, + drained: Option, +) -> Result { + match catalog_entry::descriptor_stamp::stamp( + entry.clone(), + &shared.hlc_clock, + shared.credentials.catalog(), + ) { + Ok(stamped) => Ok(stamped), + Err(error) => Err(end_unproposed_drain(shared, drained, error).await), + } +} + +/// End the drain a DDL started when it fails before its entry is proposed. +/// A drain has no wall-clock expiry, and no apply of this entry will end it. +/// Returns `error`, the failure that stopped the DDL. +async fn end_unproposed_drain( + shared: &SharedState, + drained: Option, + error: Error, +) -> Error { + if let Some(descriptor_id) = drained + && let Err(end) = crate::control::lease::end_drain_async( + shared, + descriptor_id, + nodedb_cluster::DrainOwner::Ddl, + ) + .await { - stamped_owned = catalog_entry::descriptor_stamp::stamp( - entry.clone(), - &shared.hlc_clock, - shared.credentials.catalog(), + tracing::warn!( + error = %end, + "metadata propose: the drain of a DDL that failed before its propose did not end" ); - &stamped_owned - } else { - entry - }; + } + error +} - let payload = catalog_entry::encode(entry)?; +/// Wrap `entry` as a `CatalogDdl` metadata entry. +/// +/// Carries the statement's audit context when the statement boundary +/// installed one. Internal callers (descriptor lease grant/release, drain +/// proposer) run outside that scope and emit the plain `CatalogDdl` variant. +pub(super) fn catalog_ddl_entry(entry: &CatalogEntry) -> Result { + catalog_ddl_entry_with( + entry, + crate::control::server::shared::session::audit_context::current(), + ) +} - // Attach J.4 audit context when the pgwire statement boundary - // installed one. Internal callers (descriptor lease grant/release, - // drain proposer) run outside that scope and emit the plain - // `CatalogDdl` variant — they have no SQL text to log. - let catalog_entry = match crate::control::server::shared::session::audit_context::current() { +/// Wrap `entry` as a `CatalogDdl` metadata entry carrying `audit`, the +/// context of the statement that issued it. +pub(super) fn catalog_ddl_entry_with( + entry: &CatalogEntry, + audit: Option, +) -> Result { + let payload = catalog_entry::encode(entry)?; + Ok(match audit { Some(ctx) => MetadataEntry::CatalogDdlAudited { payload, auth_user_id: ctx.auth_user_id, @@ -164,40 +201,49 @@ fn propose_and_apply_locally( sql_text: ctx.sql_text, }, None => MetadataEntry::CatalogDdl { payload }, - }; - let metadata_entry = MetadataEntry::DdlPrepared { - token: distributed_ddl_guard.token(), - entry: Box::new(catalog_entry), - }; - let raw = encode_entry(&metadata_entry).map_err(|e| Error::Config { + }) +} + +/// Propose `entry` under the preparation lease `token` and wait until this +/// node applied it. Returns the log index. Fails when another lease owner +/// superseded `token` before the apply. +/// +/// `timeout` bounds a stall, not the whole wait, as +/// [`super::wait::wait_tracking_progress_async`] describes. +pub(super) async fn propose_prepared( + shared: &SharedState, + handle: &dyn MetadataRaftHandle, + token: u64, + entry: MetadataEntry, + timeout: Duration, +) -> Result { + let raw = encode_entry(&MetadataEntry::DdlPrepared { + token, + entry: Box::new(entry), + }) + .map_err(|e| Error::Config { detail: format!("metadata entry encode: {e}"), })?; - - let log_index = handle.propose(raw)?; - + let log_index = handle.propose_async(raw).await?; let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - // `wait_for` blocks the calling thread on a Condvar. When the - // caller is already inside a tokio task (pgwire handlers always - // are), parking the worker without telling tokio starves every - // other task that lands on it — including the raft tick that - // would otherwise bump the watcher. Wrap the blocking section - // in `block_in_place` so tokio reassigns a fresh worker. - let outcome = tokio::task::block_in_place(|| watcher.wait_for(log_index, timeout)); + let outcome = + wait_tracking_progress_async(std::sync::Arc::clone(&watcher), log_index, timeout, || { + shared.metadata_apply_progress.load(Ordering::Acquire) + }) + .await?; match outcome { WaitOutcome::Reached - if shared.metadata_ddl_applied_token.load(Ordering::Acquire) - == distributed_ddl_guard.token() => + if shared.metadata_ddl_applied_token.load(Ordering::Acquire) == token => { - Ok(ProposeOutcome::Replicated { log_index }) + Ok(log_index) } WaitOutcome::Reached => Err(Error::Config { detail: "metadata DDL preparation ownership was superseded before apply".into(), }), WaitOutcome::TimedOut => Err(Error::Config { detail: format!( - "metadata propose timed out after {:?} waiting for log index {} (current: {})", - timeout, - log_index, + "metadata propose timed out after {timeout:?} without apply progress waiting \ + for log index {log_index} (current: {})", watcher.current() ), }), @@ -206,3 +252,87 @@ fn propose_and_apply_locally( }), } } + +#[cfg(test)] +mod tests { + use std::sync::Arc; + use std::sync::atomic::AtomicU64; + + use nodedb_cluster::AppliedIndexWatcher; + + use super::*; + + const WINDOW: Duration = Duration::from_millis(60); + + /// Progress that keeps moving for many windows never times out; the wait + /// ends when the entry applies. + #[tokio::test] + async fn steady_progress_never_times_out() { + let watcher = Arc::new(AppliedIndexWatcher::new()); + let progress = Arc::new(AtomicU64::new(0)); + let driver = { + let (watcher, progress) = (Arc::clone(&watcher), Arc::clone(&progress)); + std::thread::spawn(move || { + for _ in 0..20 { + std::thread::sleep(Duration::from_millis(20)); + progress.fetch_add(1, Ordering::Release); + } + watcher.bump(5); + }) + }; + let outcome = wait_tracking_progress_async(Arc::clone(&watcher), 5, WINDOW, || { + progress.load(Ordering::Acquire) + }) + .await + .expect("the wait finishes"); + driver.join().unwrap(); + assert!(outcome.is_reached(), "{outcome:?}"); + } + + /// A stall of one window times the wait out. + #[tokio::test] + async fn a_stall_times_out() { + let watcher = Arc::new(AppliedIndexWatcher::new()); + let started = std::time::Instant::now(); + let outcome = wait_tracking_progress_async(watcher, 5, WINDOW, || 7) + .await + .expect("the wait finishes"); + assert!(matches!(outcome, WaitOutcome::TimedOut), "{outcome:?}"); + assert!(started.elapsed() < WINDOW * 4); + } + + /// A one-node cluster stamps a create, commits it through its metadata + /// group, and applies it before the propose returns: the row holds its + /// first descriptor version and a non-zero incarnation. + #[tokio::test(flavor = "multi_thread")] + async fn a_one_node_cluster_stamps_and_applies_through_its_metadata_group() { + let cluster = crate::control::cluster::test_one_node::boot().await; + + let outcome = propose_catalog_entry_async(&cluster.state, &orders()) + .await + .expect("propose"); + assert!(outcome.is_replicated(), "{outcome:?}"); + assert_orders_applied(&cluster.state); + cluster.shutdown().await; + } + + fn orders() -> CatalogEntry { + use crate::control::security::catalog::StoredCollection; + CatalogEntry::PutCollection(Box::new(StoredCollection::new(7, "orders", "admin"))) + } + + /// The create of [`orders`] landed with its first descriptor version and a + /// non-zero incarnation. + fn assert_orders_applied(state: &SharedState) { + use nodedb_types::{DatabaseId, Hlc}; + let row = state + .credentials + .catalog() + .get_committed_collection(DatabaseId::DEFAULT, 7, "orders") + .expect("read") + .expect("the propose applied the row"); + assert_eq!(row.descriptor_version, 1); + assert_ne!(row.incarnation, Hlc::ZERO); + assert_ne!(row.modification_hlc, Hlc::ZERO); + } +} diff --git a/nodedb/src/control/metadata_proposer/catalog_batch.rs b/nodedb/src/control/metadata_proposer/catalog_batch.rs new file mode 100644 index 000000000..4e2dc3122 --- /dev/null +++ b/nodedb/src/control/metadata_proposer/catalog_batch.rs @@ -0,0 +1,209 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Propose several catalog entries as one metadata commit. + +use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry}; + +use crate::control::catalog_entry::{self, CatalogEntry}; +use crate::control::security::catalog::SystemCatalog; +use crate::control::server::shared::session::ddl_buffer; +use crate::control::state::SharedState; +use crate::error::Error; + +use super::catalog::{catalog_ddl_entry, propose_prepared}; +use super::ddl_prepare::{acquire_ddl_prepare_lease_async, lock_ddl_preparation_async}; +use super::handle::MetadataRaftHandle; +use super::timeouts::{DEFAULT_DRAIN_TIMEOUT, DEFAULT_PROPOSE_TIMEOUT}; + +/// What happened to the entries a batch plan built. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BatchOutcome { + /// Replicated through the metadata group and applied here at this index. + Replicated { log_index: u64 }, + /// Every entry joined the open DDL transaction. + Buffered, + /// The plan built no entries: nothing was proposed, buffered, or applied. + Empty, +} + +/// The entries a batch plan built, and what happened to them. +pub struct CatalogBatch { + pub outcome: BatchOutcome, + /// Stamped as applied on the `Replicated` path. + pub entries: Vec, +} + +impl CatalogBatch { + fn empty() -> Self { + Self { + outcome: BatchOutcome::Empty, + entries: Vec::new(), + } + } +} + +/// Build entries with `plan` and propose them as one metadata commit, on any +/// runtime flavor. +/// +/// `plan` runs under the DDL preparation lock and lease, so no other DDL +/// changes the catalog between the plan and the commit. The batch applies at +/// one log index on every node, entry by entry in plan order, and a restart +/// replays it as one unit. +/// +/// Every wait is awaited: the preparation lock and lease, the descriptor +/// drains, the commit's apply, and the authorization barrier. +pub async fn propose_catalog_batch_async( + shared: &SharedState, + plan: impl FnOnce(&SystemCatalog) -> crate::Result>, +) -> Result { + let catalog = shared.credentials.catalog(); + if ddl_buffer::is_active() { + let entries = plan(catalog)?; + if entries.is_empty() { + return Ok(CatalogBatch::empty()); + } + for entry in &entries { + ddl_buffer::try_buffer(entry.clone()); + } + return Ok(CatalogBatch { + outcome: BatchOutcome::Buffered, + entries, + }); + } + let handle = shared.metadata_raft_handle()?; + + // The guards drop before the authorization barrier below, as on the + // single-entry path. + let Some((log_index, entries)) = propose_replicated(shared, handle.as_ref(), plan).await? + else { + return Ok(CatalogBatch::empty()); + }; + if entries.iter().any(CatalogEntry::bears_authorization) { + crate::control::security::auth_lease::authorization_barrier( + shared, + vec![nodedb_cluster::GroupCoverage { + group_id: METADATA_GROUP_ID, + through: log_index, + }], + ) + .await?; + } + Ok(CatalogBatch { + outcome: BatchOutcome::Replicated { log_index }, + entries, + }) +} + +/// Plan, drain, stamp, and propose a batch under the preparation lock and +/// lease, and wait until this node applied it. Returns its log index and its +/// stamped entries, or `None` for a plan with no entries. Both guards are +/// released before it returns. +async fn propose_replicated( + shared: &SharedState, + handle: &dyn MetadataRaftHandle, + plan: impl FnOnce(&SystemCatalog) -> crate::Result>, +) -> Result)>, Error> { + let _local_ddl_guard = lock_ddl_preparation_async(shared).await; + let lease = acquire_ddl_prepare_lease_async(shared, handle).await?; + let proposed: Result)>, Error> = async { + let Some(entries) = plan_drain_and_stamp(shared, plan).await? else { + return Ok::<_, Error>(None); + }; + let mut wrapped = Vec::with_capacity(entries.len()); + for entry in &entries { + wrapped.push(catalog_ddl_entry(entry)?); + } + // The wait follows apply progress: a batch of slow purges keeps it + // alive, and only a stall of `DEFAULT_PROPOSE_TIMEOUT` ends it. + let log_index = propose_prepared( + shared, + handle, + lease.token(), + MetadataEntry::Batch { entries: wrapped }, + DEFAULT_PROPOSE_TIMEOUT, + ) + .await?; + Ok::<_, Error>(Some((log_index, entries))) + } + .await; + lease.release().await; + proposed +} + +/// Plan, drain, and stamp a batch. The caller holds the DDL preparation lock +/// and lease. `None` is a plan with no entries. +/// +/// The batch's apply ends every drain it started. A failure before the batch +/// is proposed ends them here: a drain has no wall-clock expiry. +async fn plan_drain_and_stamp( + shared: &SharedState, + plan: impl FnOnce(&SystemCatalog) -> crate::Result>, +) -> Result>, Error> { + let catalog = shared.credentials.catalog(); + let entries = plan(catalog)?; + if entries.is_empty() { + return Ok(None); + } + if let Err(error) = drain_batch(shared, &entries).await { + return Err(end_batch_drains(shared, &entries, error).await); + } + match catalog_entry::descriptor_stamp::stamp_batch(entries.clone(), &shared.hlc_clock, catalog) + { + Ok(stamped) => Ok(Some(stamped)), + Err(error) => Err(end_batch_drains(shared, &entries, error).await), + } +} + +/// End the drain of every entry of a batch that failed before it was +/// proposed. Returns `error`, the failure that stopped the batch. +/// +/// Every node installed the drains, so each DDL drain ends through the +/// metadata group. +async fn end_batch_drains(shared: &SharedState, entries: &[CatalogEntry], error: Error) -> Error { + for entry in entries { + if let Err(end) = end_replicated_drain(shared, entry).await { + tracing::warn!( + kind = entry.kind(), + error = %end, + "metadata batch: the drain of an unapplied entry did not end" + ); + } + } + error +} + +/// End, through the metadata group, the DDL drain [`drain_batch`] started for +/// `entry`. Ending a drain that is not active is a no-op on every node. +async fn end_replicated_drain(shared: &SharedState, entry: &CatalogEntry) -> Result<(), Error> { + match crate::control::lease::descriptor_id_and_prior_version(entry, shared) { + Some((descriptor_id, prior_version)) if prior_version > 0 => { + crate::control::lease::end_drain_async( + shared, + descriptor_id, + nodedb_cluster::DrainOwner::Ddl, + ) + .await + } + _ => Ok(()), + } +} + +/// Drain the prior version of every descriptor the batch changes. +async fn drain_batch(shared: &SharedState, entries: &[CatalogEntry]) -> Result<(), Error> { + for entry in entries { + if let Some((descriptor_id, prior_version)) = + crate::control::lease::descriptor_id_and_prior_version(entry, shared) + && prior_version > 0 + { + crate::control::lease::drain_for_ddl_async( + shared, + descriptor_id, + prior_version, + DEFAULT_DRAIN_TIMEOUT, + 0, + ) + .await?; + } + } + Ok(()) +} diff --git a/nodedb/src/control/metadata_proposer/ddl_owner.rs b/nodedb/src/control/metadata_proposer/ddl_owner.rs new file mode 100644 index 000000000..70de3f1a2 --- /dev/null +++ b/nodedb/src/control/metadata_proposer/ddl_owner.rs @@ -0,0 +1,95 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The owner of the DDL preparation lease, and when the metadata leader can +//! take the lease back. +//! +//! Every replica applies the same `DdlPrepareAcquire` and `DdlPrepareRelease` +//! entries, so every replica agrees on the owner at each log index. A +//! `DdlPrepared`, `DdlPendingPropose` or `DdlPendingFinalize` applies only +//! while its token owns the lease. An entry the old owner proposes after a +//! reclaim applies after the reclaim's release in log order, so it is a no-op +//! on every replica. The token is the fence. The grace below decides when a +//! reclaim stops a live owner's DDL, never whether a stale entry applies. + +use std::time::{Duration, Instant}; + +use crate::control::state::SharedState; + +/// How long an owner holds the lease before the metadata leader reclaims it +/// from a live owner. The fallback for an owner that is alive but stuck. +pub(super) const DDL_PREPARE_LEASE: Duration = Duration::from_secs(60); + +/// The current owner of the DDL preparation lease on this replica. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct DdlPrepareOwner { + /// The fencing token the owner's prepared entries carry. + pub token: u64, + /// The node that proposed the acquire. + pub node_id: u64, + /// When this replica applied the acquire, or seeded it at boot. Local and + /// monotonic: it paces the stuck-owner fallback only. + pub acquired_at: Instant, +} + +/// Why the metadata leader takes the lease from its owner. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum ReclaimCause { + /// The owner held the lease past [`DDL_PREPARE_LEASE`]. + Expired, + /// The owner's node left the topology. + OwnerLeft, + /// The owner's node is SWIM-Dead past `DEAD_HOLDER_LEASE_GRACE`, and the + /// leader heard no Raft response from it for `DEAD_HOLDER_RAFT_SILENCE`. + /// By then the owner self-fenced, or it still hears the leader. + OwnerDead, +} + +/// Why this node, as metadata leader, can reclaim the lease `owner` holds, +/// or `None`. Only the leader decides: its Raft contact samples are the ones +/// that show a dead owner silent. +pub(crate) fn reclaim_cause(shared: &SharedState, owner: &DdlPrepareOwner) -> Option { + if !shared.is_metadata_leader() { + return None; + } + if owner.acquired_at.elapsed() >= DDL_PREPARE_LEASE { + return Some(ReclaimCause::Expired); + } + if owner.node_id == shared.node_id { + return None; + } + if let Some(topology) = shared.cluster_topology.as_ref() + && !topology + .read() + .unwrap_or_else(|p| p.into_inner()) + .contains(owner.node_id) + { + return Some(ReclaimCause::OwnerLeft); + } + let leader_term = shared + .lease_runtime + .metadata_leader_term + .get() + .and_then(|term| term()); + shared + .lease_runtime + .holder_liveness + .dead_holder_released(owner.node_id, leader_term, Instant::now()) + .then_some(ReclaimCause::OwnerDead) +} + +/// Whether `token` owns the preparation lease on this replica. +pub(crate) fn owns_ddl_lease(shared: &SharedState, token: u64) -> bool { + shared + .metadata_ddl_owner + .lock() + .unwrap_or_else(|p| p.into_inner()) + .is_some_and(|owner| owner.token == token) +} + +/// The current owner on this replica, if the lease is held. +pub(crate) fn current_owner(shared: &SharedState) -> Option { + *shared + .metadata_ddl_owner + .lock() + .unwrap_or_else(|p| p.into_inner()) +} diff --git a/nodedb/src/control/metadata_proposer/ddl_prepare.rs b/nodedb/src/control/metadata_proposer/ddl_prepare.rs index 7f5310325..9fefdee02 100644 --- a/nodedb/src/control/metadata_proposer/ddl_prepare.rs +++ b/nodedb/src/control/metadata_proposer/ddl_prepare.rs @@ -3,11 +3,10 @@ //! The DDL preparation lease: the local lock and the replicated lease that //! serialize descriptor preparation across the cluster. +use std::sync::Arc; use std::sync::atomic::Ordering; use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; -use tokio::runtime::RuntimeFlavor; - use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry, WaitOutcome, encode_entry}; use crate::control::state::SharedState; @@ -15,9 +14,12 @@ use crate::error::Error; use super::handle::MetadataRaftHandle; use super::timeouts::DEFAULT_PROPOSE_TIMEOUT; +use super::wait::wait_applied; -const DDL_PREPARE_LEASE: Duration = Duration::from_secs(60); -const DDL_PREPARE_WAIT: Duration = Duration::from_secs(70); +/// How long a proposer waits for the lease. It outlasts the leader's +/// stuck-owner fallback, so a reclaim always frees the lease first. +const DDL_PREPARE_WAIT: Duration = + super::ddl_owner::DDL_PREPARE_LEASE.saturating_add(Duration::from_secs(10)); fn wall_now_ns() -> u64 { SystemTime::now() @@ -27,24 +29,36 @@ fn wall_now_ns() -> u64 { .min(u64::MAX as u128) as u64 } -fn propose_metadata_and_wait( - shared: &SharedState, - handle: &dyn MetadataRaftHandle, - entry: &MetadataEntry, +/// Poll interval while another token holds the preparation lease. +const DDL_PREPARE_POLL: Duration = Duration::from_millis(10); + +/// A fresh preparation-lease token for this node. +fn next_token(shared: &SharedState) -> u64 { + let sequence = shared + .metadata_ddl_token_seq + .fetch_add(1, Ordering::Relaxed); + shared.node_id.wrapping_mul(0x9e37_79b9_7f4a_7c15) ^ wall_now_ns().rotate_left(17) ^ sequence +} + +fn encode_metadata(entry: &MetadataEntry) -> Result, Error> { + encode_entry(entry).map_err(|e| Error::Config { + detail: format!("metadata entry encode: {e}"), + }) +} + +/// The result of waiting `timeout` for log index `index` to apply. +fn applied_or_error( + outcome: WaitOutcome, + index: u64, timeout: Duration, + current: u64, ) -> Result { - let raw = encode_entry(entry).map_err(|e| Error::Config { - detail: format!("metadata entry encode: {e}"), - })?; - let index = handle.propose(raw)?; - let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); - let outcome = tokio::task::block_in_place(|| watcher.wait_for(index, timeout)); match outcome { WaitOutcome::Reached => Ok(index), WaitOutcome::TimedOut => Err(Error::Config { detail: format!( - "metadata propose timed out after {timeout:?} waiting for log index {index} (current: {})", - watcher.current() + "metadata propose timed out after {timeout:?} waiting for log index {index} \ + (current: {current})" ), }), WaitOutcome::GroupGone => Err(Error::Config { @@ -53,140 +67,197 @@ fn propose_metadata_and_wait( } } -/// RAII ownership of the metadata-Raft-serialized descriptor preparation lease. -/// The matching release is itself replicated, so another node cannot stamp from -/// the same prior catalog version until this guard is dropped and that release +pub(super) async fn propose_metadata_and_wait_async( + shared: &SharedState, + handle: &dyn MetadataRaftHandle, + entry: &MetadataEntry, + timeout: Duration, +) -> Result { + let index = handle.propose_async(encode_metadata(entry)?).await?; + let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); + let outcome = wait_applied(Arc::clone(&watcher), index, timeout).await?; + applied_or_error(outcome, index, timeout, watcher.current()) +} + +/// What the acquire loop does next, from the current lease owner. +enum OwnerStep { + /// `token` holds the lease. + Acquired, + /// No owner: propose the acquire again. + Retry, + /// Another token holds the lease: poll again. The metadata leader's + /// reclaim loop frees a lease whose owner died or got stuck. + Wait, +} + +fn owner_step(shared: &SharedState, token: u64, deadline: Instant) -> Result { + match super::ddl_owner::current_owner(shared) { + Some(owner) if owner.token == token => Ok(OwnerStep::Acquired), + None => Ok(OwnerStep::Retry), + Some(_) if Instant::now() < deadline => Ok(OwnerStep::Wait), + Some(owner) => Err(Error::Config { + detail: format!( + "metadata DDL preparation lease timed out after {DDL_PREPARE_WAIT:?}: node {} \ + still holds it", + owner.node_id + ), + }), + } +} + +/// Release the preparation lease `token` and await the release's apply here. +/// The background lease releaser calls it for a lease dropped unreleased. +pub(crate) async fn release_ddl_prepare_token( + shared: &SharedState, + token: u64, +) -> Result<(), Error> { + let handle = shared.metadata_raft_handle()?; + propose_metadata_and_wait_async( + shared, + handle.as_ref(), + &MetadataEntry::DdlPrepareRelease { token }, + DEFAULT_PROPOSE_TIMEOUT, + ) + .await + .map(|_| ()) +} + +/// The metadata-Raft-serialized descriptor preparation lease an async +/// proposer holds. The matching release is itself replicated, so another +/// node cannot stamp from the same prior catalog version until that release /// has applied. -pub(crate) struct DdlPrepareGuard<'a> { +/// +/// [`Self::release`] releases it and awaits the release's apply. A lease +/// dropped unreleased, as when its proposer's future is cancelled, hands its +/// release to the background lease releaser and never blocks. A release that +/// never applies ends when the metadata leader reclaims the lease (see +/// [`super::ddl_reclaim`]). +pub(crate) struct DdlPrepareLease<'a> { shared: &'a SharedState, handle: &'a dyn MetadataRaftHandle, token: u64, + released: bool, } -impl DdlPrepareGuard<'_> { +impl DdlPrepareLease<'_> { pub(crate) fn token(&self) -> u64 { self.token } -} -impl Drop for DdlPrepareGuard<'_> { - fn drop(&mut self) { - if let Err(error) = propose_metadata_and_wait( + /// Release the lease and wait until the release applied here. A failed + /// release is logged: the metadata leader reclaims the lease once + /// [`super::ddl_owner::DDL_PREPARE_LEASE`] passed. + pub(crate) async fn release(mut self) { + self.released = true; + if let Err(error) = propose_metadata_and_wait_async( self.shared, self.handle, &MetadataEntry::DdlPrepareRelease { token: self.token }, DEFAULT_PROPOSE_TIMEOUT, - ) { + ) + .await + { tracing::error!(token = self.token, %error, "metadata DDL lease release failed"); } } } -pub(crate) fn acquire_ddl_prepare_lease<'a>( +impl Drop for DdlPrepareLease<'_> { + fn drop(&mut self) { + if !self.released { + self.shared.lease_runtime.releaser.submit( + crate::control::lease::releaser::ReleaseRequest::DdlPrepare { token: self.token }, + ); + } + } +} + +/// Take the preparation lease from async code. +pub(crate) async fn acquire_ddl_prepare_lease_async<'a>( shared: &'a SharedState, handle: &'a dyn MetadataRaftHandle, -) -> Result, Error> { - let sequence = shared - .metadata_ddl_token_seq - .fetch_add(1, Ordering::Relaxed); - let token = shared.node_id.wrapping_mul(0x9e37_79b9_7f4a_7c15) - ^ wall_now_ns().rotate_left(17) - ^ sequence; +) -> Result, Error> { + let token = next_token(shared); let deadline = Instant::now() + DDL_PREPARE_WAIT; loop { - propose_metadata_and_wait( + propose_metadata_and_wait_async( shared, handle, - &MetadataEntry::DdlPrepareAcquire { token }, + &MetadataEntry::DdlPrepareAcquire { + token, + node_id: shared.node_id, + }, DEFAULT_PROPOSE_TIMEOUT, - )?; + ) + .await?; loop { - let owner = *shared - .metadata_ddl_owner - .lock() - .map_err(|_| Error::Config { - detail: "metadata DDL owner lock poisoned".into(), - })?; - match owner { - Some((current, _)) if current == token => { - return Ok(DdlPrepareGuard { + match owner_step(shared, token, deadline)? { + OwnerStep::Acquired => { + return Ok(DdlPrepareLease { shared, handle, token, + released: false, }); } - Some((current, acquired_at)) - if shared.is_metadata_leader() - && acquired_at.elapsed() >= DDL_PREPARE_LEASE => - { - // Cancel the dead owner's pending record before releasing its - // lease, so it never lingers visible-but-unresolved past the lease. - if shared.pending_ddl.contains(current) { - propose_metadata_and_wait( - shared, - handle, - &MetadataEntry::DdlPendingCancel { token: current }, - DEFAULT_PROPOSE_TIMEOUT, - )?; - } - propose_metadata_and_wait( - shared, - handle, - &MetadataEntry::DdlPrepareRelease { token: current }, - DEFAULT_PROPOSE_TIMEOUT, - )?; - break; - } - None => break, - Some(_) if Instant::now() < deadline => { - // Reached from async tasks (ILP batch flush -> - // `propose_catalog_entry`), so hand the worker back to - // tokio rather than parking it: the lease owner this - // polls for is released by a raft apply that needs a - // worker to make progress. - tokio::task::block_in_place(|| { - std::thread::sleep(Duration::from_millis(10)); - }); - } - Some(_) => { - return Err(Error::Config { - detail: "metadata DDL preparation lease timed out".into(), - }); - } + OwnerStep::Retry => break, + OwnerStep::Wait => tokio::time::sleep(DDL_PREPARE_POLL).await, } } } } -/// Take the local DDL preparation lock, handing the wait back to tokio when -/// the caller is on a multi-thread worker. -/// -/// The holder keeps this lock across the distributed preparation lease, the -/// descriptor drain and the local apply wait — each already wrapped in -/// `block_in_place`, but that only tells tokio about the waits *inside* the -/// lock, never about the wait *for* it. A bare `lock()` on a worker therefore -/// removes that worker from the runtime silently, including from the raft -/// apply work the current holder needs in order to finish, which turns -/// contention into a self-sustaining stall. -/// -/// `block_in_place` is a passthrough outside a multi-thread worker (plain sync -/// callers, blocking-pool threads) and panics on the current-thread runtime, -/// so it is applied only where it is both legal and meaningful — mirroring -/// `lease::drain_propose::poll_leases_drained`. -pub(super) fn lock_ddl_preparation( +/// Take the local DDL preparation lock from async code. The guard is `Send`, +/// so the holder can await its post-apply while it holds the lock. +pub(crate) async fn lock_ddl_preparation_async( shared: &SharedState, -) -> Result, Error> { - let acquire = || { - shared.metadata_ddl_lock.lock().map_err(|_| Error::Config { - detail: "metadata DDL preparation lock poisoned".into(), - }) - }; - match tokio::runtime::Handle::try_current() { - Ok(handle) if handle.runtime_flavor() == RuntimeFlavor::MultiThread => { - tokio::task::block_in_place(acquire) +) -> tokio::sync::MutexGuard<'_, ()> { + shared.metadata_ddl_lock.lock().await +} + +#[cfg(test)] +mod tests { + use super::*; + + fn owner(shared: &SharedState) -> Option { + let current = *shared + .metadata_ddl_owner + .lock() + .unwrap_or_else(|p| p.into_inner()); + current.map(|owner| owner.token) + } + + /// A preparation lease dropped unreleased, as a cancelled DDL drops it, + /// is released by the background releaser without blocking the drop. + #[tokio::test] + async fn a_dropped_preparation_lease_is_released_in_the_background() { + let cluster = crate::control::cluster::test_one_node::boot().await; + let state = Arc::clone(&cluster.state); + let token = { + let handle = state.metadata_raft_handle().expect("metadata raft handle"); + let lease = acquire_ddl_prepare_lease_async(&state, handle.as_ref()) + .await + .expect("take the preparation lease"); + assert_eq!(owner(&state), Some(lease.token())); + lease.token() + }; + assert_eq!( + owner(&state), + Some(token), + "the drop itself proposes nothing" + ); + + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + while owner(&state).is_some() { + assert!( + tokio::time::Instant::now() < deadline, + "the background releaser did not release the dropped lease" + ); + tokio::time::sleep(Duration::from_millis(10)).await; } - _ => acquire(), + drop(state); + cluster.shutdown().await; } } diff --git a/nodedb/src/control/metadata_proposer/ddl_reclaim.rs b/nodedb/src/control/metadata_proposer/ddl_reclaim.rs new file mode 100644 index 000000000..76ed82c13 --- /dev/null +++ b/nodedb/src/control/metadata_proposer/ddl_reclaim.rs @@ -0,0 +1,118 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The metadata leader's reclaim of a DDL preparation lease whose owner died, +//! left, or held it past its lease. +//! +//! One loop runs on every node and acts only while the node leads the +//! metadata group. A proposer waiting for the lease never reclaims it itself: +//! a waiter on a follower cannot judge the owner dead, and the leader can have +//! no DDL of its own. The loop covers both. + +use std::sync::{Arc, Weak}; +use std::time::Duration; + +use nodedb_cluster::MetadataEntry; + +use crate::control::shutdown::{ShutdownPhase, spawn_loop}; +use crate::control::state::SharedState; +use crate::error::Error; + +use super::ddl_owner::{current_owner, reclaim_cause}; +use super::ddl_prepare::propose_metadata_and_wait_async; +use super::handle::MetadataRaftHandle; +use super::timeouts::DEFAULT_PROPOSE_TIMEOUT; + +/// How often the leader checks the lease owner. +const RECLAIM_POLL: Duration = Duration::from_millis(250); + +/// Spawn the reclaim loop. `start_raft` calls it once the metadata raft +/// handle is installed. +pub(crate) fn spawn_ddl_lease_reclaimer(shared: &Arc) { + let weak = Arc::downgrade(shared); + spawn_loop( + &shared.loop_registry, + &shared.shutdown, + "ddl_lease_reclaimer", + ShutdownPhase::DrainingControlPlane, + move |mut shutdown| async move { + let mut tick = tokio::time::interval(RECLAIM_POLL); + loop { + tokio::select! { + _ = shutdown.wait_cancelled() => break, + _ = tick.tick() => { + if !reclaim_if_due(&weak).await { + break; + } + } + } + } + }, + ); +} + +/// Reclaim the lease when its owner qualifies. Returns `false` once the +/// state is gone. +async fn reclaim_if_due(shared: &Weak) -> bool { + let Some(state) = shared.upgrade() else { + return false; + }; + let Some(owner) = current_owner(&state) else { + return true; + }; + let Some(cause) = reclaim_cause(&state, &owner) else { + return true; + }; + let handle = match state.metadata_raft_handle() { + Ok(handle) => Arc::clone(handle), + Err(error) => { + tracing::warn!(%error, "DDL lease reclaim: no metadata raft handle"); + return true; + } + }; + match reclaim_ddl_prepare_lease(&state, handle.as_ref(), owner.token).await { + Ok(()) => tracing::info!( + token = owner.token, + owner_node = owner.node_id, + ?cause, + "reclaimed the DDL preparation lease" + ), + Err(error) => tracing::warn!( + token = owner.token, + owner_node = owner.node_id, + ?cause, + %error, + "DDL lease reclaim did not apply; the next check retries" + ), + } + true +} + +/// Cancel `token`'s pending DDL record, when it has one, then release its +/// lease, and await both applies here. +/// +/// The cancel goes first, so a pending record never outlives its lease. Both +/// entries are idempotent, and the release applies only while `token` still +/// owns the lease, so a reclaim that races the owner's own release is a no-op. +pub(crate) async fn reclaim_ddl_prepare_lease( + shared: &SharedState, + handle: &dyn MetadataRaftHandle, + token: u64, +) -> Result<(), Error> { + if shared.pending_ddl.contains(token) { + propose_metadata_and_wait_async( + shared, + handle, + &MetadataEntry::DdlPendingCancel { token }, + DEFAULT_PROPOSE_TIMEOUT, + ) + .await?; + } + propose_metadata_and_wait_async( + shared, + handle, + &MetadataEntry::DdlPrepareRelease { token }, + DEFAULT_PROPOSE_TIMEOUT, + ) + .await + .map(|_| ()) +} diff --git a/nodedb/src/control/metadata_proposer/handle.rs b/nodedb/src/control/metadata_proposer/handle.rs index 48bf80222..bf060605e 100644 --- a/nodedb/src/control/metadata_proposer/handle.rs +++ b/nodedb/src/control/metadata_proposer/handle.rs @@ -16,19 +16,27 @@ use crate::error::Error; /// [`nodedb_cluster::METADATA_GROUP_ID`]); callers of [`Self::propose`] /// look it up there rather than receiving it through this handle. pub trait MetadataRaftHandle: Send + Sync { - /// Propose a raw encoded `MetadataEntry` to the metadata group. - /// Returns its assigned log index on success. - fn propose(&self, bytes: Vec) -> Result; + /// Propose a raw encoded `MetadataEntry` to the metadata group. Resolves + /// to its assigned log index. Runs on any runtime flavor. + fn propose_async<'a>(&'a self, bytes: Vec) -> ProposeFuture<'a>; } +/// The future [`MetadataRaftHandle::propose_async`] returns. +pub type ProposeFuture<'a> = + std::pin::Pin> + Send + 'a>>; + /// Concrete impl wrapping `nodedb_cluster::RaftLoop`. /// /// Holds the loop weakly: this handle lives on `SharedState`, which is /// itself kept alive transitively by the `RaftLoop`, so a strong -/// reference here would close a cycle that pins both forever and blocks +/// reference here closes a cycle that pins both forever and blocks /// clean shutdown. The loop is kept alive by its own spawned tasks; /// `upgrade` therefore succeeds throughout normal operation and only /// fails once the loop has been dropped on shutdown. +/// +/// Every entry it proposes carries the metadata leader's HLC stamp, taken as +/// the leader appends it. A restore keeps an entry only when that stamp is +/// below the restore point's watermark, the rule it applies to data writes. pub struct RaftLoopProposerHandle { raft_loop: Weak< nodedb_cluster::RaftLoop< @@ -53,27 +61,36 @@ impl RaftLoopProposerHandle { } } -impl MetadataRaftHandle for RaftLoopProposerHandle { - fn propose(&self, bytes: Vec) -> Result { - // The cluster crate's `propose_to_metadata_group_via_leader` - // is async because it may need to forward to the metadata - // leader over QUIC. The trait method is sync because every - // caller (catalog DDL handlers, lease grant/release helpers) - // is itself sync but runs inside a tokio task. Wrap in - // `block_in_place` + the current runtime's `block_on` so the - // forwarding QUIC round-trip drives without starving the - // raft tick that produces the leader_hint. - // `upgrade` fails only once the raft loop has been dropped on - // shutdown; a request racing shutdown then fails cleanly with a - // typed error instead of panicking. - let raft_loop = self.raft_loop.upgrade().ok_or_else(|| Error::Config { +impl RaftLoopProposerHandle { + /// The live raft loop. `upgrade` fails only once the raft loop has been + /// dropped on shutdown; a request racing shutdown then fails cleanly with + /// a typed error instead of panicking. + fn live_loop( + &self, + ) -> Result< + Arc< + nodedb_cluster::RaftLoop< + crate::control::cluster::SpscCommitApplier, + crate::control::LocalPlanExecutor, + >, + >, + Error, + > { + self.raft_loop.upgrade().ok_or_else(|| Error::Config { detail: "metadata propose: cluster not running".into(), - })?; - tokio::task::block_in_place(|| { - tokio::runtime::Handle::current() - .block_on(raft_loop.propose_to_metadata_group_via_leader(bytes)) }) - .map_err(metadata_propose_error) + } +} + +impl MetadataRaftHandle for RaftLoopProposerHandle { + fn propose_async<'a>(&'a self, bytes: Vec) -> ProposeFuture<'a> { + Box::pin(async move { + let raft_loop = self.live_loop()?; + raft_loop + .propose_stamped_to_metadata_group_via_leader(bytes) + .await + .map_err(metadata_propose_error) + }) } } @@ -83,11 +100,11 @@ fn metadata_propose_error(error: ClusterError) -> Error { // An election in progress is transient, not a failure of this // proposal. Keep it typed rather than flattening it into a generic // config error, so callers can wait the election out instead of - // failing the statement — a node that has just restarted answers + // failing the statement — a node that has recently restarted answers // every metadata proposal this way for a moment. - ClusterError::Raft(RaftError::NotLeader { leader_hint: None }) => { - Error::MetadataLeaderUnavailable - } + ClusterError::Raft(RaftError::NotLeader { + leader_hint: None, .. + }) => Error::MetadataLeaderUnavailable, // A typed verdict keeps its class. ClusterError::DataPlane { code } => Error::DataPlane(code.into()), ClusterError::ShardExecution { error, .. } | ClusterError::StreamTerminal { error, .. } => { @@ -96,6 +113,7 @@ fn metadata_propose_error(error: ClusterError) -> Error { other @ (ClusterError::Raft( RaftError::NotLeader { leader_hint: Some(_), + .. } | RaftError::LogCompacted { .. } | RaftError::CompactionAheadOfApplied { .. } @@ -142,6 +160,7 @@ fn metadata_propose_error(error: ClusterError) -> Error { | ClusterError::SpatialGather(_) | ClusterError::Bm25Gather(_) | ClusterError::TsGather(_) + | ClusterError::ShufflePush(_) | ClusterError::RemoteUntyped { .. }) => Error::Config { detail: format!("metadata propose: {other}"), }, diff --git a/nodedb/src/control/metadata_proposer/mod.rs b/nodedb/src/control/metadata_proposer/mod.rs index 6a12b572a..c06c8a0e2 100644 --- a/nodedb/src/control/metadata_proposer/mod.rs +++ b/nodedb/src/control/metadata_proposer/mod.rs @@ -1,38 +1,56 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Synchronous `propose-and-wait-for-local-apply` helper for -//! replicated catalog DDL. +//! `propose-and-wait-for-local-apply` helpers for replicated catalog DDL. //! -//! The sole entry point pgwire DDL handlers use to write a -//! [`crate::control::catalog_entry::CatalogEntry`] through the metadata raft group (group 0). It is -//! deliberately sync — pgwire DDL handlers are not async, and -//! `tokio::task::block_in_place`-style wrapping keeps the blocking -//! wait from starving the tokio runtime. +//! The sole entry point DDL handlers use to write a +//! [`crate::control::catalog_entry::CatalogEntry`] through the metadata raft +//! group (group 0): `propose_catalog_entry_async` and +//! `propose_catalog_batch_async`. Both run on any runtime flavor. Every wait +//! is awaited: the preparation lock and lease, the descriptor drain, the +//! metadata commit's apply, and the authorization barrier. +//! +//! A preparation lease dropped unreleased, as when its proposer's future is +//! cancelled, hands its release to the background lease releaser. A release +//! that never applies ends when the metadata leader reclaims the lease: at +//! once when its owner left, once its owner is dead past the grace, and after +//! the lease time when its owner is alive but stuck. +//! +//! Every node runs a metadata raft group, a one-node cluster included. +//! `start_raft` installs its handle before any listener opens. A state with +//! no handle refuses the propose with a typed error. No caller writes the +//! catalog itself. //! //! Semantics: //! -//! 1. If no cluster is configured (`shared.metadata_raft` not -//! installed), returns `ProposeOutcome::LocalOnly`. The caller's -//! single-node direct-write path stays authoritative. -//! 2. If this node is the metadata-group leader, proposes the -//! entry, blocks until its local applied watermark reaches the -//! assigned log index (5s default timeout), and returns the +//! 1. If this node is the metadata-group leader, proposes the entry, waits +//! until its local applied watermark reaches the assigned log index for +//! as long as the apply moves (a 5s stall ends the wait), and returns the //! log index on success. -//! 3. If this node is NOT the leader, returns +//! 2. If this node is NOT the leader, returns //! `Error::Config { detail: "metadata propose: not leader ..." }`. //! Gateway-side redirection will make this transparent. pub mod catalog; +pub mod catalog_batch; +pub mod ddl_owner; pub mod ddl_prepare; +pub(crate) mod ddl_reclaim; pub mod handle; pub mod replicated_entries; pub mod timeouts; +pub(crate) mod wait; -pub use catalog::{propose_catalog_entry, propose_catalog_entry_with_timeout}; -pub(crate) use ddl_prepare::{DdlPrepareGuard, acquire_ddl_prepare_lease}; -pub use handle::{MetadataRaftHandle, RaftLoopProposerHandle}; +pub use catalog::propose_catalog_entry_async; +pub use catalog_batch::{BatchOutcome, CatalogBatch, propose_catalog_batch_async}; +pub use ddl_owner::DdlPrepareOwner; +pub(crate) use ddl_prepare::{ + DdlPrepareLease, acquire_ddl_prepare_lease_async, lock_ddl_preparation_async, + release_ddl_prepare_token, +}; +pub use handle::{MetadataRaftHandle, ProposeFuture, RaftLoopProposerHandle}; pub use replicated_entries::{ - propose_surrogate_hwm, propose_surrogate_reserve, propose_sync_peer_bind, - propose_sync_producer_fence, propose_sync_producer_register, + propose_cursor_commit, propose_cursor_commit_audited, propose_database_id_reserve, + propose_restore_point, propose_surrogate_hwm, propose_surrogate_reserve, + propose_sync_peer_bind, propose_sync_producer_fence, propose_sync_producer_register, }; pub use timeouts::{DEFAULT_DRAIN_TIMEOUT, DEFAULT_PROPOSE_TIMEOUT}; diff --git a/nodedb/src/control/metadata_proposer/replicated_entries.rs b/nodedb/src/control/metadata_proposer/replicated_entries.rs index f45af8b4e..0e2cce34b 100644 --- a/nodedb/src/control/metadata_proposer/replicated_entries.rs +++ b/nodedb/src/control/metadata_proposer/replicated_entries.rs @@ -1,38 +1,41 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Metadata entries that replicate node-local registries: surrogate -//! allocation and Lite sync producers. +//! Metadata entries that replicate node-local registries: surrogate and +//! database-id allocation, Lite sync producers, and the system cursors of +//! the Event Plane lanes. //! -//! Each one proposes the entry and waits for its commit on this node. In -//! single-node mode (no `metadata_raft` installed) each returns `Ok(0)`: the -//! local write already persisted the state. +//! Each one proposes the entry and awaits its commit on this node, on any +//! runtime flavor. Every node runs a metadata group, a one-node cluster +//! included. + +use std::sync::Arc; use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry, encode_entry}; +use crate::control::catalog_entry::CatalogEntry; use crate::control::state::SharedState; use crate::error::Error; +use crate::event::cdc::consumer_group::OffsetCommit; use super::timeouts::DEFAULT_PROPOSE_TIMEOUT; /// Propose `entry` and wait until this node reaches its log index. `label` -/// names the entry in the errors. Returns `Ok(0)` when no cluster runs. -fn propose_and_wait( +/// names the entry in the errors. +async fn propose_and_wait( shared: &SharedState, entry: &MetadataEntry, label: &str, ) -> Result { - let Some(handle) = shared.metadata_raft.get() else { - return Ok(0); - }; + let handle = shared.metadata_raft_handle()?; let raw = encode_entry(entry).map_err(|e| Error::Config { detail: format!("{label} encode: {e}"), })?; - let log_index = handle.propose(raw)?; + let log_index = handle.propose_async(raw).await?; let watcher = shared.applied_index_watcher(METADATA_GROUP_ID); let outcome = - tokio::task::block_in_place(|| watcher.wait_for(log_index, DEFAULT_PROPOSE_TIMEOUT)); + super::wait::wait_applied(Arc::clone(&watcher), log_index, DEFAULT_PROPOSE_TIMEOUT).await?; if !outcome.is_reached() { return Err(Error::Config { detail: format!("{label} propose timed out waiting for log index {log_index}"), @@ -42,34 +45,43 @@ fn propose_and_wait( Ok(log_index) } +/// Propose a cluster restore point at watermark `hlc` and wait until this +/// node reaches its log index, which is the point's id. +pub async fn propose_restore_point( + shared: &SharedState, + hlc: u64, + created_at_ms: u64, +) -> Result { + propose_and_wait( + shared, + &MetadataEntry::RestorePoint { hlc, created_at_ms }, + "restore_point", + ) + .await +} + /// Propose a surrogate high-watermark advance to the metadata Raft group /// and wait for it to be applied locally. /// -/// In single-node / no-cluster mode (no `metadata_raft` installed), -/// returns `Ok(0)` immediately — the WAL-only path on `SharedState` is -/// still sufficient. In cluster mode this is called by the leader-side -/// flush path instead of (or in addition to) the local WAL record, so -/// every follower's `SurrogateRegistry` advances to the same hwm via the -/// Raft commit. +/// The leader-side flush path calls this in addition to the local WAL +/// record, so every follower's `SurrogateRegistry` advances to the same hwm +/// via the Raft commit. /// /// `hwm` is the highest surrogate that has been issued so far on this /// node. Followers apply the entry by calling /// `SurrogateRegistry::restore_hwm(hwm)` (idempotent, monotonic). -pub fn propose_surrogate_hwm(shared: &SharedState, hwm: u32) -> Result { +pub async fn propose_surrogate_hwm(shared: &SharedState, hwm: u32) -> Result { propose_and_wait( shared, &MetadataEntry::SurrogateAlloc { hwm }, "surrogate_alloc", ) + .await } /// Propose a HiLo surrogate batch reservation to the metadata Raft group /// and wait for the commit (returns the assigned log index). /// -/// In single-node / no-cluster mode (no `metadata_raft` installed), -/// returns `Ok(0)` immediately — single-node uses the local `alloc_one` -/// path and never reaches here. Kept as a safety guard only. -/// /// The carved `[start, end)` range is NOT decided here: it is computed /// at apply time on every node by advancing the global watermark in /// identical log order (see `MetadataEntry::SurrogateReserve`). The @@ -81,7 +93,7 @@ pub fn propose_surrogate_hwm(shared: &SharedState, hwm: u32) -> Result Result { + propose_and_wait( + shared, + &MetadataEntry::DatabaseIdReserve { + node_id, + request_id, + }, + "database_id_reserve", + ) + .await +} + +/// Propose a consumer offset commit and wait until this node applied it. +/// +/// Every node raises each offset to the highest one committed, so commits +/// converge in any order. A commit stamps no descriptor version, so it takes +/// no DDL preparation lease. A node that dies holding that lease then never +/// stalls another node's cursor. +/// +/// The entry carries the running statement's audit context, if any: a user +/// `COMMIT OFFSET` is audited, a system cursor is not. +pub async fn propose_cursor_commit( + shared: &SharedState, + commit: OffsetCommit, +) -> Result { + propose_cursor_commit_audited( + shared, + commit, + crate::control::server::shared::session::audit_context::current(), + ) + .await +} + +/// [`propose_cursor_commit`] with the audit context of the statement that +/// issued the commit, for a commit a transaction buffered until COMMIT. +pub async fn propose_cursor_commit_audited( + shared: &SharedState, + commit: OffsetCommit, + audit: Option, +) -> Result { + let entry = super::catalog::catalog_ddl_entry_with( + &CatalogEntry::CommitConsumerOffsets(Box::new(commit)), + audit, + )?; + propose_and_wait(shared, &entry, "cursor_commit").await } #[cfg(test)] diff --git a/nodedb/src/control/metadata_proposer/timeouts.rs b/nodedb/src/control/metadata_proposer/timeouts.rs index c1fe63ddb..97e1c41f2 100644 --- a/nodedb/src/control/metadata_proposer/timeouts.rs +++ b/nodedb/src/control/metadata_proposer/timeouts.rs @@ -4,9 +4,8 @@ use std::time::Duration; -/// Default upper bound on how long a single -/// `propose_catalog_entry` call will block before returning an -/// error. +/// Default window of no apply progress after which a metadata propose +/// returns an error. pub const DEFAULT_PROPOSE_TIMEOUT: Duration = Duration::from_secs(5); /// Default upper bound on how long a DDL drain will wait for @@ -15,7 +14,5 @@ pub const DEFAULT_PROPOSE_TIMEOUT: Duration = Duration::from_secs(5); /// so an existing lease gets at least one full lifetime to /// expire naturally. 35 seconds matches the 300s lease duration /// plus a 30-second grace minus the typical 5-minute default -/// cut down for test budget — in production -/// `propose_catalog_entry_with_drain_timeout` can pass a longer -/// value if an operator is willing to wait. +/// cut down for test budget. pub const DEFAULT_DRAIN_TIMEOUT: Duration = Duration::from_secs(35); diff --git a/nodedb/src/control/metadata_proposer/wait.rs b/nodedb/src/control/metadata_proposer/wait.rs new file mode 100644 index 000000000..17bd5759d --- /dev/null +++ b/nodedb/src/control/metadata_proposer/wait.rs @@ -0,0 +1,54 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Waits on the metadata group's applied index. +//! +//! The applied-index watcher parks its caller on a condition variable. Each +//! park runs on the blocking pool, so no runtime flavor loses a worker to it. + +use std::sync::Arc; +use std::time::Duration; + +use nodedb_cluster::{AppliedIndexWatcher, WaitOutcome}; + +use crate::error::Error; + +/// Wait until `watcher` reaches `index` or `timeout` passes, on the blocking +/// pool. +pub(crate) async fn wait_applied( + watcher: Arc, + index: u64, + timeout: Duration, +) -> Result { + tokio::task::spawn_blocking(move || watcher.wait_for(index, timeout)) + .await + .map_err(|error| Error::Config { + detail: format!("metadata apply wait for log index {index} did not finish: {error}"), + }) +} + +/// Wait until `watcher` reaches `log_index`, for as long as the apply moves. +/// Each window parks on the blocking pool. +/// +/// Each window of `stall` that ends with neither the applied index nor +/// `progress()` changed ends the wait with `TimedOut`. Any change inside a +/// window opens another one, so a slow apply that keeps moving never times +/// out, and a stuck one times out after one quiet window. +pub(super) async fn wait_tracking_progress_async( + watcher: Arc, + log_index: u64, + stall: Duration, + mut progress: impl FnMut() -> u64, +) -> Result { + let mut seen = (watcher.current(), progress()); + loop { + let outcome = wait_applied(Arc::clone(&watcher), log_index, stall).await?; + if !matches!(outcome, WaitOutcome::TimedOut) { + return Ok(outcome); + } + let now = (watcher.current(), progress()); + if now == seen { + return Ok(outcome); + } + seen = now; + } +} diff --git a/nodedb/src/control/metrics/prometheus/engines.rs b/nodedb/src/control/metrics/prometheus/engines.rs index b4b72aa55..de83b9248 100644 --- a/nodedb/src/control/metrics/prometheus/engines.rs +++ b/nodedb/src/control/metrics/prometheus/engines.rs @@ -299,6 +299,12 @@ impl SystemMetrics { "Change events dropped", self.change_events_dropped.load(Ordering::Relaxed), ); + counter( + out, + "nodedb_committed_publishes_dropped_total", + "Committed transaction messages whose topic was dropped before delivery", + self.committed_publishes_dropped.load(Ordering::Relaxed), + ); // ── Checkpoints ── counter( @@ -307,5 +313,140 @@ impl SystemMetrics { "Checkpoints completed", self.checkpoints.load(Ordering::Relaxed), ); + + // ── WAL archive ── + counter( + out, + "nodedb_wal_archive_segments_total", + "WAL segments uploaded to the archive", + self.wal_archive_segments_total.load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_wal_archive_bytes_total", + "WAL bytes uploaded to the archive", + self.wal_archive_bytes_total.load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_wal_archive_failures_total", + "WAL archive listings or uploads that failed", + self.wal_archive_failures_total.load(Ordering::Relaxed), + ); + gauge( + out, + "nodedb_wal_archive_rpo_gap_bytes", + "Local WAL bytes not yet in the archive", + self.wal_archive_rpo_gap_bytes.load(Ordering::Relaxed), + ); + gauge( + out, + "nodedb_wal_archive_rpo_gap_seconds", + "Age of the oldest local WAL segment not yet in the archive", + self.wal_archive_rpo_gap_seconds.load(Ordering::Relaxed), + ); + + // ── PITR base snapshots ── + gauge( + out, + "nodedb_pitr_base_last_success_timestamp_seconds", + "Unix time the last base snapshot completed", + self.pitr_base_last_success_timestamp_seconds + .load(Ordering::Relaxed), + ); + gauge( + out, + "nodedb_pitr_base_snapshots", + "Base snapshots this node life holds after retention", + self.pitr_base_snapshots.load(Ordering::Relaxed), + ); + gauge( + out, + "nodedb_pitr_base_last_failure_timestamp_seconds", + "Unix time the last base snapshot run failed", + self.pitr_base_last_failure_timestamp_seconds + .load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_pitr_base_failures_total", + "Base snapshot runs that failed", + self.pitr_base_failures_total.load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_pitr_wal_segments_collected_total", + "Archived WAL segments deleted because no kept base needs them", + self.pitr_wal_segments_collected_total + .load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_pitr_base_chunks_uploaded_total", + "Snapshot chunks uploaded because the store held no match", + self.pitr_base_chunks_uploaded_total.load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_pitr_base_chunk_bytes_uploaded_total", + "Plaintext bytes of the snapshot chunks uploaded", + self.pitr_base_chunk_bytes_uploaded_total + .load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_pitr_base_chunks_reused_total", + "Distinct snapshot chunks a base reused from the store", + self.pitr_base_chunks_reused_total.load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_pitr_base_chunks_collected_total", + "Snapshot chunks deleted because no kept base lists them", + self.pitr_base_chunks_collected_total + .load(Ordering::Relaxed), + ); + + // ── Scheduled logical backups ── + counter( + out, + "nodedb_backup_schedule_runs_total", + "Scheduled backup runs that completed", + self.backup_schedule_runs_total.load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_backup_schedule_failures_total", + "Scheduled backup runs that failed", + self.backup_schedule_failures_total.load(Ordering::Relaxed), + ); + gauge( + out, + "nodedb_backup_schedule_last_success_timestamp_seconds", + "Unix time the last scheduled backup completed", + self.backup_schedule_last_success_timestamp_seconds + .load(Ordering::Relaxed), + ); + gauge( + out, + "nodedb_backup_schedule_last_failure_timestamp_seconds", + "Unix time the last scheduled backup failed", + self.backup_schedule_last_failure_timestamp_seconds + .load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_backup_schedule_envelopes_deleted_total", + "Backup envelopes deleted by keep retention", + self.backup_schedule_envelopes_deleted_total + .load(Ordering::Relaxed), + ); + counter( + out, + "nodedb_backup_schedule_ticks_skipped_total", + "Scheduled backup ticks skipped because the schedule mark read was not confirmed", + self.backup_schedule_ticks_skipped_total + .load(Ordering::Relaxed), + ); } } diff --git a/nodedb/src/control/metrics/system/fields.rs b/nodedb/src/control/metrics/system/fields.rs index a21da10da..486843dcc 100644 --- a/nodedb/src/control/metrics/system/fields.rs +++ b/nodedb/src/control/metrics/system/fields.rs @@ -22,6 +22,46 @@ pub struct SystemMetrics { pub wal_segment_count: AtomicU64, pub wal_segment_bytes: AtomicU64, + // ── WAL archive ── + pub wal_archive_segments_total: AtomicU64, + pub wal_archive_bytes_total: AtomicU64, + pub wal_archive_failures_total: AtomicU64, + /// Bytes of local WAL the archive does not hold, active segment included. + pub wal_archive_rpo_gap_bytes: AtomicU64, + /// Seconds since the lowest unarchived local segment was created. + pub wal_archive_rpo_gap_seconds: AtomicU64, + + // ── PITR base snapshots ── + /// Unix seconds when the last base snapshot completed. 0 before the first. + pub pitr_base_last_success_timestamp_seconds: AtomicU64, + /// Base snapshots this node life holds after retention. + pub pitr_base_snapshots: AtomicU64, + /// Unix seconds when the last base snapshot run failed. 0 before the first. + pub pitr_base_last_failure_timestamp_seconds: AtomicU64, + pub pitr_base_failures_total: AtomicU64, + /// Archived WAL segments deleted because no kept base needs them. + pub pitr_wal_segments_collected_total: AtomicU64, + /// Snapshot chunks put to the store because no stored chunk matched. + pub pitr_base_chunks_uploaded_total: AtomicU64, + pub pitr_base_chunk_bytes_uploaded_total: AtomicU64, + /// Distinct snapshot chunks a base reused from the store. + pub pitr_base_chunks_reused_total: AtomicU64, + /// Snapshot chunks deleted because no kept base lists them. + pub pitr_base_chunks_collected_total: AtomicU64, + + // ── Scheduled logical backups ── + pub backup_schedule_runs_total: AtomicU64, + pub backup_schedule_failures_total: AtomicU64, + /// Unix seconds when the last scheduled backup completed. 0 before the first. + pub backup_schedule_last_success_timestamp_seconds: AtomicU64, + /// Unix seconds when the last scheduled backup failed. 0 before the first. + pub backup_schedule_last_failure_timestamp_seconds: AtomicU64, + /// Backup envelopes deleted by `keep` retention. + pub backup_schedule_envelopes_deleted_total: AtomicU64, + /// Scheduler ticks skipped because the schedule mark read was not + /// confirmed. + pub backup_schedule_ticks_skipped_total: AtomicU64, + // ── Raft / replication ── pub raft_apply_lag: AtomicU64, pub raft_commit_index: AtomicU64, @@ -67,7 +107,7 @@ pub struct SystemMetrics { pub vector_builds_started: AtomicU64, /// HNSW builds installed on their core. pub vector_builds_completed: AtomicU64, - /// HNSW builds that failed or could not be read. + /// HNSW builds that failed or were unreadable. pub vector_builds_failed: AtomicU64, /// Times a core found its builder queue full and kept the job waiting. pub vector_builds_deferred: AtomicU64, @@ -142,12 +182,15 @@ pub struct SystemMetrics { pub active_subscriptions: AtomicU64, pub active_listen_channels: AtomicU64, pub change_events_delivered: AtomicU64, - /// Global CDC drop counter (sum across all streams). Kept for backward - /// compatibility with existing dashboards that query this name without labels. + /// Global CDC drop counter (sum across all streams). Dashboards query + /// this name without labels. pub change_events_dropped: AtomicU64, /// Per-stream CDC drop counters. Key: `(tenant_id, stream_name)`. /// Rendered as `nodedb_cdc_events_dropped_total{tenant="",stream=""}`. pub cdc_events_dropped_by_stream: RwLock>, + /// Committed transaction messages not delivered because their topic was + /// dropped after the transaction committed. + pub committed_publishes_dropped: AtomicU64, // ── Backpressure ── /// Per-engine Critical-pressure fire count. diff --git a/nodedb/src/control/mod.rs b/nodedb/src/control/mod.rs index 697cdc12b..f8f152013 100644 --- a/nodedb/src/control/mod.rs +++ b/nodedb/src/control/mod.rs @@ -38,13 +38,13 @@ pub mod notify_bus; pub(crate) mod orchestrated_write; pub mod otel; pub mod pending_ddl; +pub mod pitr; pub mod planner; pub mod promql; pub mod propose_outcome; pub mod request_tracker; pub mod rolling_upgrade; pub mod router; -pub mod scatter_gather; pub mod security; pub mod sequence; pub mod server; @@ -64,6 +64,7 @@ pub mod update_from_join_orchestrator; pub mod vshard_admission; pub mod wal_catchup; pub mod wal_replication; +pub mod write_gate; pub mod write_resolve; pub use exec_receiver::LocalPlanExecutor; diff --git a/nodedb/src/control/orchestrated_write.rs b/nodedb/src/control/orchestrated_write.rs index c618c76e7..e15400aa3 100644 --- a/nodedb/src/control/orchestrated_write.rs +++ b/nodedb/src/control/orchestrated_write.rs @@ -4,18 +4,15 @@ //! `UPDATE ... FROM`, `INSERT ... SELECT`): the resolved plan the orchestrator //! built lands on its target vShard's owner and on every replica. //! -//! On a cluster the plan proposes through Raft exactly as a plain replicated -//! write does; the proposer forwards to the group leader, so the coordinator -//! never has to own the vShard. Standalone, the plan dispatches to the local -//! Data Plane and mints the redo record the write funnel would have minted. +//! The plan proposes through Raft exactly as a plain replicated write does, +//! on a one-node cluster too; the proposer forwards to the group leader, so +//! the coordinator never has to own the vShard. use std::sync::atomic::Ordering; use nodedb_types::{DatabaseId, TenantId}; use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; -use crate::control::maintenance::clone_materializer::dispatch_local; -use crate::control::server::dispatch_utils::publish_origin_change_events; use crate::control::state::SharedState; use crate::control::wal_replication::{ ReplicableWrite, propose_replicated_entry, to_replicated_entry, @@ -36,25 +33,7 @@ pub(crate) async fn apply_orchestrated_write( collection: &str, plan: PhysicalPlan, ) -> crate::Result { - let Some(proposer) = state.async_raft_proposer() else { - let resp = dispatch_local(state, tenant_id, database_id, collection, plan, None).await?; - // `dispatch_local` bypasses the funnel's post-apply redo minting, so a - // vector-indexed target's write-set arrives unconsumed. Without it a - // WAL-only restart rebuilds the index from pre-write records. No-op - // on a target with no write-set. - crate::control::server::wal_dispatch::mint_dispatch_local_redo( - state - .wal - .appender(crate::wal::manager::NO_APPLY_KEY) - // `dispatch_local` runs the write as a client write. - .with_event_source(crate::event::EventSource::User), - tenant_id, - database_id, - collection, - &resp, - )?; - return Ok(resp); - }; + let proposer = state.async_raft_proposer()?; // `collection` is the plan's database-qualified name. let vshard_id = @@ -84,9 +63,6 @@ pub(crate) async fn apply_orchestrated_write( read_version_lsn: write_version, write_set: Vec::new(), }; - // The proposing node handled this write exactly once, so it is - // the one node that publishes the CDC change event. - publish_origin_change_events(state, tenant_id, database_id, &plan, &response); Ok(response) } Err(crate::Error::DataPlane(code)) => Ok(data_plane_verdict(request_id, code)), diff --git a/nodedb/src/control/otel/receiver.rs b/nodedb/src/control/otel/receiver.rs index 2cb8b75d3..be80ee279 100644 --- a/nodedb/src/control/otel/receiver.rs +++ b/nodedb/src/control/otel/receiver.rs @@ -110,8 +110,13 @@ pub async fn receive_metrics( } let payload = lines.join("\n"); + // A stored count below the lines sent is the lines the + // resolve rejected. match ingest_ilp(&state, &identity, &peer_addr, &payload).await { - Ok(n) => accepted += n, + Ok(n) => { + accepted += n; + rejected += (lines.len() as u64).saturating_sub(n); + } Err(_) => rejected += lines.len() as u64, } } diff --git a/nodedb/src/control/pending_ddl.rs b/nodedb/src/control/pending_ddl.rs index 8ee703706..d53cf045e 100644 --- a/nodedb/src/control/pending_ddl.rs +++ b/nodedb/src/control/pending_ddl.rs @@ -2,12 +2,10 @@ //! Node-local table of in-flight `DdlPendingPropose` records. //! -//! Rebuilt entirely by metadata Raft log replay, the same way -//! `SharedState::metadata_ddl_owner` and `MetadataCache` are — never -//! persisted on its own. A record is inserted on `DdlPendingPropose` -//! apply and removed on the matching `DdlPendingFinalize` / -//! `DdlPendingCancel` apply (see -//! `control::cluster::metadata_applier::pending_ddl`). +//! A record is inserted on `DdlPendingPropose` apply and removed on the +//! matching `DdlPendingFinalize` / `DdlPendingCancel` apply (see +//! `control::cluster::metadata_applier::pending_ddl`). Each apply writes its +//! `SystemCatalog` row first, and boot seeds this table from those rows. use std::collections::HashMap; use std::sync::Mutex; @@ -66,6 +64,15 @@ impl PendingDdlTable { .remove(&token) } + /// Drop every record. A metadata snapshot install calls this before + /// loading the persisted records. + pub fn clear(&self) { + self.records + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .clear(); + } + /// True when a pending record exists for `token`. pub fn contains(&self, token: u64) -> bool { self.records diff --git a/nodedb/src/control/pitr/chunk_gc.rs b/nodedb/src/control/pitr/chunk_gc.rs new file mode 100644 index 000000000..2b6302ab9 --- /dev/null +++ b/nodedb/src/control/pitr/chunk_gc.rs @@ -0,0 +1,108 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Garbage collection of snapshot chunks that no kept manifest lists. +//! +//! Bases share chunks, so deleting a base deletes only its manifest. A chunk +//! goes once no pin names it: neither a kept base nor a base in progress. + +use std::sync::Arc; + +use object_store::ObjectStore; +use tracing::info; + +use super::pins::ColdPins; +use crate::storage::snapshot_writer::{delete_chunk, list_chunk_ids}; + +/// Delete every chunk in `store` that `pins` does not pin. Returns the number +/// deleted. +/// +/// Each delete holds the read lock across its pin check and the delete. A +/// base pins its ids under the write lock before it checks the store, so a +/// chunk it relies on is either pinned first or deleted before it looks. +pub async fn collect_unreferenced_chunks( + store: &Arc, + pins: &tokio::sync::RwLock, +) -> crate::Result { + let mut collected = 0; + for id in list_chunk_ids(store).await? { + let guard = pins.read().await; + if guard.is_pinned(&id) { + continue; + } + delete_chunk(store, &id).await?; + drop(guard); + collected += 1; + } + if collected > 0 { + info!(collected, "snapshot chunks no kept base lists collected"); + } + Ok(collected) +} + +#[cfg(test)] +mod tests { + use object_store::memory::InMemory; + use object_store::{ObjectStoreExt, PutPayload}; + + use super::*; + use crate::control::pitr::pins::pending_pin_name; + use crate::storage::snapshot_writer::chunk_path; + + fn id(digit: char) -> String { + digit.to_string().repeat(64) + } + + async fn store_with(ids: &[String]) -> Arc { + let store: Arc = Arc::new(InMemory::new()); + for id in ids { + store + .put(&chunk_path(id), PutPayload::from_static(b"chunk")) + .await + .unwrap(); + } + store + } + + async fn stored(store: &Arc) -> Vec { + let mut ids = list_chunk_ids(store).await.unwrap(); + ids.sort(); + ids + } + + #[tokio::test] + async fn a_chunk_goes_only_when_no_kept_base_lists_it() { + let (shared, old, new) = (id('a'), id('b'), id('c')); + let store = store_with(&[shared.clone(), old.clone(), new.clone()]).await; + let kept = vec![shared.clone(), new.clone()]; + let pins = + tokio::sync::RwLock::new(ColdPins::from_bases([("snap-2", kept.as_slice())], true)); + assert_eq!(collect_unreferenced_chunks(&store, &pins).await.unwrap(), 1); + assert_eq!(stored(&store).await, [shared, new]); + } + + #[tokio::test] + async fn a_chunk_of_a_base_in_progress_survives() { + let pending_id = id('d'); + let store = store_with(std::slice::from_ref(&pending_id)).await; + let pins = tokio::sync::RwLock::new(ColdPins::default()); + let pending = pending_pin_name(); + pins.write().await.pin(&pending, vec![pending_id.clone()]); + // Retention rebuilds the kept pins while the base is still written. + pins.write().await.rebuild([], true); + + assert_eq!(collect_unreferenced_chunks(&store, &pins).await.unwrap(), 0); + assert_eq!(stored(&store).await, [pending_id]); + + pins.write().await.unpin(&pending); + assert_eq!(collect_unreferenced_chunks(&store, &pins).await.unwrap(), 1); + assert!(stored(&store).await.is_empty()); + } + + #[tokio::test] + async fn incomplete_pins_collect_nothing() { + let store = store_with(&[id('e')]).await; + let pins = tokio::sync::RwLock::new(ColdPins::from_bases([], false)); + assert_eq!(collect_unreferenced_chunks(&store, &pins).await.unwrap(), 0); + assert_eq!(stored(&store).await.len(), 1); + } +} diff --git a/nodedb/src/control/pitr/cycle.rs b/nodedb/src/control/pitr/cycle.rs new file mode 100644 index 000000000..bde8e758d --- /dev/null +++ b/nodedb/src/control/pitr/cycle.rs @@ -0,0 +1,661 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One base snapshot run after the capture: write the base, apply retention, +//! garbage-collect unlisted chunks and archived WAL, and build the new +//! catalog. + +use std::num::NonZeroUsize; +use std::sync::Arc; + +use object_store::ObjectStore; +use tracing::warn; + +use super::chunk_gc::collect_unreferenced_chunks; +use super::node_life::NodeLife; +use super::pins::pending_pin_name; +use super::retention::{ListedBase, list_snapshots, plan_retention, wal_floor}; +use super::state::PitrState; +use super::wal_gc::collect_archived_wal; +use crate::data::snapshot::NodeSnapshot; +use crate::storage::cold::ColdStorage; +use crate::storage::snapshot::{SnapshotCatalog, SnapshotMeta}; +use crate::storage::snapshot_writer::{ + BasePlan, CdcParams, ChunkUploads, WrittenBase, delete_snapshot, list_cold_segments, + write_base_snapshot, +}; + +/// Everything a run needs besides the captured images. +pub struct BaseCycle<'a> { + /// The snapshot store root. Bases go under the node life directory. + pub root: &'a Arc, + /// Cold storage: the WAL archive and the tiered segments a base relies + /// on. `None` references no segment and skips WAL garbage collection. + pub cold: Option<&'a ColdStorage>, + pub life: &'a NodeLife, + /// Holds the cold and chunk pins the run installs and releases, and the + /// catalog that names the parent of the new base. + pub state: &'a PitrState, + pub encryption_key: &'a nodedb_wal::crypto::WalEncryptionKey, + pub retention: NonZeroUsize, + pub chunk_params: CdcParams, +} + +/// What one run did. +#[derive(Debug)] +pub struct CycleOutcome { + pub base: SnapshotMeta, + pub uploads: ChunkUploads, + /// The catalog after retention. `None` when the listing failed: the + /// caller then adds `base` to the catalog it holds. + pub catalog: Option, + pub deleted_bases: usize, + pub collected_chunks: u64, + pub collected_segments: u64, + /// The first retention or garbage-collection error. The base itself is + /// written. + pub cleanup_error: Option, +} + +impl BaseCycle<'_> { + /// Write a base from `cores` and `node`, then apply retention and + /// garbage collection. + /// + /// Archived WAL is collected only below the floor of every base still in + /// the store, a base whose delete failed included. Neither WAL nor chunks + /// are collected while a snapshot prefix holds a manifest that does not + /// load: that prefix can be a base still needing them. + /// + /// The cold keys and chunk ids of a base are pinned before the base + /// relies on them, and released only when retention deletes it. + pub async fn run( + &self, + cores: Vec<(usize, Vec)>, + node: NodeSnapshot, + ) -> crate::Result { + let store = self.life.snapshot_store(self.root); + let pending = pending_pin_name(); + let written = self.write_pinned(&store, &pending, cores, node).await; + let Pinned { + written, + cold_segments, + chunk_ids, + } = { + let mut cold_pins = self.state.cold_pins().write().await; + let mut chunk_pins = self.state.chunk_pins().write().await; + match written { + Ok(pinned) => { + cold_pins.rename(&pending, &pinned.written.prefix); + chunk_pins.rename(&pending, &pinned.written.prefix); + pinned + } + Err(e) => { + cold_pins.unpin(&pending); + chunk_pins.unpin(&pending); + return Err(e); + } + } + }; + let WrittenBase { + meta: base, + prefix, + uploads, + } = written; + let mut outcome = CycleOutcome { + base: base.clone(), + uploads, + catalog: None, + deleted_bases: 0, + collected_chunks: 0, + collected_segments: 0, + cleanup_error: None, + }; + + let listed = match list_snapshots(&store, self.encryption_key).await { + Ok(listed) => listed, + Err(e) => { + outcome.cleanup_error = Some(e); + return Ok(outcome); + } + }; + let mut bases = listed.bases; + if !bases.iter().any(|listed_base| listed_base.prefix == prefix) { + bases.push(ListedBase { + prefix: prefix.clone(), + meta: base, + cold_segments, + chunk_ids, + }); + } + + let plan = plan_retention(bases, &prefix, self.retention); + let mut remaining = plan.keep; + for old in plan.delete { + match delete_snapshot(&store, &old.prefix).await { + Ok(()) => { + self.state.cold_pins().write().await.unpin(&old.prefix); + self.state.chunk_pins().write().await.unpin(&old.prefix); + outcome.deleted_bases += 1; + } + Err(e) => { + warn!(prefix = %old.prefix, error = %e, "expired base snapshot not deleted"); + remaining.push(old); + keep_first(&mut outcome.cleanup_error, e); + } + } + } + let complete = listed.unreadable.is_empty(); + self.state.cold_pins().write().await.rebuild( + remaining + .iter() + .map(|kept| (kept.prefix.as_str(), kept.cold_segments.as_slice())), + complete, + ); + self.state.chunk_pins().write().await.rebuild( + remaining + .iter() + .map(|kept| (kept.prefix.as_str(), kept.chunk_ids.as_slice())), + complete, + ); + let mut remaining: Vec = + remaining.into_iter().map(|kept| kept.meta).collect(); + + if !complete { + let e = crate::Error::Storage { + engine: "snapshot".into(), + detail: format!( + "archived WAL and snapshot chunks kept: snapshot prefixes {:?} hold a \ + manifest that does not load", + listed.unreadable + ), + }; + keep_first(&mut outcome.cleanup_error, e); + } else { + match collect_unreferenced_chunks(&store, self.state.chunk_pins()).await { + Ok(collected) => outcome.collected_chunks = collected, + Err(e) => keep_first(&mut outcome.cleanup_error, e), + } + if let (Some(cold), Some(floor)) = (self.cold, wal_floor(&remaining)) { + match collect_archived_wal(cold, self.life, floor).await { + Ok(collected) => outcome.collected_segments = collected, + Err(e) => keep_first(&mut outcome.cleanup_error, e), + } + } + } + + remaining.sort_by_key(|meta| (meta.applied_high_lsn, meta.created_at_us)); + let mut catalog = SnapshotCatalog::new(); + for meta in remaining { + catalog.add(meta); + } + outcome.catalog = Some(catalog); + Ok(outcome) + } + + /// Pin the cold keys, plan the base, pin its chunk ids, and write it. + /// Every pin is held under `pending`, and the caller moves or releases + /// it, on error too. + async fn write_pinned( + &self, + store: &Arc, + pending: &str, + cores: Vec<(usize, Vec)>, + node: NodeSnapshot, + ) -> crate::Result { + // Listed under the write lock, so no cold delete lands between the + // listing and the pin. + let cold_segments = { + let mut pins = self.state.cold_pins().write().await; + let keys = match self.cold { + Some(cold) => list_cold_segments(&cold.object_store(), cold.prefix()).await?, + None => Vec::new(), + }; + pins.pin(pending, keys.clone()); + keys + }; + let key = self.encryption_key.clone(); + let params = self.chunk_params; + let plan = tokio::task::spawn_blocking(move || { + BasePlan::new(cores, &node, cold_segments, &key, params) + }) + .await + .map_err(|e| crate::Error::Internal { + detail: format!("base snapshot planning did not finish: {e}"), + })??; + let chunk_ids = plan.chunk_ids(); + // Pinned before the store is checked, so no chunk this base finds + // present is collected before its manifest lands. + self.state + .chunk_pins() + .write() + .await + .pin(pending, chunk_ids.clone()); + let parent = newest_base(&self.state.catalog()); + let written = write_base_snapshot( + store, + &plan, + parent, + &self.life.node_name(), + self.encryption_key, + ) + .await?; + Ok(Pinned { + written, + cold_segments: plan.cold_segments().to_vec(), + chunk_ids, + }) + } +} + +/// A written base with the keys it pinned. +struct Pinned { + written: WrittenBase, + cold_segments: Vec, + chunk_ids: Vec, +} + +/// The id of the newest base in `catalog`. +fn newest_base(catalog: &SnapshotCatalog) -> Option { + catalog + .all() + .iter() + .max_by_key(|meta| (meta.applied_high_lsn, meta.created_at_us)) + .map(|meta| meta.snapshot_id) +} + +fn keep_first(slot: &mut Option, error: crate::Error) { + if slot.is_none() { + *slot = Some(error); + } +} + +#[cfg(test)] +mod tests { + use std::path::Path; + + use object_store::memory::InMemory; + + use super::*; + use crate::control::security::catalog::SystemCatalog; + use crate::data::snapshot::{CoreSnapshot, SnapshotComponent, SnapshotFile}; + use crate::storage::cold::ColdStorageConfig; + use crate::storage::snapshot_node::capture_node_state; + use crate::storage::snapshot_writer::{list_chunk_ids, rebuild_catalog}; + use crate::types::Lsn; + use crate::types::replay_stamp::ReplayStamp; + + fn key() -> nodedb_wal::crypto::WalEncryptionKey { + nodedb_wal::crypto::WalEncryptionKey::from_bytes(&[0x5A; 32]).unwrap() + } + + /// Small enough that a test image spans many chunks. + const SMALL_CHUNKS: CdcParams = CdcParams { + min: 64, + max: 1024, + mask_bits: 7, + }; + + /// One core image whose files hold every record through `floor`. + fn cores(floor: u64) -> Vec<(usize, Vec)> { + cores_with(floor, Vec::new()) + } + + /// One core image holding `bytes` as a file. + fn cores_with(floor: u64, bytes: Vec) -> Vec<(usize, Vec)> { + let snapshot = CoreSnapshot { + stamp: ReplayStamp::through(floor), + files: vec![SnapshotFile { + component: SnapshotComponent::Kv, + path: "kv-ckpt/core-0/data".into(), + bytes, + }], + dirs: Vec::new(), + }; + vec![(0, snapshot.to_bytes().unwrap())] + } + + fn noise(len: usize, seed: u64) -> Vec { + let mut state = seed | 1; + (0..len) + .map(|_| { + state ^= state << 13; + state ^= state >> 7; + state ^= state << 17; + (state >> 24) as u8 + }) + .collect() + } + + struct Fixture { + data_dir: tempfile::TempDir, + cold_dir: tempfile::TempDir, + root: Arc, + cold: ColdStorage, + life: NodeLife, + pitr: PitrState, + key: nodedb_wal::crypto::WalEncryptionKey, + } + + impl Fixture { + async fn new() -> Self { + let data_dir = tempfile::tempdir().unwrap(); + let cold_dir = tempfile::tempdir().unwrap(); + let cold = ColdStorage::new(ColdStorageConfig { + local_dir: Some(cold_dir.path().to_path_buf()), + ..Default::default() + }) + .unwrap(); + let life = NodeLife::resolve(3, data_dir.path().to_path_buf()) + .await + .unwrap(); + Self { + data_dir, + cold_dir, + root: Arc::new(InMemory::new()), + cold, + life, + pitr: PitrState::default(), + key: key(), + } + } + + fn cycle(&self, retention: usize) -> BaseCycle<'_> { + BaseCycle { + root: &self.root, + cold: Some(&self.cold), + life: &self.life, + state: &self.pitr, + encryption_key: &self.key, + retention: NonZeroUsize::new(retention).unwrap(), + chunk_params: SMALL_CHUNKS, + } + } + + fn node_image(&self) -> NodeSnapshot { + let system = SystemCatalog::open(&self.data_dir.path().join("system.redb")).unwrap(); + capture_node_state(self.data_dir.path(), &system, None, &[]).unwrap() + } + + /// Archive a segment starting at each LSN, with its checksum marker. + async fn archive_segments(&self, first_lsns: &[u64]) { + let wal_dir = self.data_dir.path().join("wal"); + std::fs::create_dir_all(&wal_dir).unwrap(); + for &first_lsn in first_lsns { + let path = nodedb_wal::segment::segment_path(&wal_dir, first_lsn); + std::fs::write(&path, first_lsn.to_le_bytes()).unwrap(); + self.cold + .upload_wal_segment( + &path, + self.life.node_id, + self.life.incarnation.as_str(), + first_lsn, + &[], + ) + .await + .unwrap(); + } + } + + async fn archived(&self) -> Vec { + let remote = self + .cold + .archived_wal_segments(self.life.node_id, self.life.incarnation.as_str(), 0) + .await + .unwrap(); + let mut lsns: Vec = remote + .into_iter() + .filter(|(_, seg)| seg.size.is_some() && !seg.crc32c.is_empty()) + .map(|(lsn, _)| lsn) + .collect(); + lsns.sort_unstable(); + lsns + } + + async fn rebuilt(&self) -> SnapshotCatalog { + rebuild_catalog(&self.life.snapshot_store(&self.root), &self.key).await + } + + fn cold_dir(&self) -> &Path { + self.cold_dir.path() + } + } + + fn begin_lsns(catalog: &SnapshotCatalog) -> Vec { + catalog.all().iter().map(|m| m.begin_lsn.as_u64()).collect() + } + + #[tokio::test] + async fn a_run_writes_a_base_the_rebuilt_catalog_lists() { + let fx = Fixture::new().await; + let outcome = fx.cycle(2).run(cores(40), fx.node_image()).await.unwrap(); + assert!( + outcome.cleanup_error.is_none(), + "{:?}", + outcome.cleanup_error + ); + assert_eq!(outcome.base.begin_lsn, Lsn::new(40)); + assert_eq!(begin_lsns(&outcome.catalog.unwrap()), [40]); + + let rebuilt = fx.rebuilt().await; + assert_eq!(rebuilt.all(), std::slice::from_ref(&outcome.base)); + assert!(rebuilt.find_base(Lsn::new(40)).is_some()); + // The base lives under the node life directory, not at the root. + assert!(rebuild_catalog(&fx.root, &fx.key).await.is_empty()); + } + + #[tokio::test] + async fn retention_keeps_n_bases_and_deletes_older_ones() { + let fx = Fixture::new().await; + for floor in [10, 20, 30] { + fx.cycle(2) + .run(cores(floor), fx.node_image()) + .await + .unwrap(); + } + let outcome = fx.cycle(2).run(cores(40), fx.node_image()).await.unwrap(); + assert!( + outcome.cleanup_error.is_none(), + "{:?}", + outcome.cleanup_error + ); + assert_eq!(outcome.deleted_bases, 1); + assert_eq!(begin_lsns(&outcome.catalog.unwrap()), [30, 40]); + assert_eq!(begin_lsns(&fx.rebuilt().await), [30, 40]); + } + + #[tokio::test] + async fn the_last_base_is_never_deleted() { + let fx = Fixture::new().await; + for floor in [10, 20] { + let outcome = fx + .cycle(1) + .run(cores(floor), fx.node_image()) + .await + .unwrap(); + assert!( + outcome.cleanup_error.is_none(), + "{:?}", + outcome.cleanup_error + ); + assert_eq!(begin_lsns(&outcome.catalog.unwrap()), [floor]); + } + assert_eq!(begin_lsns(&fx.rebuilt().await), [20]); + } + + #[tokio::test] + async fn the_wal_chain_from_the_oldest_kept_base_stays_continuous() { + let fx = Fixture::new().await; + fx.archive_segments(&[1, 100, 200, 300, 400]).await; + for floor in [150, 250] { + fx.cycle(1) + .run(cores(floor), fx.node_image()) + .await + .unwrap(); + } + // The oldest kept base replays from 250, inside the segment at 200. + let left = fx.archived().await; + assert_eq!(left, [200, 300, 400]); + assert!(left[0] <= 250, "the segment holding the floor is kept"); + + // Every object below the floor is gone, markers included. + let node_dir = fx + .cold_dir() + .join(crate::wal::archiver::wal_archive_node_prefix( + "data/", + fx.life.node_id, + fx.life.incarnation.as_str(), + )); + let mut names: Vec = std::fs::read_dir(&node_dir) + .unwrap() + .map(|entry| entry.unwrap().file_name().to_string_lossy().into_owned()) + .collect(); + names.sort(); + assert_eq!( + names.len(), + 6, + "three segments and their markers: {names:?}" + ); + assert!( + names[0].starts_with("wal-00000000000000000200.seg"), + "{names:?}" + ); + } + + #[tokio::test] + async fn an_abandoned_write_is_removed_and_does_not_block_collection() { + let fx = Fixture::new().await; + fx.archive_segments(&[1, 100, 200]).await; + use object_store::ObjectStoreExt; + let store = fx.life.snapshot_store(&fx.root); + store + .put( + &object_store::path::Path::from("snap-000999-lsn00000000000000000001/core-0.snap"), + object_store::PutPayload::from_static(b"partial"), + ) + .await + .unwrap(); + let outcome = fx.cycle(1).run(cores(150), fx.node_image()).await.unwrap(); + assert!( + outcome.cleanup_error.is_none(), + "{:?}", + outcome.cleanup_error + ); + assert_eq!(fx.archived().await, [100, 200]); + } + + /// Put a cold-tier segment object under the cold store's `segments/`. + async fn put_cold_segment(fx: &Fixture, name: &str) -> String { + use object_store::ObjectStoreExt; + let key = format!("{}segments/{name}", fx.cold.prefix()); + fx.cold + .object_store() + .put( + &object_store::path::Path::from(key.as_str()), + object_store::PutPayload::from_static(b"seg"), + ) + .await + .unwrap(); + key + } + + #[tokio::test] + async fn a_base_pins_its_cold_segments_until_retention_retires_it() { + use object_store::ObjectStoreExt; + let fx = Fixture::new().await; + let key = put_cold_segment(&fx, "a.seg").await; + fx.cycle(1).run(cores(10), fx.node_image()).await.unwrap(); + assert!(fx.pitr.cold_pins().read().await.is_pinned(&key)); + + // The segment leaves the store, so the next base does not reference it. + fx.cold + .object_store() + .delete(&object_store::path::Path::from(key.as_str())) + .await + .unwrap(); + let outcome = fx.cycle(1).run(cores(20), fx.node_image()).await.unwrap(); + assert_eq!(outcome.deleted_bases, 1); + assert!(!fx.pitr.cold_pins().read().await.is_pinned(&key)); + } + + impl Fixture { + /// Run one cycle, then install its catalog as the base task does. + async fn run(&self, retention: usize, cores: Vec<(usize, Vec)>) -> CycleOutcome { + let outcome = self + .cycle(retention) + .run(cores, self.node_image()) + .await + .unwrap(); + assert!( + outcome.cleanup_error.is_none(), + "{:?}", + outcome.cleanup_error + ); + if let Some(catalog) = &outcome.catalog { + self.pitr.replace_catalog(catalog.clone(), None); + } + outcome + } + + async fn stored_chunks(&self) -> std::collections::BTreeSet { + let store = self.life.snapshot_store(&self.root); + list_chunk_ids(&store).await.unwrap().into_iter().collect() + } + + async fn listed_chunks(&self) -> std::collections::BTreeSet { + let store = self.life.snapshot_store(&self.root); + let listed = list_snapshots(&store, &self.key).await.unwrap(); + listed + .bases + .into_iter() + .flat_map(|base| base.chunk_ids) + .collect() + } + } + + #[tokio::test] + async fn retention_deletes_a_chunk_only_when_no_kept_base_lists_it() { + let fx = Fixture::new().await; + let mut data = noise(30_000, 3); + let first = fx.run(1, cores_with(10, data.clone())).await; + let first_chunks = fx.listed_chunks().await; + + data[15_000] ^= 0xFF; + let second = fx.run(1, cores_with(10, data)).await; + assert_eq!(second.base.parent_id, Some(first.base.snapshot_id)); + assert!(second.uploads.reused > 0, "{:?}", second.uploads); + assert_eq!(second.deleted_bases, 1); + assert!(second.collected_chunks > 0); + + // The store holds exactly the chunks the kept base lists: the shared + // ones stayed, the ones only the deleted base listed went. + let kept = fx.listed_chunks().await; + assert_eq!(fx.stored_chunks().await, kept); + assert!(first_chunks.intersection(&kept).count() > 0); + assert!(first_chunks.difference(&kept).count() > 0); + } + + #[tokio::test] + async fn a_chunk_listed_by_a_base_in_progress_survives_retention() { + let fx = Fixture::new().await; + fx.run(1, cores_with(10, noise(8_000, 5))).await; + let in_progress = fx.listed_chunks().await; + let pending = pending_pin_name(); + fx.pitr + .chunk_pins() + .write() + .await + .pin(&pending, in_progress.iter().cloned().collect()); + + let second = fx.run(1, cores_with(20, noise(8_000, 6))).await; + assert_eq!(second.deleted_bases, 1); + assert!(fx.stored_chunks().await.is_superset(&in_progress)); + + fx.pitr.chunk_pins().write().await.unpin(&pending); + fx.run(1, cores_with(30, noise(8_000, 6))).await; + let kept = fx.listed_chunks().await; + assert_eq!(fx.stored_chunks().await, kept); + assert!( + in_progress.difference(&kept).count() > 0, + "released chunks went" + ); + } +} diff --git a/nodedb/src/control/pitr/mod.rs b/nodedb/src/control/pitr/mod.rs new file mode 100644 index 000000000..4c9b745a0 --- /dev/null +++ b/nodedb/src/control/pitr/mod.rs @@ -0,0 +1,23 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod chunk_gc; +pub mod cycle; +pub mod node_life; +pub mod on_demand; +pub mod pins; +pub mod raft_archive; +pub mod restore_point; +pub mod restore_seal; +pub mod retention; +pub mod state; +pub mod task; +pub mod wal_gc; + +pub use cycle::{BaseCycle, CycleOutcome}; +pub use node_life::NodeLife; +pub use on_demand::{archive_wal_now, force_base_after_install, take_base_now}; +pub use pins::ColdPins; +pub use raft_archive::spawn_metadata_log_archiver; +pub use restore_seal::seal_restored_generation; +pub use state::{PitrFailure, PitrState}; +pub use task::{spawn_base_snapshot_task, wire_pitr}; diff --git a/nodedb/src/control/pitr/node_life.rs b/nodedb/src/control/pitr/node_life.rs new file mode 100644 index 000000000..bf81b488f --- /dev/null +++ b/nodedb/src/control/pitr/node_life.rs @@ -0,0 +1,85 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One node life: the node id plus the incarnation the WAL archiver keys on. +//! Base snapshots and archived WAL of one life share the pair, so a wiped data +//! directory never mixes its bases with an earlier life's WAL. + +use std::path::PathBuf; +use std::sync::Arc; + +use object_store::ObjectStore; +use object_store::prefix::PrefixStore; + +use crate::wal::archiver::{Incarnation, load_or_mint_incarnation}; + +/// Directory of one life's base snapshots in the snapshot store: +/// `{node_id}/{incarnation}`. +pub fn snapshot_dir(node_id: u64, incarnation: &str) -> String { + format!("{node_id}/{incarnation}") +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct NodeLife { + pub node_id: u64, + pub incarnation: Incarnation, +} + +impl NodeLife { + /// Read the incarnation the archiver uses, or mint it. Runs the file I/O + /// on a blocking thread. + pub async fn resolve(node_id: u64, data_dir: PathBuf) -> crate::Result { + let incarnation = tokio::task::spawn_blocking(move || load_or_mint_incarnation(&data_dir)) + .await + .map_err(|e| crate::Error::Internal { + detail: format!("node incarnation read did not finish: {e}"), + })??; + Ok(Self { + node_id, + incarnation, + }) + } + + /// Directory of this life's base snapshots. + pub fn snapshot_dir(&self) -> String { + snapshot_dir(self.node_id, self.incarnation.as_str()) + } + + /// `root` scoped to [`Self::snapshot_dir`]. + pub fn snapshot_store(&self, root: &Arc) -> Arc { + Arc::new(PrefixStore::new(Arc::clone(root), self.snapshot_dir())) + } + + /// The `created_by` name stamped on this life's bases. + pub fn node_name(&self) -> String { + format!("node-{}", self.node_id) + } +} + +#[cfg(test)] +mod tests { + use object_store::memory::InMemory; + use object_store::path::Path as ObjectPath; + use object_store::{ObjectStoreExt, PutPayload}; + + use super::*; + + #[tokio::test] + async fn the_snapshot_store_writes_under_the_node_life_directory() { + let dir = tempfile::tempdir().unwrap(); + let life = NodeLife::resolve(7, dir.path().to_path_buf()) + .await + .unwrap(); + let again = NodeLife::resolve(7, dir.path().to_path_buf()) + .await + .unwrap(); + assert_eq!(life, again, "the incarnation is stable across calls"); + + let root: Arc = Arc::new(InMemory::new()); + life.snapshot_store(&root) + .put(&ObjectPath::from("snap/x"), PutPayload::from_static(b"1")) + .await + .unwrap(); + let full = format!("7/{}/snap/x", life.incarnation.as_str()); + assert!(root.head(&ObjectPath::from(full)).await.is_ok()); + } +} diff --git a/nodedb/src/control/pitr/on_demand.rs b/nodedb/src/control/pitr/on_demand.rs new file mode 100644 index 000000000..7fd530ae6 --- /dev/null +++ b/nodedb/src/control/pitr/on_demand.rs @@ -0,0 +1,80 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! PITR work run on demand, outside its schedule: one base snapshot, and one +//! WAL archive pass. + +use std::sync::Arc; +use std::time::Duration; + +use crate::config::server::PitrSettings; +use crate::control::state::SharedState; +use crate::storage::snapshot::SnapshotMeta; +use crate::wal::archiver::WalArchiver; + +/// The time an on-demand base can take. +const ON_DEMAND_BASE_DEADLINE: Duration = Duration::from_secs(600); + +/// Take one base snapshot now, with the default retention. +pub async fn take_base_now(state: &Arc) -> crate::Result { + let life = state + .pitr + .life() + .cloned() + .ok_or_else(|| crate::Error::Config { + detail: "a base snapshot needs PITR wired at boot; set pitr.enabled = true".into(), + })?; + let retention = PitrSettings::default().retention()?; + super::task::run_once(state, &life, retention, ON_DEMAND_BASE_DEADLINE).await +} + +/// Take a base once the gateway is open, because a Raft snapshot install +/// replaced rows no WAL record carries. A restore to a target at or after the +/// install starts from a base taken after it. Does nothing with PITR off. +pub fn force_base_after_install(state: &Arc) { + if state.pitr.life().is_none() { + return; + } + let state = Arc::clone(state); + tokio::spawn(async move { + let phase = crate::control::startup::StartupPhase::GatewayEnable; + if state.startup.await_phase(phase).await.is_err() { + return; + } + if let Err(error) = take_base_now(&state).await { + tracing::warn!( + %error, + "the base after a Raft snapshot install failed; a restore to a later target \ + waits for the next scheduled base" + ); + } + }); +} + +/// Seal the active WAL segment and upload every sealed segment the archive +/// lacks. Returns whether the pass uploaded everything it found sealed. +pub async fn archive_wal_now(state: &SharedState) -> crate::Result { + let cold = state + .cold_storage + .clone() + .ok_or_else(|| crate::Error::Config { + detail: "the WAL archive needs [cold_storage]".into(), + })?; + state.wal.seal_active_segment()?; + let mut archiver = WalArchiver::new( + state.node_id, + state.data_dir.clone(), + cold, + state.system_metrics.clone(), + ); + let Some(listed) = archiver.tick(&state.wal).await else { + return Ok(false); + }; + let cursor = archiver.cursor(); + Ok(cursor.is_some_and(|cursor| { + listed + .segments + .iter() + .filter(|seg| seg.first_lsn < listed.active_first_lsn) + .all(|seg| cursor.is_archived(seg.first_lsn)) + })) +} diff --git a/nodedb/src/control/pitr/pins.rs b/nodedb/src/control/pitr/pins.rs new file mode 100644 index 000000000..65333055a --- /dev/null +++ b/nodedb/src/control/pitr/pins.rs @@ -0,0 +1,176 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Object keys pinned by kept base snapshots. +//! +//! One pin set holds the cold-store keys the bases reference, another the +//! chunk ids they list. A deleter checks the pins first and skips a pinned +//! key. It holds the read lock of the set across the check and the delete, +//! and a base pins its keys under the write lock before it relies on them, +//! so no key is deleted between the two. + +use std::collections::HashMap; +use std::collections::hash_map::Entry; +use std::sync::atomic::{AtomicU64, Ordering}; + +/// Prefix of the name a base still being written pins its keys under. A real +/// base prefix starts with `snap-`. +pub const PENDING_BASE: &str = "pending"; + +/// A pin name no other base in progress holds. +pub fn pending_pin_name() -> String { + static NEXT: AtomicU64 = AtomicU64::new(0); + format!("{PENDING_BASE}-{}", NEXT.fetch_add(1, Ordering::Relaxed)) +} + +#[derive(Debug)] +pub struct ColdPins { + /// Keys each base references, by base prefix. + by_base: HashMap>, + /// How many bases reference each key. + counts: HashMap, + /// `false` while a snapshot prefix's manifest does not load. Its keys are + /// unknown, so every key counts as pinned. + complete: bool, +} + +impl Default for ColdPins { + fn default() -> Self { + Self { + by_base: HashMap::new(), + counts: HashMap::new(), + complete: true, + } + } +} + +impl ColdPins { + /// Pins built from every base, as `(prefix, cold keys)`. + pub fn from_bases<'a>( + bases: impl IntoIterator, + complete: bool, + ) -> Self { + let mut pins = Self { + complete, + ..Self::default() + }; + for (base, keys) in bases { + pins.pin(base, keys.to_vec()); + } + pins + } + + /// Replace every kept base's pins with `bases`. Pins of bases still being + /// written stay: a rebuild from the store cannot see them yet. + pub fn rebuild<'a>( + &mut self, + bases: impl IntoIterator, + complete: bool, + ) { + let mut rebuilt = Self::from_bases(bases, complete); + for (base, keys) in self.by_base.drain() { + if base.starts_with(PENDING_BASE) { + rebuilt.pin(&base, keys); + } + } + *self = rebuilt; + } + + /// Pin `keys` for `base`, replacing what `base` pinned before. + pub fn pin(&mut self, base: &str, keys: Vec) { + self.unpin(base); + for key in &keys { + *self.counts.entry(key.clone()).or_default() += 1; + } + self.by_base.insert(base.to_owned(), keys); + } + + /// Release every key `base` pinned. + pub fn unpin(&mut self, base: &str) { + let Some(keys) = self.by_base.remove(base) else { + return; + }; + for key in keys { + if let Entry::Occupied(mut count) = self.counts.entry(key) { + *count.get_mut() -= 1; + if *count.get() == 0 { + count.remove(); + } + } + } + } + + /// Move the pins of `from` to `to`. + pub fn rename(&mut self, from: &str, to: &str) { + if let Some(keys) = self.by_base.remove(from) { + self.by_base.insert(to.to_owned(), keys); + } + } + + pub fn is_pinned(&self, key: &str) -> bool { + !self.complete || self.counts.contains_key(key) + } + + /// Keys pinned by any base. + pub fn len(&self) -> usize { + self.counts.len() + } + + pub fn is_empty(&self) -> bool { + self.counts.is_empty() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn keys(names: &[&str]) -> Vec { + names.iter().map(|n| n.to_string()).collect() + } + + #[test] + fn a_key_stays_pinned_until_its_last_base_unpins() { + let mut pins = ColdPins::default(); + pins.pin("snap-1", keys(&["a", "b"])); + pins.pin("snap-2", keys(&["b"])); + pins.unpin("snap-1"); + assert!(!pins.is_pinned("a")); + assert!(pins.is_pinned("b")); + pins.unpin("snap-2"); + assert!(pins.is_empty()); + } + + #[test] + fn a_pending_base_keeps_its_keys_through_the_rename() { + let mut pins = ColdPins::default(); + pins.pin(PENDING_BASE, keys(&["a"])); + pins.rename(PENDING_BASE, "snap-9"); + assert!(pins.is_pinned("a")); + pins.unpin(PENDING_BASE); + assert!(pins.is_pinned("a")); + pins.unpin("snap-9"); + assert!(!pins.is_pinned("a")); + } + + #[test] + fn a_rebuild_keeps_the_pins_of_a_base_in_progress() { + let mut pins = ColdPins::default(); + let pending = pending_pin_name(); + assert_ne!(pending, pending_pin_name()); + pins.pin(&pending, keys(&["new"])); + pins.pin("snap-1", keys(&["old"])); + let kept = keys(&["kept"]); + pins.rebuild([("snap-2", kept.as_slice())], true); + assert!(pins.is_pinned("new")); + assert!(pins.is_pinned("kept")); + assert!(!pins.is_pinned("old")); + pins.unpin(&pending); + assert!(!pins.is_pinned("new")); + } + + #[test] + fn incomplete_pins_pin_every_key() { + let pins = ColdPins::from_bases([], false); + assert!(pins.is_pinned("anything")); + } +} diff --git a/nodedb/src/control/pitr/raft_archive.rs b/nodedb/src/control/pitr/raft_archive.rs new file mode 100644 index 000000000..34a7bf4be --- /dev/null +++ b/nodedb/src/control/pitr/raft_archive.rs @@ -0,0 +1,302 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The metadata Raft log archiver. +//! +//! A restore rebuilds the metadata group from a base's catalog images plus +//! the metadata log entries after them. The log compacts, so every committed +//! entry is copied to cold storage first: the archiver sets the group's +//! compaction ceiling to the highest index it holds, and the log never +//! compacts past it. The same rule the WAL archiver enforces for WAL +//! segments. +//! +//! The archiver writes under the metadata timeline the catalog names, and +//! reads it on every pass: a node that joins a restored cluster learns the +//! timeline from the snapshot it installs. +//! +//! Each pass records the life's frontier: the newest stamp among the entries +//! it holds. A leader stamps each entry above every stamp before it (see +//! `MultiRaft::propose_stamped_metadata`), so every entry stamped at or below +//! the frontier is archived. On a log with no recent stamp, the leader's +//! archiver proposes an `ArchiveMark` entry, so the frontier keeps up with +//! the clock. + +use std::sync::atomic::{AtomicU64, Ordering}; +use std::sync::{Arc, Mutex}; +use std::time::Duration; + +use nodedb_cluster::multi_raft::MultiRaft; +use nodedb_cluster::{METADATA_GROUP_ID, MetadataEntry, encode_entry, entry_stamp}; +use tracing::{debug, warn}; + +use crate::control::pitr::NodeLife; +use crate::control::security::credential::CredentialStore; +use crate::control::shutdown::{ShutdownPhase, spawn_loop}; +use crate::control::state::SharedState; +use crate::storage::cold::ColdStorage; +use crate::storage::raft_log_archive::{ + ArchiveFrontier, ArchiveLife, ArchivedLogChunk, ArchivedLogEntry, chunk_key, fetch_chunk, + frontier_key, list_chunks, put_chunk, put_frontier, seal_chunk, +}; + +/// How often the archiver copies new committed entries. +const ARCHIVE_INTERVAL: Duration = Duration::from_secs(10); + +/// Most entries one archive object holds. +const MAX_ENTRIES_PER_CHUNK: u64 = 4096; + +/// Copy group 0's committed log to cold storage, and bound its compaction +/// by what the archive holds. Starts nothing with PITR off, no cold storage, +/// or no WAL key: the archive is encrypted with the WAL key. +pub fn spawn_metadata_log_archiver(shared: &Arc, multi_raft: Arc>) { + let Some(life) = shared.pitr.life().cloned() else { + return; + }; + let Some(cold) = shared.cold_storage.clone() else { + warn!("the metadata log is not archived: PITR is on, and this node has no cold storage"); + return; + }; + let Some(wal_key) = shared.wal.encryption_key().cloned() else { + warn!( + "the metadata log is not archived: the archive needs the WAL key, and this node has none" + ); + return; + }; + // Nothing is archived yet, so the log compacts nothing until the first + // pass copies it. + let ceiling = Arc::new(AtomicU64::new(0)); + multi_raft + .lock() + .unwrap_or_else(|p| p.into_inner()) + .set_compaction_ceiling(METADATA_GROUP_ID, Arc::clone(&ceiling)); + let mut archiver = MetadataLogArchiver { + cold, + multi_raft, + wal_key, + ceiling, + life, + clock: Arc::clone(&shared.hlc_clock), + credentials: Arc::clone(&shared.credentials), + held: None, + }; + spawn_loop( + &shared.loop_registry, + &shared.shutdown, + "pitr_metadata_log_archive", + ShutdownPhase::DrainingControlPlane, + move |mut shutdown| async move { + loop { + if let Err(e) = archiver.pass().await { + warn!(error = %e, "metadata log archive pass failed; compaction stays held"); + } + tokio::select! { + _ = shutdown.wait_cancelled() => break, + _ = tokio::time::sleep(ARCHIVE_INTERVAL) => {} + } + } + }, + ); +} + +struct MetadataLogArchiver { + cold: Arc, + multi_raft: Arc>, + wal_key: nodedb_wal::crypto::WalEncryptionKey, + ceiling: Arc, + life: NodeLife, + /// The node HLC. The leader stamps metadata entries from it. + clock: Arc, + /// The catalog, which names the metadata timeline. + credentials: Arc, + /// What the archive of the current timeline holds. `None` until read + /// from the archive. + held: Option, +} + +/// What this life's archive of one timeline holds. +#[derive(Debug, Clone, Copy)] +struct Held { + timeline: u64, + /// Highest index archived. + through: u64, + /// Newest stamp among the archived entries. + newest_stamp: Option, + /// The frontier last written. + frontier: Option, +} + +/// What a pass reads under the `MultiRaft` lock. +struct Batch { + prev_term: u64, + entries: Vec, +} + +impl MetadataLogArchiver { + /// Archive every committed entry the archive lacks, record the frontier, + /// and on an idle log propose the entry that moves it. + async fn pass(&mut self) -> crate::Result<()> { + let timeline = self.credentials.catalog().load_metadata_timeline()?; + let incarnation = self.life.incarnation.as_str().to_string(); + let life = ArchiveLife { + timeline, + node_id: self.life.node_id, + incarnation: &incarnation, + }; + let mut held = match self.held.take() { + Some(held) if held.timeline == timeline => held, + _ => self.load_held(life).await?, + }; + let drained = self.drain(life, &mut held).await; + self.held = Some(held); + drained?; + self.mark_idle_log() + } + + /// Read what this life's archive of `life.timeline` holds. + async fn load_held(&self, life: ArchiveLife<'_>) -> crate::Result { + let store = self.cold.object_store(); + let chunks = list_chunks(&store, self.cold.prefix(), life).await?; + let Some(last) = chunks.last() else { + return Ok(Held { + timeline: life.timeline, + through: 0, + newest_stamp: None, + frontier: None, + }); + }; + let chunk = fetch_chunk(&store, &last.key, &self.wal_key).await?; + Ok(Held { + timeline: life.timeline, + through: last.last, + newest_stamp: newest_stamp(&chunk.entries), + frontier: None, + }) + } + + /// Archive every committed entry after `held.through`, then write the + /// frontier when it moved. + async fn drain(&self, life: ArchiveLife<'_>, held: &mut Held) -> crate::Result<()> { + let prefix = self.cold.prefix(); + let store = self.cold.object_store(); + self.ceiling.store(held.through, Ordering::Release); + while let Some(batch) = self.read_batch(held.through)? { + let (Some(first), Some(last)) = (batch.entries.first(), batch.entries.last()) else { + break; + }; + let (first, last) = (first.index, last.index); + let key = chunk_key(prefix, life, first, last); + let entries: Vec = batch + .entries + .into_iter() + .map(|entry| ArchivedLogEntry { + index: entry.index, + term: entry.term, + data: entry.data, + }) + .collect(); + let batch_stamp = newest_stamp(&entries); + let chunk = ArchivedLogChunk { + group_id: METADATA_GROUP_ID, + prev_term: batch.prev_term, + entries, + }; + let sealed = seal_chunk(&chunk, &key, &self.life.node_name(), &self.wal_key)?; + put_chunk(&store, &key, sealed).await?; + held.through = last; + held.newest_stamp = held.newest_stamp.max(batch_stamp); + self.ceiling.store(last, Ordering::Release); + debug!(first, last, "metadata log entries archived"); + } + let Some(stamped_through_ns) = held.newest_stamp else { + return Ok(()); + }; + let frontier = ArchiveFrontier { + through: held.through, + stamped_through_ns, + }; + if held.frontier != Some(frontier) { + let key = frontier_key(prefix, life); + put_frontier( + &store, + &key, + frontier, + &self.life.node_name(), + &self.wal_key, + ) + .await?; + held.frontier = Some(frontier); + } + Ok(()) + } + + /// On the leader of a log whose newest stamp is older than one archive + /// interval, propose an `ArchiveMark`. Its stamp moves the frontier to the + /// clock once a later pass archives it. + fn mark_idle_log(&self) -> crate::Result<()> { + let mut mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + if !mr.is_group_leader(METADATA_GROUP_ID) { + return Ok(()); + } + let now = self.clock.now().wall_ns; + let interval = u64::try_from(ARCHIVE_INTERVAL.as_nanos()).unwrap_or(u64::MAX); + if mr + .newest_metadata_stamp() + .is_some_and(|stamp| now.saturating_sub(stamp) < interval) + { + return Ok(()); + } + let bytes = + encode_entry(&MetadataEntry::ArchiveMark).map_err(|e| crate::Error::Internal { + detail: format!("encode metadata archive mark: {e}"), + })?; + mr.propose_stamped_metadata(&bytes) + .map_err(|e| crate::Error::ColdStorage { + detail: format!("propose metadata archive mark: {e}"), + })?; + Ok(()) + } + + /// The committed entries after `archived`, at most one object's worth. + /// `None` when there are none. + fn read_batch(&self, archived: u64) -> crate::Result> { + let mr = self.multi_raft.lock().unwrap_or_else(|p| p.into_inner()); + let Some(first_available) = mr.first_available_index(METADATA_GROUP_ID) else { + return Ok(None); + }; + let mut lo = archived.saturating_add(1); + if lo < first_available { + // A log that starts at a snapshot holds nothing below it: this + // life's archive starts there, and other lives hold the earlier + // entries. Entries leaving after this life archived some mean a + // snapshot install replaced them, and the archive has a hole. + if archived > 0 { + warn!( + archived_through = archived, + first_available, + "metadata log entries left the log unarchived; the archive has a hole" + ); + } + lo = first_available; + } + let Some(prev_term) = mr.log_term_at(METADATA_GROUP_ID, lo - 1) else { + return Ok(None); + }; + let hi = lo.saturating_add(MAX_ENTRIES_PER_CHUNK - 1); + let entries = mr + .read_committed_entries(METADATA_GROUP_ID, lo, hi) + .map_err(|e| crate::Error::ColdStorage { + detail: format!("read metadata log entries {lo}..={hi}: {e}"), + })?; + if entries.is_empty() { + return Ok(None); + } + Ok(Some(Batch { prev_term, entries })) + } +} + +/// The newest stamp among `entries`. +fn newest_stamp(entries: &[ArchivedLogEntry]) -> Option { + entries + .iter() + .filter_map(|entry| entry_stamp(&entry.data)) + .max() +} diff --git a/nodedb/src/control/pitr/restore_point/create.rs b/nodedb/src/control/pitr/restore_point/create.rs new file mode 100644 index 000000000..e3c7cc772 --- /dev/null +++ b/nodedb/src/control/pitr/restore_point/create.rs @@ -0,0 +1,51 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Create and list cluster restore points. + +use crate::control::metadata_proposer::propose_restore_point; +use crate::control::security::catalog::restore_points::StoredRestorePoint; +use crate::control::state::SharedState; + +#[derive(Debug, thiserror::Error)] +pub enum RestorePointError { + #[error(transparent)] + Node(#[from] crate::Error), +} + +impl From for crate::Error { + fn from(e: RestorePointError) -> Self { + match e { + RestorePointError::Node(inner) => inner, + } + } +} + +/// Create a cluster restore point at this node's HLC now, and wait until this +/// node applied it. Every node then cuts each group it hosts at the point. +pub async fn create_restore_point( + state: &SharedState, +) -> Result { + let hlc = state.hlc_clock.now().wall_ns; + // Parks the proposal after the watermark is taken until the test releases + // it: a test orders a metadata entry stamped at or above the watermark + // before the point. + #[cfg(feature = "failpoints")] + crate::control::fail_gate::wait("restore_point::after_watermark").await; + let created_at_ms = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|d| u64::try_from(d.as_millis()).unwrap_or(u64::MAX)) + .unwrap_or(0); + let id = propose_restore_point(state, hlc, created_at_ms).await?; + Ok(StoredRestorePoint { + id, + hlc, + created_at_ms, + }) +} + +/// Every restore point this node applied, oldest first. +pub fn list_restore_points( + state: &SharedState, +) -> Result, RestorePointError> { + Ok(state.credentials.catalog().list_restore_points()?) +} diff --git a/nodedb/src/control/pitr/restore_point/mod.rs b/nodedb/src/control/pitr/restore_point/mod.rs new file mode 100644 index 000000000..f9c9ceb5c --- /dev/null +++ b/nodedb/src/control/pitr/restore_point/mod.rs @@ -0,0 +1,11 @@ +// SPDX-License-Identifier: BUSL-1.1 + +pub mod create; +pub mod record; +pub mod seed; +pub mod task; + +pub use create::{RestorePointError, create_restore_point, list_restore_points}; +pub use record::{record_group_point, sequencer_hook, spawn_node_cut}; +pub use seed::{RecordedCut, load_recorded_cuts, persist_cut_floor, persist_until_durable}; +pub use task::spawn_restore_point_task; diff --git a/nodedb/src/control/pitr/restore_point/record.rs b/nodedb/src/control/pitr/restore_point/record.rs new file mode 100644 index 000000000..cbffeb47f --- /dev/null +++ b/nodedb/src/control/pitr/restore_point/record.rs @@ -0,0 +1,137 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Each node's part of a cluster restore point. +//! +//! The metadata entry that creates a point applies on every node in log +//! order. Each node then records the metadata group's place, and cuts every +//! data group it hosts, plus the Calvin sequencer, at the point's watermark. +//! Every replica of a group applies the cut barrier at the same log index and +//! records it, so the group's place at the point is that index on every node. +//! A group can hold more than one barrier for a point: every replica proposes +//! its own. The lowest index is the group's place. + +use std::sync::{Arc, Weak}; + +use nodedb_cluster::METADATA_GROUP_ID; +use nodedb_cluster::calvin::{RestorePointHook, SEQUENCER_GROUP_ID, SequencerRestorePoint}; +use nodedb_types::Hlc; +use nodedb_wal::record::RestorePointPayload; +use tracing::{error, info, warn}; + +use crate::control::state::SharedState; +use crate::wal::manager::NO_APPLY_KEY; + +/// Tenant the restore point's cut barriers are framed under. A barrier +/// orders every entry of its group, whatever tenant writes it. No step reads +/// the tenant of a barrier: the proposal stamps and forwards opaque bytes, +/// the apply's barrier arms ignore it, and a barrier raises no tenant mark. +const CUT_FRAME_TENANT: u64 = 0; + +/// Record one group's place at a restore point in this node's WAL, durably, +/// and move this node's clock past the point's watermark. A data group's +/// record lists the vShards it homes now. A failure is logged: a cluster +/// restore to the point then starts this node's replica of the group with no +/// log. +pub fn record_group_point(state: &SharedState, mut point: RestorePointPayload) { + state + .hlc_clock + .update(Hlc::new(point.hlc.saturating_add(1), 0)); + if point.group_id != METADATA_GROUP_ID + && point.group_id != SEQUENCER_GROUP_ID + && let Some(routing) = &state.cluster_routing + { + point.vshards = routing + .read() + .unwrap_or_else(|p| p.into_inner()) + .vshards_for_group(point.group_id); + } + let recorded = state + .wal + .appender(NO_APPLY_KEY) + .append_restore_point(&point) + .and_then(|_| state.wal.sync()); + match recorded { + Ok(()) => info!( + restore_point = point.id, + group_id = point.group_id, + applied_index = point.applied_index, + "restore point recorded for raft group" + ), + Err(e) => error!( + restore_point = point.id, + group_id = point.group_id, + error = %e, + "restore point not recorded in the WAL; a cluster restore to it starts this \ + group with no log on this node" + ), + } +} + +/// The hook the sequencer state machine calls when it applies a restore +/// point's cut marker. The WAL append runs on a blocking thread: the hook +/// runs on the Raft tick thread, which must not do I/O. +pub fn sequencer_hook(shared: Weak) -> RestorePointHook { + Arc::new(move |point: SequencerRestorePoint| { + let Some(state) = shared.upgrade() else { + return; + }; + let payload = RestorePointPayload { + id: point.id, + hlc: point.hlc, + group_id: SEQUENCER_GROUP_ID, + applied_index: point.index, + term: 0, + next_epoch: point.next_epoch, + epoch_system_ms: point + .epoch_system_ms + .and_then(|ms| u64::try_from(ms).ok()) + .unwrap_or(0), + vshards: Vec::new(), + }; + let record = move || { + record_group_point(&state, payload); + seal_point_segment(&state, point.id); + }; + match tokio::runtime::Handle::try_current() { + Ok(runtime) => { + runtime.spawn_blocking(record); + } + Err(_) => record(), + } + }) +} + +/// Cut every data group this node hosts, and the Calvin sequencer, at the +/// point's watermark. Runs in the background: the metadata apply that +/// starts it must not wait on other groups. The WAL segment holding the +/// point's records is then sealed, so the archiver uploads it. +pub fn spawn_node_cut(state: Arc, id: u64, hlc: u64) { + tokio::spawn(async move { + match crate::control::backup::cut::cut_at_point(&state, CUT_FRAME_TENANT, hlc, id).await { + Ok(()) => info!(restore_point = id, "restore point cut taken on this node"), + Err(e) => warn!( + restore_point = id, + error = %e, + "restore point cut did not finish on this node; a cluster restore to it \ + starts every group this node did not record with no log" + ), + } + let sealing = Arc::clone(&state); + if let Err(e) = tokio::task::spawn_blocking(move || seal_point_segment(&sealing, id)).await + { + warn!(restore_point = id, error = %e, "restore point segment seal did not run"); + } + }); +} + +/// Seal the active WAL segment after the records of restore point `id`. +fn seal_point_segment(state: &SharedState, id: u64) { + if let Err(e) = state.wal.seal_active_segment() { + warn!( + restore_point = id, + error = %e, + "WAL segment not sealed after the restore point; the archiver uploads its \ + records once the segment fills" + ); + } +} diff --git a/nodedb/src/control/pitr/restore_point/seed.rs b/nodedb/src/control/pitr/restore_point/seed.rs new file mode 100644 index 000000000..f10d22eb0 --- /dev/null +++ b/nodedb/src/control/pitr/restore_point/seed.rs @@ -0,0 +1,238 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The cut barriers this node applied, kept in the system catalog. +//! +//! Every entry a group's log places after a cut barrier records a commit HLC +//! above the barrier's watermark, on every replica. The apply loop tracks +//! the barriers it applies. After a restart it applies again the entries +//! above its durable applied index, and some of those follow a barrier it +//! applied before the restart. The catalog rows give those barriers back. A +//! WAL checkpoint never removes them. A barrier finishes its apply only once +//! its row is durable. + +use std::time::Duration; + +use tracing::{error, info, warn}; + +use crate::control::cluster::metadata_applier::{MetadataApplyWedge, WedgeReport, classify}; + +use crate::control::security::catalog::SystemCatalog; +use crate::control::security::catalog::cut_floors::StoredBarrier; +use crate::control::state::SharedState; + +/// One group's cut barrier. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct RecordedCut { + pub group_id: u64, + /// Log index of the barrier. + pub barrier_index: u64, + /// The barrier's watermark HLC. + pub hlc: u64, +} + +/// Every cut barrier `catalog` holds. +pub fn load_recorded_cuts(catalog: &SystemCatalog) -> crate::Result> { + Ok(catalog + .load_cut_floors()? + .into_iter() + .map(|(group_id, barrier)| RecordedCut { + group_id, + barrier_index: barrier.index, + hlc: barrier.watermark, + }) + .collect()) +} + +/// Durably record the cut barrier group `group_id` applies at `barrier_index` +/// with watermark `hlc`. +pub fn persist_cut_floor( + state: &SharedState, + group_id: u64, + barrier_index: u64, + hlc: u64, +) -> crate::Result<()> { + let barrier = StoredBarrier { + index: barrier_index, + watermark: hlc, + }; + let durable = state.pitr.durable_applied(group_id); + state + .credentials + .catalog() + .put_cut_floor(group_id, barrier, durable) +} + +/// First wait between two attempts of a floor write. +const FIRST_RETRY: Duration = Duration::from_millis(10); + +/// Longest wait between two attempts of a floor write. +const LAST_RETRY: Duration = Duration::from_secs(1); + +/// Entry kind a floor-write wedge report names. +const BARRIER_ENTRY_KIND: &str = "CutBarrier"; + +/// Run `attempt` until it succeeds, waiting longer after each failure. The +/// barrier apply awaits this: no later entry of the group starts, and the +/// barrier does not settle, until the floor is durable. +/// +/// A failure `classify` names permanent wedges the node at once. A transient +/// failure wedges it once the wait between attempts reaches `LAST_RETRY`. +/// The wedge goes on `wedge`, the marker the readiness probe reads. The log +/// names the first failure, the wedge, and the recovery, never each retry. +/// A later success clears the report this barrier recorded. +pub async fn persist_until_durable( + wedge: &MetadataApplyWedge, + group_id: u64, + barrier_index: u64, + mut attempt: impl FnMut() -> crate::Result<()>, +) { + let mut wait = FIRST_RETRY; + let mut failures = 0u64; + let mut wedged = false; + // The report this barrier recorded. A report another writer holds is + // never cleared here. + let mut recorded: Option = None; + loop { + match attempt() { + Ok(()) => { + if let Some(report) = recorded.as_ref() { + wedge.clear(report); + } + if failures > 0 { + info!( + group_id, + barrier_index, failures, "cut barrier floor persisted; the group resumes" + ); + } + return; + } + Err(e) => { + failures += 1; + if failures == 1 { + warn!( + group_id, + barrier_index, + error = %e, + "cut barrier floor not persisted; the group holds at the barrier and retries" + ); + } + let stalled = classify(&e).is_permanent() || wait >= LAST_RETRY; + if stalled && !wedged { + wedged = true; + let report = WedgeReport { + raft_index: barrier_index, + last_applied_watermark: barrier_index.saturating_sub(1), + entry_kind: BARRIER_ENTRY_KIND.to_owned(), + error: format!("group {group_id}: cut barrier floor not persisted: {e}"), + }; + error!( + group_id, + barrier_index, + failures, + error = %e, + "cut barrier floor write keeps failing; the group is halted and this node \ + is no longer ready" + ); + if wedge.record(report.clone()) { + recorded = Some(report); + } + } + } + } + tokio::time::sleep(wait).await; + wait = (wait * 2).min(LAST_RETRY); + } +} + +#[cfg(all(test, feature = "failpoints"))] +mod tests { + use std::time::Duration; + + use super::*; + + #[tokio::test] + async fn the_barrier_holds_until_its_floor_persists() { + let dir = tempfile::tempdir().unwrap(); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).unwrap(); + let wedge = MetadataApplyWedge::default(); + let barrier = StoredBarrier { + index: 10, + watermark: 100, + }; + let fail = nodedb_types::fail_point::FailGuard::fail( + "cut_floor::before_persist", + "injected floor write error", + ); + let persist = persist_until_durable(&wedge, 1, 10, || catalog.put_cut_floor(1, barrier, 0)); + tokio::pin!(persist); + assert!( + tokio::time::timeout(Duration::from_millis(100), persist.as_mut()) + .await + .is_err(), + "the barrier does not finish while its floor write fails" + ); + assert!(catalog.load_cut_floors().unwrap().is_empty()); + assert!( + !wedge.is_wedged(), + "a floor write that failed briefly does not wedge the node" + ); + + // The waits between attempts reach LAST_RETRY after about 1.3 s. + assert!( + tokio::time::timeout(Duration::from_secs(3), persist.as_mut()) + .await + .is_err(), + "the barrier does not finish while its floor write fails" + ); + let report = wedge + .report() + .expect("a floor write that keeps failing wedges the node"); + assert_eq!(report.raft_index, 10); + assert_eq!(report.last_applied_watermark, 9); + assert_eq!(report.entry_kind, BARRIER_ENTRY_KIND); + + drop(fail); + tokio::time::timeout(Duration::from_secs(5), persist) + .await + .expect("the barrier finishes once the floor write succeeds"); + assert_eq!(catalog.load_cut_floors().unwrap(), [(1, barrier)]); + assert!( + !wedge.is_wedged(), + "the recovery clears the barrier's report" + ); + } + + #[tokio::test] + async fn a_recovered_floor_keeps_another_writers_report() { + let dir = tempfile::tempdir().unwrap(); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).unwrap(); + let wedge = MetadataApplyWedge::default(); + let applier = WedgeReport { + raft_index: 3, + last_applied_watermark: 2, + entry_kind: "DdlPrepared".into(), + error: "applier".into(), + }; + wedge.record(applier.clone()); + let barrier = StoredBarrier { + index: 10, + watermark: 100, + }; + let fail = nodedb_types::fail_point::FailGuard::fail( + "cut_floor::before_persist", + "injected floor write error", + ); + let persist = persist_until_durable(&wedge, 1, 10, || catalog.put_cut_floor(1, barrier, 0)); + tokio::pin!(persist); + assert!( + tokio::time::timeout(Duration::from_secs(3), persist.as_mut()) + .await + .is_err() + ); + drop(fail); + tokio::time::timeout(Duration::from_secs(5), persist) + .await + .expect("the barrier finishes once the floor write succeeds"); + assert_eq!(wedge.report(), Some(applier)); + } +} diff --git a/nodedb/src/control/pitr/restore_point/task.rs b/nodedb/src/control/pitr/restore_point/task.rs new file mode 100644 index 000000000..49732dfde --- /dev/null +++ b/nodedb/src/control/pitr/restore_point/task.rs @@ -0,0 +1,64 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The periodic restore point task. + +use std::sync::Arc; +use std::time::Duration; + +use tracing::{info, warn}; + +use crate::control::shutdown::{ShutdownPhase, spawn_loop}; +use crate::control::startup::StartupPhase; +use crate::control::state::SharedState; + +use super::create::create_restore_point; + +/// Create a restore point every `interval`. `None` starts nothing. Only the +/// node that runs the cluster's singleton workers creates them, so the +/// cluster takes one point per interval. +pub fn spawn_restore_point_task(shared: &Arc, interval: Option) { + let Some(interval) = interval else { + return; + }; + let task_state = Arc::clone(shared); + spawn_loop( + &shared.loop_registry, + &shared.shutdown, + "pitr_restore_point", + ShutdownPhase::DrainingControlPlane, + move |mut shutdown| async move { + let state = task_state; + let ready = tokio::select! { + ready = state.startup.await_phase(StartupPhase::GatewayEnable) => ready.is_ok(), + _ = shutdown.wait_cancelled() => false, + }; + if !ready { + return; + } + info!( + interval_secs = interval.as_secs(), + "restore point task started" + ); + let mut tick = + tokio::time::interval_at(tokio::time::Instant::now() + interval, interval); + tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + loop { + tokio::select! { + _ = shutdown.wait_cancelled() => break, + _ = tick.tick() => {} + } + if !state.is_singleton_worker() { + continue; + } + match create_restore_point(&state).await { + Ok(point) => info!( + restore_point = point.id, + hlc = point.hlc, + "periodic restore point created" + ), + Err(e) => warn!(error = %e, "periodic restore point not created"), + } + } + }, + ); +} diff --git a/nodedb/src/control/pitr/restore_seal.rs b/nodedb/src/control/pitr/restore_seal.rs new file mode 100644 index 000000000..8a0b51e95 --- /dev/null +++ b/nodedb/src/control/pitr/restore_seal.rs @@ -0,0 +1,35 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Seal the restore generation of a node's first boot after a cluster +//! restore, before its Raft groups start. A later restore then takes a newer +//! generation, above every term and epoch this cluster reaches. + +use tracing::info; + +use crate::control::state::SharedState; +use crate::storage::restore_generation::{ + read_generation_marker, remove_generation_marker, seal_generation, +}; + +/// Seal the generation the data directory was restored at, if any. Refuses +/// the boot when the node has no cold storage to seal it in. +pub async fn seal_restored_generation(state: &SharedState) -> crate::Result<()> { + let data_dir = state.data_dir.clone(); + let Some(generation) = read_generation_marker(&data_dir)? else { + return Ok(()); + }; + let cold = state + .cold_storage + .as_ref() + .ok_or_else(|| crate::Error::Config { + detail: format!( + "{} was restored at generation {generation}, and its first boot seals the \ + generation in cold storage; configure [cold_storage]", + data_dir.display() + ), + })?; + seal_generation(&cold.object_store(), cold.prefix(), generation).await?; + remove_generation_marker(&data_dir)?; + info!(generation, "cluster restore generation sealed"); + Ok(()) +} diff --git a/nodedb/src/control/pitr/retention.rs b/nodedb/src/control/pitr/retention.rs new file mode 100644 index 000000000..7b79a3205 --- /dev/null +++ b/nodedb/src/control/pitr/retention.rs @@ -0,0 +1,220 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Base snapshot retention: list the bases of one node life, keep the newest +//! N, and name the WAL floor the oldest kept base replays from. + +use std::collections::HashMap; +use std::num::NonZeroUsize; +use std::sync::Arc; + +use object_store::{ObjectStore, ObjectStoreExt}; +use tracing::warn; + +use crate::storage::snapshot::SnapshotMeta; +use crate::storage::snapshot_writer::{ + CHUNK_DIR, SnapshotManifest, delete_snapshot, discover_snapshots, manifest_key, +}; +use crate::types::Lsn; + +/// One base snapshot in the store. +#[derive(Debug, Clone)] +pub struct ListedBase { + pub prefix: String, + pub meta: SnapshotMeta, + /// Cold-store keys the base references. + pub cold_segments: Vec, + /// Chunk ids the manifest lists, each once. + pub chunk_ids: Vec, +} + +impl ListedBase { + pub fn from_manifest(prefix: String, manifest: SnapshotManifest) -> Self { + let mut chunk_ids: Vec = manifest.chunk_ids().map(str::to_owned).collect(); + chunk_ids.sort_unstable(); + chunk_ids.dedup(); + Self { + prefix, + meta: manifest.meta, + cold_segments: manifest.cold_segments, + chunk_ids, + } + } +} + +/// Every snapshot prefix of one node life, by what its manifest says. +#[derive(Debug, Default)] +pub struct Listed { + /// Every snapshot, incremental ones included: each is a full base. + pub bases: Vec, + /// Prefixes whose manifest exists but does not load. + pub unreadable: Vec, + /// Prefixes without a manifest, deleted by this listing. + pub abandoned: usize, +} + +/// List every snapshot prefix in `store`. +/// +/// A prefix without a manifest is a write that never finished, and it is +/// deleted. One task writes a node life's snapshots, one at a time, so no +/// write is in flight while the listing runs. The chunk directory is no +/// snapshot, and chunk garbage collection owns it. +pub async fn list_snapshots( + store: &Arc, + encryption_key: &nodedb_wal::crypto::WalEncryptionKey, +) -> crate::Result { + let prefixes = store + .list_with_delimiter(None) + .await + .map_err(|e| crate::Error::Storage { + engine: "snapshot".into(), + detail: format!("list snapshot prefixes: {e}"), + })? + .common_prefixes; + let mut loaded: HashMap = discover_snapshots(store, encryption_key) + .await + .into_iter() + .collect(); + + let mut listed = Listed::default(); + for prefix in prefixes { + let prefix = prefix.as_ref().trim_end_matches('/').to_string(); + if prefix == CHUNK_DIR { + continue; + } + match loaded.remove(&prefix) { + Some(manifest) => listed + .bases + .push(ListedBase::from_manifest(prefix, manifest)), + None => { + let manifest = manifest_key(&prefix); + match store.head(&manifest).await { + Err(object_store::Error::NotFound { .. }) => { + delete_snapshot(store, &prefix).await?; + listed.abandoned += 1; + } + _ => { + warn!(prefix = %prefix, "snapshot manifest does not load"); + listed.unreadable.push(prefix); + } + } + } + } + } + Ok(listed) +} + +/// Which bases retention keeps and which it deletes. +#[derive(Debug, Default)] +pub struct RetentionPlan { + /// Kept bases, oldest first. + pub keep: Vec, + /// Bases to delete, oldest first. + pub delete: Vec, +} + +/// Keep the newest `keep` bases, `fresh` always among them. +/// +/// Bases order by `applied_high_lsn`, then creation time, then id. `fresh` is +/// the base the current run wrote. It is kept even if an older base orders +/// after it, so a run never deletes the base it wrote. +pub fn plan_retention(bases: Vec, fresh: &str, keep: NonZeroUsize) -> RetentionPlan { + let (fresh_base, mut older): (Vec<_>, Vec<_>) = + bases.into_iter().partition(|base| base.prefix == fresh); + older.sort_by_key(|base| { + let meta = &base.meta; + (meta.applied_high_lsn, meta.created_at_us, meta.snapshot_id) + }); + let kept_older = keep.get().saturating_sub(fresh_base.len()); + let split = older.len().saturating_sub(kept_older); + let mut keep = older.split_off(split); + keep.extend(fresh_base); + RetentionPlan { + keep, + delete: older, + } +} + +/// The LSN the oldest remaining base replays WAL from, or `None` with no base. +/// Archived WAL wholly below it is needed by no remaining base. +pub fn wal_floor(bases: &[SnapshotMeta]) -> Option { + bases.iter().map(|meta| meta.begin_lsn).min() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::storage::snapshot::{SNAPSHOT_FORMAT_VERSION, SnapshotKind}; + + fn base(id: u64, begin: u64, applied_high: u64) -> ListedBase { + ListedBase { + prefix: format!("snap-{id}"), + meta: SnapshotMeta { + format_version: SNAPSHOT_FORMAT_VERSION, + snapshot_id: id, + begin_lsn: Lsn::new(begin), + end_lsn: Lsn::new(begin), + applied_high_lsn: Lsn::new(applied_high), + created_at_us: id, + created_by: "node-1".into(), + kind: SnapshotKind::Base, + parent_id: None, + data_bytes: 0, + }, + cold_segments: Vec::new(), + chunk_ids: Vec::new(), + } + } + + fn ids(bases: &[ListedBase]) -> Vec { + bases.iter().map(|base| base.meta.snapshot_id).collect() + } + + fn n(keep: usize) -> NonZeroUsize { + NonZeroUsize::new(keep).unwrap() + } + + #[test] + fn retention_keeps_the_newest_n_and_deletes_the_rest() { + let bases = vec![ + base(3, 30, 35), + base(1, 10, 15), + base(4, 40, 45), + base(2, 20, 25), + ]; + let plan = plan_retention(bases, "snap-4", n(2)); + assert_eq!(ids(&plan.keep), [3, 4]); + assert_eq!(ids(&plan.delete), [1, 2], "deleted oldest first"); + } + + #[test] + fn the_only_base_is_never_deleted() { + let plan = plan_retention(vec![base(1, 10, 15)], "snap-1", n(1)); + assert_eq!(ids(&plan.keep), [1]); + assert!(plan.delete.is_empty()); + } + + #[test] + fn fewer_bases_than_the_retention_are_all_kept() { + let plan = plan_retention(vec![base(1, 10, 15), base(2, 20, 25)], "snap-2", n(7)); + assert_eq!(ids(&plan.keep), [1, 2]); + assert!(plan.delete.is_empty()); + } + + #[test] + fn the_fresh_base_is_kept_even_when_it_orders_first() { + let bases = vec![base(1, 10, 15), base(2, 20, 25), base(9, 5, 8)]; + let plan = plan_retention(bases, "snap-9", n(2)); + assert_eq!(ids(&plan.keep), [2, 9]); + assert_eq!(ids(&plan.delete), [1]); + } + + #[test] + fn the_wal_floor_is_the_lowest_begin_lsn_of_any_remaining_base() { + let metas: Vec = [base(2, 20, 25), base(3, 18, 35)] + .into_iter() + .map(|base| base.meta) + .collect(); + assert_eq!(wal_floor(&metas), Some(Lsn::new(18))); + assert_eq!(wal_floor(&[]), None); + } +} diff --git a/nodedb/src/control/pitr/state.rs b/nodedb/src/control/pitr/state.rs new file mode 100644 index 000000000..0a78a6d1b --- /dev/null +++ b/nodedb/src/control/pitr/state.rs @@ -0,0 +1,231 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! PITR state shared across the Control Plane: this node life, the base +//! snapshot catalog, the cold keys and chunk ids kept bases pin, and the +//! outcome of the last base snapshot run. + +use std::collections::HashMap; +use std::sync::atomic::Ordering; +use std::sync::{Arc, Mutex, OnceLock, RwLock}; + +use super::node_life::NodeLife; +use super::pins::ColdPins; +use super::restore_point::RecordedCut; +use crate::control::metrics::SystemMetrics; +use crate::storage::snapshot::SnapshotCatalog; + +/// The last base snapshot run that failed. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct PitrFailure { + pub at_unix_secs: u64, + pub detail: String, +} + +#[derive(Default)] +pub struct PitrState { + /// Set at boot when PITR is on. + life: OnceLock, + /// Set at boot on a cluster node. A base captures it beside the system catalog. + cluster_catalog: OnceLock>, + /// Bases of this node life that retention kept. + catalog: RwLock, + /// Cold keys the kept bases reference. Async so a deleter holds it across + /// its check and its delete. + cold_pins: tokio::sync::RwLock, + /// Chunk ids the kept bases and the bases in progress list. + chunk_pins: tokio::sync::RwLock, + last_failure: Mutex>, + /// The cut barriers the system catalog held at boot. + recorded_cuts: OnceLock>, + /// Each group's durable applied index, as this life saved it. + durable_applied: Mutex>, +} + +impl PitrState { + /// Install this node life. A second install keeps the first. + pub fn install_life(&self, life: NodeLife) { + let _ = self.life.set(life); + } + + pub fn life(&self) -> Option<&NodeLife> { + self.life.get() + } + + /// Install the cluster catalog. A second install keeps the first. + pub fn install_cluster_catalog(&self, catalog: Arc) { + let _ = self.cluster_catalog.set(catalog); + } + + pub fn cluster_catalog(&self) -> Option<&Arc> { + self.cluster_catalog.get() + } + + /// Install the cut barriers read from the system catalog at boot. A + /// second install keeps the first. + pub fn install_recorded_cuts(&self, cuts: Vec) { + let _ = self.recorded_cuts.set(cuts); + } + + /// The cut barriers the system catalog held at boot. + pub fn recorded_cuts(&self) -> &[RecordedCut] { + self.recorded_cuts.get().map_or(&[], Vec::as_slice) + } + + /// Note that `group_id`'s durable applied index reached `index`. + pub fn note_durable_applied(&self, group_id: u64, index: u64) { + let mut held = self + .durable_applied + .lock() + .unwrap_or_else(|p| p.into_inner()); + let slot = held.entry(group_id).or_insert(0); + *slot = (*slot).max(index); + } + + /// `group_id`'s durable applied index as this life saved it, `0` before + /// the first save. + pub fn durable_applied(&self, group_id: u64) -> u64 { + self.durable_applied + .lock() + .unwrap_or_else(|p| p.into_inner()) + .get(&group_id) + .copied() + .unwrap_or(0) + } + + /// Every cold object delete or overwrite holds the read lock across its + /// pin check and the operation. A base pins its keys under the write lock. + pub fn cold_pins(&self) -> &tokio::sync::RwLock { + &self.cold_pins + } + + /// Every chunk delete holds the read lock across its pin check and the + /// delete. A base pins its chunk ids under the write lock before it + /// checks which chunks the store holds. + pub fn chunk_pins(&self) -> &tokio::sync::RwLock { + &self.chunk_pins + } + + /// A copy of the catalog, for restore planning. + pub fn catalog(&self) -> SnapshotCatalog { + self.catalog + .read() + .unwrap_or_else(|p| p.into_inner()) + .clone() + } + + /// Replace the catalog and publish its base count. + pub fn replace_catalog(&self, catalog: SnapshotCatalog, metrics: Option<&SystemMetrics>) { + let bases = base_count(&catalog); + *self.catalog.write().unwrap_or_else(|p| p.into_inner()) = catalog; + if let Some(metrics) = metrics { + metrics.pitr_base_snapshots.store(bases, Ordering::Relaxed); + } + } + + pub fn last_failure(&self) -> Option { + self.last_failure + .lock() + .unwrap_or_else(|p| p.into_inner()) + .clone() + } + + /// Record a base that completed at `at_unix_secs`. + pub fn record_success(&self, metrics: Option<&SystemMetrics>, at_unix_secs: u64) { + if let Some(metrics) = metrics { + metrics + .pitr_base_last_success_timestamp_seconds + .store(at_unix_secs, Ordering::Relaxed); + } + } + + /// Record a run that failed at `at_unix_secs`. + pub fn record_failure( + &self, + metrics: Option<&SystemMetrics>, + error: &crate::Error, + at_unix_secs: u64, + ) { + *self.last_failure.lock().unwrap_or_else(|p| p.into_inner()) = Some(PitrFailure { + at_unix_secs, + detail: error.to_string(), + }); + if let Some(metrics) = metrics { + metrics + .pitr_base_last_failure_timestamp_seconds + .store(at_unix_secs, Ordering::Relaxed); + metrics + .pitr_base_failures_total + .fetch_add(1, Ordering::Relaxed); + } + } +} + +/// Every snapshot is a full base, incremental ones included. +fn base_count(catalog: &SnapshotCatalog) -> u64 { + catalog.len() as u64 +} + +/// Seconds since the Unix epoch. +pub fn unix_now_secs() -> u64 { + std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .unwrap_or_default() + .as_secs() +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::storage::snapshot::{SNAPSHOT_FORMAT_VERSION, SnapshotKind, SnapshotMeta}; + use crate::types::Lsn; + + fn base(id: u64) -> SnapshotMeta { + SnapshotMeta { + format_version: SNAPSHOT_FORMAT_VERSION, + snapshot_id: id, + begin_lsn: Lsn::new(id), + end_lsn: Lsn::new(id), + applied_high_lsn: Lsn::new(id), + created_at_us: id, + created_by: "node-1".into(), + kind: SnapshotKind::Base, + parent_id: None, + data_bytes: 0, + } + } + + #[test] + fn metrics_follow_the_catalog_and_every_outcome() { + let state = PitrState::default(); + let metrics = SystemMetrics::new(); + let mut catalog = SnapshotCatalog::new(); + catalog.add(base(1)); + catalog.add(base(2)); + state.replace_catalog(catalog, Some(&metrics)); + assert_eq!(metrics.pitr_base_snapshots.load(Ordering::Relaxed), 2); + assert_eq!(state.catalog().len(), 2); + + state.record_success(Some(&metrics), 100); + assert_eq!( + metrics + .pitr_base_last_success_timestamp_seconds + .load(Ordering::Relaxed), + 100 + ); + + let error = crate::Error::Config { + detail: "no key".into(), + }; + state.record_failure(Some(&metrics), &error, 200); + let failure = state.last_failure().unwrap(); + assert_eq!(failure.at_unix_secs, 200); + assert!(failure.detail.contains("no key"), "{}", failure.detail); + assert_eq!(metrics.pitr_base_failures_total.load(Ordering::Relaxed), 1); + assert_eq!( + metrics + .pitr_base_last_failure_timestamp_seconds + .load(Ordering::Relaxed), + 200 + ); + } +} diff --git a/nodedb/src/control/pitr/task.rs b/nodedb/src/control/pitr/task.rs new file mode 100644 index 000000000..4d65cc29d --- /dev/null +++ b/nodedb/src/control/pitr/task.rs @@ -0,0 +1,325 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The periodic base snapshot task, and its boot wiring. + +use std::num::NonZeroUsize; +use std::sync::Arc; +use std::time::Duration; + +use tracing::{info, warn}; + +use super::cycle::{BaseCycle, CycleOutcome}; +use super::node_life::NodeLife; +use super::pins::ColdPins; +use super::retention::list_snapshots; +use super::state::unix_now_secs; +use crate::config::server::PitrSettings; +use crate::control::server::exchange::capture_base_on_local_cores; +use crate::control::shutdown::{ShutdownPhase, spawn_loop}; +use crate::control::startup::StartupPhase; +use crate::control::state::SharedState; +use crate::storage::snapshot::{SnapshotCatalog, SnapshotMeta}; +use crate::storage::snapshot_node::capture_node_snapshot; +use crate::storage::snapshot_writer::CdcParams; + +/// Wire PITR at boot: resolve this node life with the archiver's incarnation, +/// install the cluster catalog, and rebuild the base catalog, cold pins, and +/// chunk pins from the store. +/// +/// A listing that fails, or a manifest that does not load, leaves the pins +/// incomplete: every cold key and chunk counts as pinned until a run lists +/// cleanly. +pub async fn wire_pitr( + state: &SharedState, + cluster_catalog: Arc, +) -> crate::Result<()> { + let life = NodeLife::resolve(state.node_id, state.data_dir.clone()).await?; + state.pitr.install_cluster_catalog(cluster_catalog); + let key = wal_key(state)?; + let store = life.snapshot_store(&state.snapshot_storage); + let mut catalog = SnapshotCatalog::new(); + let (pins, chunk_pins) = match list_snapshots(&store, &key).await { + Ok(listed) => { + let mut metas: Vec<_> = listed.bases.iter().map(|base| base.meta.clone()).collect(); + metas.sort_by_key(|meta| (meta.applied_high_lsn, meta.created_at_us)); + for meta in metas { + catalog.add(meta); + } + let complete = listed.unreadable.is_empty(); + let pins = ColdPins::from_bases( + listed + .bases + .iter() + .map(|base| (base.prefix.as_str(), base.cold_segments.as_slice())), + complete, + ); + let chunk_pins = ColdPins::from_bases( + listed + .bases + .iter() + .map(|base| (base.prefix.as_str(), base.chunk_ids.as_slice())), + complete, + ); + (pins, chunk_pins) + } + Err(e) => { + warn!( + error = %e, + "PITR snapshot listing failed; every cold key and chunk stays pinned" + ); + state + .pitr + .record_failure(state.system_metrics.as_deref(), &e, unix_now_secs()); + ( + ColdPins::from_bases([], false), + ColdPins::from_bases([], false), + ) + } + }; + info!( + node_id = life.node_id, + incarnation = life.incarnation.as_str(), + snapshots = catalog.len(), + pinned_cold_keys = pins.len(), + pinned_chunks = chunk_pins.len(), + "PITR base snapshot catalog rebuilt" + ); + *state.pitr.cold_pins().write().await = pins; + *state.pitr.chunk_pins().write().await = chunk_pins; + state + .pitr + .replace_catalog(catalog, state.system_metrics.as_deref()); + state.pitr.install_life(life); + Ok(()) +} + +/// Spawn the base snapshot task when PITR is on. +/// +/// The first base is due one interval after the newest base in the catalog, +/// or at once when there is none. Runs start once the gateway is open, so a +/// capture never races boot recovery. +pub fn spawn_base_snapshot_task(shared: &Arc, settings: &PitrSettings) { + if !settings.enabled { + return; + } + let interval = settings.base_snapshot_interval(); + let retention = match settings.retention() { + Ok(retention) => retention, + Err(e) => { + shared + .pitr + .record_failure(shared.system_metrics.as_deref(), &e, unix_now_secs()); + warn!(error = %e, "PITR base snapshots not started"); + return; + } + }; + let task_state = Arc::clone(shared); + spawn_loop( + &shared.loop_registry, + &shared.shutdown, + "pitr_base_snapshot", + ShutdownPhase::DrainingControlPlane, + move |mut shutdown| async move { + let state = task_state; + let ready = tokio::select! { + ready = state.startup.await_phase(StartupPhase::GatewayEnable) => ready.is_ok(), + _ = shutdown.wait_cancelled() => false, + }; + if !ready { + return; + } + let Some(life) = state.pitr.life().cloned() else { + let e = crate::Error::Internal { + detail: "PITR is on but boot wired no node life".into(), + }; + state + .pitr + .record_failure(state.system_metrics.as_deref(), &e, unix_now_secs()); + warn!(error = %e, "PITR base snapshots not started"); + return; + }; + let first = first_delay(&state.pitr.catalog(), unix_now_secs(), interval); + info!( + interval_secs = interval.as_secs(), + retention = retention.get(), + first_in_secs = first.as_secs(), + "PITR base snapshot task started" + ); + let mut tick = tokio::time::interval_at(tokio::time::Instant::now() + first, interval); + tick.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + loop { + tokio::select! { + _ = shutdown.wait_cancelled() => break, + _ = tick.tick() => {} + } + // A failure is recorded and logged; the next tick retries. + let _ = run_once(&state, &life, retention, interval).await; + } + }, + ); +} + +/// Take one base and record the outcome in state and metrics. Returns the +/// base. +pub(super) async fn run_once( + state: &Arc, + life: &NodeLife, + retention: NonZeroUsize, + interval: Duration, +) -> Result { + let metrics = state.system_metrics.as_deref(); + let outcome = match take_base(state, life, retention, interval).await { + Ok(outcome) => outcome, + Err(e) => { + warn!(error = %e, "PITR base snapshot failed"); + state.pitr.record_failure(metrics, &e, unix_now_secs()); + return Err(e); + } + }; + let CycleOutcome { + base, + uploads, + catalog, + deleted_bases, + collected_chunks, + collected_segments, + cleanup_error, + } = outcome; + let now = unix_now_secs(); + let catalog = catalog.unwrap_or_else(|| { + let mut held = state.pitr.catalog(); + held.add(base.clone()); + held + }); + state.pitr.replace_catalog(catalog, metrics); + state.pitr.record_success(metrics, now); + if let Some(metrics) = metrics { + use std::sync::atomic::Ordering::Relaxed; + metrics + .pitr_wal_segments_collected_total + .fetch_add(collected_segments, Relaxed); + metrics + .pitr_base_chunks_uploaded_total + .fetch_add(uploads.uploaded, Relaxed); + metrics + .pitr_base_chunk_bytes_uploaded_total + .fetch_add(uploads.uploaded_bytes, Relaxed); + metrics + .pitr_base_chunks_reused_total + .fetch_add(uploads.reused, Relaxed); + metrics + .pitr_base_chunks_collected_total + .fetch_add(collected_chunks, Relaxed); + } + info!( + snapshot_id = base.snapshot_id, + parent_id = ?base.parent_id, + begin_lsn = base.begin_lsn.as_u64(), + applied_high_lsn = base.applied_high_lsn.as_u64(), + chunks_uploaded = uploads.uploaded, + chunks_reused = uploads.reused, + deleted_bases, + collected_chunks, + collected_segments, + "PITR base snapshot taken" + ); + if let Some(e) = cleanup_error { + warn!(error = %e, "PITR retention did not finish; the next run retries"); + state.pitr.record_failure(metrics, &e, now); + } + Ok(base) +} + +/// Capture every core and the node-level stores, then run the cycle. +async fn take_base( + state: &Arc, + life: &NodeLife, + retention: NonZeroUsize, + interval: Duration, +) -> crate::Result { + let key = wal_key(state)?; + // A base that outlasts the interval overlaps the next one. + let deadline = std::time::Instant::now() + interval; + let cores = capture_base_on_local_cores(state, deadline).await?; + let node = capture_node_snapshot( + state.data_dir.clone(), + Arc::clone(state), + state.pitr.cluster_catalog().cloned(), + deadline.saturating_duration_since(std::time::Instant::now()), + ) + .await?; + BaseCycle { + root: &state.snapshot_storage, + cold: state.cold_storage.as_deref(), + life, + state: &state.pitr, + encryption_key: &key, + retention, + chunk_params: CdcParams::DEFAULT, + } + .run(cores, node) + .await +} + +fn wal_key(state: &SharedState) -> crate::Result { + state + .wal + .encryption_key() + .cloned() + .ok_or_else(|| crate::Error::Config { + detail: "PITR base snapshots are encrypted with the WAL key, and this node has \ + none. Add [encryption] key_path or set pitr.enabled = false" + .into(), + }) +} + +/// Time until the next base is due: one interval after the newest base, or +/// zero with no base. +fn first_delay(catalog: &SnapshotCatalog, now_unix_secs: u64, interval: Duration) -> Duration { + let Some(newest_us) = catalog.all().iter().map(|meta| meta.created_at_us).max() else { + return Duration::ZERO; + }; + let age = Duration::from_secs(now_unix_secs.saturating_sub(newest_us / 1_000_000)); + interval.saturating_sub(age) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::storage::snapshot::{SNAPSHOT_FORMAT_VERSION, SnapshotKind, SnapshotMeta}; + use crate::types::Lsn; + + fn catalog_created_at(secs: &[u64]) -> SnapshotCatalog { + let mut catalog = SnapshotCatalog::new(); + for (id, &at) in secs.iter().enumerate() { + catalog.add(SnapshotMeta { + format_version: SNAPSHOT_FORMAT_VERSION, + snapshot_id: id as u64, + begin_lsn: Lsn::new(1), + end_lsn: Lsn::new(1), + applied_high_lsn: Lsn::new(1), + created_at_us: at * 1_000_000, + created_by: "node-1".into(), + kind: SnapshotKind::Base, + parent_id: None, + data_bytes: 0, + }); + } + catalog + } + + #[test] + fn with_no_base_the_first_run_is_immediate() { + let delay = first_delay(&SnapshotCatalog::new(), 1_000, Duration::from_secs(60)); + assert_eq!(delay, Duration::ZERO); + } + + #[test] + fn a_restart_keeps_the_schedule_of_the_newest_base() { + let catalog = catalog_created_at(&[100, 940]); + let delay = first_delay(&catalog, 1_000, Duration::from_secs(100)); + assert_eq!(delay, Duration::from_secs(40)); + let overdue = first_delay(&catalog, 5_000, Duration::from_secs(100)); + assert_eq!(overdue, Duration::ZERO); + } +} diff --git a/nodedb/src/control/pitr/wal_gc.rs b/nodedb/src/control/pitr/wal_gc.rs new file mode 100644 index 000000000..f742eec00 --- /dev/null +++ b/nodedb/src/control/pitr/wal_gc.rs @@ -0,0 +1,89 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Garbage collection of archived WAL segments no remaining base needs. +//! +//! A segment is named by its first LSN, so its last LSN is below the next +//! segment's first LSN. A segment is collectable only when a later archived +//! segment starts at or below the floor: then every record it holds is below +//! the floor. The segment holding the floor itself is always kept. + +use tracing::info; + +use super::node_life::NodeLife; +use crate::storage::cold::ColdStorage; +use crate::types::Lsn; + +/// First LSNs of the segments wholly below `floor`, ascending. +/// +/// `first_lsns` must be ascending. The result is always a prefix of it, so the +/// segments left form an unbroken suffix of the archive. +pub fn collectable_segments(first_lsns: &[u64], floor: Lsn) -> Vec { + first_lsns + .windows(2) + .take_while(|pair| pair[1] <= floor.as_u64()) + .map(|pair| pair[0]) + .collect() +} + +/// Delete every archived segment of `life` wholly below `floor`, oldest first. +/// +/// Stops at the first failed delete, so the archive keeps an unbroken suffix. +/// Returns the number of segments deleted. +pub async fn collect_archived_wal( + cold: &ColdStorage, + life: &NodeLife, + floor: Lsn, +) -> crate::Result { + let incarnation = life.incarnation.as_str(); + let archived = cold + .archived_wal_segments(life.node_id, incarnation, 0) + .await?; + let mut first_lsns: Vec = archived.keys().copied().collect(); + first_lsns.sort_unstable(); + + let mut collected = 0; + for first_lsn in collectable_segments(&first_lsns, floor) { + let markers = archived + .get(&first_lsn) + .map_or(&[][..], |remote| remote.crc32c.as_slice()); + cold.delete_archived_wal_segment(life.node_id, incarnation, first_lsn, markers) + .await?; + collected += 1; + } + if collected > 0 { + info!( + node_id = life.node_id, + floor = floor.as_u64(), + collected, + "archived WAL below the oldest kept base collected" + ); + } + Ok(collected) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn only_segments_wholly_below_the_floor_are_collectable() { + let segments = [1, 100, 200, 300, 400]; + // 250 lies in the segment starting at 200: it and all above stay. + assert_eq!(collectable_segments(&segments, Lsn::new(250)), [1, 100]); + // A floor on a segment's first LSN keeps that segment. + assert_eq!(collectable_segments(&segments, Lsn::new(200)), [1, 100]); + assert_eq!(collectable_segments(&segments, Lsn::new(199)), [1]); + } + + #[test] + fn the_newest_segment_is_never_collectable() { + assert!(collectable_segments(&[5], Lsn::new(u64::MAX)).is_empty()); + assert_eq!(collectable_segments(&[5, 9], Lsn::new(u64::MAX)), [5]); + assert!(collectable_segments(&[], Lsn::new(10)).is_empty()); + } + + #[test] + fn a_floor_below_the_archive_collects_nothing() { + assert!(collectable_segments(&[100, 200], Lsn::new(50)).is_empty()); + } +} diff --git a/nodedb/src/control/planner/calvin/abort_error.rs b/nodedb/src/control/planner/calvin/abort_error.rs index fe5b31e1f..6eef969e8 100644 --- a/nodedb/src/control/planner/calvin/abort_error.rs +++ b/nodedb/src/control/planner/calvin/abort_error.rs @@ -9,12 +9,21 @@ use crate::Error; /// Pick the error for an ABORT verdict from the reason the verdict recorded. /// /// A participant error never validated a read-set, so it must not be reported -/// as a serialization conflict. `None` — a pre-existing durable verdict with no -/// recorded reason — keeps the historical serialization-conflict reading. -pub fn calvin_abort_error(reason: Option) -> Error { +/// as a serialization conflict. +pub fn calvin_abort_error(reason: AbortReason) -> Error { match reason { - Some(AbortReason::ParticipantError) => Error::CalvinParticipantError, - Some(AbortReason::SerializationConflict) | None => Error::CalvinSerializationConflict, + AbortReason::ParticipantError => Error::CalvinParticipantError, + // Planned against a collection a purge and a same-name create replaced. + AbortReason::CollectionSuperseded => Error::RetryableSchemaChanged { + descriptor: "a collection the transaction writes was dropped and recreated".into(), + }, + // The state a participant checked moved since the coordinator read + // it. A retrying coordinator reads again; any other caller reports a + // conflict the client retries. A multi-part transaction whose parts a + // leader change lost wrote nothing, and a retry succeeds the same way. + AbortReason::SerializationConflict + | AbortReason::PredictionDrift + | AbortReason::PartsLost => Error::CalvinSerializationConflict, } } @@ -25,19 +34,23 @@ mod tests { #[test] fn participant_error_is_not_reported_as_a_serialization_conflict() { assert!(matches!( - calvin_abort_error(Some(AbortReason::ParticipantError)), + calvin_abort_error(AbortReason::ParticipantError), Error::CalvinParticipantError )); } #[test] - fn a_stale_read_set_and_a_reasonless_legacy_verdict_stay_serialization_conflicts() { + fn a_superseded_collection_asks_for_a_retry() { assert!(matches!( - calvin_abort_error(Some(AbortReason::SerializationConflict)), - Error::CalvinSerializationConflict + calvin_abort_error(AbortReason::CollectionSuperseded), + Error::RetryableSchemaChanged { .. } )); + } + + #[test] + fn a_stale_read_set_is_a_serialization_conflict() { assert!(matches!( - calvin_abort_error(None), + calvin_abort_error(AbortReason::SerializationConflict), Error::CalvinSerializationConflict )); } diff --git a/nodedb/src/control/planner/calvin/dependent_recon.rs b/nodedb/src/control/planner/calvin/dependent_recon.rs index 4ff3ae86c..27fb732fd 100644 --- a/nodedb/src/control/planner/calvin/dependent_recon.rs +++ b/nodedb/src/control/planner/calvin/dependent_recon.rs @@ -8,32 +8,36 @@ //! //! - [`plan_needs_implicit_edge_recon`] — the detection gate: given a task set, //! return the collection + database of the first dependent-predicate task -//! (`BulkUpdate`/`BulkDelete`) whose target collection `has_implicit_edges`, -//! else `None`. It does NOT check the not-in-txn-block or registry-available -//! guards — those are per-protocol / session-state concerns and stay at the -//! call sites. +//! (`BulkUpdate`/`BulkDelete`), CRDT document delete or TRUNCATE whose +//! target collection `has_implicit_edges`, else `None`. It does NOT check +//! the not-in-txn-block or registry-available guards — those are +//! per-protocol / session-state concerns and stay at the call sites. //! - [`dispatch_dependent_edge_recon`] — the OLLP orchestration body: pre-exec -//! recon scan → derive mirrored EdgeDelete/EdgePut tasks → atomic Calvin -//! submit → OLLP drift-retry loop. It returns a protocol-neutral +//! recon scan → derive mirrored EdgeDelete/EdgePut tasks, and for a delete +//! the node guards and incident-edge deletes → atomic Calvin submit → OLLP +//! drift-retry loop. It returns a protocol-neutral //! [`DependentReconOutcome`]; each protocol synthesises its own command tags //! from the original task list AFTER this returns `Ok`. use crate::Error; use crate::control::cluster::calvin::executor::ollp::error::OllpError; -use crate::control::planner::calvin::preexec::{PreexecScan, run_preexec_scan}; -use crate::control::planner::calvin::tx_class::collection_name_from_plan; +use crate::control::planner::calvin::preexec::PreexecScan; use crate::control::planner::calvin::{ - DependentOutcome, DependentRetryArgs, build_dependent_tx_class, - build_single_vshard_dependent_tx_class, is_dependent_predicate, predicate_class_for_filters, - run_dependent_with_retry, submit_calvin_routed_assign, + DependentOutcome, DependentRetryArgs, build_single_vshard_dependent_tx_class, + is_dependent_predicate, predicate_class_for_filters, run_dependent_with_retry, + submit_calvin_routed_assign, }; use crate::control::planner::implicit_edges::{ EdgeUpdateCtx, append_implicit_edge_delete_tasks, append_implicit_edge_update_tasks, }; -use crate::control::state::{CalvinApplyResult, SharedState}; +use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId}; use nodedb_physical::physical_plan::OllpPredictedEdge; +use super::dependent_recon_finish::finish_committed; +use super::dependent_recon_node_edges::{ + NodeDeletePlan, node_delete_tasks, planned_edge_deletes, reconnoitre, +}; use super::dependent_recon_plan::{inject_ollp_predicted_edges, inject_ollp_surrogates}; use super::dependent_recon_predicate::{ EdgeLifecycle, classify_edge_lifecycle, extract_bulk_predicate_info, @@ -56,15 +60,15 @@ pub struct DependentReconOutcome { pub apply_result: Option, } -/// Detect whether `tasks` carry a dependent predicate on an implicit-edge- -/// bearing collection, requiring the OLLP/Calvin recon path. +/// Detect whether `tasks` carry a dependent predicate or a TRUNCATE on an +/// implicit-edge-bearing collection, requiring the edge recon path. /// /// Returns `Some((collection, database_id))` of the FIRST dependent-predicate -/// task (`BulkUpdate`/`BulkDelete`) whose target collection has +/// task (`BulkUpdate`/`BulkDelete`) or TRUNCATE whose target collection has /// `has_implicit_edges` set in the catalog, else `None`. /// /// A genuine catalog READ error propagates as a typed [`crate::Error`]: -/// misrouting a delete on a real I/O fault would silently skip edge cleanup +/// misrouting a delete on a real I/O fault will silently skip edge cleanup /// (dangling edges). An ABSENT catalog (`None`) or absent collection row /// (`Ok(None)`) is treated as non-edge-bearing and yields `None`. /// @@ -76,15 +80,25 @@ pub fn plan_needs_implicit_edge_recon( tasks: &[PhysicalTask], tenant_id: TenantId, ) -> crate::Result> { - let Some(dep_task) = tasks.iter().find(|t| is_dependent_predicate(&t.plan)) else { + let Some(dep_task) = tasks.iter().find(|t| is_edge_recon_plan(&t.plan)) else { return Ok(None); }; - let coll = collection_name_from_plan(&dep_task.plan); + // Every edge recon plan names its database-qualified collection. + let coll = dep_task + .plan + .collection() + .ok_or_else(|| Error::Internal { + detail: "internal invariant break: an edge recon plan names no collection".into(), + })? + .to_owned(); let db = dep_task.database_id; let edge_bearing = { + // The plan names the collection database-qualified. The catalog keys + // collections by the bare name. + let bare = crate::control::target_identity::naming::bare_collection_name(db, &coll); let catalog = state.credentials.catalog(); catalog - .get_collection(db, tenant_id.as_u64(), &coll)? + .get_collection(db, tenant_id.as_u64(), &bare)? .map(|c| c.has_implicit_edges) .unwrap_or(false) }; @@ -95,6 +109,15 @@ pub fn plan_needs_implicit_edge_recon( } } +/// Whether `plan` takes the edge recon path when its collection is +/// edge-bearing: a dependent predicate write, a CRDT document delete, or a +/// TRUNCATE. +pub fn is_edge_recon_plan(plan: &nodedb_physical::physical_plan::PhysicalPlan) -> bool { + is_dependent_predicate(plan) + || super::dependent_recon_crdt::is_crdt_doc_delete(plan) + || super::edge_truncate::is_truncate(plan) +} + /// Drive the implicit-edge OLLP/Calvin reconnaissance dispatch for `tasks`. /// /// The coordinator owns the OLLP retry loop. This: @@ -104,12 +127,16 @@ pub fn plan_needs_implicit_edge_recon( /// reconciles them against the SET clause — overrides parsed ONCE here as /// they are constant across retries). /// 2. Runs an initial pre-execution reconnaissance scan to predict the matched -/// surrogate set + the implicit edges of any matched edge documents. +/// surrogate set + the implicit edges of any matched edge documents. A +/// delete also reads each matched row's node identity and the node's +/// incident edges in the collection. /// 3. Submits a Calvin transaction (routed to the sequencer-group leader via /// `submit_calvin_routed_assign`) that mirrors the doc write together with -/// the derived EdgeDelete/EdgePut tasks, ATOMICALLY. -/// 4. On a POST-EXEC predicate-drift mismatch, re-scans (FRESH reconnaissance) -/// and resubmits, via [`run_dependent_with_retry`]. +/// the derived EdgeDelete/EdgePut tasks, ATOMICALLY. A delete adds one +/// `NodeEdgeGuard` per node and one `EdgeDelete` per incident edge. +/// 4. On a POST-EXEC predicate-drift mismatch or a guard's drift abort, +/// re-scans (FRESH reconnaissance) and resubmits, via +/// [`run_dependent_with_retry`]. /// /// Returns a protocol-neutral [`DependentReconOutcome`]; the caller synthesises /// its own per-task command tags from the original task list. All errors are @@ -118,39 +145,24 @@ pub fn plan_needs_implicit_edge_recon( /// `database_id` is supplied by the caller (it comes from the detection gate, /// [`plan_needs_implicit_edge_recon`]) so it does not have to be re-derived. /// -/// `allow_single_vshard` selects the participant floor of the `TxClass` this -/// builds: `false` (the normal multi-shard OLLP callers — the pgwire and -/// native predicate-dispatch gates) uses the strict -/// [`build_dependent_tx_class`], which rejects a write set that collapses to -/// one vshard. `true` is the explicit opt-in used ONLY by the contended -/// single-collection predicate-write routing path -/// (`route_write_to_calvin`'s dependent-predicate branch, reached when the -/// write-admission gate returns `RouteToCalvin`): it uses -/// [`build_single_vshard_dependent_tx_class`] so a single-collection -/// `BulkUpdate`/`BulkDelete` that legitimately resolves to one vshard -/// sequences through the scheduler instead of being rejected. +/// The `TxClass` this builds accepts a write set on one vShard +/// ([`build_single_vshard_dependent_tx_class`]). A delete whose row, node +/// guard and edges all home on one vShard is a legitimate one-vShard +/// transaction, and so is a contended single-collection predicate write +/// routed here by the write-admission gate. pub async fn dispatch_authorized_dependent_edge_recon( state: &SharedState, authorized: crate::control::server::shared::authorization::AuthorizedTaskSet, identity: &crate::control::security::identity::AuthenticatedIdentity, tenant_id: TenantId, database_id: DatabaseId, - allow_single_vshard: bool, ) -> crate::Result { let tasks = authorized .into_tasks() .into_iter() .map(|task| task.into_physical_task()) .collect(); - dispatch_dependent_edge_recon_inner( - state, - tasks, - Some(identity), - tenant_id, - database_id, - allow_single_vshard, - ) - .await + dispatch_dependent_edge_recon_inner(state, tasks, Some(identity), tenant_id, database_id).await } pub(crate) async fn dispatch_dependent_edge_recon( @@ -158,17 +170,8 @@ pub(crate) async fn dispatch_dependent_edge_recon( tasks: Vec, tenant_id: TenantId, database_id: DatabaseId, - allow_single_vshard: bool, ) -> crate::Result { - dispatch_dependent_edge_recon_inner( - state, - tasks, - None, - tenant_id, - database_id, - allow_single_vshard, - ) - .await + dispatch_dependent_edge_recon_inner(state, tasks, None, tenant_id, database_id).await } async fn dispatch_dependent_edge_recon_inner( @@ -177,8 +180,31 @@ async fn dispatch_dependent_edge_recon_inner( identity: Option<&crate::control::security::identity::AuthenticatedIdentity>, tenant_id: TenantId, database_id: DatabaseId, - allow_single_vshard: bool, ) -> crate::Result { + // A TRUNCATE empties the collection and tombstones its edges on every + // vShard, in one transaction. + if let Some(truncated) = + super::edge_truncate::dispatch_truncate(state, &tasks, identity, tenant_id).await? + { + return Ok(DependentReconOutcome { + tasks_dispatched: tasks.len() as u64, + apply_result: truncated.apply_result, + }); + } + // A CRDT document delete tombstones its node's edges in its transaction. + if !tasks.iter().any(|t| is_dependent_predicate(&t.plan)) + && let Some(outcome) = super::dependent_recon_crdt::dispatch_crdt_doc_deletes( + state, + &tasks, + identity, + tenant_id, + database_id, + ) + .await? + { + return Ok(outcome); + } + let orchestrator = state.ollp_orchestrator.get(); let registry = state .calvin_completion_registry @@ -205,12 +231,13 @@ async fn dispatch_dependent_edge_recon_inner( let edge_mode = classify_edge_lifecycle(&dep_task.plan)?; // Initial reconnaissance — the first prediction the loop submits. - let initial_predicted = run_preexec_scan( + let initial_predicted = reconnoitre( state, tenant_id, database_id, &dep_collection, dep_filter_bytes.clone(), + &edge_mode, ) .await?; @@ -226,6 +253,7 @@ async fn dispatch_dependent_edge_recon_inner( let submit = |predicted: &PreexecScan| { let surrogates = predicted.surrogates.clone(); let edges = predicted.edges.clone(); + let node_edges = predicted.node_edges.clone(); let tasks = &tasks; let dep_collection = &dep_collection; let edge_mode = &edge_mode; @@ -262,8 +290,11 @@ async fn dispatch_dependent_edge_recon_inner( .collect(); let mut edge_tasks: Vec = Vec::new(); + // Node-delete guards run before every other task of the + // transaction, so each compares the edges the transaction found. + let mut guard_tasks: Vec = Vec::new(); match edge_mode { - EdgeLifecycle::Delete => { + EdgeLifecycle::Delete { .. } => { append_implicit_edge_delete_tasks( state, &mut edge_tasks, @@ -275,6 +306,23 @@ async fn dispatch_dependent_edge_recon_inner( ) .await .map_err(|e| OllpError::Terminal(Box::new(e)))?; + // Each deleted row is a graph node: its incident edges in + // this collection are tombstoned in this transaction. + let (guards, deletes) = node_delete_tasks( + state, + tenant_id, + database_id, + NodeDeletePlan { + collection: dep_collection, + guarded: &node_edges, + deleted: &node_edges, + already_deleted: &planned_edge_deletes(&edge_tasks), + }, + ) + .await + .map_err(|e| OllpError::Terminal(Box::new(e)))?; + guard_tasks = guards; + edge_tasks.extend(deletes); } EdgeLifecycle::Update(overrides) => { append_implicit_edge_update_tasks( @@ -295,7 +343,8 @@ async fn dispatch_dependent_edge_recon_inner( } } - let mut submission_tasks: Vec = tasks.to_vec(); + let mut submission_tasks: Vec = guard_tasks; + submission_tasks.extend(tasks.iter().cloned()); submission_tasks.extend(edge_tasks); if let Some(identity) = identity { let emitter = crate::control::security::audit::ArcAuditEmitter( @@ -328,31 +377,22 @@ async fn dispatch_dependent_edge_recon_inner( // only touch the original BulkUpdate/BulkDelete // doc tasks (no-ops on any other plan); the // edge-delete tasks are appended AFTER, so they - // are untouched. The tx_builder may run more than + // are untouched. The tx_builder can run more than // once, so clone the predicted sets per task. inject_ollp_surrogates(&mut t.plan, surrogates.clone()); inject_ollp_predicted_edges(&mut t.plan, predicted_edges.clone()); t }) .collect(); - let built = if allow_single_vshard { - build_single_vshard_dependent_tx_class( - &modified_tasks, - tenant_id, - dep_collection, - &surrogates, - &[], - ) - } else { - build_dependent_tx_class( - &modified_tasks, - tenant_id, - dep_collection, - &surrogates, - &[], - ) - }; - built.map_err(|e| OllpError::Terminal(Box::new(e))) + let tx_class = build_single_vshard_dependent_tx_class( + &modified_tasks, + tenant_id, + dep_collection, + &surrogates, + &[], + ) + .map_err(|e| OllpError::Terminal(Box::new(e)))?; + Ok(tx_class) }, // Retryable: a routed submit races a leader change. The cause // travels so exhaustion names it instead of claiming drift. @@ -366,18 +406,19 @@ async fn dispatch_dependent_edge_recon_inner( } }; - // `rescan`: FRESH reconnaissance on each post-exec mismatch. + // `rescan`: FRESH reconnaissance on each post-exec mismatch or drift. let rescan = || { - run_preexec_scan( + reconnoitre( state, tenant_id, database_id, &dep_collection, dep_filter_bytes.clone(), + &edge_mode, ) }; - let completed_txn = match run_dependent_with_retry(DependentRetryArgs { + let (completed_txn, ack_results) = match run_dependent_with_retry(DependentRetryArgs { registry, orchestrator: orc, predicate_class_hash: pred_class, @@ -389,7 +430,10 @@ async fn dispatch_dependent_edge_recon_inner( }) .await? { - DependentOutcome::Committed(txn_id) => txn_id, + DependentOutcome::Committed { + txn_id, + ack_results, + } => (txn_id, ack_results), // The predicate matched no rows and nothing else in the batch writes, // so no entry was sequenced. The statement reports zero rows affected. DependentOutcome::NoOp => { @@ -399,50 +443,5 @@ async fn dispatch_dependent_edge_recon_inner( }); } }; - - // A write to a permission-tree source is acknowledged only once it binds - // every node. Tree sources live in the default database. - let sources = state.authorization_fence.sources(); - let binds_authorization = database_id == crate::types::DatabaseId::DEFAULT - && tasks.iter().any(|task| { - task.plan - .named_collections() - .iter() - .any(|collection| sources.is_source_collection(collection)) - }); - if binds_authorization { - crate::control::security::auth_lease::calvin_write_barrier(state).await?; - } - - // Completion fired: the scheduler deposited the applied Response (with any - // RETURNING rows) into the sidecar before proposing the ack that woke the - // retry loop, so the entry is present now if this write carried RETURNING. - // Drain it (removing the entry) for the caller to shape into DATA-ROWs; a - // `Conflict` (>1 RETURNING participant) fails loudly rather than returning a - // partial cross-shard union. - let drained = state - .calvin - .apply_results - .lock() - .unwrap_or_else(|p| p.into_inner()) - .remove(&completed_txn); - let apply_result = match drained { - Some(CalvinApplyResult::Single { response, .. }) => { - // An installed txn whose reply failed to render deposits it as an - // error for the statement. - crate::control::local_dispatch::reject_data_plane_error(&response)?; - Some(response) - } - Some(CalvinApplyResult::Conflict) => { - return Err(Error::Internal { - detail: "multi-participant cross-shard RETURNING not supported".to_owned(), - }); - } - None => None, - }; - - Ok(DependentReconOutcome { - tasks_dispatched: tasks.len() as u64, - apply_result, - }) + finish_committed(state, &tasks, completed_txn, &ack_results).await } diff --git a/nodedb/src/control/planner/calvin/dependent_recon_crdt.rs b/nodedb/src/control/planner/calvin/dependent_recon_crdt.rs new file mode 100644 index 000000000..238961314 --- /dev/null +++ b/nodedb/src/control/planner/calvin/dependent_recon_crdt.rs @@ -0,0 +1,305 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A CRDT document delete on the edge recon path. +//! +//! A CRDT document is keyed by its document id, and that id names its graph +//! node. Deleting documents of an edge-bearing CRDT collection runs as one +//! Calvin transaction: a `NodeEdgeGuard` per node whose document is stored, +//! a `NodePresenceGuard` on the collection's vShard, the `DocDelete`s, and +//! an `EdgeDelete` per edge of the collection incident on each stored +//! document's node. +//! +//! A delete of a missing document removes nothing, so its node's edges +//! stay. The planner reads which documents are stored, and the presence +//! guard proves none appeared or vanished since. A guard that finds its +//! state changed aborts the transaction with a drift verdict, and the +//! coordinator reads again and resubmits. + +use std::collections::BTreeSet; + +use nodedb_physical::physical_plan::{CrdtOp, GraphOp, PhysicalPlan}; +use nodedb_physical::physical_task::PhysicalTask; + +use super::dependent_recon::DependentReconOutcome; +use super::dependent_recon_finish::finish_committed; +use super::dependent_recon_node_edges::{ + NodeDeletePlan, NodeIncidentEdges, node_delete_tasks, read_node_incident_edges, +}; +use super::edge_truncate::collection_is_edge_bearing; +use super::{ + DependentOutcome, DependentRetryArgs, build_single_vshard_tx_class, + predicate_class_for_filters, run_dependent_with_retry, submit_calvin_routed_assign, +}; +use crate::Error; +use crate::control::cluster::calvin::executor::ollp::error::OllpError; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::graph_dispatch::{VShardRead, read_on_vshard}; +use crate::control::state::SharedState; +use crate::types::VShardId; +use crate::types::{DatabaseId, TenantId}; + +/// Whether `plan` is a CRDT document delete. +pub fn is_crdt_doc_delete(plan: &PhysicalPlan) -> bool { + matches!(plan, PhysicalPlan::Crdt(CrdtOp::DocDelete { .. })) +} + +/// Dispatch the CRDT document deletes among `tasks` with their nodes' edge +/// tombstones. `None` when `tasks` holds no CRDT document delete. +pub(super) async fn dispatch_crdt_doc_deletes( + state: &SharedState, + tasks: &[PhysicalTask], + identity: Option<&AuthenticatedIdentity>, + tenant_id: TenantId, + database_id: DatabaseId, +) -> crate::Result> { + // The node every deleted document id names, bound or not. An id unbound + // at planning names no stored document, so the read finds it absent and + // the presence guard names it: a writer that binds and stores it before + // the delete's turn fails the guard, and the retry deletes it with its + // edges. `template` is the first delete: its vShard is the collection's. + let mut template: Option<(&PhysicalTask, String)> = None; + let mut nodes = Vec::new(); + for task in tasks { + if let PhysicalPlan::Crdt(CrdtOp::DocDelete { + collection: coll, + document_id, + .. + }) = &task.plan + { + template.get_or_insert_with(|| (task, coll.as_str().to_string())); + nodes.push(document_id.clone()); + } + } + let Some((template, collection)) = template else { + return Ok(None); + }; + if !collection_is_edge_bearing(state, tenant_id, database_id, &collection)? { + nodes.clear(); + } + nodes.sort_unstable(); + nodes.dedup(); + + let orc = state + .ollp_orchestrator + .get() + .ok_or(Error::SequencerUnavailable)?; + let registry = state + .calvin_completion_registry + .get() + .ok_or(Error::SequencerUnavailable)?; + let pred_class = predicate_class_for_filters(&[], &collection); + let read = || { + read_crdt_nodes( + state, + CrdtNodes { + tenant_id, + database_id, + collection: &collection, + vshard: template.vshard_id, + }, + &nodes, + ) + }; + let initial_predicted = read().await?; + + let submit = |predicted: &CrdtDeleteRead| { + let predicted = predicted.clone(); + let collection = &collection; + async move { + let (guards, deletes) = node_delete_tasks( + state, + tenant_id, + database_id, + NodeDeletePlan { + collection, + guarded: &predicted.edges, + deleted: &predicted.edges, + already_deleted: &Default::default(), + }, + ) + .await + .map_err(|e| OllpError::Terminal(Box::new(e)))?; + let mut submission: Vec = guards; + submission.extend(presence_guard(template, collection, &predicted)); + submission.extend(tasks.iter().cloned()); + submission.extend(deletes); + if let Some(identity) = identity { + let emitter = crate::control::security::audit::ArcAuditEmitter( + std::sync::Arc::clone(&state.audit), + ); + submission = crate::control::server::shared::authorization::authorize_task_set( + identity, + &submission, + &state.permissions, + &state.roles, + &emitter, + ) + .map_err(|e| OllpError::Terminal(Box::new(e.into())))? + .into_tasks() + .into_iter() + .map(|task| task.into_physical_task()) + .collect(); + } + orc.submit_with_retry_via( + pred_class, + tenant_id, + || { + let tx_class = build_single_vshard_tx_class(&submission, tenant_id, &[]) + .map_err(|e| OllpError::Terminal(Box::new(e)))?; + Ok(Some(tx_class)) + }, + |tx_class| async move { + submit_calvin_routed_assign(state, tx_class) + .await + .map_err(|e| OllpError::Retryable(Box::new(e))) + }, + ) + .await + } + }; + + let outcome = run_dependent_with_retry(DependentRetryArgs { + registry, + orchestrator: orc, + predicate_class_hash: pred_class, + timeout: std::time::Duration::from_secs(state.tuning.network.default_deadline_secs), + ollp_max_retries: orc.ollp_max_retries() as u32, + initial_predicted, + submit, + rescan: read, + }) + .await?; + match outcome { + DependentOutcome::Committed { + txn_id, + ack_results, + } => finish_committed(state, tasks, txn_id, &ack_results) + .await + .map(Some), + DependentOutcome::NoOp => Ok(Some(DependentReconOutcome { + tasks_dispatched: 0, + apply_result: None, + })), + } +} + +/// What the planner read of a CRDT delete's nodes: which documents are +/// stored, and the incident edges of the stored ones. +#[derive(Debug, Clone, Default, PartialEq, Eq)] +struct CrdtDeleteRead { + present: Vec, + absent: Vec, + edges: Vec, +} + +/// Where a CRDT delete's nodes are read. +#[derive(Clone, Copy)] +struct CrdtNodes<'a> { + tenant_id: TenantId, + database_id: DatabaseId, + collection: &'a str, + /// The collection's vShard, which stores its documents. + vshard: VShardId, +} + +/// Read which of `nodes` have a stored document, on the collection's +/// vShard, and the incident edges of those that do. A delete of a missing +/// document removes nothing, so its node's edges stay. +async fn read_crdt_nodes( + state: &SharedState, + at: CrdtNodes<'_>, + nodes: &[String], +) -> crate::Result { + if nodes.is_empty() { + return Ok(CrdtDeleteRead::default()); + } + let plan = PhysicalPlan::Graph(GraphOp::NodePresenceRead { + collection: nodedb_types::QualifiedCollection::from_stored(at.collection.to_string()), + vshard: at.vshard.as_u32(), + ids: nodes.to_vec(), + }); + let payload = read_on_vshard( + state, + VShardRead { + tenant_id: at.tenant_id, + database_id: at.database_id, + vshard_id: at.vshard.as_u32(), + txn_id: None, + linearizable: true, + }, + plan, + ) + .await?; + let stored: BTreeSet = if payload.as_bytes().is_empty() { + BTreeSet::new() + } else { + zerompk::from_msgpack::>(payload.as_bytes()) + .map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("crdt delete presence read: {e}"), + })? + .into_iter() + .collect() + }; + let (present, absent): (Vec, Vec) = nodes + .iter() + .cloned() + .partition(|node| stored.contains(node)); + let edges = read_node_incident_edges( + state, + at.tenant_id, + at.database_id, + at.collection, + &present, + None, + ) + .await?; + Ok(CrdtDeleteRead { + present, + absent, + edges, + }) +} + +/// The presence guard of `read` on the collection's vShard, beside the +/// delete `template`. `None` when the delete names no bound node. +fn presence_guard( + template: &PhysicalTask, + collection: &str, + read: &CrdtDeleteRead, +) -> Option { + if read.present.is_empty() && read.absent.is_empty() { + return None; + } + Some(PhysicalTask { + plan: PhysicalPlan::Graph(GraphOp::NodePresenceGuard { + collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), + vshard: template.vshard_id.as_u32(), + present: read.present.clone(), + absent: read.absent.clone(), + }), + ..template.clone() + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn only_a_crdt_document_delete_takes_this_path() { + let delete = PhysicalPlan::Crdt(CrdtOp::DocDelete { + collection: nodedb_types::QualifiedCollection::from_stored("c".to_string()), + document_id: "a".to_string(), + surrogate: None, + returning: None, + rls_filters: Vec::new(), + }); + assert!(is_crdt_doc_delete(&delete)); + let read = PhysicalPlan::Crdt(CrdtOp::Read { + collection: nodedb_types::QualifiedCollection::from_stored("c".to_string()), + document_id: "a".to_string(), + }); + assert!(!is_crdt_doc_delete(&read)); + } +} diff --git a/nodedb/src/control/planner/calvin/dependent_recon_finish.rs b/nodedb/src/control/planner/calvin/dependent_recon_finish.rs new file mode 100644 index 000000000..5f6b210f2 --- /dev/null +++ b/nodedb/src/control/planner/calvin/dependent_recon_finish.rs @@ -0,0 +1,67 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The outcome of a committed edge recon transaction. + +use nodedb_physical::physical_task::PhysicalTask; + +use super::dependent_recon::DependentReconOutcome; +use super::submit::local::{ReplyFold, with_reported_results}; +use crate::Error; +use crate::control::state::{CalvinApplyResult, SharedState}; + +/// The outcome of a committed recon transaction over `tasks`: waits for the +/// authorization barrier when a task writes a permission-tree source, then +/// drains the applied response the scheduler deposited. +pub(super) async fn finish_committed( + state: &SharedState, + tasks: &[PhysicalTask], + completed_txn: nodedb_cluster::calvin::TxnId, + ack_results: &[Vec], +) -> crate::Result { + // A write to a permission-tree source is acknowledged only once it binds + // every node. + let sources = state.authorization_fence.sources(); + let binds_authorization = tasks.iter().any(|task| { + task.plan + .named_collections() + .iter() + .any(|collection| sources.is_source_collection(collection)) + }); + if binds_authorization { + crate::control::security::auth_lease::calvin_write_barrier(state).await?; + } + + // Completion fired: the scheduler deposited the applied Response (with any + // RETURNING rows) into the sidecar before proposing the ack that woke the + // retry loop, so the entry is present now if this write carried RETURNING. + // Drain it (removing the entry) for the caller to shape into DATA-ROWs; a + // `Conflict` (>1 RETURNING participant) fails loudly rather than returning a + // partial cross-shard union. + let drained = state.calvin.apply_results.take(&completed_txn); + let applied = match drained { + Some(CalvinApplyResult::Single { + response, + has_returning, + }) => { + // An installed txn whose reply failed to render deposits it as an + // error for the statement. + crate::control::local_dispatch::reject_data_plane_error(&response)?; + Some((response, has_returning)) + } + Some(CalvinApplyResult::Conflict) => { + return Err(Error::Internal { + detail: "multi-participant cross-shard RETURNING not supported".to_owned(), + }); + } + None => None, + }; + // A participant on a node that holds no replica of it reports its + // counts and RETURNING rows through its completion ack. + let fold = ReplyFold::of_plans(tasks.iter().map(|task| &task.plan)); + let apply_result = with_reported_results(applied, ack_results, fold)?; + + Ok(DependentReconOutcome { + tasks_dispatched: tasks.len() as u64, + apply_result, + }) +} diff --git a/nodedb/src/control/planner/calvin/dependent_recon_node_edges.rs b/nodedb/src/control/planner/calvin/dependent_recon_node_edges.rs new file mode 100644 index 000000000..dbe90cc81 --- /dev/null +++ b/nodedb/src/control/planner/calvin/dependent_recon_node_edges.rs @@ -0,0 +1,518 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Incident-edge reconnaissance and task generation for a node delete. +//! +//! A document row of an edge-bearing collection is also a graph node, keyed +//! by the row's identity. Deleting the row tombstones every live edge of the +//! same collection incident on that node, in the delete's own Calvin +//! transaction: +//! +//! - The recon reads the node's incident edges on the node's key vShard, +//! which holds every edge incident on the node. +//! - The transaction carries one `EdgeDelete` per edge. Each one stamps the +//! ordinal the transaction decides, identically on both homes of the edge. +//! - The transaction carries one `NodeEdgeGuard` per node, on the node's key +//! vShard. It refuses the transaction with `OllpRetryRequired` when the +//! node's live edges differ from the ones the recon read. The coordinator +//! then reads again and resubmits. +//! +//! Edges of other collections that name the node are not touched. + +use std::collections::{BTreeSet, HashMap}; + +use nodedb_physical::physical_plan::{BatchEdge, GraphOp, PhysicalPlan}; +use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; +use nodedb_types::graph::Direction; +use nodedb_types::{QualifiedCollection, Surrogate, SystemTimeScope, Value}; + +use super::dependent_recon_predicate::EdgeLifecycle; +use super::preexec::{PreexecRequest, PreexecScan, RowIdentityRead, run_preexec_scan}; +use crate::control::server::graph_dispatch::read_on_key_owner; +use crate::control::server::surrogate_exchange::assign_surrogates_routed; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, RecordHomes, TenantId, TraceId, TxnId, VShardId}; + +/// One graph edge as `(src, label, dst)`. +pub type EdgeTriple = (String, String, String); + +/// The live edges of one collection incident on one node, as the recon read +/// them. `edges` is sorted and holds no duplicates. +#[derive(Clone, Debug, PartialEq, Eq)] +pub struct NodeIncidentEdges { + pub node: String, + pub edges: Vec, +} + +/// Run the dependent write's reconnaissance: the predicate scan, and for a +/// delete, the incident edges of every matched row's node. The initial +/// prediction and every rescan after drift both come from here. +pub(super) async fn reconnoitre( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + filters: Vec, + lifecycle: &EdgeLifecycle, +) -> crate::Result { + let identity = match lifecycle { + EdgeLifecycle::Delete { + declared_primary_key, + } => RowIdentityRead::Read { + declared_primary_key: declared_primary_key.as_deref(), + }, + EdgeLifecycle::Update(_) => RowIdentityRead::Skip, + }; + let mut scan = run_preexec_scan( + state, + tenant_id, + database_id, + PreexecRequest { + collection, + filters, + prefilter: None, + identity, + txn_id: None, + }, + ) + .await?; + if let EdgeLifecycle::Delete { .. } = lifecycle { + scan.node_edges = read_node_incident_edges( + state, + tenant_id, + database_id, + collection, + &scan.identities, + None, + ) + .await?; + } + Ok(scan) +} + +/// Read the live edges of `collection` incident on each node in `nodes`. +/// +/// `collection` is the database-qualified name. Each node's edges are read +/// on the leader of its key vShard, out-edges and in-edges separately. A +/// self-loop appears in both reads and is kept once. With `txn_id`, the read +/// also folds in the edge writes that session transaction staged. +pub(super) async fn read_node_incident_edges( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, + nodes: &[String], + txn_id: Option, +) -> crate::Result> { + let mut out = Vec::with_capacity(nodes.len()); + for node in nodes { + let mut edges = BTreeSet::new(); + for direction in [Direction::Out, Direction::In] { + // The current-state edge-store read, the source the node's guard + // checks. The CSR can lag it, and a read that disagrees with the + // guard refuses every retry. + let plan = PhysicalPlan::Graph(GraphOp::TemporalNeighbors { + collection: QualifiedCollection::from_stored(collection.to_string()), + node_id: node.clone(), + edge_label: None, + direction, + system_time: SystemTimeScope::Current, + valid_at_ms: None, + rls_filters: Vec::new(), + }); + let payload = + read_on_key_owner(state, tenant_id, database_id, node, plan, txn_id, true).await?; + for (label, other) in decode_neighbors(payload.as_bytes())? { + edges.insert(match direction { + Direction::In => (other, label, node.clone()), + Direction::Out | Direction::Both => (node.clone(), label, other), + }); + } + } + out.push(NodeIncidentEdges { + node: node.clone(), + edges: edges.into_iter().collect(), + }); + } + Ok(out) +} + +/// Decode a neighbors payload, a msgpack array of `{label, node}` maps. +fn decode_neighbors(payload: &[u8]) -> crate::Result> { + if payload.is_empty() { + return Ok(Vec::new()); + } + let value = + nodedb_types::value_from_msgpack(payload).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("node delete incident-edge read: {e}"), + })?; + let Value::Array(entries) = value else { + return Err(crate::Error::Serialization { + format: "msgpack".into(), + detail: "node delete incident-edge read: payload is not an array".into(), + }); + }; + let mut out = Vec::with_capacity(entries.len()); + for entry in &entries { + let Value::Object(fields) = entry else { + continue; + }; + if let (Some(label), Some(node)) = ( + fields.get("label").and_then(Value::as_str), + fields.get("node").and_then(Value::as_str), + ) { + out.push((label.to_string(), node.to_string())); + } + } + Ok(out) +} + +/// The `(src, label, dst)` of every `EdgeDelete` among `tasks`. +pub(super) fn planned_edge_deletes(tasks: &[PhysicalTask]) -> BTreeSet { + tasks + .iter() + .filter_map(|task| match &task.plan { + PhysicalPlan::Graph(GraphOp::EdgeDelete { + src_id, + label, + dst_id, + .. + }) => Some((src_id.clone(), label.clone(), dst_id.clone())), + _ => None, + }) + .collect() +} + +/// What a node delete removes and what its guards check. +pub(super) struct NodeDeletePlan<'a> { + /// The database-qualified collection the rows and edges belong to. + pub collection: &'a str, + /// Each node's live edges in the store. One `NodeEdgeGuard` per node + /// checks the store still holds exactly these. + pub guarded: &'a [NodeIncidentEdges], + /// Each node's edges the transaction tombstones: the guarded edges, plus + /// the edges a session transaction staged itself. + pub deleted: &'a [NodeIncidentEdges], + /// Edges another task of the transaction already deletes. + pub already_deleted: &'a BTreeSet, +} + +/// Build the node-delete tasks of `plan`: one `NodeEdgeGuard` per guarded +/// node, and one `EdgeDelete` per deleted edge that `already_deleted` does +/// not hold. Returns `(guards, deletes)`. +/// +/// Guards come first in a transaction's task list. A guard compares the +/// store as the transaction found it, before any of the transaction's own +/// deletes. +pub(super) async fn node_delete_tasks( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + plan: NodeDeletePlan<'_>, +) -> crate::Result<(Vec, Vec)> { + // Every endpoint's surrogate, resolved in one batch. The edge's insert + // bound each endpoint in the collection, so none is minted here. + let endpoints: BTreeSet<&str> = plan + .guarded + .iter() + .chain(plan.deleted) + .flat_map(|node| node.edges.iter()) + .flat_map(|(src, _, dst)| [src.as_str(), dst.as_str()]) + .collect(); + let endpoints: Vec<&str> = endpoints.into_iter().collect(); + let key = nodedb_types::CollectionKey::from_qualified_str(database_id, plan.collection)?; + let keys: Vec<&[u8]> = endpoints.iter().map(|node| node.as_bytes()).collect(); + let surrogates = assign_surrogates_routed(state, key, tenant_id, &keys, TraceId::ZERO).await?; + let bound: HashMap<&str, Surrogate> = endpoints.into_iter().zip(surrogates).collect(); + build_node_delete_tasks(tenant_id, database_id, plan, &bound) +} + +/// [`node_delete_tasks`] with every endpoint's surrogate in `bound`. +fn build_node_delete_tasks( + tenant_id: TenantId, + database_id: DatabaseId, + plan: NodeDeletePlan<'_>, + bound: &HashMap<&str, Surrogate>, +) -> crate::Result<(Vec, Vec)> { + let NodeDeletePlan { + collection, + guarded, + deleted: to_delete, + already_deleted, + } = plan; + let surrogate_of = |node: &str| { + bound + .get(node) + .copied() + .ok_or_else(|| crate::Error::Internal { + detail: format!("node delete: no surrogate resolved for endpoint '{node}'"), + }) + }; + + let qualified = QualifiedCollection::from_stored(collection.to_string()); + let mut deletes = Vec::new(); + let mut deleted: BTreeSet<&EdgeTriple> = BTreeSet::new(); + for node in to_delete { + for edge in &node.edges { + let (src, label, dst) = edge; + // An edge between two deleted nodes, or one an edge document's + // own delete already covers, is deleted once. + if already_deleted.contains(edge) || !deleted.insert(edge) { + continue; + } + let (src_surrogate, dst_surrogate) = (surrogate_of(src)?, surrogate_of(dst)?); + deletes.push(PhysicalTask { + tenant_id, + vshard_id: RecordHomes::edge(src, dst).owner(), + database_id, + plan: PhysicalPlan::Graph(GraphOp::EdgeDelete { + collection: qualified.clone(), + src_id: src.clone(), + label: label.clone(), + dst_id: dst.clone(), + src_surrogate, + dst_surrogate, + // The node's own delete is what removes the edge, and the + // policy on this collection decided that delete before + // this task was derived. + rls_write_check: nodedb_types::RlsWriteCheck::decided_earlier_in_request(), + }), + post_set_op: PostSetOp::None, + txn_id: None, + }); + } + } + let mut guards = Vec::with_capacity(guarded.len()); + for node in guarded { + let mut expected = Vec::with_capacity(node.edges.len()); + for (src, label, dst) in &node.edges { + expected.push(BatchEdge { + collection: qualified.clone(), + src_id: src.clone(), + label: label.clone(), + dst_id: dst.clone(), + src_surrogate: surrogate_of(src)?, + dst_surrogate: surrogate_of(dst)?, + }); + } + guards.push(PhysicalTask { + tenant_id, + vshard_id: VShardId::from_key(node.node.as_bytes()), + database_id, + plan: PhysicalPlan::Graph(GraphOp::NodeEdgeGuard { + collection: qualified.clone(), + node_id: node.node.clone(), + expected, + }), + post_set_op: PostSetOp::None, + txn_id: None, + }); + } + Ok((guards, deletes)) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn triple(src: &str, dst: &str) -> EdgeTriple { + (src.to_string(), "KNOWS".to_string(), dst.to_string()) + } + + /// A node on another key vShard than `node`. + fn node_homed_apart_from(node: &str) -> String { + let home = VShardId::from_key(node.as_bytes()); + (0..) + .map(|i| format!("peer{i}")) + .find(|peer| VShardId::from_key(peer.as_bytes()) != home) + .unwrap_or_default() + } + + /// A node delete whose incident edge's other endpoint homes on another + /// vShard guards the node on its key home and deletes the edge from the + /// owner, and the transaction takes both homes of the edge as + /// participants. Both homes then apply one `EdgeDelete` under the one + /// ordinal the transaction decides. + #[test] + fn a_cross_shard_incident_edge_is_deleted_on_both_homes() { + let peer = node_homed_apart_from("alice"); + let nodes = vec![NodeIncidentEdges { + node: "alice".to_string(), + edges: vec![triple("alice", &peer)], + }]; + let bound: HashMap<&str, Surrogate> = HashMap::from([ + ("alice", Surrogate::new(1)), + (peer.as_str(), Surrogate::new(2)), + ]); + let (guards, deletes) = build_node_delete_tasks( + TenantId::new(1), + DatabaseId::DEFAULT, + NodeDeletePlan { + collection: "users", + guarded: &nodes, + deleted: &nodes, + already_deleted: &BTreeSet::new(), + }, + &bound, + ) + .expect("tasks"); + + assert_eq!(guards.len(), 1); + assert_eq!(guards[0].vshard_id, VShardId::from_key(b"alice")); + let PhysicalPlan::Graph(GraphOp::NodeEdgeGuard { expected, .. }) = &guards[0].plan else { + panic!("expected a node guard, got {:?}", guards[0].plan); + }; + assert_eq!(expected.len(), 1); + assert_eq!( + (expected[0].src_surrogate, expected[0].dst_surrogate), + (Surrogate::new(1), Surrogate::new(2)) + ); + + assert_eq!(deletes.len(), 1); + let homes = RecordHomes::edge("alice", &peer); + assert_eq!(deletes[0].vshard_id, homes.owner()); + + let mut tasks = guards; + tasks.extend(deletes); + let tx = super::super::build_single_vshard_tx_class(&tasks, TenantId::new(1), &[]) + .expect("tx class"); + let participants: Vec = tx + .participating_vshards() + .iter() + .map(|v| v.as_u32()) + .collect(); + for home in [homes.owner(), homes.second()] { + assert!( + participants.contains(&home.as_u32()), + "home {home:?} participates: {participants:?}" + ); + } + } + + /// An edge between two deleted nodes, or one another task already + /// deletes, is deleted once. Each node still guards all its edges. + #[test] + fn a_shared_or_covered_edge_is_deleted_once() { + let nodes = vec![ + NodeIncidentEdges { + node: "a".to_string(), + edges: vec![triple("a", "b"), triple("a", "c")], + }, + NodeIncidentEdges { + node: "b".to_string(), + edges: vec![triple("a", "b")], + }, + ]; + let bound: HashMap<&str, Surrogate> = HashMap::from([ + ("a", Surrogate::new(1)), + ("b", Surrogate::new(2)), + ("c", Surrogate::new(3)), + ]); + let covered = BTreeSet::from([triple("a", "c")]); + let (guards, deletes) = build_node_delete_tasks( + TenantId::new(1), + DatabaseId::DEFAULT, + NodeDeletePlan { + collection: "users", + guarded: &nodes, + deleted: &nodes, + already_deleted: &covered, + }, + &bound, + ) + .expect("tasks"); + assert_eq!(guards.len(), 2); + assert_eq!( + planned_edge_deletes(&deletes), + BTreeSet::from([triple("a", "b")]) + ); + assert_eq!(deletes.len(), 1); + } + + /// An endpoint the resolver did not bind is an error, never a zero + /// surrogate. + #[test] + fn an_unbound_endpoint_is_an_error() { + let nodes = vec![NodeIncidentEdges { + node: "a".to_string(), + edges: vec![triple("a", "b")], + }]; + let bound: HashMap<&str, Surrogate> = HashMap::from([("a", Surrogate::new(1))]); + assert!( + build_node_delete_tasks( + TenantId::new(1), + DatabaseId::DEFAULT, + NodeDeletePlan { + collection: "users", + guarded: &nodes, + deleted: &nodes, + already_deleted: &BTreeSet::new(), + }, + &bound, + ) + .is_err() + ); + } + + fn entry(label: &str, node: &str) -> Value { + Value::Object(HashMap::from([ + ("label".to_string(), Value::String(label.to_string())), + ("node".to_string(), Value::String(node.to_string())), + ])) + } + + #[test] + fn a_neighbors_payload_decodes_to_label_node_pairs() { + let payload = nodedb_types::value_to_msgpack(&Value::Array(vec![ + entry("KNOWS", "bob"), + entry("LIKES", "carol"), + ])) + .expect("encode"); + assert_eq!( + decode_neighbors(&payload).expect("decode"), + vec![ + ("KNOWS".to_string(), "bob".to_string()), + ("LIKES".to_string(), "carol".to_string()), + ] + ); + assert!(decode_neighbors(&[]).expect("empty").is_empty()); + } + + #[test] + fn a_neighbors_payload_that_is_not_an_array_is_an_error() { + let payload = nodedb_types::value_to_msgpack(&entry("KNOWS", "bob")).expect("encode"); + assert!(decode_neighbors(&payload).is_err()); + } + + #[test] + fn planned_edge_deletes_names_only_edge_deletes() { + let task = |plan| PhysicalTask { + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(0), + database_id: DatabaseId::DEFAULT, + plan, + post_set_op: PostSetOp::None, + txn_id: None, + }; + let tasks = vec![ + task(PhysicalPlan::Graph(GraphOp::EdgeDelete { + collection: QualifiedCollection::from_stored("c".to_string()), + src_id: "a".to_string(), + label: "L".to_string(), + dst_id: "b".to_string(), + src_surrogate: nodedb_types::Surrogate::new(1), + dst_surrogate: nodedb_types::Surrogate::new(2), + rls_write_check: nodedb_types::RlsWriteCheck::decided_earlier_in_request(), + })), + task(PhysicalPlan::Graph(GraphOp::NodeEdgeGuard { + collection: QualifiedCollection::from_stored("c".to_string()), + node_id: "a".to_string(), + expected: Vec::new(), + })), + ]; + assert_eq!( + planned_edge_deletes(&tasks), + BTreeSet::from([("a".to_string(), "L".to_string(), "b".to_string())]) + ); + } +} diff --git a/nodedb/src/control/planner/calvin/dependent_recon_predicate.rs b/nodedb/src/control/planner/calvin/dependent_recon_predicate.rs index 965cc8ed1..e3ec15847 100644 --- a/nodedb/src/control/planner/calvin/dependent_recon_predicate.rs +++ b/nodedb/src/control/planner/calvin/dependent_recon_predicate.rs @@ -9,10 +9,13 @@ use crate::control::planner::implicit_edges::{EdgeFieldOverrides, parse_edge_fie use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; /// The implicit-edge lifecycle a dependent (OLLP) Calvin task drives, derived -/// once from the dependent task's plan variant. `Update` carries the SET-clause -/// overrides (parsed once — they are constant across retries). +/// once from the dependent task's plan variant. `Delete` carries the column +/// each deleted row's node identity is read from. `Update` carries the +/// SET-clause overrides (parsed once — they are constant across retries). pub(super) enum EdgeLifecycle { - Delete, + Delete { + declared_primary_key: Option, + }, Update(EdgeFieldOverrides), } @@ -53,12 +56,18 @@ pub(super) fn extract_bulk_predicate_info(plan: &PhysicalPlan) -> (String, Vec crate::Result { match plan { - PhysicalPlan::Document(DocumentOp::BulkDelete { .. }) => Ok(EdgeLifecycle::Delete), + PhysicalPlan::Document(DocumentOp::BulkDelete { + declared_primary_key, + .. + }) => Ok(EdgeLifecycle::Delete { + declared_primary_key: declared_primary_key.clone(), + }), PhysicalPlan::Document(DocumentOp::BulkUpdate { updates, .. }) => { Ok(EdgeLifecycle::Update(parse_edge_field_overrides(updates)?)) } diff --git a/nodedb/src/control/planner/calvin/dispatch.rs b/nodedb/src/control/planner/calvin/dispatch.rs index f365ce63e..71841cdc3 100644 --- a/nodedb/src/control/planner/calvin/dispatch.rs +++ b/nodedb/src/control/planner/calvin/dispatch.rs @@ -3,7 +3,7 @@ //! Calvin dispatch classification and routing for cross-shard writes. //! //! This module is the single chokepoint for deciding whether a set of -//! [`PhysicalTask`]s should be dispatched via: +//! [`PhysicalTask`]s must be dispatched via: //! //! - The single-shard fast path (existing path, no Calvin involvement). //! - Calvin static dispatch (all write keys known upfront). @@ -15,7 +15,7 @@ //! //! # Note on predicate_class //! -//! The ideal implementation of `predicate_class` would serialize the `Filter` +//! The ideal implementation of `predicate_class` will serialize the `Filter` //! AST via zerompk and normalize bound parameter values to their type tags. //! However, `nodedb_sql::types::Filter` does not derive `zerompk::ToMessagePack` //! or `zerompk::FromMessagePack`. As a declared fallback, `predicate_class` @@ -81,18 +81,21 @@ pub fn is_dependent_predicate(plan: &PhysicalPlan) -> bool { /// `TxClass` read_set's participants. An entry carries the plan's /// database-qualified name, so it is de-qualified into a [`CollectionKey`] /// before hashing. Each read retains its session database so classification -/// and the database-scoped transaction class agree. A read with no extractable -/// collection contributes nothing. +/// and the database-scoped transaction class agree. An entry homed on a vShard +/// (a cross-shard graph read) contributes that vShard, with or without a +/// collection. An unhomed read with no extractable collection contributes +/// nothing. pub fn read_vshards_of(reads: &[ReadSetEntry]) -> crate::Result> { reads .iter() - .filter(|e| !e.collection.is_empty()) - .map(|e| { - Ok( + .filter(|e| e.home.is_some() || !e.collection.is_empty()) + .map(|e| match e.home { + Some(home) => Ok(home.as_u32()), + None => Ok( CollectionKey::from_qualified_str(e.database_id, &e.collection)? .vshard() .as_u32(), - ) + ), }) .collect() } @@ -132,7 +135,7 @@ pub fn classify_dispatch(tasks: &[PhysicalTask], read_vshards: &BTreeSet) - }, 1 => DispatchClass::SingleShard { // The single vShard is a write shard whenever any write ran (the - // common case: `last_vshard` is set). It could instead be a lone + // common case: `last_vshard` is set). It can instead be a lone // read shard with no writes — unreachable via the COMMIT path, which // only classifies a non-empty write buffer — so the `unwrap_or_else` // is a defensive fallback upholding the no-panic contract. @@ -378,10 +381,9 @@ mod tests { #[test] fn classify_dispatch_multi_shard_counts_newly_widened_crdt_apply_write() { - // Before the `is_write_plan` widening, `CrdtOp::Apply` was misclassified - // as a read: `classify_dispatch` would have counted zero write vshards - // for this pair and returned `SingleShard`, silently dropping Calvin's - // cross-shard atomicity for a real two-vshard CRDT write. + // `CrdtOp::Apply` is a write: `classify_dispatch` counts both write + // vshards for this pair, so a two-vshard CRDT write keeps Calvin's + // cross-shard atomicity. let tasks = vec![crdt_apply_task(3), crdt_apply_task(7)]; let class = classify_dispatch(&tasks, &BTreeSet::new()); match class { @@ -443,7 +445,7 @@ mod tests { fn classify_dispatch_read_widened_multi_shard() { // A single-WRITE-shard batch (shard 5) that READS shard 8 classifies as // MultiShard{5,8}: the read vShard widens the participant set exactly as a - // write vShard would. + // write vShard does. let tasks = vec![doc_insert_task(5)]; let read_vshards: BTreeSet = [8u32].into_iter().collect(); let class = classify_dispatch(&tasks, &read_vshards); @@ -489,7 +491,7 @@ mod tests { // WHY this must stay `MultiShard`: only the `MultiShard` branch of COMMIT // flushes through the Calvin barrier (`run_commit_calvin`), which validates // B's read slice on B's OWNING node using the real per-shard `read_lsn`. If a - // foreign read failed to widen the class, COMMIT would take the `SingleShard` + // foreign read failed to widen the class, COMMIT will take the `SingleShard` // branch and run only the local-WAL `si_conflict_abort`, which never sees a // stale read on the remote owner — silently committing a non-serializable // cross-node transaction. This test guarantees a future refactor of @@ -513,6 +515,8 @@ mod tests { read_lsn: Lsn::new(1), read_version_lsn: Lsn::ZERO, origin: ReadOrigin::Session, + home: None, + home_node: 0, }; // The homing step under test: a foreign-collection read must home to a diff --git a/nodedb/src/control/planner/calvin/dispatch_multi.rs b/nodedb/src/control/planner/calvin/dispatch_multi.rs index c9ba806a3..f9e9a2337 100644 --- a/nodedb/src/control/planner/calvin/dispatch_multi.rs +++ b/nodedb/src/control/planner/calvin/dispatch_multi.rs @@ -15,7 +15,7 @@ //! //! The OLLP (dependent-predicate) variant is intentionally NOT handled here: it //! is still tied to the local `OllpOrchestrator` and completion registry and is -//! not yet leader-routed. Callers that may carry a dependent predicate must +//! not yet leader-routed. Callers that can carry a dependent predicate must //! route that case through their own OLLP path. use crate::bridge::envelope::Response; @@ -26,25 +26,37 @@ use crate::control::planner::calvin::{ use crate::control::server::shared::authorization::AuthorizedTaskSet; use crate::control::server::shared::session::read_set::ReadSetEntry; use crate::control::state::SharedState; +use crate::event::EventSource; use crate::types::TenantId; use nodedb_physical::physical_task::PhysicalTask; -/// Submit an externally authorized strict atomic static Calvin task set. -pub async fn dispatch_authorized_strict_atomic_tasks_to_calvin( - state: &SharedState, - authorized: AuthorizedTaskSet, - tenant_id: TenantId, - position: TxnDispatchPosition, - reads: &[ReadSetEntry], - lock_owner: Option, -) -> crate::Result> { - let tasks: Vec = authorized - .into_tasks() - .into_iter() - .map(|task| task.into_physical_task()) - .collect(); - dispatch_strict_atomic_tasks_to_calvin(state, &tasks, tenant_id, position, reads, lock_owner) - .await +/// The sources a Calvin transaction's writes commit under. +#[derive(Debug, Clone)] +pub(crate) struct TxnProvenance { + /// The transaction's own source. + pub event_source: EventSource, + /// Indexes into the task set of the tasks a trigger body buffered. Their + /// rows commit under `Trigger`. + pub body_tasks: Vec, + /// The messages the transaction's trigger bodies published, as + /// [`crate::wal::RedoPublish::encode_all`] wrote them. Empty for none. + pub publishes: Vec, + /// The encoded dedup key of the cross-shard request the transaction + /// applies, with the vShard the request addresses. `None` for any other + /// transaction. + pub applied_key: Option<(Vec, u32)>, +} + +impl TxnProvenance { + /// A client transaction no trigger body joined. + pub(crate) fn client() -> Self { + Self { + event_source: EventSource::User, + body_tasks: Vec::new(), + publishes: Vec::new(), + applied_key: None, + } + } } /// Submit one trusted internal strict atomic static Calvin task set. @@ -60,6 +72,8 @@ pub async fn dispatch_authorized_strict_atomic_tasks_to_calvin( /// On success, `Some(Response)` means the scheduler retained a materialized /// applied primary response; `None` means no response was retained. This is not /// an affected-row envelope and callers must apply their own operation semantics. +/// +/// Every participant stamps `provenance` on the transaction's writes. pub(crate) async fn dispatch_strict_atomic_tasks_to_calvin( state: &SharedState, tasks: &[PhysicalTask], @@ -67,12 +81,18 @@ pub(crate) async fn dispatch_strict_atomic_tasks_to_calvin( position: TxnDispatchPosition, reads: &[ReadSetEntry], lock_owner: Option, + provenance: TxnProvenance, ) -> crate::Result> { + let resolved = crate::control::write_resolve::resolve_tasks_for_log(state, tasks).await?; + let tasks = resolved.as_deref().unwrap_or(tasks); let mut tx_class = admit_strict_atomic_tasks(tasks, tenant_id, position, reads)?; - if state.sequencer_inbox.get().is_none() { - return Err(crate::Error::SequencerUnavailable); - } tx_class.set_lock_owner(lock_owner); + tx_class.set_event_source(provenance.event_source.wal_code()); + tx_class.set_body_plans(provenance.body_tasks); + tx_class.set_publishes(provenance.publishes); + if let Some((applied_key, vshard)) = provenance.applied_key { + tx_class.set_applied_key(applied_key, vshard); + } submit_calvin_routed(state, tx_class).await } @@ -149,7 +169,7 @@ pub async fn dispatch_authorized_tasks_to_calvin( /// Drive the legacy trusted-internal strict Calvin multi-shard path for `tasks`. /// -/// This compatibility API preserves its historical multi-shard-only contract. +/// This API serves multi-shard dispatch only. /// Its strict branch delegates to [`dispatch_strict_atomic_tasks_to_calvin`], /// while best-effort remains rejected because this helper has no non-atomic /// dispatch implementation. @@ -167,7 +187,13 @@ pub(crate) async fn dispatch_tasks_to_calvin( DispatchClass::MultiShard { .. } => { admit_legacy_multi_shard_dispatch(cross_shard_mode, position)?; dispatch_strict_atomic_tasks_to_calvin( - state, tasks, tenant_id, position, reads, lock_owner, + state, + tasks, + tenant_id, + position, + reads, + lock_owner, + TxnProvenance::client(), ) .await } @@ -221,6 +247,8 @@ mod tests { read_lsn: Lsn::new(1), read_version_lsn: Lsn::new(1), origin: ReadOrigin::Session, + home: None, + home_node: 0, } } diff --git a/nodedb/src/control/planner/calvin/edge_sequencing.rs b/nodedb/src/control/planner/calvin/edge_sequencing.rs new file mode 100644 index 000000000..bd1d5b23c --- /dev/null +++ b/nodedb/src/control/planner/calvin/edge_sequencing.rs @@ -0,0 +1,192 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Every edge write runs as a Calvin transaction. +//! +//! The edge versions of a collection live in one ordering domain: the +//! Calvin sequence. An edge version is applied at its transaction's ordinal. +//! A TRUNCATE's cut is its own transaction's ordinal, and it hides every +//! version applied below it. So on every replica, and under any clock skew, +//! a TRUNCATE hides exactly the versions sequenced before it. +//! +//! A data-group Raft entry cannot give that. Each replica stamps it from its +//! own clock, and each replica applies it at its own point among the Calvin +//! flushes. It also takes no Calvin lock, so a node delete's guard does not +//! order against it. +//! +//! The Raft propose seams (`propose_replicated_entry`, `propose_sync_write`) +//! hand every edge write here. A session COMMIT that buffered an edge write +//! commits through Calvin. A RESTORE re-issues its edge versions as Calvin +//! transactions of its own, which carry the restore's mark (see +//! `backup::restore::redo_reissue`). + +use std::future::Future; +use std::pin::Pin; + +use nodedb_physical::physical_plan::{GraphOp, PhysicalPlan}; +use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; + +use super::submit::submit_calvin_routed_write; +use super::tx_class::build_single_vshard_tx_class; +use crate::bridge::envelope::Response; +use crate::control::state::SharedState; +use crate::control::wal_replication::ReplicatedEntry; +use crate::event::EventSource; +use crate::types::{DatabaseId, TenantId, VShardId}; + +/// Whether `plan` writes an edge. Exhaustive over `GraphOp`: a new graph op +/// is a compile error here. +pub fn is_edge_write(plan: &PhysicalPlan) -> bool { + let PhysicalPlan::Graph(op) = plan else { + return false; + }; + match op { + GraphOp::EdgePut { .. } + | GraphOp::EdgeDelete { .. } + | GraphOp::EdgePutBatch { .. } + | GraphOp::EdgeDeleteBatch { .. } => true, + // Guards and a TRUNCATE's edge share run only inside a Calvin + // transaction already. Label writes version no edge. + GraphOp::NodeEdgeGuard { .. } + | GraphOp::NodePresenceGuard { .. } + | GraphOp::TruncateEdges { .. } + | GraphOp::SetNodeLabels { .. } + | GraphOp::RemoveNodeLabels { .. } + | GraphOp::ResolveEdgeDelete(_) + | GraphOp::Hop { .. } + | GraphOp::Neighbors { .. } + | GraphOp::NeighborsMulti { .. } + | GraphOp::Path { .. } + | GraphOp::Subgraph { .. } + | GraphOp::RagFusion { .. } + | GraphOp::Algo { .. } + | GraphOp::Match { .. } + | GraphOp::MatchContinuation { .. } + | GraphOp::MatchVarLenResume { .. } + | GraphOp::BspSuperstep(_) + | GraphOp::WccSuperstep(_) + | GraphOp::TemporalNeighbors { .. } + | GraphOp::TemporalAlgorithm { .. } + | GraphOp::Stats { .. } + | GraphOp::NodePresenceRead { .. } => false, + } +} + +/// Whether any task among `tasks` writes an edge. +pub fn writes_edges(tasks: &[PhysicalTask]) -> bool { + tasks.iter().any(|task| is_edge_write(&task.plan)) +} + +/// One edge write and the scope it runs in. +pub struct EdgeWrite { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + /// The vShard the write was addressed to. The transaction's participants + /// are the edges' endpoint homes. + pub vshard_id: VShardId, + pub plan: PhysicalPlan, + /// The source every participant stamps on the write's events. + pub event_source: EventSource, +} + +/// Run `write` as a single-write Calvin transaction and return its applied +/// response. A Data-Plane refusal comes back as `Err(Error::DataPlane)`, the +/// shape a proposed write returns. +/// +/// Returns a boxed future: the Calvin submit can reach a Raft propose seam +/// that calls back here, and the box gives that cycle a finite size. +pub fn sequence_edge_write<'a>( + state: &'a SharedState, + write: EdgeWrite, +) -> Pin> + Send + 'a>> { + Box::pin(async move { + let EdgeWrite { + tenant_id, + database_id, + vshard_id, + plan, + event_source, + } = write; + let task = PhysicalTask { + tenant_id, + vshard_id, + database_id, + plan, + post_set_op: PostSetOp::None, + txn_id: None, + }; + let mut tx_class = + build_single_vshard_tx_class(std::slice::from_ref(&task), tenant_id, &[])?; + tx_class.set_event_source(event_source.wal_code()); + // Every participant of a graph-only transaction reports its answer + // in its completion ack, so the count arrives on any node. + let response = submit_calvin_routed_write(state, tx_class).await?; + crate::control::local_dispatch::reject_data_plane_error(&response)?; + Ok(response) + }) +} + +/// Run the edge write `entry` carries as a Calvin transaction. `None` when +/// `entry` writes no edge: it keeps its Raft entry. +pub async fn sequence_replicated_edge_write( + state: &SharedState, + entry: &ReplicatedEntry, +) -> crate::Result> { + let Some(plan) = crate::control::wal_replication::decode::edge_write_plan(&entry.write)? else { + return Ok(None); + }; + let response = sequence_edge_write( + state, + EdgeWrite { + tenant_id: TenantId::new(entry.tenant_id), + database_id: DatabaseId::new(entry.database_id), + vshard_id: VShardId::new(entry.vshard_id), + plan, + event_source: entry.event_source.into(), + }, + ) + .await?; + Ok(Some(response)) +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_physical::physical_plan::BatchEdge; + use nodedb_types::{QualifiedCollection, Surrogate}; + + fn batch_edge() -> BatchEdge { + BatchEdge { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "g"), + src_id: "a".into(), + label: "L".into(), + dst_id: "b".into(), + src_surrogate: Surrogate::new(1), + dst_surrogate: Surrogate::new(2), + } + } + + /// Single and batched edge writes take the Calvin route. A guard, a + /// TRUNCATE share and a label write do not: they version no edge + /// outside a Calvin transaction. + #[test] + fn every_edge_write_and_only_an_edge_write_is_sequenced() { + let put = PhysicalPlan::Graph(GraphOp::EdgePutBatch { + edges: vec![batch_edge()], + }); + let delete = PhysicalPlan::Graph(GraphOp::EdgeDeleteBatch { + edges: vec![batch_edge()], + }); + assert!(is_edge_write(&put)); + assert!(is_edge_write(&delete)); + let share = PhysicalPlan::Graph(GraphOp::TruncateEdges { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "g"), + vshard: 0, + }); + let labels = PhysicalPlan::Graph(GraphOp::SetNodeLabels { + node_id: "a".into(), + labels: vec!["L".into()], + }); + assert!(!is_edge_write(&share)); + assert!(!is_edge_write(&labels)); + } +} diff --git a/nodedb/src/control/planner/calvin/edge_truncate.rs b/nodedb/src/control/planner/calvin/edge_truncate.rs new file mode 100644 index 000000000..917d9a754 --- /dev/null +++ b/nodedb/src/control/planner/calvin/edge_truncate.rs @@ -0,0 +1,277 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! TRUNCATE of an edge-bearing collection, as one Calvin transaction. +//! +//! The transaction holds the rows' truncate on the collection's vShard and +//! one `GraphOp::TruncateEdges` on every vShard. It spans every vShard, so +//! it travels as a multi-part transaction. It commits or aborts whole: no +//! reader sees the rows gone and an edge left, and no crash leaves either +//! half owed. +//! +//! - Each vShard's share records a cut of the collection at the +//! transaction's ordinal. The cut writes no edge version and reads no +//! stored edge: every read hides the collection's versions applied below +//! it. Every edge write runs as a Calvin transaction +//! ([`super::edge_sequencing`]), so every edge version is applied at a +//! Calvin ordinal. An edge sequenced before the TRUNCATE is hidden, and +//! one sequenced after it stays, on every replica, under any clock skew +//! and in any order a core applies the transactions in. +//! - A RESTORE's edge version keeps its historical system time and is +//! applied at the RESTORE's ordinal. A TRUNCATE sequenced before the +//! RESTORE leaves it, and one sequenced after the RESTORE hides it. +//! +//! A collection that never held an edge truncates its rows alone, as an +//! autocommit write. + +use nodedb_physical::physical_plan::{DocumentOp, GraphOp, PhysicalPlan}; +use nodedb_physical::physical_task::PhysicalTask; + +use crate::bridge::envelope::Response; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::dispatch_utils::{ + AutocommitWrite, dispatch_authorized_durable_write, dispatch_durable_autocommit_write, +}; +use crate::control::server::shared::clone_write::{ + CloneCheckedOutcome, InterceptAndAuthorizeParams, intercept_and_authorize, +}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; + +use super::submit::submit_calvin_routed; +use super::tx_class::build_static_tx_class; + +/// Whether `plan` is a TRUNCATE. +pub fn is_truncate(plan: &PhysicalPlan) -> bool { + matches!(plan, PhysicalPlan::Document(DocumentOp::Truncate { .. })) +} + +/// The answer to a TRUNCATE. +pub(super) struct TruncateOutcome { + /// The applied response of the rows' truncate, when a participant this + /// node hosts deposited one. + pub apply_result: Option, +} + +/// Run the TRUNCATE among `tasks`: with the collection's edges in one +/// Calvin transaction when an edge was ever written into it. `None` when +/// `tasks` holds no TRUNCATE. +/// +/// With `identity`, the TRUNCATE passes the clone-write gate and +/// authorization. Without it, the caller already authorized it. +pub(super) async fn dispatch_truncate( + state: &SharedState, + tasks: &[PhysicalTask], + identity: Option<&AuthenticatedIdentity>, + tenant_id: TenantId, +) -> crate::Result> { + let Some(task) = tasks.iter().find(|task| is_truncate(&task.plan)) else { + return Ok(None); + }; + let PhysicalPlan::Document(DocumentOp::Truncate { collection, .. }) = &task.plan else { + return Ok(None); + }; + let collection = collection.clone(); + + // The statement holds the collection's lease, so the flag read here + // stays true or false until the TRUNCATE's outcome. + let edge_bearing = + collection_is_edge_bearing(state, task.tenant_id, task.database_id, collection.as_str())?; + + // The clone-write gate and authorization of the rows' truncate. The + // lease holds until the transaction's outcome. + let (rows_task, _lease) = match identity { + Some(identity) => { + let emitter = crate::control::security::audit::ArcAuditEmitter(std::sync::Arc::clone( + &state.audit, + )); + match intercept_and_authorize(InterceptAndAuthorizeParams { + state, + task: task.clone(), + identity, + tenant_id, + permissions: &state.permissions, + roles: &state.roles, + emitter: &emitter, + }) + .await? + { + CloneCheckedOutcome::Handled(response) if !edge_bearing => { + return Ok(Some(TruncateOutcome { + apply_result: Some(response), + })); + } + CloneCheckedOutcome::Handled(_) => { + return Err(crate::Error::BadRequest { + detail: format!( + "TRUNCATE of '{collection}' writes through a clone and cuts \ + graph edges, which one transaction does not carry together" + ), + }); + } + CloneCheckedOutcome::Proceed(checked) if !edge_bearing => { + let response = + dispatch_authorized_durable_write(state, checked, TraceId::ZERO).await?; + crate::control::local_dispatch::reject_data_plane_error(&response)?; + return Ok(Some(TruncateOutcome { + apply_result: Some(response), + })); + } + CloneCheckedOutcome::Proceed(checked) => { + let (authorized, lease) = checked.into_parts(); + (authorized.into_physical_task(), Some(lease)) + } + } + } + None if !edge_bearing => { + let response = dispatch_rows_plan(state, task).await?; + crate::control::local_dispatch::reject_data_plane_error(&response)?; + return Ok(Some(TruncateOutcome { + apply_result: Some(response), + })); + } + None => (task.clone(), None), + }; + + let edge_shares = edge_share_tasks(&rows_task, &collection); + let edge_shares = match identity { + Some(identity) => authorize(state, identity, edge_shares)?, + None => edge_shares, + }; + let mut submission = Vec::with_capacity(edge_shares.len() + 1); + submission.push(rows_task); + submission.extend(edge_shares); + let tx_class = build_static_tx_class(&submission, tenant_id, &[])?; + let apply_result = submit_calvin_routed(state, tx_class).await?; + Ok(Some(TruncateOutcome { apply_result })) +} + +/// One `TruncateEdges` of `collection` per vShard, beside `rows_task`. +fn edge_share_tasks( + rows_task: &PhysicalTask, + collection: &nodedb_types::QualifiedCollection, +) -> Vec { + (0..nodedb_cluster::routing::VSHARD_COUNT) + .map(|vshard| PhysicalTask { + vshard_id: VShardId::new(vshard), + plan: PhysicalPlan::Graph(GraphOp::TruncateEdges { + collection: collection.clone(), + vshard, + }), + ..rows_task.clone() + }) + .collect() +} + +/// Authorize the edge shares as `identity`. +fn authorize( + state: &SharedState, + identity: &AuthenticatedIdentity, + tasks: Vec, +) -> crate::Result> { + let emitter = + crate::control::security::audit::ArcAuditEmitter(std::sync::Arc::clone(&state.audit)); + Ok( + crate::control::server::shared::authorization::authorize_task_set( + identity, + &tasks, + &state.permissions, + &state.roles, + &emitter, + )? + .into_tasks() + .into_iter() + .map(|task| task.into_physical_task()) + .collect(), + ) +} + +/// Dispatch the rows' truncate `task` as a durable autocommit write. +async fn dispatch_rows_plan(state: &SharedState, task: &PhysicalTask) -> crate::Result { + dispatch_durable_autocommit_write( + state, + AutocommitWrite { + tenant_id: task.tenant_id, + database_id: task.database_id, + vshard_id: task.vshard_id, + plan: task.plan.clone(), + trace_id: TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + }, + ) + .await +} + +/// Whether an edge was ever written into `collection`, the +/// database-qualified name. +pub(super) fn collection_is_edge_bearing( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: &str, +) -> crate::Result { + let bare = + crate::control::target_identity::naming::bare_collection_name(database_id, collection); + Ok(state + .credentials + .catalog() + .get_collection(database_id, tenant_id.as_u64(), &bare)? + .is_some_and(|coll| coll.has_implicit_edges)) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn truncate_task() -> PhysicalTask { + PhysicalTask { + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(3), + database_id: DatabaseId::DEFAULT, + plan: PhysicalPlan::Document(DocumentOp::Truncate { + collection: nodedb_types::QualifiedCollection::from_stored("c".to_string()), + restart_identity: false, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }), + post_set_op: nodedb_physical::physical_task::PostSetOp::None, + txn_id: None, + } + } + + #[test] + fn only_a_truncate_is_a_truncate() { + let task = truncate_task(); + assert!(is_truncate(&task.plan)); + let share = PhysicalPlan::Graph(GraphOp::TruncateEdges { + collection: nodedb_types::QualifiedCollection::from_stored("c".to_string()), + vshard: 1, + }); + assert!(!is_truncate(&share)); + } + + /// A TRUNCATE carries one edge share per vShard, each on the vShard it + /// names, so the transaction's participants are every vShard. + #[test] + fn a_truncate_carries_one_edge_share_per_vshard() { + let task = truncate_task(); + let collection = nodedb_types::QualifiedCollection::from_stored("c".to_string()); + let shares = edge_share_tasks(&task, &collection); + assert_eq!(shares.len(), nodedb_cluster::routing::VSHARD_COUNT as usize); + for (vshard, share) in (0u32..).zip(&shares) { + assert_eq!(share.vshard_id, VShardId::new(vshard)); + assert!(matches!( + &share.plan, + PhysicalPlan::Graph(GraphOp::TruncateEdges { vshard: named, .. }) if *named == vshard + )); + } + let mut submission = vec![task]; + submission.extend(shares); + let tx_class = build_static_tx_class(&submission, TenantId::new(1), &[]) + .expect("a TRUNCATE transaction class"); + assert_eq!( + tx_class.participating_vshards().len(), + nodedb_cluster::routing::VSHARD_COUNT as usize + ); + } +} diff --git a/nodedb/src/control/planner/calvin/mod.rs b/nodedb/src/control/planner/calvin/mod.rs index 3ee8e6ed3..a37011a80 100644 --- a/nodedb/src/control/planner/calvin/mod.rs +++ b/nodedb/src/control/planner/calvin/mod.rs @@ -3,11 +3,17 @@ pub mod abort_error; pub mod cross_shard_mode; pub mod dependent_recon; +pub mod dependent_recon_crdt; +mod dependent_recon_finish; +pub mod dependent_recon_node_edges; pub mod dependent_recon_plan; mod dependent_recon_predicate; pub mod dispatch; pub mod dispatch_multi; +pub mod edge_sequencing; +pub mod edge_truncate; pub mod explain; +pub mod node_delete_txn; pub mod predicate; pub mod preexec; pub mod reservation; @@ -21,21 +27,23 @@ pub use abort_error::calvin_abort_error; pub use cross_shard_mode::CrossShardTxnMode; pub(crate) use dependent_recon::dispatch_dependent_edge_recon; pub use dependent_recon::{ - DependentReconOutcome, dispatch_authorized_dependent_edge_recon, plan_needs_implicit_edge_recon, + DependentReconOutcome, dispatch_authorized_dependent_edge_recon, is_edge_recon_plan, + plan_needs_implicit_edge_recon, }; pub use dispatch::{ classify_dispatch, is_dependent_predicate, is_write_plan, predicate_class, read_vshards_of, }; -pub(crate) use dispatch_multi::dispatch_strict_atomic_tasks_to_calvin; -pub use dispatch_multi::{ - dispatch_authorized_strict_atomic_tasks_to_calvin, dispatch_authorized_tasks_to_calvin, +pub use dispatch_multi::dispatch_authorized_tasks_to_calvin; +pub(crate) use dispatch_multi::{TxnProvenance, dispatch_strict_atomic_tasks_to_calvin}; +pub use edge_sequencing::{ + EdgeWrite, is_edge_write, sequence_edge_write, sequence_replicated_edge_write, writes_edges, }; pub use explain::calvin_explain_preamble; pub use predicate::predicate_class_for_filters; pub use retry_loop::{DependentOutcome, DependentRetryArgs, run_dependent_with_retry}; pub use submit::{ RoutedAssignment, submit_and_await_calvin, submit_and_await_calvin_with_timeout, - submit_calvin_routed, submit_calvin_routed_assign, + submit_calvin_routed, submit_calvin_routed_assign, submit_calvin_routed_write, }; pub use tx_class::{ build_dependent_tx_class, build_single_vshard_dependent_tx_class, build_single_vshard_tx_class, diff --git a/nodedb/src/control/planner/calvin/node_delete_txn.rs b/nodedb/src/control/planner/calvin/node_delete_txn.rs new file mode 100644 index 000000000..fbb01d825 --- /dev/null +++ b/nodedb/src/control/planner/calvin/node_delete_txn.rs @@ -0,0 +1,244 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Node-delete edge tasks for a delete inside a session transaction. +//! +//! An explicit transaction does not take the OLLP reconnaissance route, so +//! the delete's edge tasks are derived at the statement and buffered with it: +//! +//! - One `EdgeDelete` per edge an edge document's `_from`/`_to` mirrors. +//! - One `EdgeDelete` per edge of the collection incident on each deleted +//! row's node, including edges the transaction staged itself. +//! - One `NodeEdgeGuard` per node over the node's edges in the store. COMMIT +//! checks it, and a node whose edges changed since the statement fails the +//! transaction with a serialization error the client retries. + +use nodedb_physical::physical_plan::{CrdtOp, DocumentOp, PhysicalPlan}; +use nodedb_physical::physical_task::PhysicalTask; + +use super::dependent_recon_node_edges::{ + NodeDeletePlan, NodeIncidentEdges, node_delete_tasks, planned_edge_deletes, + read_node_incident_edges, +}; +use super::preexec::{PreexecRequest, RowIdentityRead, run_preexec_scan}; +use crate::control::planner::implicit_edges::append_implicit_edge_delete_tasks; +use crate::control::state::SharedState; +use crate::types::{TraceId, TxnId}; + +/// The edge tasks a delete `task` inside transaction `txn_id` carries: a +/// `BulkDelete`, a `PointDelete` of a bound surrogate, or a CRDT +/// `DocDelete` of a bound document. Empty for any +/// other task, and for a delete on a collection no edge was ever written +/// into. A TRUNCATE of an edge-bearing collection is refused. +/// +/// Guards come first in the returned list. +pub async fn txn_node_delete_tasks( + state: &SharedState, + task: &PhysicalTask, + txn_id: Option, +) -> crate::Result> { + let (tenant_id, database_id) = (task.tenant_id, task.database_id); + // The rows the statement deletes: a predicate, one row by surrogate, or + // one CRDT document by its id. + let (collection, target) = match &task.plan { + PhysicalPlan::Document(DocumentOp::BulkDelete { + collection, + filters, + .. + }) => ( + collection, + DeleteTarget::Rows { + filters: filters.clone(), + prefilter: None, + }, + ), + PhysicalPlan::Document(DocumentOp::PointDelete { + collection, + surrogate: Some(surrogate), + .. + }) => ( + collection, + DeleteTarget::Rows { + filters: Vec::new(), + prefilter: Some([*surrogate].into_iter().collect()), + }, + ), + PhysicalPlan::Crdt(CrdtOp::DocDelete { + collection, + document_id, + surrogate: Some(_), + .. + }) => (collection, DeleteTarget::Document(document_id.clone())), + // A TRUNCATE's edge shares stage nothing: each tombstones its + // vShard's edges at the transaction's turn. A read later in the same + // transaction block still sees the edges, so the block refuses it. + PhysicalPlan::Document(DocumentOp::Truncate { collection, .. }) => { + if super::edge_truncate::collection_is_edge_bearing( + state, + tenant_id, + database_id, + collection.as_str(), + )? { + return Err(crate::Error::BadRequest { + detail: format!( + "TRUNCATE of '{collection}' cannot run inside a transaction block: \ + graph edges were written into it, and the block's later reads would \ + still see them. Run the TRUNCATE outside the transaction." + ), + }); + } + return Ok(Vec::new()); + } + _ => return Ok(Vec::new()), + }; + let bare = crate::control::target_identity::naming::bare_collection_name( + database_id, + collection.as_str(), + ); + let Some(coll) = + state + .credentials + .catalog() + .get_collection(database_id, tenant_id.as_u64(), &bare)? + else { + return Ok(Vec::new()); + }; + if !coll.has_implicit_edges { + return Ok(Vec::new()); + } + let declared_primary_key = coll.declared_primary_key; + + // The rows as this transaction sees them, read before the delete stages. + let (identities, scanned_edges) = match target { + DeleteTarget::Rows { filters, prefilter } => { + let scan = run_preexec_scan( + state, + tenant_id, + database_id, + PreexecRequest { + collection: collection.as_str(), + filters, + prefilter, + identity: RowIdentityRead::Read { + declared_primary_key: declared_primary_key.as_deref(), + }, + txn_id, + }, + ) + .await?; + (scan.identities, scan.edges) + } + DeleteTarget::Document(document_id) => (vec![document_id], Vec::new()), + }; + let mut edge_tasks = Vec::new(); + append_implicit_edge_delete_tasks( + state, + &mut edge_tasks, + tenant_id, + database_id, + TraceId::ZERO, + collection.as_str(), + &scanned_edges, + ) + .await?; + + // The guards check the store. The deletes also cover the edges this + // transaction staged. + let stored = read_node_incident_edges( + state, + tenant_id, + database_id, + collection.as_str(), + &identities, + None, + ) + .await?; + let seen = match txn_id { + Some(_) => { + read_node_incident_edges( + state, + tenant_id, + database_id, + collection.as_str(), + &identities, + txn_id, + ) + .await? + } + None => Vec::new(), + }; + let deleted = union_by_node(&stored, &seen); + let (mut guards, deletes) = node_delete_tasks( + state, + tenant_id, + database_id, + NodeDeletePlan { + collection: collection.as_str(), + guarded: &stored, + deleted: &deleted, + already_deleted: &planned_edge_deletes(&edge_tasks), + }, + ) + .await?; + guards.extend(edge_tasks); + guards.extend(deletes); + Ok(guards) +} + +/// What a transaction delete removes. +enum DeleteTarget { + /// The rows a scan with these filters matches. + Rows { + filters: Vec, + prefilter: Option, + }, + /// One CRDT document, whose id names its node. + Document(String), +} + +/// Each node of `a` with its edges in `a` or `b`. `a` and `b` read the same +/// nodes in the same order. +fn union_by_node(a: &[NodeIncidentEdges], b: &[NodeIncidentEdges]) -> Vec { + a.iter() + .map(|node| { + let mut edges = node.edges.clone(); + if let Some(other) = b.iter().find(|other| other.node == node.node) { + edges.extend(other.edges.iter().cloned()); + edges.sort_unstable(); + edges.dedup(); + } + NodeIncidentEdges { + node: node.node.clone(), + edges, + } + }) + .collect() +} + +#[cfg(test)] +mod tests { + use super::*; + + fn edge(src: &str, dst: &str) -> (String, String, String) { + (src.to_string(), "L".to_string(), dst.to_string()) + } + + #[test] + fn staged_edges_join_the_stored_edges_of_the_same_node() { + let stored = vec![NodeIncidentEdges { + node: "a".to_string(), + edges: vec![edge("a", "b")], + }]; + let seen = vec![NodeIncidentEdges { + node: "a".to_string(), + edges: vec![edge("a", "b"), edge("c", "a")], + }]; + assert_eq!( + union_by_node(&stored, &seen), + vec![NodeIncidentEdges { + node: "a".to_string(), + edges: vec![edge("a", "b"), edge("c", "a")], + }] + ); + assert_eq!(union_by_node(&stored, &[]), stored); + } +} diff --git a/nodedb/src/control/planner/calvin/preexec.rs b/nodedb/src/control/planner/calvin/preexec.rs index d1a30a067..31c7dedb2 100644 --- a/nodedb/src/control/planner/calvin/preexec.rs +++ b/nodedb/src/control/planner/calvin/preexec.rs @@ -17,11 +17,13 @@ //! comparison in the executor is order-independent. No `SystemTime::now()`, //! no unseeded RNG, no `HashMap` iteration order dependency. -use nodedb_types::TenantId; +use nodedb_types::{ + DEFAULT_IDENTITY_COLUMN, ROWID_COLUMN, RowIdentity, StorageKey, Surrogate, TenantId, Value, +}; -use crate::control::server::dispatch_utils::dispatch_to_data_plane; +use super::dependent_recon_node_edges::NodeIncidentEdges; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId}; +use crate::types::{DatabaseId, TraceId, TxnId}; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; /// One implicit graph edge surfaced from the pre-execution reconnaissance scan. @@ -39,7 +41,7 @@ use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; pub struct ScannedEdge { /// Surrogate of the edge DOCUMENT (the `_from`/`_to`-carrying schemaless /// doc), parsed from the row's `id`. Carried so the data plane can - /// validate edge CONTENT (not just the matched surrogate set) against the + /// validate edge CONTENT (not only the matched surrogate set) against the /// actual stored docs at execution time, closing the recon→execute TOCTOU /// on `_from`/`_to`/`_type`. pub surrogate: u32, @@ -48,32 +50,63 @@ pub struct ScannedEdge { pub label: Option, /// The document's `weight` as stored, when present and finite. Carried so an /// UPDATE that moves or relabels the edge re-creates it with the SAME weight - /// the matching INSERT mirrored — otherwise the re-created edge would silently + /// the matching INSERT mirrored — otherwise the re-created edge will silently /// revert to the default unit weight. `None` when absent or non-finite. pub weight: Option, } +/// Whether the recon scan reads each matched row's identity, and which +/// column holds it. +#[derive(Clone, Copy)] +pub enum RowIdentityRead<'a> { + /// The caller needs no row identity. + Skip, + /// Read the identity a delete keys the row's graph node by: the declared + /// primary key column, else `id`, else `_rowid`, else the surrogate. + Read { + declared_primary_key: Option<&'a str>, + }, +} + /// Result of the OLLP pre-execution reconnaissance scan. /// /// `surrogates` is the sorted set of matched document surrogates used for OLLP -/// write-set verification (unchanged from the prior `Vec` return). -/// `edges` carries the implicit edges of any matched edge documents so their -/// auto-created graph edges can be cleaned up atomically in the same Calvin -/// transaction. Edge order is irrelevant; surrogates remain sorted. +/// write-set verification. `edges` carries the implicit edges of any matched +/// edge documents so their auto-created graph edges can be cleaned up +/// atomically in the same Calvin transaction. `identities` holds each matched +/// row's identity, in surrogate order, when the scan read it. `node_edges` +/// holds the live edges incident on each matched row's node, when the caller +/// read them. Edge order is irrelevant; surrogates remain sorted. +#[derive(Default)] pub struct PreexecScan { pub surrogates: Vec, pub edges: Vec, + pub identities: Vec, + pub node_edges: Vec, +} + +/// What a pre-execution scan reads. +pub struct PreexecRequest<'a> { + /// The database-qualified collection name. + pub collection: &'a str, + /// Serialized filter predicates. Empty matches every row. + pub filters: Vec, + /// When `Some`, only rows whose surrogate the bitmap holds. + pub prefilter: Option, + pub identity: RowIdentityRead<'a>, + /// The session transaction whose staged rows the scan also sees. `None` + /// outside a transaction block. + pub txn_id: Option, } -/// Dispatch a pre-execution scan for the given collection and serialized -/// filter bytes. Returns the sorted list of matching surrogate u32 values plus -/// the implicit edges of any matched edge documents. +/// Dispatch a pre-execution scan for `request`. Returns the sorted list of +/// matching surrogate u32 values plus the implicit edges of any matched edge +/// documents, and each row's identity when `request.identity` asks for it. /// -/// When a gateway is wired (cluster mode), the scan is routed through -/// `gateway.execute` so it reaches the owning vshard leader — a bare local -/// data-plane dispatch on a coordinator that does not host the shard would -/// return an empty result, causing OLLP convergence to fail. In single-node -/// deployments (no gateway) the scan falls back to `dispatch_to_data_plane`. +/// The scan is routed through the gateway so it reaches the owning vshard +/// leader — a bare local data-plane dispatch on a coordinator that does not +/// host the shard will return an empty result, causing OLLP convergence to +/// fail. /// /// Returns `Err` on dispatch failure (SPSC timeout, serialization error, etc.). /// Returns an empty `PreexecScan` if no documents match. @@ -81,35 +114,52 @@ pub async fn run_preexec_scan( shared: &SharedState, tenant_id: TenantId, database_id: DatabaseId, - collection: &str, - filter_bytes: Vec, + request: PreexecRequest<'_>, ) -> crate::Result { - let vshard_id = nodedb_types::CollectionKey::from_bare(database_id, collection).vshard(); - + let PreexecRequest { + collection, + filters: filter_bytes, + prefilter, + identity, + txn_id, + } = request; + // `id` (the hex surrogate) is always included regardless of projection. + // `_from`/`_to`/`_type`/`weight` surface the implicit edge of any matched + // edge document — its auto-created graph edge must be kept consistent in + // the same Calvin txn, including its `weight` when the edge is moved or + // relabeled. A delete also projects the columns the row's identity is + // read from, so the Data Plane returns their stored values. + let mut projection = vec![ + "_from".to_string(), + "_to".to_string(), + "_type".to_string(), + "weight".to_string(), + ]; + if let RowIdentityRead::Read { + declared_primary_key, + } = identity + { + projection.push( + declared_primary_key + .unwrap_or(DEFAULT_IDENTITY_COLUMN) + .to_string(), + ); + projection.push(ROWID_COLUMN.to_string()); + } let scan_plan = PhysicalPlan::Document(DocumentOp::Scan { - collection: nodedb_types::QualifiedCollection::new(database_id, collection), + // `collection` is the dependent plan's database-qualified name. + collection: nodedb_types::QualifiedCollection::from_stored(collection.to_string()), filters: filter_bytes, limit: usize::MAX, offset: 0, sort_keys: vec![], distinct: false, - // `id` (the hex surrogate) is always included regardless of - // projection. We additionally request `_from`/`_to`/`_type`/`weight` so - // the recon scan can surface the implicit edge of any matched edge - // document — its auto-created graph edge must be kept consistent in the - // same Calvin txn, including its `weight` when the edge is moved or - // relabeled. - projection: vec![ - "_from".to_string(), - "_to".to_string(), - "_type".to_string(), - "weight".to_string(), - ], + projection, computed_columns: vec![], window_functions: vec![], system_time: nodedb_types::SystemTimeScope::Current, valid_at_ms: None, - prefilter: None, + prefilter, }); // Route the recon scan to the vshard's OWNER (leader/replica), not the @@ -119,56 +169,32 @@ pub async fn run_preexec_scan( // (the predicted set comes back empty, so the dependent write never // converges). The gateway routes the read to the owning node exactly like // a normal `SELECT` and returns the same msgpack payload shape, so - // `decode_scan` applies unchanged. In single-node deployments without a - // gateway, fall back to the local data-plane dispatch. - if let Some(gateway) = shared.gateway.get() { - let gw_ctx = crate::control::gateway::core::QueryContext { - tenant_id, - trace_id: TraceId::ZERO, - database_id, - txn_id: None, - }; - // A shard verdict keeps its own typed error. - let payloads = gateway.execute_internal(&gw_ctx, scan_plan).await?; - // A single-collection scan routes to one vshard → one payload. An - // absent payload means zero matching rows. - let payload = payloads.into_iter().next().unwrap_or_default(); - return Ok(decode_scan(&payload)); - } - - let response = dispatch_to_data_plane( - shared, + // `decode_scan` applies unchanged. + let gateway = shared.installed_gateway()?; + let gw_ctx = crate::control::gateway::core::QueryContext { tenant_id, + trace_id: TraceId::ZERO, database_id, - vshard_id, - scan_plan, - TraceId::ZERO, - ) - .await?; - - scan_from_response(&response) -} - -/// Decode a local scan response. -/// -/// A shard verdict keeps its own typed error, so a scan the statement's -/// deadline cut short reports the deadline rather than a storage fault. -/// `reject_data_plane_error` passes only a `NotFound` refusal. Its payload is -/// empty, so it decodes as no matches, the answer the gateway path gives. -fn scan_from_response(response: &crate::bridge::envelope::Response) -> crate::Result { - crate::control::local_dispatch::reject_data_plane_error(response)?; - Ok(decode_scan(&response.payload)) + txn_id, + linearizable: true, + }; + // A shard verdict keeps its own typed error. + let payloads = gateway.execute_internal(&gw_ctx, scan_plan).await?; + // A single-collection scan routes to one vshard → one payload. An + // absent payload means zero matching rows. + let payload = payloads.into_iter().next().unwrap_or_default(); + decode_scan(&payload, identity) } /// Decode the msgpack scan response payload into a sorted list of surrogate u32 /// values plus the implicit edges of any matched edge documents. /// -/// Each row in the response is a msgpack map with an `id` field whose value is -/// an 8-character lowercase hex string encoding the document's u32 surrogate -/// (e.g. `"0000002a"` → `42u32`). Rows whose `id` cannot be parsed are silently -/// skipped for the surrogate set — they are legacy non-surrogate documents that -/// predate the surrogate-keyed storage format and do not participate in OLLP -/// verification. +/// Each row in the response is a msgpack map `{"id": .., "data": {..}}` +/// (`encode_raw_document_rows`). `id` is an 8-character lowercase hex string +/// encoding the document's u32 surrogate (e.g. `"0000002a"` → `42u32`). Rows +/// whose `id` cannot be parsed are skipped: they are legacy non-surrogate +/// documents that predate the surrogate-keyed storage format and do not +/// participate in OLLP verification. /// /// Additionally, for any row carrying BOTH `_from` and `_to` as strings, an /// implicit [`ScannedEdge`] is recorded (with the raw `_type` as `label`, or @@ -177,162 +203,154 @@ fn scan_from_response(response: &crate::bridge::envelope::Response) -> crate::Re /// /// The surrogate output is sorted ascending so the comparison with /// `ollp_predicted_surrogates` in the executor is a simple equality check on -/// sorted slices. Edge order is irrelevant. -fn decode_scan(payload: &[u8]) -> PreexecScan { +/// sorted slices. Identities follow the same order. Edge order is irrelevant. +fn decode_scan(payload: &[u8], identity: RowIdentityRead<'_>) -> crate::Result { if payload.is_empty() { - return PreexecScan { - surrogates: vec![], - edges: vec![], - }; + return Ok(PreexecScan::default()); } - - // Transcode msgpack → JSON string and parse the fields. - // This avoids introducing a zerompk-level partial decode dependency - // into the Control Plane layer: we let the existing transcoder convert - // the payload to a JSON array, then pull the fields out per row. - let json_str = nodedb_types::msgpack_to_json_string(payload) - .unwrap_or_else(|_| String::from_utf8_lossy(payload).into_owned()); - - decode_scan_json(&json_str) + let rows = + nodedb_types::value_from_msgpack(payload).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("OLLP reconnaissance scan payload: {e}"), + })?; + Ok(decode_rows(&rows, identity)) } -/// Pure decode of the transcoded JSON-array scan payload. Split from -/// [`decode_scan`] so it is unit-testable without hand-rolling msgpack. -fn decode_scan_json(json_str: &str) -> PreexecScan { - use sonic_rs::{JsonContainerTrait, JsonValueTrait}; - - let mut surrogates = Vec::new(); +/// Decode the rows of a scan response. Split from [`decode_scan`] so tests +/// build rows as values. +fn decode_rows(rows: &Value, identity: RowIdentityRead<'_>) -> PreexecScan { + let mut matched: Vec<(u32, Option)> = Vec::new(); let mut edges = Vec::new(); + let Value::Array(rows) = rows else { + return PreexecScan::default(); + }; + for row in rows { + let Value::Object(row) = row else { + continue; + }; + // A row whose `id` is not a parseable 8-hex surrogate is a legacy + // non-surrogate document: it is excluded from the surrogate set AND + // cannot be a surrogate-keyed edge doc, so any edge it carries is + // skipped too. + let Some(surrogate) = row + .get("id") + .and_then(Value::as_str) + .and_then(StorageKey::parse) + .map(|key| key.surrogate().as_u32()) + else { + continue; + }; + let data = row.get("data"); + let field = |name: &str| match data { + Some(Value::Object(fields)) => fields.get(name), + _ => None, + }; + let row_identity = match identity { + RowIdentityRead::Skip => None, + RowIdentityRead::Read { + declared_primary_key, + } => Some( + RowIdentity::of_row_value( + data.unwrap_or(&Value::Null), + declared_primary_key, + StorageKey::for_surrogate(Surrogate::new(surrogate)), + ) + .into_string(), + ), + }; + matched.push((surrogate, row_identity)); - if let Ok(rows) = sonic_rs::from_str::(json_str) - && rows.is_array() - { - for row in rows.as_array().into_iter().flatten() { - // Parse the row's surrogate ONCE and reuse it for both the - // surrogate set and the edge record. A row whose `id` is not a - // parseable 8-hex surrogate is a legacy non-surrogate document: it - // is excluded from the surrogate set AND cannot be a surrogate-keyed - // edge doc, so any edge it carries is skipped too. - let surrogate = row - .get("id") - .and_then(|id_val| id_val.as_str()) - .filter(|id_str| id_str.len() == 8) - .and_then(|id_str| u32::from_str_radix(id_str, 16).ok()); - - if let Some(surrogate) = surrogate { - surrogates.push(surrogate); - } - - // The raw-document scan encoder nests the document's user fields - // under a `data` object alongside the top-level `id` - // (`encode_raw_document_rows`): `{"id": "..", "data": {..fields..}}`. - // An edge document carries BOTH `_from` and `_to` as strings inside - // `data`. - let data = row.get("data"); - let from = data - .as_ref() - .and_then(|d| d.get("_from")) - .and_then(|v| v.as_str()); - let to = data - .as_ref() - .and_then(|d| d.get("_to")) - .and_then(|v| v.as_str()); - // Only record an edge when the row is BOTH a surrogate-keyed doc - // AND carries both endpoints — content-drift validation keys edges - // by surrogate, so an unparseable-id edge row is not a real edge. - if let (Some(surrogate), Some(from), Some(to)) = (surrogate, from, to) { - let label = data - .as_ref() - .and_then(|d| d.get("_type")) - .and_then(|v| v.as_str()) - .map(str::to_string); - // A finite numeric `weight` mirrors the doc's mirrored edge - // weight; non-finite / absent / non-numeric → `None` (unit - // weight). `as_f64` covers both integer and float JSON numbers. - let weight = data - .as_ref() - .and_then(|d| d.get("weight")) - .and_then(|v| v.as_f64()) - .filter(|w| w.is_finite()); - edges.push(ScannedEdge { - surrogate, - from: from.to_string(), - to: to.to_string(), - label, - weight, - }); - } + // An edge document carries BOTH `_from` and `_to` as strings inside + // `data`. + let from = field("_from").and_then(Value::as_str); + let to = field("_to").and_then(Value::as_str); + if let (Some(from), Some(to)) = (from, to) { + let label = field("_type").and_then(Value::as_str).map(str::to_string); + // A finite numeric `weight` mirrors the doc's mirrored edge + // weight; non-finite / absent / non-numeric → `None` (unit + // weight). + let weight = field("weight") + .and_then(|v| match v { + Value::Float(w) => Some(*w), + Value::Integer(w) => Some(*w as f64), + _ => None, + }) + .filter(|w| w.is_finite()); + edges.push(ScannedEdge { + surrogate, + from: from.to_string(), + to: to.to_string(), + label, + weight, + }); } } - surrogates.sort_unstable(); - PreexecScan { surrogates, edges } + matched.sort_unstable_by_key(|(surrogate, _)| *surrogate); + let surrogates = matched.iter().map(|(surrogate, _)| *surrogate).collect(); + let identities = matched + .into_iter() + .filter_map(|(_, identity)| identity) + .collect(); + PreexecScan { + surrogates, + edges, + identities, + node_edges: Vec::new(), + } } #[cfg(test)] mod tests { - use super::*; - use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; - use crate::types::{Lsn, RequestId}; + use std::collections::HashMap; - fn refusal(code: ErrorCode) -> Response { - Response { - request_id: RequestId::new(1), - status: Status::Error, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: Lsn::ZERO, - error_code: Some(Box::new(code)), - read_set_valid: None, - read_version_lsn: Lsn::ZERO, - write_set: Vec::new(), - } - } + use super::*; - /// A refused scan keeps its code, never a storage error. - #[test] - fn a_refused_scan_keeps_its_code() { - let code = ErrorCode::Unsupported { - detail: "not on this engine".into(), - }; - match scan_from_response(&refusal(code.clone())) { - Err(crate::Error::DataPlane(kept)) => assert_eq!(kept, code), - Err(other) => panic!("expected the typed refusal, got {other:?}"), - Ok(_) => panic!("a refused scan must fail"), - } + fn row(id: &str, fields: &[(&str, Value)]) -> Value { + let data: HashMap = fields + .iter() + .map(|(name, value)| ((*name).to_string(), value.clone())) + .collect(); + Value::Object(HashMap::from([ + ("id".to_string(), Value::String(id.to_string())), + ("data".to_string(), Value::Object(data)), + ])) } - /// A `NotFound` refusal means the shard holds no slice of the collection, - /// so the scan matched nothing. - #[test] - fn a_not_found_scan_matches_nothing() { - let scan = scan_from_response(&refusal(ErrorCode::NotFound)) - .expect("a NotFound refusal reads as an empty scan"); - assert!(scan.surrogates.is_empty()); - assert!(scan.edges.is_empty()); + fn text(s: &str) -> Value { + Value::String(s.to_string()) } #[test] fn decode_empty_payload_returns_empty() { - let scan = decode_scan(&[]); + let scan = decode_scan(&[], RowIdentityRead::Skip).expect("decode"); assert!(scan.surrogates.is_empty()); assert!(scan.edges.is_empty()); + assert!(scan.identities.is_empty()); } #[test] - fn decode_json_extracts_surrogates_and_edges() { + fn decode_extracts_surrogates_and_edges() { // Two edge rows + one non-edge row. `id` is the 8-hex surrogate; an edge // row additionally carries `_from`/`_to` (and optionally `_type`). - let json = r#"[ - {"id":"0000002a","data":{"_from":"a","_to":"b","_type":"ROAD","weight":5.0}}, - {"id":"0000000b","data":{"_from":"c","_to":"d"}}, - {"id":"00000001","data":{"name":"alice"}} - ]"#; - let scan = decode_scan_json(json); + let rows = Value::Array(vec![ + row( + "0000002a", + &[ + ("_from", text("a")), + ("_to", text("b")), + ("_type", text("ROAD")), + ("weight", Value::Float(5.0)), + ], + ), + row("0000000b", &[("_from", text("c")), ("_to", text("d"))]), + row("00000001", &[("name", text("alice"))]), + ]); + let scan = decode_rows(&rows, RowIdentityRead::Skip); // Surrogates: all three rows parse; sorted ascending. assert_eq!(scan.surrogates, vec![1, 11, 42]); + assert!(scan.identities.is_empty()); // Edges: only the two rows with BOTH _from and _to. assert_eq!(scan.edges.len(), 2); @@ -343,10 +361,7 @@ mod tests { .expect("edge a->b present"); assert_eq!(road.to, "b"); assert_eq!(road.label.as_deref(), Some("ROAD")); - // The edge carries the document's surrogate (id "0000002a" → 42). assert_eq!(road.surrogate, 42); - // A finite numeric `weight` is surfaced so a moved/relabeled edge keeps - // it. assert_eq!(road.weight, Some(5.0)); let untyped = scan .edges @@ -355,27 +370,54 @@ mod tests { .expect("edge c->d present"); assert_eq!(untyped.to, "d"); assert_eq!(untyped.label, None); - // id "0000000b" → 11. assert_eq!(untyped.surrogate, 11); - // No `weight` field → `None` (unit weight). assert_eq!(untyped.weight, None); } #[test] - fn decode_json_row_without_both_endpoints_is_not_an_edge() { - // `_from` present but no `_to` → surrogate extracted, no edge recorded. - // The fixture must use the wire shape emitted by `encode_raw_document_rows`: - // `{"id": "..", "data": {..fields..}}`. A top-level `_from` would test - // the "no `data` field" branch instead of the intended one. - let json = r#"[{"id":"00000005","data":{"_from":"x"}}]"#; - let scan = decode_scan_json(json); + fn decode_row_without_both_endpoints_is_not_an_edge() { + let rows = Value::Array(vec![row("00000005", &[("_from", text("x"))])]); + let scan = decode_rows(&rows, RowIdentityRead::Skip); assert_eq!(scan.surrogates, vec![5]); assert!(scan.edges.is_empty()); } + #[test] + fn decode_reads_the_identity_a_delete_keys_the_node_by() { + let rows = Value::Array(vec![ + row( + "00000009", + &[("sku", Value::Integer(42)), ("id", text("x"))], + ), + row("00000003", &[("_rowid", Value::Integer(3))]), + row("00000007", &[]), + ]); + let declared = decode_rows( + &rows, + RowIdentityRead::Read { + declared_primary_key: Some("sku"), + }, + ); + assert_eq!(declared.surrogates, vec![3, 7, 9]); + // Sorted with the surrogates: `_rowid`, then the decimal surrogate, + // then the declared key. + assert_eq!(declared.identities, vec!["3", "7", "42"]); + + let by_id = decode_rows( + &rows, + RowIdentityRead::Read { + declared_primary_key: None, + }, + ); + assert_eq!(by_id.identities, vec!["3", "7", "x"]); + } + + #[test] + fn a_payload_that_is_not_msgpack_is_an_error() { + assert!(decode_scan(&[0xc1], RowIdentityRead::Skip).is_err()); + } + // Format-coupled coverage of the surrogate decode lives in // `tests/executor_tests/test_ollp_verification.rs`, which exercises the // decoder against real scan-response payloads emitted by the Data Plane. - // Keeping a hand-rolled msgpack mock in this unit test would duplicate - // the wire format and break on every response_codec change. } diff --git a/nodedb/src/control/planner/calvin/reservation.rs b/nodedb/src/control/planner/calvin/reservation.rs index 0a321083f..191d01ed4 100644 --- a/nodedb/src/control/planner/calvin/reservation.rs +++ b/nodedb/src/control/planner/calvin/reservation.rs @@ -71,7 +71,9 @@ pub(crate) async fn submit_local_reserve_read( /// /// Routing logic mirrors [`super::submit::submit_calvin_routed_assign`] /// exactly: -/// - **Not cluster mode** OR **leader is self**: submit locally. +/// - **No `cluster_transport`**: this `SharedState` never ran `start_raft`, +/// so no sequencer runs here. Return `SequencerUnavailable`. +/// - **Leader is self**: submit locally. /// - **No leader elected (0 / none)**: return a typed error — never submit /// locally, since a non-leader submit is silently discarded. /// - **Leader is a remote node**: register the leader's address from the live @@ -86,13 +88,10 @@ pub(crate) async fn submit_reserve_read( ) -> crate::Result { let local_timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); - // Not cluster mode — single-node is the only sequencer member, hence the - // leader. Submit locally. - let (Some(transport), Some(_routing)) = ( - state.cluster_transport.as_ref(), - state.cluster_routing.as_ref(), - ) else { - return submit_local_reserve_read(state, key, vshard, owner, local_timeout).await; + // Every running server has a cluster transport, the synthesized one-node + // cluster included. Without one, `start_raft` never ran here. + let Some(transport) = state.cluster_transport.as_ref() else { + return Err(Error::SequencerUnavailable); }; // Resolve the sequencer-group leader from THIS node's live Raft status. @@ -114,7 +113,7 @@ pub(crate) async fn submit_reserve_read( }); } - // Leader is self: submit locally (a self-RPC would be a pointless extra + // Leader is self: submit locally (a self-RPC will be a pointless extra // hop and the local inbox is the one that gets the assignment). if leader == state.node_id { return submit_local_reserve_read(state, key, vshard, owner, local_timeout).await; @@ -155,7 +154,7 @@ pub(crate) async fn submit_reserve_read( // The leader-side handler holds this RPC open until the reservation is // assigned (up to `deadline_remaining_ms`). The generic short `rpc_timeout` - // would abort the call long before that, so bound the response read by the + // will abort the call long before that, so bound the response read by the // forwarded deadline plus a margin for the round-trip itself. let read_timeout = Duration::from_millis(deadline_remaining_ms.saturating_add(2_000)); match transport @@ -223,11 +222,10 @@ pub(crate) async fn release_reservation( vshard: u32, reason: ReleaseReason, ) -> crate::Result<()> { - let (Some(transport), Some(_routing)) = ( - state.cluster_transport.as_ref(), - state.cluster_routing.as_ref(), - ) else { - return submit_local_release(state, owner, vshard, reason).await; + // Without a cluster transport `start_raft` never ran here, so no + // reservation this release can name was ever granted. + let Some(transport) = state.cluster_transport.as_ref() else { + return Err(Error::SequencerUnavailable); }; let status_fn = match state.raft_status_fn.get() { diff --git a/nodedb/src/control/planner/calvin/retry_loop.rs b/nodedb/src/control/planner/calvin/retry_loop.rs index 6a451acee..0bea546d1 100644 --- a/nodedb/src/control/planner/calvin/retry_loop.rs +++ b/nodedb/src/control/planner/calvin/retry_loop.rs @@ -9,7 +9,7 @@ //! under predicate drift. The scheduler's only job on mismatch is to release the //! aborted attempt's locks and signal the completion registry so this loop wakes. -use nodedb_cluster::calvin::{AttemptOutcome, CalvinCompletionRegistry, TxnId}; +use nodedb_cluster::calvin::{AbortReason, AttemptOutcome, CalvinCompletionRegistry, TxnId}; use crate::control::cluster::calvin::executor::ollp::error::OllpError; use crate::control::cluster::calvin::executor::ollp::orchestrator::OllpOrchestrator; @@ -36,9 +36,13 @@ fn pre_admission_cause(err: OllpError) -> OllpExhaustedCause { /// Terminal outcome of the dependent-read retry loop. #[derive(Debug)] pub enum DependentOutcome { - /// The dependent transaction committed; carries its `TxnId` so the caller - /// can drain the applied response the scheduler deposited. - Committed(TxnId), + /// The dependent transaction committed. The caller drains the applied + /// response the scheduler deposited under `txn_id`, and reads each + /// participant's report from `ack_results`. + Committed { + txn_id: TxnId, + ack_results: Vec>, + }, /// The batch decided an empty write set — the predicate matched no rows and /// no other task in it writes — so no Calvin entry was proposed. The /// statement's result is zero rows affected. @@ -128,8 +132,8 @@ where // registry, which receives the replicated completion ack on every // sequencer-group member. let txn_id = TxnId::new(assignment.epoch, assignment.position); - let completion_rx = registry.register_completion(txn_id, assignment.participants); - let outcome = tokio::time::timeout(timeout, completion_rx) + let completion_rx = registry.register_completion_report(txn_id, assignment.participants); + let report = tokio::time::timeout(timeout, completion_rx) .await .map_err(|_| Error::Internal { detail: "timed out waiting for Calvin completion".into(), @@ -138,17 +142,26 @@ where detail: "Calvin completion channel closed".into(), })?; - match outcome { + match report.outcome { // Return the completed txn's id so the caller can drain the applied - // Response (RETURNING rows) the scheduler deposited before the ack. - AttemptOutcome::Completed => return Ok(DependentOutcome::Committed(txn_id)), - // Terminal, NON-retryable: the global cross-shard verdict was ABORT. - // A committed verdict is not OLLP predicate drift — a fresh - // reconnaissance cannot change it — so surface it to the client - // immediately instead of burning retries. The verdict's reason picks - // the error: a stale read-set is SQLSTATE 40001, a participant error - // is not. - AttemptOutcome::Aborted { reason } => { + // Response (RETURNING rows) the scheduler deposited before the ack, + // and the participants' reports for a participant on another node. + AttemptOutcome::Completed => { + return Ok(DependentOutcome::Committed { + txn_id, + ack_results: report.ack_results, + }); + } + // Terminal, NON-retryable: the global cross-shard verdict was ABORT + // for a reason a fresh reconnaissance cannot change, so surface it + // to the client immediately instead of burning retries. The + // verdict's reason picks the error: a stale read-set is SQLSTATE + // 40001, a participant error is not. `PredictionDrift` and + // `PartsLost` retry, in the arm below: a multi-part transaction + // whose parts a leader change lost staged nothing anywhere. + AttemptOutcome::Aborted { reason } + if reason != AbortReason::PredictionDrift && reason != AbortReason::PartsLost => + { return Err(calvin_abort_error(reason)); } // Terminal, NON-retryable: the scheduler rejected the transaction's @@ -160,11 +173,14 @@ where detail: format!("calvin transaction routing failed: {detail}"), }); } - AttemptOutcome::Mismatch => { - // POST-EXEC predicate drift. The scheduler already released the - // aborted attempt's locks before signalling the registry, so a - // FRESH reconnaissance is safe — and necessary, since the stale - // prediction can never converge under drift. + AttemptOutcome::Mismatch | AttemptOutcome::Aborted { .. } => { + // POST-EXEC predicate drift: an OLLP mismatch, or an abort + // verdict because a participant found state other than the + // reconnaissance predicted and wrote nothing. The scheduler + // already released the aborted attempt's locks before + // signalling the registry, so a FRESH reconnaissance is safe — + // and necessary, since the stale prediction can never converge + // under drift. if retry >= ollp_max_retries { return Err(Error::OllpExhausted { retries: ollp_max_retries.min(u8::MAX as u32) as u8, @@ -189,16 +205,16 @@ mod tests { //! without a live server/executor. The coordinator's injected `submit` closure //! returns a `RoutedAssignment` carrying a deterministic `(epoch, position)` — //! exactly the `(epoch, position)` the loop feeds to - //! `register_completion(TxnId::new(epoch, position))`. The closure also forwards + //! `register_completion_report(TxnId::new(epoch, position))`. The closure also forwards //! that same `TxnId` to the fake over an mpsc channel; the fake then either calls //! `note_ollp_mismatch(txn)` (first K submissions → `Mismatch`) or //! `note_completion_ack(txn, 1)` (submission K+1, 1 participant → fires //! `Completed`). The `rescan` closure increments a counter and returns a fresh //! prediction vec. //! - //! Since the routed `submit` now returns the assignment itself (the leader - //! assigns and replies with `(epoch, position)`), the loop no longer calls - //! `register_submission` / the fake no longer needs `note_assigned`. The + //! The routed `submit` returns the assignment itself (the leader + //! assigns and replies with `(epoch, position)`), so the loop never calls + //! `register_submission` and the fake needs no `note_assigned`. The //! `(epoch, position)` source is deterministic on the test side: `epoch = //! inbox_seq`, `position = 0`. Both the `RoutedAssignment` returned by `submit` //! AND the fake's `note_completion_ack` / `note_ollp_mismatch` use that same @@ -206,7 +222,7 @@ mod tests { //! //! Determinism: a current-thread runtime plus a bounded fake channel with enough //! capacity that the closure's `send().await` never yields control to the fake - //! before `register_completion` runs. + //! before `register_completion_report` runs. use super::*; @@ -228,7 +244,7 @@ mod tests { /// Spawn the fake scheduler. It reads `TxnId` events (the same `TxnId` the loop /// registers for completion) and, for the first `mismatch_count` events, signals /// an OLLP mismatch; on the next event it acks completion with a single - /// participant (which fires `Completed`). The loop no longer calls + /// participant (which fires `Completed`). The loop never calls /// `register_submission`, so `note_assigned` is not needed. fn spawn_fake_scheduler( registry: Arc, @@ -315,7 +331,7 @@ mod tests { }; assert!( - matches!(result, Ok(DependentOutcome::Committed(_))), + matches!(result, Ok(DependentOutcome::Committed { .. })), "expected Committed, got {result:?}" ); assert_eq!( @@ -459,7 +475,7 @@ mod tests { }; assert!( - matches!(result, Ok(DependentOutcome::Committed(_))), + matches!(result, Ok(DependentOutcome::Committed { .. })), "expected Committed, got {result:?}" ); assert_eq!( @@ -618,4 +634,101 @@ mod tests { ); }); } + + /// Run the loop against a fake that answers the first attempt with an + /// ABORT verdict for `reason` and commits every later attempt. Returns + /// the result, the submit count and the rescan count. + fn run_with_first_abort(reason: AbortReason) -> (crate::Result, u64, u32) { + let rt = tokio::runtime::Builder::new_current_thread() + .enable_all() + .build() + .expect("build current-thread runtime"); + rt.block_on(async { + let registry = CalvinCompletionRegistry::new_detached(); + let orchestrator = zero_backoff_orchestrator(); + let (tx, mut rx) = tokio::sync::mpsc::channel::(16); + let fake = { + let registry = Arc::clone(®istry); + tokio::spawn(async move { + let mut first = true; + while let Some(txn) = rx.recv().await { + if first { + registry.note_verdict( + txn, + nodedb_cluster::calvin::VerdictOutcome::Abort(reason), + ); + first = false; + } + registry.note_completion_ack(txn, 1); + } + }) + }; + let seq = Arc::new(AtomicU64::new(1)); + let submit_calls = Arc::new(AtomicU64::new(0)); + let rescan_calls = Arc::new(AtomicU32::new(0)); + let result = { + let seq = Arc::clone(&seq); + let submit_calls = Arc::clone(&submit_calls); + let rescan_calls = Arc::clone(&rescan_calls); + let tx = tx.clone(); + run_dependent_with_retry(DependentRetryArgs { + registry: ®istry, + orchestrator: &orchestrator, + predicate_class_hash: 0xABCD, + timeout: std::time::Duration::from_secs(5), + ollp_max_retries: 5, + initial_predicted: vec![1], + submit: move |_predicted: &Vec| { + let seq = Arc::clone(&seq); + let submit_calls = Arc::clone(&submit_calls); + let tx = tx.clone(); + async move { + submit_calls.fetch_add(1, Ordering::SeqCst); + let assignment = fake_assignment(seq.fetch_add(1, Ordering::SeqCst)); + let txn = TxnId::new(assignment.epoch, assignment.position); + tx.send(txn).await.expect("fake recv alive"); + Ok::, OllpError>(Some(assignment)) + } + }, + rescan: move || { + let rescan_calls = Arc::clone(&rescan_calls); + async move { + rescan_calls.fetch_add(1, Ordering::SeqCst); + Ok(vec![2]) + } + }, + }) + .await + }; + drop(tx); + let _ = fake.await; + ( + result, + submit_calls.load(Ordering::SeqCst), + rescan_calls.load(Ordering::SeqCst), + ) + }) + } + + #[test] + fn a_prediction_drift_abort_rescans_and_retries() { + let (result, submits, rescans) = run_with_first_abort(AbortReason::PredictionDrift); + assert!( + matches!(result, Ok(DependentOutcome::Committed { .. })), + "expected Committed, got {result:?}" + ); + assert_eq!(submits, 2, "one drift abort + one commit → two submits"); + assert_eq!(rescans, 1, "the drift abort reads again once"); + } + + #[test] + fn a_participant_error_abort_is_terminal() { + let (result, submits, rescans) = run_with_first_abort(AbortReason::ParticipantError); + assert!( + matches!(result, Err(Error::CalvinParticipantError)), + "expected CalvinParticipantError, got {result:?}" + ); + assert_eq!(submits, 1); + assert_eq!(rescans, 0); + } } diff --git a/nodedb/src/control/planner/calvin/submit/assign.rs b/nodedb/src/control/planner/calvin/submit/assign.rs index 49d414d9e..4459347ad 100644 --- a/nodedb/src/control/planner/calvin/submit/assign.rs +++ b/nodedb/src/control/planner/calvin/submit/assign.rs @@ -12,6 +12,7 @@ use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; use nodedb_cluster::calvin::types::TxClass; use nodedb_cluster::{RaftRpc, SubmitCalvinInboxRequest, SubmitCalvinInboxResponse}; +use super::stream::{PartStream, StreamTarget, stream_parts}; use crate::Error; use crate::control::cluster::warm_peers::register_peers_from_topology; use crate::control::state::SharedState; @@ -42,8 +43,24 @@ pub struct RoutedAssignment { /// call it after decoding the wire bytes, mirroring how `hook.rs` delegates to /// `submit_and_await_calvin_with_timeout`. pub(crate) async fn submit_local_assign( + state: &SharedState, + mut tx_class: TxClass, + timeout: Duration, +) -> crate::Result { + super::local::raise_metadata_floor(state, &mut tx_class); + super::local::stamp_incarnations(state, &mut tx_class)?; + let stream = super::parts::split_into_parts(state, &mut tx_class)?; + assign_prepared(state, tx_class, stream, timeout).await +} + +/// Submit a stamped `tx_class` to this node's sequencer, await its +/// assignment, then stream its parts when it carries them as parts. +/// +/// PRECONDITION: this node is the sequencer-group leader. +async fn assign_prepared( state: &SharedState, tx_class: TxClass, + stream: Option, timeout: Duration, ) -> crate::Result { let inbox = state @@ -55,19 +72,21 @@ pub(crate) async fn submit_local_assign( .get() .ok_or(Error::SequencerUnavailable)?; - let inbox_seq = inbox.submit(tx_class).map_err(|e| Error::BadRequest { - detail: format!("Calvin sequencer rejected transaction: {e}"), - })?; - - let assignment_rx = registry.register_submission(inbox_seq); - let (epoch, position, participants) = tokio::time::timeout(timeout, assignment_rx) - .await - .map_err(|_| Error::Internal { - detail: "timed out waiting for Calvin sequencer assignment".to_owned(), - })? - .map_err(|_| Error::Internal { - detail: "Calvin sequencer assignment channel closed".to_owned(), - })?; + let (inbox_seq, assignment_rx) = + inbox + .submit_with(tx_class, registry) + .map_err(|e| Error::BadRequest { + detail: format!("Calvin sequencer rejected transaction: {e}"), + })?; + let (epoch, position, participants) = + super::local::await_assignment(registry, inbox_seq, assignment_rx, timeout).await?; + // A lost stream aborts the transaction, and the caller's completion + // wait reports it. + if let Some(stream) = &stream { + stream_parts(state, StreamTarget::Local, stream, true) + .await + .into_result()?; + } Ok(RoutedAssignment { inbox_seq, @@ -83,9 +102,10 @@ pub(crate) async fn submit_local_assign( /// /// The OLLP dependent sibling of [`super::routed::submit_calvin_routed`]. Routing logic mirrors /// it exactly: -/// - **Not cluster mode** (no `cluster_transport` / `cluster_routing`) OR -/// **leader is self**: submit-and-assign locally — single-node / this node IS -/// the sequencer leader. +/// - **No `cluster_transport`**: this `SharedState` never ran `start_raft`, +/// so no sequencer runs here. Return `SequencerUnavailable`. +/// - **Leader is self**: submit-and-assign locally — this node IS the +/// sequencer leader. /// - **No leader elected (0 / none)**: return a typed error — never submit /// locally, since a non-leader submit is silently discarded. /// - **Leader is a remote node**: register the leader's address from the live @@ -95,17 +115,17 @@ pub(crate) async fn submit_local_assign( /// `crate::Error`. pub async fn submit_calvin_routed_assign( state: &SharedState, - tx_class: TxClass, + mut tx_class: TxClass, ) -> crate::Result { + super::local::raise_metadata_floor(state, &mut tx_class); + super::local::stamp_incarnations(state, &mut tx_class)?; + let stream = super::parts::split_into_parts(state, &mut tx_class)?; let local_timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); - // Not cluster mode — single-node is the only sequencer member, hence the - // leader. Submit-and-assign locally. - let (Some(transport), Some(_routing)) = ( - state.cluster_transport.as_ref(), - state.cluster_routing.as_ref(), - ) else { - return submit_local_assign(state, tx_class, local_timeout).await; + // Every running server has a cluster transport, the synthesized one-node + // cluster included. Without one, `start_raft` never ran here. + let Some(transport) = state.cluster_transport.as_ref() else { + return Err(Error::SequencerUnavailable); }; // Resolve the sequencer-group leader from THIS node's live Raft status. @@ -128,10 +148,10 @@ pub async fn submit_calvin_routed_assign( }); } - // Leader is self: submit-and-assign locally (a self-RPC would be a pointless + // Leader is self: submit-and-assign locally (a self-RPC will be a pointless // extra hop and the local registry is the one that gets the assignment). if leader == state.node_id { - return submit_local_assign(state, tx_class, local_timeout).await; + return assign_prepared(state, tx_class, stream, local_timeout).await; } // Remote leader: ensure its address is registered before dispatch, then send @@ -159,10 +179,10 @@ pub async fn submit_calvin_routed_assign( // The leader-side handler holds this RPC open until the transaction is // assigned (up to `deadline_remaining_ms`). The generic short `rpc_timeout` - // would abort the call long before that, so bound the response read by the + // will abort the call long before that, so bound the response read by the // forwarded deadline plus a margin for the round-trip itself. let read_timeout = Duration::from_millis(deadline_remaining_ms.saturating_add(2_000)); - match transport + let assignment = match transport .send_rpc_with_read_timeout(leader, RaftRpc::SubmitCalvinInboxRequest(req), read_timeout) .await { @@ -172,23 +192,38 @@ pub async fn submit_calvin_routed_assign( position, participants, error: None, - })) => Ok(RoutedAssignment { + })) => RoutedAssignment { inbox_seq, epoch, position, participants: participants as usize, - }), + }, Ok(RaftRpc::SubmitCalvinInboxResponse(SubmitCalvinInboxResponse { error: Some(e), .. - })) => Err(Error::Internal { - detail: format!("calvin-inbox failed on sequencer leader node {leader}: {e:?}"), - }), - Ok(other) => Err(Error::Internal { - detail: format!("calvin-inbox: unexpected reply from node {leader}: {other:?}"), - }), - Err(e) => Err(Error::Internal { - detail: format!("calvin-inbox RPC to sequencer leader node {leader} failed: {e}"), - }), + })) => { + return Err(Error::Internal { + detail: format!("calvin-inbox failed on sequencer leader node {leader}: {e:?}"), + }); + } + Ok(other) => { + return Err(Error::Internal { + detail: format!("calvin-inbox: unexpected reply from node {leader}: {other:?}"), + }); + } + Err(e) => { + return Err(Error::Internal { + detail: format!("calvin-inbox RPC to sequencer leader node {leader} failed: {e}"), + }); + } + }; + // The leader opened the stream before it reported the assignment. A lost + // stream aborts the transaction, and the caller's completion wait + // reports it. + if let Some(stream) = &stream { + stream_parts(state, StreamTarget::Remote(leader), stream, true) + .await + .into_result()?; } + Ok(assignment) } diff --git a/nodedb/src/control/planner/calvin/submit/edge_slices.rs b/nodedb/src/control/planner/calvin/submit/edge_slices.rs new file mode 100644 index 000000000..23658f2bd --- /dev/null +++ b/nodedb/src/control/planner/calvin/submit/edge_slices.rs @@ -0,0 +1,186 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Slice a Calvin transaction's edge batches by the homes of their edges. +//! +//! An edge lives on both endpoint homes: its source's key home, which owns +//! it, and its destination's. A batch edge op routes to the homes of all its +//! edges, so one batch can span hundreds of vShards, past what one sequencer +//! entry targets, and every home will stage every edge of it. +//! +//! Before a transaction is submitted, each batch is split into one batch per +//! home pair `(source home, destination home)`. Each slice routes to at most +//! two vShards, so no task spans more vShards than one part allows. Each +//! home stages exactly the edges it holds: both homes of an edge receive it, +//! each with its own copy. +//! +//! Every edge of a slice has the same owner, so the slice's count is +//! counted once, at that owner (see `stage_calvin_plan`). The coordinator of +//! a transaction with no write of its own sums its homes' answers (see +//! `ReplyFold`). + +use std::collections::BTreeMap; + +use nodedb_cluster::calvin::types::{EngineKeySet, TxClass}; +use nodedb_physical::physical_plan::{BatchEdge, GraphOp, PhysicalPlan}; + +use crate::Error; +use crate::types::{RecordHomes, VShardId}; + +/// Split every edge batch of `tx_class` that spans more than one home pair. +/// A class with no edge write, or one already split into parts, is left as +/// it is. `body_plans` follows the split: a slice of a body task is a body +/// task. +pub(crate) fn slice_edge_batches(tx_class: &mut TxClass) -> crate::Result<()> { + if tx_class.is_multi_part() || !writes_edges(tx_class) { + return Ok(()); + } + let plans = + nodedb_physical::physical_plan::wire::decode_batch(&tx_class.plans).map_err(|e| { + Error::Serialization { + format: "msgpack".into(), + detail: format!("calvin edge slicing: plan decode: {e}"), + } + })?; + if !plans.iter().any(|plan| batch_slices(plan).is_some()) { + return Ok(()); + } + let mut sliced: Vec = Vec::with_capacity(plans.len()); + let mut body_plans: Vec = Vec::with_capacity(tx_class.body_plans.len()); + for (task, plan) in (0u32..).zip(plans) { + let body = tx_class.body_plans.contains(&task); + let slices = batch_slices(&plan).unwrap_or_else(|| vec![plan]); + for slice in slices { + if body { + body_plans.push(index_of(sliced.len())?); + } + sliced.push(slice); + } + } + tx_class.plans = nodedb_physical::physical_plan::wire::encode_batch(&sliced).map_err(|e| { + Error::Serialization { + format: "msgpack".into(), + detail: format!("calvin edge slicing: plan encode: {e}"), + } + })?; + tx_class.body_plans = body_plans; + Ok(()) +} + +/// Whether `tx_class` writes an edge: its write set names edge keys. +pub(crate) fn writes_edges(tx_class: &TxClass) -> bool { + tx_class + .write_set + .0 + .iter() + .any(|keys| matches!(keys, EngineKeySet::Edge { .. })) +} + +/// The per-home-pair slices of `plan` when it is an edge batch over more +/// than one home pair, else `None`. +fn batch_slices(plan: &PhysicalPlan) -> Option> { + let (edges, delete) = match plan { + PhysicalPlan::Graph(GraphOp::EdgePutBatch { edges }) => (edges, false), + PhysicalPlan::Graph(GraphOp::EdgeDeleteBatch { edges }) => (edges, true), + _ => return None, + }; + let mut groups: BTreeMap<(u32, u32), Vec> = BTreeMap::new(); + for edge in edges { + groups + .entry(home_pair(edge)) + .or_default() + .push(edge.clone()); + } + if groups.len() < 2 { + return None; + } + Some( + groups + .into_values() + .map(|edges| { + PhysicalPlan::Graph(if delete { + GraphOp::EdgeDeleteBatch { edges } + } else { + GraphOp::EdgePutBatch { edges } + }) + }) + .collect(), + ) +} + +/// `(source home, destination home)` of `edge`. The source home owns it. +fn home_pair(edge: &BatchEdge) -> (u32, u32) { + ( + RecordHomes::edge_owner(&edge.src_id).as_u32(), + VShardId::from_key(edge.dst_id.as_bytes()).as_u32(), + ) +} + +fn index_of(len: usize) -> crate::Result { + u32::try_from(len).map_err(|_| Error::BadRequest { + detail: format!("a transaction of {len} tasks is more than one transaction indexes"), + }) +} + +#[cfg(test)] +mod tests { + use std::collections::BTreeSet; + + use nodedb_types::{DatabaseId, QualifiedCollection, Surrogate}; + + use super::*; + use crate::control::cluster::calvin::scheduler::driver::core::routing::{ + PlanRouting, plan_vshard_in_database, + }; + + fn edge(src: String, dst: String) -> BatchEdge { + BatchEdge { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "g"), + src_id: src, + label: "L".into(), + dst_id: dst, + src_surrogate: Surrogate::new(1), + dst_surrogate: Surrogate::new(2), + } + } + + /// A batch over 400 home pairs splits into one slice per pair. Each + /// slice routes to at most two vShards, every edge lands in exactly one + /// slice, and every edge of a slice has the same owner. + #[test] + fn a_wide_batch_splits_into_slices_of_one_home_pair() { + let edges: Vec = (0..400) + .map(|i| edge(format!("src{i}"), format!("dst{i}"))) + .collect(); + let pairs: BTreeSet<(u32, u32)> = edges.iter().map(home_pair).collect(); + let plan = PhysicalPlan::Graph(GraphOp::EdgePutBatch { + edges: edges.clone(), + }); + let slices = batch_slices(&plan).expect("a wide batch splits"); + assert_eq!(slices.len(), pairs.len()); + let mut seen = 0; + for slice in &slices { + let PhysicalPlan::Graph(GraphOp::EdgePutBatch { edges }) = slice else { + panic!("a slice stays a put batch"); + }; + seen += edges.len(); + let owners: BTreeSet = edges + .iter() + .map(|e| RecordHomes::edge_owner(&e.src_id).as_u32()) + .collect(); + assert_eq!(owners.len(), 1, "one owner per slice"); + match plan_vshard_in_database(slice, DatabaseId::DEFAULT) { + PlanRouting::Vshards(homes) => assert!(homes.len() <= 2), + _ => panic!("a slice routes to its homes"), + } + } + assert_eq!(seen, edges.len(), "every edge lands in one slice"); + + let one_pair = PhysicalPlan::Graph(GraphOp::EdgePutBatch { + edges: vec![edge("a".into(), "b".into())], + }); + assert!( + batch_slices(&one_pair).is_none(), + "one home pair stays whole" + ); + } +} diff --git a/nodedb/src/control/planner/calvin/submit/local.rs b/nodedb/src/control/planner/calvin/submit/local.rs index 30d51eac6..27574e6d3 100644 --- a/nodedb/src/control/planner/calvin/submit/local.rs +++ b/nodedb/src/control/planner/calvin/submit/local.rs @@ -10,10 +10,13 @@ use std::time::Duration; use nodedb_cluster::calvin::types::TxClass; -use nodedb_cluster::calvin::{AttemptOutcome, TxnId}; +use nodedb_cluster::calvin::{ + Assignment, AssignmentReceiver, AttemptOutcome, CalvinCompletionRegistry, TxnId, +}; +use super::stream::{PartStream, StreamTarget, stream_parts}; use crate::Error; -use crate::bridge::envelope::Response; +use crate::bridge::envelope::{PhysicalPlan, Response}; use crate::control::planner::calvin::abort_error::calvin_abort_error; use crate::control::state::{CalvinApplyResult, SharedState}; @@ -41,6 +44,87 @@ pub(super) fn synthetic_returning_response(payload_bytes: Vec) -> Response { } } +/// Raise `tx_class`'s metadata floor to this node's metadata floor, which +/// covers the batch being applied now (see `AppliedIndexWatcher::floor`). +/// +/// The coordinator stamps the catalog it planned against. The leader raises +/// it again, which only makes replicas wait longer. Each replica's scheduler +/// holds the transaction until its own metadata apply reached the floor. +pub(crate) fn raise_metadata_floor(state: &SharedState, tx_class: &mut TxClass) { + let floor = state + .applied_index_watcher(nodedb_cluster::METADATA_GROUP_ID) + .floor(); + tx_class.metadata_floor = tx_class.metadata_floor.max(floor); +} + +/// Stamp the incarnation this node's catalog holds for every user collection +/// `tx_class`'s plans name. The coordinator stamps first, against the catalog +/// it planned with. A forwarded submit arrives stamped and keeps its stamps. +pub(crate) fn stamp_incarnations(state: &SharedState, tx_class: &mut TxClass) -> crate::Result<()> { + // A split class was stamped by its coordinator before the split, and no + // longer carries its plans in `plans`. + if !tx_class.incarnations.is_empty() || tx_class.is_multi_part() { + return Ok(()); + } + let plans = + nodedb_physical::physical_plan::wire::decode_batch(&tx_class.plans).map_err(|e| { + Error::Serialization { + format: "msgpack".into(), + detail: format!("calvin incarnation stamp: plan decode: {e}"), + } + })?; + let mut named = std::collections::BTreeSet::new(); + for plan in &plans { + // Array cell writes route by the array's own incarnation. + if matches!(plan, PhysicalPlan::Array(_) | PhysicalPlan::ClusterArray(_)) { + continue; + } + named.extend(plan.named_collections().into_iter().map(str::to_owned)); + } + let catalog = state.credentials.catalog(); + let tenant_id = tx_class.tenant_id.as_u64(); + let mut incarnations = Vec::with_capacity(named.len()); + for collection in named { + let incarnation = catalog.incarnation_of(tx_class.database_id, tenant_id, &collection)?; + incarnations.push(nodedb_cluster::calvin::types::CalvinIncarnation { + collection, + incarnation, + }); + } + tx_class.set_incarnations(incarnations); + Ok(()) +} + +/// Await the sequencer assignment of submission `inbox_seq`, bounded by +/// `timeout`. +/// +/// A closed channel means the sequencer rejected or discarded the submission +/// without sequencing it: a read/write cycle in its epoch, a leadership +/// change, or a halted sequencer. Nothing applied, so the caller gets a +/// retryable refusal. On timeout the registration is dropped, so no sender +/// stays behind. A timed-out submission can still be sequenced later. +pub(crate) async fn await_assignment( + registry: &CalvinCompletionRegistry, + inbox_seq: u64, + assignment_rx: AssignmentReceiver, + timeout: Duration, +) -> crate::Result { + match tokio::time::timeout(timeout, assignment_rx).await { + Ok(Ok(assignment)) => Ok(assignment), + Ok(Err(_)) => Err(Error::RetryableRefusal { + reason: "the Calvin sequencer did not sequence the transaction; nothing was \ + applied" + .to_owned(), + }), + Err(_) => { + registry.drop_assignment(inbox_seq); + Err(Error::Internal { + detail: "timed out waiting for Calvin sequencer assignment".to_owned(), + }) + } + } +} + /// Submit `tx_class` to THIS node's Calvin sequencer inbox and await completion. /// /// PRECONDITION: this node is the sequencer-group leader (its service assigns; @@ -63,10 +147,29 @@ pub async fn submit_and_await_calvin( /// bounded by the coordinator's remaining deadline rather than this node's full /// default deadline. pub async fn submit_and_await_calvin_with_timeout( + state: &SharedState, + mut tx_class: TxClass, + timeout: Duration, +) -> crate::Result> { + raise_metadata_floor(state, &mut tx_class); + stamp_incarnations(state, &mut tx_class)?; + let stream = super::parts::split_into_parts(state, &mut tx_class)?; + submit_prepared_and_await(state, tx_class, stream, timeout).await +} + +/// Submit a stamped `tx_class` to this node's sequencer, stream its parts +/// when it carries them as parts, and await its completion. +/// +/// PRECONDITION: this node is the sequencer-group leader. +pub(crate) async fn submit_prepared_and_await( state: &SharedState, tx_class: TxClass, + stream: Option, timeout: Duration, ) -> crate::Result> { + #[cfg(feature = "failpoints")] + crate::control::fail_gate::after_calvin_stamp(&tx_class).await; + let fold = ReplyFold::of_tx_class(&tx_class)?; let inbox = state .sequencer_inbox .get() @@ -77,38 +180,41 @@ pub async fn submit_and_await_calvin_with_timeout( .ok_or(Error::SequencerUnavailable)?; // A write to a permission-tree source is acknowledged only once it binds - // every node. Tree sources live in the default database. - let binds_authorization = tx_class.database_id == crate::types::DatabaseId::DEFAULT - && tx_class - .write_set - .participating_vshards_in_database(tx_class.database_id) - .map_err(|e| Error::BadRequest { - detail: format!("Calvin transaction write set: {e}"), - })? - .iter() - .any(|vshard| { - state - .authorization_fence - .sources() - .is_source_vshard(vshard.as_u32()) - }); - - let inbox_seq = inbox.submit(tx_class).map_err(|e| Error::BadRequest { - detail: format!("Calvin sequencer rejected transaction: {e}"), - })?; - - let assignment_rx = registry.register_submission(inbox_seq); - let (epoch, position, participants) = tokio::time::timeout(timeout, assignment_rx) - .await - .map_err(|_| Error::Internal { - detail: "timed out waiting for Calvin sequencer assignment".to_owned(), + // every node. + let binds_authorization = tx_class + .write_set + .participating_vshards_in_database(tx_class.database_id) + .map_err(|e| Error::BadRequest { + detail: format!("Calvin transaction write set: {e}"), })? - .map_err(|_| Error::Internal { - detail: "Calvin sequencer assignment channel closed".to_owned(), - })?; + .iter() + .any(|vshard| { + state + .authorization_fence + .sources() + .is_source_vshard(vshard.as_u32()) + }); + + let (inbox_seq, assignment_rx) = + inbox + .submit_with(tx_class, registry) + .map_err(|e| Error::BadRequest { + detail: format!("Calvin sequencer rejected transaction: {e}"), + })?; + let (epoch, position, participants) = + await_assignment(registry, inbox_seq, assignment_rx, timeout).await?; + // The header holds its locks on every participant until the parts + // arrive. A lost stream aborts the transaction, and the completion + // below reports it. + if let Some(stream) = &stream { + stream_parts(state, StreamTarget::Local, stream, true) + .await + .into_result()?; + } - let completion_rx = registry.register_completion(TxnId::new(epoch, position), participants); - let outcome = tokio::time::timeout(timeout, completion_rx) + let completion_rx = + registry.register_completion_report(TxnId::new(epoch, position), participants); + let report = tokio::time::timeout(timeout, completion_rx) .await .map_err(|_| { let err = Error::Internal { @@ -130,9 +236,10 @@ pub async fn submit_and_await_calvin_with_timeout( .map_err(|_| Error::Internal { detail: "Calvin completion channel closed".to_owned(), })?; + let outcome = report.outcome; // Terminal, NON-retryable: the scheduler rejected the transaction's local // plan routing and broadcast `TxnRoutingFailed`. Surface it immediately — - // falling through to the RETURNING-drain below would silently report + // falling through to the RETURNING-drain below will silently report // `Ok(None)` for a transaction that never applied. if let AttemptOutcome::Failed { detail } = &outcome { return Err(Error::Internal { @@ -141,7 +248,7 @@ pub async fn submit_and_await_calvin_with_timeout( } // Terminal, NON-retryable: the global cross-shard verdict was ABORT and the // writes were dropped. This is a fall-through chain, NOT a match — without - // this explicit check `Aborted` would fall through to the RETURNING drain + // this explicit check `Aborted` will fall through to the RETURNING drain // below and silently return `Ok(None)`, reporting COMMIT SUCCESS for a // transaction that never applied. The verdict's reason picks the error the // client retries on. @@ -172,19 +279,428 @@ pub async fn submit_and_await_calvin_with_timeout( let drained = state .calvin .apply_results - .lock() - .unwrap_or_else(|p| p.into_inner()) - .remove(&TxnId::new(epoch, position)); - match drained { - Some(CalvinApplyResult::Single { response, .. }) => { + .take(&TxnId::new(epoch, position)); + let applied = match drained { + Some(CalvinApplyResult::Single { + response, + has_returning, + }) => { // An installed txn whose reply failed to render deposits it as an // error for the statement. crate::control::local_dispatch::reject_data_plane_error(&response)?; - Ok(Some(response)) + Some((response, has_returning)) + } + Some(CalvinApplyResult::Conflict) => { + return Err(Error::Internal { + detail: "multi-participant cross-shard RETURNING not supported".to_owned(), + }); } - Some(CalvinApplyResult::Conflict) => Err(Error::Internal { + None => None, + }; + with_reported_results(applied, &report.ack_results, fold) +} + +/// The applied answer the coordinator hands back, from what every +/// participant's `CompletionAck` reported. +/// +/// The sidecar holds the applies of the participants this node hosts +/// replicas of. The acks reach this node from every participant: +/// - their timeseries install counts replace the sidecar's plain answer, +/// and stand alone when no local participant deposited one; +/// - a participant's `RETURNING` rows answer the statement when no local +/// participant deposited them. Rows past the result limit fail the +/// statement with the error a local `RETURNING` over the limit gives, and +/// two participants with rows are a cross-shard `RETURNING` union, which +/// is unsupported; +/// - a primary-write participant's plain answer, with its affected count, +/// answers the statement when no local participant deposited one. The +/// first one reported stands, as the first deposit does in the sidecar. +/// +/// So a coordinator that hosts none of the participants answers with the +/// same result as one that hosts them all. +/// +/// Under [`ReplyFold::SumOwnedEdges`] the answer is the sum of every home's +/// owned-edge count instead. +pub(crate) fn with_reported_results( + applied: Option<(Response, bool)>, + ack_results: &[Vec], + fold: ReplyFold, +) -> crate::Result> { + use crate::control::state::{AckReply, AckReturning, CalvinAckResult}; + use crate::engine::timeseries::install_counts::merge_count_payloads; + let reports: Vec = ack_results + .iter() + .filter_map(|bytes| CalvinAckResult::from_bytes(bytes)) + .collect(); + if fold == ReplyFold::SumOwnedEdges { + return sum_owned_edges(&reports); + } + let counts = reports + .iter() + .filter(|report| !report.counts.is_empty()) + .fold(None::>, |held, report| match held { + None => merge_count_payloads(&report.counts, &[]), + Some(held) => merge_count_payloads(&held, &report.counts), + }); + let mut returning = reports + .iter() + .filter_map(|report| report.returning.as_ref()); + let reported = returning.next(); + if returning.next().is_some() { + return Err(Error::Internal { detail: "multi-participant cross-shard RETURNING not supported".to_owned(), + }); + } + // A local replica's sidecar rows answer the statement, and the local + // response path enforces the result limit on them. + let local_rows = matches!(applied, Some((_, true))); + let rows = match reported { + Some(_) if local_rows => None, + None => None, + Some(AckReturning::Rows { rows }) => Some(rows.clone()), + Some(AckReturning::OverLimit { bytes, limit }) => { + return Err(Error::ExecutionLimitExceeded { + detail: format!( + "query result exceeded max_query_result_bytes ({bytes} > {limit} bytes)" + ), + }); + } + Some(AckReturning::Failed { detail }) => { + return Err(Error::Internal { + detail: format!("calvin RETURNING: {detail}"), + }); + } + }; + Ok(match (applied, rows, counts) { + (Some((response, true)), _, _) => Some(response), + (_, Some(rows), _) => Some(synthetic_returning_response(rows)), + (Some((response, false)), None, Some(counts)) => Some(Response { + payload: crate::bridge::envelope::Payload::from_vec(counts), + ..response }), - None => Ok(None), + (Some((response, false)), None, None) => Some(response), + (None, None, Some(counts)) => Some(synthetic_returning_response(counts)), + (None, None, None) => match reports.iter().find_map(|report| report.reply.as_ref()) { + Some(AckReply::Payload { payload }) => { + Some(synthetic_returning_response(payload.clone())) + } + Some(AckReply::Failed { detail }) => { + return Err(Error::Internal { + detail: format!("calvin reply: {detail}"), + }); + } + None => None, + }, + }) +} + +/// How the coordinator folds its participants' answers into the +/// statement's answer. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum ReplyFold { + /// A participant that holds the statement's own write answers it. The + /// first answer stands. + First, + /// The transaction writes edges and nothing of its own. Each home + /// counts only the edges it owns, so the sum of every home's count is + /// the statement's count, each edge counted once. + SumOwnedEdges, +} + +impl ReplyFold { + /// The fold of a transaction over `plans`. + pub(crate) fn of_plans<'a>(plans: impl IntoIterator + Clone) -> Self { + let writes_edges = plans + .clone() + .into_iter() + .any(crate::control::planner::calvin::is_edge_write); + if writes_edges + && !crate::control::planner::calvin::write_class::plans_have_user_write(plans) + { + Self::SumOwnedEdges + } else { + Self::First + } + } + + /// The fold of `tx_class`. A class split into parts reads its manifest, + /// and one with no edge key needs no decode. + pub(crate) fn of_tx_class(tx_class: &TxClass) -> crate::Result { + if !super::edge_slices::writes_edges(tx_class) { + return Ok(Self::First); + } + if let Some(manifest) = &tx_class.multi_part { + return Ok(if manifest.user_write { + Self::First + } else { + Self::SumOwnedEdges + }); + } + let plans = + nodedb_physical::physical_plan::wire::decode_batch(&tx_class.plans).map_err(|e| { + Error::Serialization { + format: "msgpack".into(), + detail: format!("calvin reply fold: plan decode: {e}"), + } + })?; + Ok(Self::of_plans(plans.iter())) + } +} + +/// The sum of every home's owned-edge count, as the statement's answer. +fn sum_owned_edges( + reports: &[crate::control::state::CalvinAckResult], +) -> crate::Result> { + use crate::control::server::shared::sql::staging_predicates::extract_affected_count; + use crate::control::state::AckReply; + let mut total: Option = None; + for report in reports { + match &report.reply { + Some(AckReply::Payload { payload }) => { + let count = extract_affected_count(payload).ok_or_else(|| Error::Internal { + detail: "calvin edge write: a home's answer carries no affected count" + .to_owned(), + })?; + total = Some(total.unwrap_or(0).saturating_add(count)); + } + Some(AckReply::Failed { detail }) => { + return Err(Error::Internal { + detail: format!("calvin reply: {detail}"), + }); + } + None => {} + } + } + let Some(total) = total else { + return Ok(None); + }; + let count = usize::try_from(total).map_err(|_| Error::Internal { + detail: format!("calvin edge write: a count of {total} edges does not fit"), + })?; + let payload = crate::data::executor::response_codec::encode_count("affected", count)?; + Ok(Some(synthetic_returning_response(payload))) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::state::{AckReturning, CalvinAckResult}; + use crate::engine::timeseries::install_counts::{TsInstallCount, TsInstallCounts}; + + fn counts(collection: &str, accepted: u64, rejected: u64) -> Vec { + TsInstallCounts::new(vec![TsInstallCount { + collection: collection.into(), + accepted, + rejected, + }]) + .to_bytes() + .expect("encode install counts") + } + + fn ack(counts: Vec, returning: Option) -> Vec { + CalvinAckResult { + counts, + returning, + reply: None, + } + .to_bytes() + .expect("encode ack result") + } + + fn replied(affected: usize) -> Vec { + let payload = crate::data::executor::response_codec::encode_count("affected", affected) + .expect("encode count"); + CalvinAckResult { + counts: Vec::new(), + returning: None, + reply: Some(crate::control::state::AckReply::Payload { payload }), + } + .to_bytes() + .expect("encode ack result") + } + + /// A Calvin write whose coordinator hosts no participant vShard answers + /// with the affected count its primary participant reported in its ack. + /// A local deposit still stands over a reported answer. + #[test] + fn a_coordinator_with_no_participant_answers_the_reported_count() { + use crate::control::server::shared::sql::staging_predicates::extract_affected_count; + let answer = + with_reported_results(None, &[ack(Vec::new(), None), replied(3)], ReplyFold::First) + .expect("a report") + .expect("an answer"); + assert_eq!(extract_affected_count(answer.payload.as_bytes()), Some(3)); + + let local_payload = crate::data::executor::response_codec::encode_count("affected", 5) + .expect("encode count"); + let local = synthetic_returning_response(local_payload); + let answer = with_reported_results(Some((local, false)), &[replied(3)], ReplyFold::First) + .expect("a report") + .expect("an answer"); + assert_eq!(extract_affected_count(answer.payload.as_bytes()), Some(5)); + } + + /// An edge write with no write of its own sums every home's owned + /// count, whatever a local home deposited: each edge counts once, at + /// its owner. + #[test] + fn an_edge_only_write_sums_its_homes_owned_counts() { + use crate::control::server::shared::sql::staging_predicates::extract_affected_count; + let local = synthetic_returning_response( + crate::data::executor::response_codec::encode_count("affected", 2) + .expect("encode count"), + ); + let answer = with_reported_results( + Some((local, false)), + &[replied(2), replied(0), replied(5)], + ReplyFold::SumOwnedEdges, + ) + .expect("a report") + .expect("an answer"); + assert_eq!(extract_affected_count(answer.payload.as_bytes()), Some(7)); + } + + fn counted(collection: &str, accepted: u64, rejected: u64) -> Vec { + ack(counts(collection, accepted, rejected), None) + } + + fn decoded(response: &Response) -> TsInstallCounts { + TsInstallCounts::from_payload(response.payload.as_bytes()).expect("install counts") + } + + /// The sequencer dropped the assignment of a submission it did not + /// sequence. The caller fails at once with a retryable refusal, long + /// before its timeout. + #[tokio::test] + async fn an_unsequenced_submission_is_a_retryable_refusal() { + let registry = CalvinCompletionRegistry::new_detached(); + let rx = registry.register_submission(7); + registry.drop_assignment(7); + match await_assignment(®istry, 7, rx, Duration::from_secs(3600)).await { + Err(Error::RetryableRefusal { .. }) => {} + other => panic!("expected a retryable refusal, got {other:?}"), + } + } + + #[tokio::test] + async fn an_assigned_submission_returns_its_assignment() { + let registry = CalvinCompletionRegistry::new_detached(); + let rx = registry.register_submission(3); + registry.note_assigned(3, TxnId::new(5, 1), 2); + let assignment = await_assignment(®istry, 3, rx, Duration::from_secs(3600)) + .await + .expect("assigned"); + assert_eq!(assignment, (5, 1, 2)); + } + + /// A wait that times out reports the timeout, not a refusal: the + /// submission can still be sequenced. + #[tokio::test] + async fn a_timed_out_wait_is_not_a_refusal() { + let registry = CalvinCompletionRegistry::new_detached(); + let rx = registry.register_submission(8); + match await_assignment(®istry, 8, rx, Duration::from_millis(1)).await { + Err(Error::Internal { detail }) => assert!(detail.contains("timed out"), "{detail}"), + other => panic!("expected the timeout error, got {other:?}"), + } + } + + /// With one replica per vShard, the coordinator's node hosts no replica of + /// either participant, so its sidecar holds nothing. The counts both + /// participants' acks carried still reach the statement. + #[test] + fn remote_participants_report_their_counts_through_their_acks() { + let answer = with_reported_results( + None, + &[counted("cpu", 2, 1), counted("mem", 1, 3)], + ReplyFold::First, + ) + .expect("a report") + .expect("an answer"); + let reported = decoded(&answer); + assert_eq!((reported.accepted, reported.rejected), (3, 4)); + assert_eq!(reported.by_collection().get("mem"), Some(&(1, 3))); + } + + /// A local participant's sidecar answer takes every participant's + /// counts. A `RETURNING` answer keeps its rows. A write whose acks carry + /// no counts keeps its answer, or has none. + #[test] + fn acked_counts_replace_a_plain_answer_and_keep_rows() { + let local = synthetic_returning_response(counts("cpu", 2, 0)); + let answer = with_reported_results( + Some((local, false)), + &[counted("cpu", 2, 0), counted("mem", 0, 2)], + ReplyFold::First, + ) + .expect("a report") + .expect("an answer"); + assert_eq!(decoded(&answer).rejected, 2); + + let rows = synthetic_returning_response(vec![0x90]); + let reported = ack( + counts("cpu", 1, 1), + Some(AckReturning::Rows { rows: vec![0x90] }), + ); + let answer = with_reported_results(Some((rows, true)), &[reported], ReplyFold::First) + .expect("a report") + .expect("an answer"); + assert_eq!(answer.payload.as_bytes(), &[0x90]); + + assert!( + with_reported_results(None, &[], ReplyFold::First) + .expect("a report") + .is_none() + ); + } + + /// A participant on another node answers a `RETURNING` statement with + /// the rows its ack carried, next to a plain participant's counts. + #[test] + fn remote_rows_answer_the_statement() { + let rows = vec![0x91, 0x01]; + let answer = with_reported_results( + Some((synthetic_returning_response(Vec::new()), false)), + &[ + ack( + counts("cpu", 1, 1), + Some(AckReturning::Rows { rows: rows.clone() }), + ), + counted("mem", 2, 0), + ], + ReplyFold::First, + ) + .expect("a report") + .expect("an answer"); + assert_eq!(answer.payload.as_bytes(), rows.as_slice()); + } + + /// Remote rows past the limit fail the statement with the error a local + /// `RETURNING` over the limit gives. Two participants with rows fail it + /// as an unsupported cross-shard union. + #[test] + fn remote_rows_past_the_limit_or_from_two_participants_fail() { + let over = ack( + Vec::new(), + Some(AckReturning::OverLimit { + bytes: 64, + limit: 16, + }), + ); + match with_reported_results(None, &[over], ReplyFold::First) { + Err(Error::ExecutionLimitExceeded { detail }) => assert_eq!( + detail, + "query result exceeded max_query_result_bytes (64 > 16 bytes)" + ), + other => panic!("expected the result limit error, got {other:?}"), + } + + let one = ack(Vec::new(), Some(AckReturning::Rows { rows: vec![0x90] })); + match with_reported_results(None, &[one.clone(), one], ReplyFold::First) { + Err(Error::Internal { detail }) => { + assert!(detail.contains("cross-shard RETURNING"), "{detail}") + } + other => panic!("expected the cross-shard error, got {other:?}"), + } } } diff --git a/nodedb/src/control/planner/calvin/submit/mod.rs b/nodedb/src/control/planner/calvin/submit/mod.rs index 134edff75..1deb2959d 100644 --- a/nodedb/src/control/planner/calvin/submit/mod.rs +++ b/nodedb/src/control/planner/calvin/submit/mod.rs @@ -31,10 +31,13 @@ //! schedulers; this module never does storage I/O or io_uring directly. pub mod assign; +pub mod edge_slices; pub mod local; +pub mod parts; pub mod routed; +pub mod stream; pub(crate) use assign::submit_local_assign; pub use assign::{RoutedAssignment, submit_calvin_routed_assign}; pub use local::{submit_and_await_calvin, submit_and_await_calvin_with_timeout}; -pub use routed::submit_calvin_routed; +pub use routed::{submit_calvin_routed, submit_calvin_routed_write}; diff --git a/nodedb/src/control/planner/calvin/submit/parts.rs b/nodedb/src/control/planner/calvin/submit/parts.rs new file mode 100644 index 000000000..45b851be0 --- /dev/null +++ b/nodedb/src/control/planner/calvin/submit/parts.rs @@ -0,0 +1,460 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Split a Calvin transaction's plans into sequencer parts. +//! +//! One sequencer entry carries at most `max_plans_bytes_per_txn` plan bytes +//! for at most `max_participating_vshards_per_txn` vShards. A transaction +//! over either travels as a multi-part transaction instead: a header with its +//! full read and write sets, and parts the coordinator streams to the +//! sequencer leader after it. The transaction stays one transaction: one +//! sequence position, every lock from the header to the last part's apply, +//! one verdict. No cap bounds its size. +//! +//! - A part holds consecutive whole tasks, within both caps. +//! - A task whose encoding alone is over the byte cap travels as a run of +//! chunk parts, each one byte range of it and nothing else. +//! - A part targets the vShards its tasks route to, by the routing every +//! participant's scheduler applies, so a participant receives exactly the +//! parts that hold its tasks. + +use std::collections::{BTreeMap, BTreeSet}; +use std::sync::LazyLock; +use std::sync::atomic::{AtomicU64, Ordering}; + +use nodedb_cluster::calvin::sequencer::config::SequencerConfig; +use nodedb_cluster::calvin::types::{ + MultiPartPlans, PartStreamId, PlanPart, StreamedPart, TaskChunk, TxClass, VShardParts, +}; +use nodedb_physical::physical_plan::PhysicalPlan; + +use super::stream::PartStream; +use crate::Error; +use crate::control::cluster::calvin::scheduler::driver::core::routing::{ + PlanRouting, plan_vshard_in_database, +}; +use crate::control::state::SharedState; + +/// Bytes a msgpack array header adds to a part's encoded tasks, at most. +const ARRAY_HEADER_BYTES: usize = 5; + +/// The next stream sequence of this process. It starts at the wall clock in +/// nanoseconds, so a restarted coordinator never reuses a name its earlier +/// process streamed under. +static NEXT_STREAM_SEQ: LazyLock = LazyLock::new(|| { + // no-determinism: names a coordinator-local stream, never in the log. + let now = std::time::SystemTime::now() + .duration_since(std::time::UNIX_EPOCH) + .map(|elapsed| u64::try_from(elapsed.as_nanos()).unwrap_or(u64::MAX)) + .unwrap_or(0); + AtomicU64::new(now) +}); + +/// Carry `tx_class`'s plans as parts when one sequencer entry cannot, and +/// return the parts to stream after the header. Edge batches are first +/// split by home pair. A class that then fits one entry, or that is already +/// split into parts, streams nothing. +pub(crate) fn split_into_parts( + state: &SharedState, + tx_class: &mut TxClass, +) -> crate::Result> { + if tx_class.is_multi_part() { + return Ok(None); + } + // A batch edge op spans the homes of all its edges. Split into one batch + // per home pair, it spans at most two, which every part can carry. + super::edge_slices::slice_edge_batches(tx_class)?; + let limits = SequencerConfig::default(); + if tx_class.plans.len() <= limits.max_plans_bytes_per_txn + && tx_class.participating_vshards().len() <= limits.max_participating_vshards_per_txn + { + return Ok(None); + } + let plans = + nodedb_physical::physical_plan::wire::decode_batch(&tx_class.plans).map_err(|e| { + Error::Serialization { + format: "msgpack".into(), + detail: format!("calvin part split: plan decode: {e}"), + } + })?; + let (mut manifest, parts) = plan_parts( + &plans, + tx_class.database_id, + &tx_class.body_plans, + PartLimits { + max_bytes: limits.max_plans_bytes_per_txn, + max_targets: limits.max_participating_vshards_per_txn, + }, + )?; + let id = PartStreamId { + node: state.node_id, + seq: NEXT_STREAM_SEQ.fetch_add(1, Ordering::Relaxed), + }; + manifest.stream = id; + tx_class.plans = Vec::new(); + tx_class.multi_part = Some(manifest); + Ok(Some(PartStream { id, parts })) +} + +/// The caps [`plan_parts`] packs within. +#[derive(Debug, Clone, Copy)] +pub(crate) struct PartLimits { + pub max_bytes: usize, + pub max_targets: usize, +} + +/// Pack `plans` into parts, each within `limits`, in task order, and the +/// manifest that names them. `body_plans` are the indexes of the tasks a +/// trigger body buffered. The manifest's stream is left for the caller. +pub(crate) fn plan_parts( + plans: &[PhysicalPlan], + database_id: crate::types::DatabaseId, + body_plans: &[u32], + limits: PartLimits, +) -> crate::Result<(MultiPartPlans, Vec)> { + let total_tasks = u32::try_from(plans.len()).map_err(|_| Error::BadRequest { + detail: format!( + "a transaction of {} tasks is more than one transaction indexes", + plans.len() + ), + })?; + let mut packer = Packer::new(limits); + for (task, plan) in (0u32..).zip(plans) { + let homes = task_homes(plan, database_id, task)?; + let encoded = zerompk::to_msgpack_vec(plan).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("calvin part split: task {task} encode: {e}"), + })?; + packer.add(task, plan, homes, encoded)?; + } + let parts = packer.finish()?; + let part_count = u32::try_from(parts.len()).map_err(|_| Error::BadRequest { + detail: format!( + "a transaction of {} parts is more than one transaction indexes", + parts.len() + ), + })?; + let mut per_vshard: BTreeMap = BTreeMap::new(); + for part in &parts { + for target in &part.targets { + *per_vshard.entry(*target).or_insert(0) += 1; + } + } + let manifest = MultiPartPlans { + stream: PartStreamId::default(), + part_count, + total_tasks, + user_write: crate::control::planner::calvin::write_class::plans_have_user_write(plans), + client_write: (0..total_tasks).any(|task| !body_plans.contains(&task)), + per_vshard: per_vshard + .into_iter() + .map(|(vshard, parts)| VShardParts { vshard, parts }) + .collect(), + }; + Ok((manifest, parts)) +} + +/// The vShards task `task` routes to, sorted. +fn task_homes( + plan: &PhysicalPlan, + database_id: crate::types::DatabaseId, + task: u32, +) -> crate::Result> { + match plan_vshard_in_database(plan, database_id) { + PlanRouting::Vshards(vshards) => Ok(vshards.iter().map(|v| v.as_u32()).collect()), + PlanRouting::ControlPlaneOnly => Err(unroutable(task, "a control-plane-only plan")), + PlanRouting::NotAWrite => Err(unroutable(task, "not a write")), + PlanRouting::Unroutable(reason) => Err(unroutable(task, reason)), + } +} + +fn unroutable(task: u32, reason: &str) -> Error { + Error::Internal { + detail: format!("calvin part split: task {task} does not route to a vShard: {reason}"), + } +} + +/// Packs consecutive tasks into parts. +struct Packer<'a> { + limits: PartLimits, + parts: Vec, + first_task: u32, + tasks: Vec<&'a PhysicalPlan>, + homes: BTreeSet, + bytes: usize, +} + +impl<'a> Packer<'a> { + fn new(limits: PartLimits) -> Self { + Self { + limits, + parts: Vec::new(), + first_task: 0, + tasks: Vec::new(), + homes: BTreeSet::new(), + bytes: ARRAY_HEADER_BYTES, + } + } + + fn add( + &mut self, + task: u32, + plan: &'a PhysicalPlan, + homes: BTreeSet, + encoded: Vec, + ) -> crate::Result<()> { + // A task routes to its one home, or to the two homes of an edge. + if homes.len() > self.limits.max_targets { + return Err(Error::Internal { + detail: format!( + "calvin part split: task {task} routes to {} vShards, more than one entry \ + targets", + homes.len() + ), + }); + } + if encoded.len().saturating_add(ARRAY_HEADER_BYTES) > self.limits.max_bytes { + if !self.tasks.is_empty() { + self.close()?; + } + self.push_chunks(task, &homes, &encoded); + return Ok(()); + } + let widened = self.homes.union(&homes).count(); + if !self.tasks.is_empty() + && (self.bytes.saturating_add(encoded.len()) > self.limits.max_bytes + || widened > self.limits.max_targets) + { + self.close()?; + } + if self.tasks.is_empty() { + self.first_task = task; + } + self.tasks.push(plan); + self.homes.extend(homes); + self.bytes = self.bytes.saturating_add(encoded.len()); + Ok(()) + } + + /// Push task `task`'s encoding as a run of chunk parts. + fn push_chunks(&mut self, task: u32, homes: &BTreeSet, encoded: &[u8]) { + let total_len = encoded.len() as u64; + let mut offset = 0u64; + for bytes in encoded.chunks(self.limits.max_bytes) { + self.push( + homes.iter().copied().collect(), + PlanPart { + first_task: task, + plans: bytes.to_vec(), + chunk: Some(TaskChunk { offset, total_len }), + }, + ); + offset += bytes.len() as u64; + } + } + + fn push(&mut self, targets: Vec, part: PlanPart) { + let index = u32::try_from(self.parts.len()).unwrap_or(u32::MAX); + self.parts.push(StreamedPart { + index, + targets, + part, + }); + } + + /// Close the open part. + fn close(&mut self) -> crate::Result<()> { + let plans = zerompk::to_msgpack_vec(&self.tasks).map_err(|e| Error::Serialization { + format: "msgpack".into(), + detail: format!("calvin part split: part encode: {e}"), + })?; + let targets = std::mem::take(&mut self.homes).into_iter().collect(); + self.push( + targets, + PlanPart { + first_task: self.first_task, + plans, + chunk: None, + }, + ); + self.tasks.clear(); + self.bytes = ARRAY_HEADER_BYTES; + Ok(()) + } + + fn finish(mut self) -> crate::Result> { + if !self.tasks.is_empty() { + self.close()?; + } + if self.parts.len() > u32::MAX as usize { + return Err(Error::BadRequest { + detail: "a transaction of more parts than one transaction indexes".to_owned(), + }); + } + Ok(self.parts) + } +} + +#[cfg(test)] +mod tests { + use nodedb_physical::physical_plan::DocumentOp; + use nodedb_physical::physical_plan::wire as plan_wire; + use nodedb_types::QualifiedCollection; + + use super::*; + use crate::types::DatabaseId; + + const LIMITS: PartLimits = PartLimits { + max_bytes: 1 << 20, + max_targets: 64, + }; + + fn truncate(collection: &str) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::Truncate { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, collection), + restart_identity: false, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }) + } + + fn homes(plan: &PhysicalPlan) -> BTreeSet { + task_homes(plan, DatabaseId::DEFAULT, 0).expect("routes") + } + + /// Every part is within both caps, the parts carry every task in order, + /// each part targets exactly the homes of its tasks, and the manifest + /// counts the parts of each vShard. + fn assert_parts_cover( + plans: &[PhysicalPlan], + manifest: &MultiPartPlans, + parts: &[StreamedPart], + ) { + assert_eq!(manifest.part_count as usize, parts.len()); + let mut decoded_tasks: Vec = Vec::new(); + let mut spanning: Vec = Vec::new(); + let mut counts: BTreeMap = BTreeMap::new(); + for (index, streamed) in (0u32..).zip(parts) { + assert_eq!(streamed.index, index); + let part = &streamed.part; + assert!( + part.plans.len() <= LIMITS.max_bytes, + "part over the byte cap" + ); + assert!( + streamed.targets.len() <= LIMITS.max_targets, + "over the target cap" + ); + assert!( + streamed.targets.windows(2).all(|w| w[0] < w[1]), + "targets sorted" + ); + for target in &streamed.targets { + *counts.entry(*target).or_insert(0) += 1; + } + let first = part.first_task as usize; + match part.chunk { + None => { + assert_eq!(first, decoded_tasks.len(), "parts are consecutive"); + let tasks = plan_wire::decode_batch(&part.plans).expect("part decodes"); + let want: BTreeSet = tasks.iter().flat_map(homes).collect(); + assert_eq!( + streamed.targets.iter().copied().collect::>(), + want + ); + decoded_tasks.extend(tasks); + } + Some(chunk) => { + assert_eq!( + chunk.offset as usize, + spanning.len(), + "chunks are contiguous" + ); + spanning.extend_from_slice(&part.plans); + if spanning.len() as u64 == chunk.total_len { + assert_eq!(first, decoded_tasks.len(), "the task is next"); + let task = plan_wire::decode(&spanning).expect("task decodes"); + assert_eq!( + streamed.targets, + homes(&task).into_iter().collect::>() + ); + decoded_tasks.push(task); + spanning.clear(); + } + } + } + } + assert!(spanning.is_empty(), "no task is left part way"); + assert_eq!(decoded_tasks.as_slice(), plans, "every task, in order"); + let manifest_counts: BTreeMap = manifest + .per_vshard + .iter() + .map(|entry| (entry.vshard, entry.parts)) + .collect(); + assert_eq!(manifest_counts, counts); + } + + #[test] + fn tasks_over_many_vshards_split_at_the_target_cap() { + let plans: Vec = (0..400).map(|i| truncate(&format!("c_{i}"))).collect(); + let distinct: BTreeSet = plans.iter().flat_map(homes).collect(); + assert!( + distinct.len() > LIMITS.max_targets, + "the tasks span over one entry" + ); + + let (manifest, parts) = + plan_parts(&plans, DatabaseId::DEFAULT, &[], LIMITS).expect("splits"); + assert!(parts.len() >= 2); + assert_parts_cover(&plans, &manifest, &parts); + assert!(manifest.user_write); + assert!(manifest.client_write); + } + + #[test] + fn plan_bytes_over_one_entry_split_at_the_byte_cap() { + let name = "x".repeat(10_000); + let plans: Vec = (0..300).map(|_| truncate(&name)).collect(); + let (manifest, parts) = + plan_parts(&plans, DatabaseId::DEFAULT, &[], LIMITS).expect("splits"); + assert!(parts.len() >= 3); + assert_parts_cover(&plans, &manifest, &parts); + } + + /// A single task over one entry travels as a run of chunk parts + /// between whole-task parts. Nothing refuses it. + #[test] + fn a_task_over_one_entry_is_split_across_chunk_parts() { + let plans = vec![ + truncate("before"), + truncate(&"x".repeat((5 << 20) / 2)), + truncate("after"), + ]; + let (manifest, parts) = + plan_parts(&plans, DatabaseId::DEFAULT, &[], LIMITS).expect("splits"); + let chunks = parts.iter().filter(|p| p.part.chunk.is_some()).count(); + assert!(chunks >= 3, "a 2.5 MiB task over three chunk parts"); + assert_parts_cover(&plans, &manifest, &parts); + } + + /// Plans over the 64 MiB RPC limit split, with no ceiling. + #[test] + fn plans_over_64_mib_split_with_no_ceiling() { + let name = "y".repeat(100_000); + let plans: Vec = (0..700).map(|_| truncate(&name)).collect(); + let (manifest, parts) = + plan_parts(&plans, DatabaseId::DEFAULT, &[], LIMITS).expect("splits"); + let bytes: usize = parts.iter().map(|p| p.part.plans.len()).sum(); + assert!(bytes > 64 << 20, "the plans exceed 64 MiB"); + assert_parts_cover(&plans, &manifest, &parts); + } + + /// Tasks a trigger body buffered do not make a client write. + #[test] + fn body_tasks_alone_are_not_a_client_write() { + let plans = vec![truncate("a"), truncate("b")]; + let (manifest, parts) = + plan_parts(&plans, DatabaseId::DEFAULT, &[0, 1], LIMITS).expect("packs"); + assert!(!manifest.client_write); + assert_eq!(parts.len(), 1); + } +} diff --git a/nodedb/src/control/planner/calvin/submit/routed.rs b/nodedb/src/control/planner/calvin/submit/routed.rs index ab3217423..18ba17b47 100644 --- a/nodedb/src/control/planner/calvin/submit/routed.rs +++ b/nodedb/src/control/planner/calvin/submit/routed.rs @@ -18,7 +18,8 @@ use crate::bridge::envelope::Response; use crate::control::cluster::warm_peers::register_peers_from_topology; use crate::control::state::SharedState; -use super::local::{submit_and_await_calvin, synthetic_returning_response}; +use super::local::{submit_prepared_and_await, synthetic_returning_response}; +use super::stream::{StreamTarget, stream_parts}; /// Backoff schedule (milliseconds) for waiting on the sequencer-group leader /// election before a cross-shard submit. Covers the brief post-startup window @@ -27,12 +28,33 @@ use super::local::{submit_and_await_calvin, synthetic_returning_response}; /// leaderless cluster surfaces a typed error rather than hanging. const SEQUENCER_LEADER_WAIT_BACKOFF_MS: &[u64] = &[50, 100, 200, 400, 800, 1000, 1000, 1000]; +/// Submit a Calvin write whose tx class holds a write, routed as +/// [`submit_calvin_routed`], and return its answer. +/// +/// A committed write always answers. Its primary participant deposits the +/// answer on its own replicas and reports it in its completion ack, which +/// every coordinator reads (see `with_reported_results`). So the affected +/// count reaches the coordinator wherever it runs, even on a node that hosts +/// none of the participants. +pub async fn submit_calvin_routed_write( + state: &SharedState, + tx_class: TxClass, +) -> crate::Result { + submit_calvin_routed(state, tx_class) + .await? + .ok_or_else(|| Error::Internal { + detail: "a committed Calvin write reported no answer; its primary participant \ + reports one in its completion ack" + .to_owned(), + }) +} + /// Submit a cross-shard Calvin `tx_class`, routing it to the sequencer-group /// leader so it is actually sequenced and acked. /// /// Routing logic (mirrors `assign_surrogate_routed`): -/// - **Not cluster mode** (no `cluster_transport` / `cluster_routing`): submit -/// locally — single-node IS the sequencer leader. +/// - **No `cluster_transport`**: this `SharedState` never ran `start_raft`, +/// so no sequencer runs here. Return `SequencerUnavailable`. /// - **Leader is self**: submit-and-await locally. /// - **Leader is a remote node**: register the leader's address from the live /// topology, then send one `SubmitCalvinTxnRequest` (carrying the @@ -44,15 +66,15 @@ const SEQUENCER_LEADER_WAIT_BACKOFF_MS: &[u64] = &[50, 100, 200, 400, 800, 1000, /// discarded. pub async fn submit_calvin_routed( state: &SharedState, - tx_class: TxClass, + mut tx_class: TxClass, ) -> crate::Result> { - // Not cluster mode — single-node is the only sequencer member, hence the - // leader. Submit-and-await locally. - let (Some(transport), Some(_routing)) = ( - state.cluster_transport.as_ref(), - state.cluster_routing.as_ref(), - ) else { - return submit_and_await_calvin(state, tx_class).await; + super::local::raise_metadata_floor(state, &mut tx_class); + super::local::stamp_incarnations(state, &mut tx_class)?; + let stream = super::parts::split_into_parts(state, &mut tx_class)?; + // Every running server has a cluster transport, the synthesized one-node + // cluster included. Without one, `start_raft` never ran here. + let Some(transport) = state.cluster_transport.as_ref() else { + return Err(Error::SequencerUnavailable); }; // Resolve the sequencer-group leader from THIS node's live Raft status. The @@ -94,10 +116,11 @@ pub async fn submit_calvin_routed( }); } - // Leader is self: submit-and-await locally (a self-RPC would be a pointless + // Leader is self: submit-and-await locally (a self-RPC will be a pointless // extra hop and the local registry is the one that completes). if leader == state.node_id { - return submit_and_await_calvin(state, tx_class).await; + let timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); + return submit_prepared_and_await(state, tx_class, stream, timeout).await; } // Remote leader: ensure its address is registered before dispatch, then send @@ -125,52 +148,58 @@ pub async fn submit_calvin_routed( // The leader-side handler holds this RPC open until the transaction is // sequenced AND completion-acked (up to `deadline_remaining_ms`). The generic - // short `rpc_timeout` (a normal request/response round-trip budget) would + // short `rpc_timeout` (a normal request/response round-trip budget) will // abort the call long before that, so bound the response read by the // forwarded deadline plus a margin for the round-trip itself. let read_timeout = Duration::from_millis(deadline_remaining_ms.saturating_add(2_000)); - match transport - .send_rpc_with_read_timeout(leader, RaftRpc::SubmitCalvinTxnRequest(req), read_timeout) - .await - { + let header = transport.send_rpc_with_read_timeout( + leader, + RaftRpc::SubmitCalvinTxnRequest(req), + read_timeout, + ); + // The header's reply comes only at completion, so the parts stream + // beside it. The leader opens the stream when it proposes the header; + // until then it answers `Unknown`, and the stream waits. + let reply = match &stream { + None => header.await, + Some(stream) => { + tokio::pin!(header); + let streaming = stream_parts(state, StreamTarget::Remote(leader), stream, false); + tokio::pin!(streaming); + tokio::select! { + reply = &mut header => reply, + end = &mut streaming => { + end.into_result()?; + header.await + } + } + } + }; + match reply { Ok(RaftRpc::SubmitCalvinTxnResponse(SubmitCalvinTxnResponse { error: None, payload_bytes, })) => { // The leader drained ITS local sidecar and forwarded the RETURNING // payload bytes over this non-Raft RPC response. Reconstruct a - // minimal Control-Plane Response carrying just that payload so the + // minimal Control-Plane Response carrying only that payload so the // coordinator emits DATA-ROW output; `None` for plain writes. Ok(payload_bytes.map(synthetic_returning_response)) } - // A Data-Plane verdict from the sequencer leader keeps its code, so a - // constraint violation on a routed write reaches the client as its own - // SQLSTATE instead of a generic internal error. - Ok(RaftRpc::SubmitCalvinTxnResponse(SubmitCalvinTxnResponse { - error: Some(TypedClusterError::DataPlane { code }), - .. - })) => Err(Error::DataPlane(code.into())), - // A constraint refusal on the sequencer leader keeps its kind, so a - // NOT NULL refusal on a routed write reaches the client as 23502 - // instead of collapsing into a generic internal error. + // An error with no class keeps the leader in its message. Ok(RaftRpc::SubmitCalvinTxnResponse(SubmitCalvinTxnResponse { - error: - Some(TypedClusterError::RejectedConstraint { - collection, - constraint, - detail, - }), + error: Some(TypedClusterError::Internal { code: 0, message }), .. - })) => Err(Error::RejectedConstraint { - collection, - constraint, - detail, + })) => Err(Error::Internal { + detail: format!("calvin-submit failed on sequencer leader node {leader}: {message}"), }), + // Every other error from the sequencer leader is rebuilt as the error + // a local submit returns: a Calvin abort stays a serialization + // conflict, a superseded collection stays retryable, and a Data-Plane + // or constraint verdict keeps its SQLSTATE. Ok(RaftRpc::SubmitCalvinTxnResponse(SubmitCalvinTxnResponse { error: Some(e), .. - })) => Err(Error::Internal { - detail: format!("calvin-submit failed on sequencer leader node {leader}: {e:?}"), - }), + })) => Err(Error::from(e)), Ok(other) => Err(Error::Internal { detail: format!("calvin-submit: unexpected reply from node {leader}: {other:?}"), }), diff --git a/nodedb/src/control/planner/calvin/submit/stream.rs b/nodedb/src/control/planner/calvin/submit/stream.rs new file mode 100644 index 000000000..d000a2b68 --- /dev/null +++ b/nodedb/src/control/planner/calvin/submit/stream.rs @@ -0,0 +1,380 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Stream a multi-part transaction's parts to the sequencer leader that took +//! its header. +//! +//! The coordinator sends the parts in order, in batches of at most +//! `MAX_PARTS_BATCH_BYTES`, so no request nears the 64 MiB RPC limit +//! whatever the transaction's size. The leader answers each batch with the +//! next part it owes and a status: +//! +//! - `Accepted`: go on from the next part it owes. +//! - `Full`: its queue is full. Wait, then go on from the next part it owes. +//! - `Unknown`: it holds no such stream. Before the leader took a first +//! part, the header can still lack a proposal, so the stream waits and +//! retries unless the caller already holds the assignment. After that, +//! the stream is lost: the leader changed, or the stream stalled. The +//! transaction then aborts with `PartsLost` and the caller retries it. +//! - `Rejected`: a part is malformed. The transaction aborts. +//! +//! A `Full` answer comes from a live leader that holds the stream, so it +//! keeps the stream alive. A stream with no answer from such a leader for +//! the network deadline stops as lost. + +use std::collections::BTreeSet; +use std::time::{Duration, Instant}; + +use nodedb_cluster::calvin::PartsOfferStatus; +use nodedb_cluster::calvin::types::{PartStreamId, StreamedPart}; +use nodedb_cluster::{CalvinPartsRequest, CalvinPartsResponse, MAX_PARTS_BATCH_BYTES, RaftRpc}; +use tracing::warn; + +use crate::Error; +use crate::control::cluster::warm_peers::register_peers_from_topology; +use crate::control::state::SharedState; + +/// The first wait after a `Full` or an early `Unknown`. +const FIRST_BACKOFF: Duration = Duration::from_millis(5); +/// The longest wait between two offers. +const MAX_BACKOFF: Duration = Duration::from_millis(200); + +/// A multi-part transaction's parts, in index order, and the stream that +/// carries them. +#[derive(Debug)] +pub struct PartStream { + pub id: PartStreamId, + pub parts: Vec, +} + +/// The node a stream goes to. +#[derive(Debug, Clone, Copy)] +pub(crate) enum StreamTarget { + /// This node leads the sequencer group. + Local, + /// The sequencer leader that took the header. + Remote(u64), +} + +/// How a stream ended. +#[derive(Debug, PartialEq, Eq)] +pub(crate) enum StreamEnd { + /// The leader took every part. + Done, + /// The leader no longer holds the stream. The transaction aborts with + /// `PartsLost`. + Lost, + /// The leader rejected a part. + Rejected(String), +} + +impl StreamEnd { + /// A rejected stream as the statement's error. A malformed part is a + /// coordinator bug, so a retry cannot fix it. + pub(crate) fn into_result(self) -> crate::Result<()> { + match self { + Self::Done | Self::Lost => Ok(()), + Self::Rejected(detail) => Err(Error::Internal { + detail: format!("calvin part stream rejected by the sequencer leader: {detail}"), + }), + } + } +} + +/// Stream `stream` to `target`. `assigned` says the caller holds the +/// header's assignment, so the leader has opened the stream. +pub(crate) async fn stream_parts( + state: &SharedState, + target: StreamTarget, + stream: &PartStream, + assigned: bool, +) -> StreamEnd { + if stream.parts.is_empty() { + return StreamEnd::Done; + } + let deadline = Duration::from_secs(state.tuning.network.default_deadline_secs.max(1)); + // no-determinism: coordinator-side liveness timer, never in the log. + let mut progress = StreamProgress::new(stream.parts.len(), assigned, deadline, Instant::now()); + let mut backoff = FIRST_BACKOFF; + loop { + let batch = batch_from(&stream.parts, progress.next); + let reply = match offer(state, target, stream.id, batch).await { + Ok(reply) => reply, + Err(error) => { + warn!(%error, "calvin part stream: offer failed; retrying"); + PartsOfferReply { + status: PartsOfferStatus::Unknown, + next_index: 0, + detail: None, + transport_error: true, + } + } + }; + let before = progress.next; + // no-determinism: coordinator-side liveness timer. + match progress.on_reply(&reply, Instant::now()) { + Step::End(end) => return end, + Step::Offer => backoff = FIRST_BACKOFF, + Step::Wait => { + if progress.next > before { + backoff = FIRST_BACKOFF; + } + tokio::time::sleep(backoff).await; + backoff = (backoff * 2).min(MAX_BACKOFF); + } + } + } +} + +/// What the stream does after a reply. +#[derive(Debug, PartialEq, Eq)] +enum Step { + /// Offer the next batch at once. + Offer, + /// Wait, then offer again. + Wait, + /// The stream ended. + End(StreamEnd), +} + +/// Where a stream stands, and when the leader last showed it alive. +#[derive(Debug)] +struct StreamProgress { + total: usize, + /// The next part the leader owes. + next: usize, + /// Whether the leader took a part: it opened the stream. + taken_any: bool, + /// Whether the caller holds the header's assignment. + assigned: bool, + deadline: Duration, + /// The last answer from a leader that holds the stream. + last_alive: Instant, +} + +impl StreamProgress { + fn new(total: usize, assigned: bool, deadline: Duration, now: Instant) -> Self { + Self { + total, + next: 0, + taken_any: false, + assigned, + deadline, + last_alive: now, + } + } + + /// Fold `reply`, answered at `now`, into the stream. + /// + /// An `Accepted` or `Full` answer comes from a leader that holds the + /// stream, so it counts as liveness: a full queue drains as the leader + /// proposes. Only an `Unknown` answer, or no answer, runs down the + /// deadline. A leader that stops proposing closes the stalled stream, + /// and its next answer is `Unknown`. + fn on_reply(&mut self, reply: &PartsOfferReply, now: Instant) -> Step { + let owed = usize::try_from(reply.next_index).unwrap_or(usize::MAX); + match reply.status { + PartsOfferStatus::Rejected => { + return Step::End(StreamEnd::Rejected( + reply.detail.clone().unwrap_or_default(), + )); + } + PartsOfferStatus::Unknown + if !reply.transport_error && (self.assigned || self.taken_any) => + { + return Step::End(StreamEnd::Lost); + } + PartsOfferStatus::Accepted | PartsOfferStatus::Full => { + self.taken_any = true; + self.last_alive = now; + self.next = owed.min(self.total); + } + PartsOfferStatus::Unknown => {} + } + if self.next >= self.total { + return Step::End(StreamEnd::Done); + } + if reply.status == PartsOfferStatus::Accepted { + return Step::Offer; + } + if now.saturating_duration_since(self.last_alive) > self.deadline { + return Step::End(StreamEnd::Lost); + } + Step::Wait + } +} + +/// The parts from `from` on that fit one request, at least one. +fn batch_from(parts: &[StreamedPart], from: usize) -> Vec { + let mut bytes = 0usize; + let mut batch = Vec::new(); + for part in parts.iter().skip(from) { + let size = part.part.plans.len(); + if !batch.is_empty() && bytes.saturating_add(size) > MAX_PARTS_BATCH_BYTES { + break; + } + bytes = bytes.saturating_add(size); + batch.push(part.clone()); + } + batch +} + +/// A leader's answer, or a transport error read as `Unknown`. +struct PartsOfferReply { + status: PartsOfferStatus, + next_index: u32, + detail: Option, + transport_error: bool, +} + +async fn offer( + state: &SharedState, + target: StreamTarget, + stream: PartStreamId, + batch: Vec, +) -> crate::Result { + match target { + StreamTarget::Local => { + let inbox = state + .sequencer_inbox + .get() + .ok_or(Error::SequencerUnavailable)?; + let offer = inbox.offer_parts(stream, batch); + Ok(PartsOfferReply { + status: offer.status, + next_index: offer.next_index, + detail: offer.detail, + transport_error: false, + }) + } + StreamTarget::Remote(leader) => offer_remote(state, leader, stream, batch).await, + } +} + +async fn offer_remote( + state: &SharedState, + leader: u64, + stream: PartStreamId, + batch: Vec, +) -> crate::Result { + let transport = state + .cluster_transport + .as_ref() + .ok_or(Error::SequencerUnavailable)?; + register_peers_from_topology(state, transport, &BTreeSet::from([leader])); + let parts_bytes = zerompk::to_msgpack_vec(&batch).map_err(|e| Error::Serialization { + format: "msgpack".to_owned(), + detail: format!("calvin part stream: batch encode: {e}"), + })?; + let request = RaftRpc::CalvinPartsRequest(CalvinPartsRequest { + stream_node: stream.node, + stream_seq: stream.seq, + parts_bytes, + }); + let read_timeout = Duration::from_secs(state.tuning.network.default_deadline_secs.max(1)); + let reply = transport + .send_rpc_with_read_timeout(leader, request, read_timeout) + .await + .map_err(|e| Error::Internal { + detail: format!("calvin part stream RPC to sequencer leader node {leader}: {e}"), + })?; + let RaftRpc::CalvinPartsResponse(CalvinPartsResponse { + status, + next_index, + detail, + }) = reply + else { + return Err(Error::Internal { + detail: format!("calvin part stream: unexpected reply from node {leader}"), + }); + }; + let status = PartsOfferStatus::from_wire(status).ok_or_else(|| Error::Internal { + detail: format!("calvin part stream: unknown status {status} from node {leader}"), + })?; + Ok(PartsOfferReply { + status, + next_index, + detail, + transport_error: false, + }) +} + +#[cfg(test)] +mod tests { + use nodedb_cluster::calvin::types::PlanPart; + + use super::*; + + fn part(index: u32, bytes: usize) -> StreamedPart { + StreamedPart { + index, + targets: vec![1], + part: PlanPart { + first_task: index, + plans: vec![0; bytes], + chunk: None, + }, + } + } + + #[test] + fn a_batch_stays_under_its_byte_cap_and_holds_at_least_one_part() { + let parts: Vec = (0..40).map(|i| part(i, 1 << 20)).collect(); + let batch = batch_from(&parts, 3); + assert_eq!(batch.first().map(|p| p.index), Some(3)); + let bytes: usize = batch.iter().map(|p| p.part.plans.len()).sum(); + assert!(bytes <= MAX_PARTS_BATCH_BYTES); + assert_eq!(batch.len(), MAX_PARTS_BATCH_BYTES >> 20); + + let huge = vec![part(0, MAX_PARTS_BATCH_BYTES + 1)]; + assert_eq!(batch_from(&huge, 0).len(), 1); + } + + fn reply(status: PartsOfferStatus, next_index: u32, transport_error: bool) -> PartsOfferReply { + PartsOfferReply { + status, + next_index, + detail: None, + transport_error, + } + } + + /// A leader that holds the stream and answers `Full` past the deadline + /// keeps it alive: its queue drains as it proposes. Only no answer runs + /// the deadline down. + #[test] + fn full_answers_keep_a_stream_alive_past_the_deadline() { + let deadline = Duration::from_secs(1); + // no-determinism: test-only synthetic timeline. + let start = Instant::now(); + let mut progress = StreamProgress::new(4, true, deadline, start); + assert_eq!( + progress.on_reply(&reply(PartsOfferStatus::Accepted, 1, false), start), + Step::Offer + ); + for second in 1..10 { + let now = start + Duration::from_secs(second); + assert_eq!( + progress.on_reply(&reply(PartsOfferStatus::Full, 1, false), now), + Step::Wait, + "a full queue is no stall" + ); + } + let alive = start + Duration::from_secs(9); + assert_eq!( + progress.on_reply( + &reply(PartsOfferStatus::Unknown, 0, true), + alive + Duration::from_millis(500) + ), + Step::Wait, + "one missed answer within the deadline waits" + ); + assert_eq!( + progress.on_reply( + &reply(PartsOfferStatus::Unknown, 0, true), + alive + Duration::from_secs(2) + ), + Step::End(StreamEnd::Lost), + "no answer for the deadline loses the stream" + ); + } +} diff --git a/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs b/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs index e001c3225..ad8130841 100644 --- a/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs +++ b/nodedb/src/control/planner/calvin/tx_class/dependent_builder.rs @@ -10,20 +10,18 @@ //! [`TxClass::new_dependent`]; that is a distinct mechanism for passive //! vshards to broadcast reads to active participants before they write). This //! builder is write-set-identity construction only, exactly like -//! `build_static_tx_class`, just sourced from a pre-exec-scan surrogate +//! `build_static_tx_class`, sourced from a pre-exec-scan surrogate //! prediction instead of statically-known plan fields. use crate::Error; use crate::control::server::shared::session::read_set::ReadSetEntry; -use crate::types::VShardId; -use nodedb_cluster::calvin::types::{EngineKeySet, ReadWriteSet, SortedVec, TxClass}; -use nodedb_physical::physical_plan::{GraphOp, PhysicalPlan}; +use nodedb_cluster::calvin::types::{ReadWriteSet, TxClass}; +use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_task::PhysicalTask; use nodedb_types::{DatabaseId, TenantId}; -use super::shared::{ - collection_name_from_plan, read_set_from, surrogate_from_plan, versioned_reads_from, -}; +use super::shared::{read_set_from, versioned_reads_from}; +use super::write_keys::task_write_keys; /// Build a **multi-vshard** `TxClass` for a dependent-read (OLLP) transaction. /// @@ -106,8 +104,6 @@ fn build_dependent_tx_class_impl( reads: &[ReadSetEntry], allow_single_vshard: bool, ) -> crate::Result> { - use std::collections::BTreeMap; - let database_id = tasks .first() .map_or(DatabaseId::DEFAULT, |task| task.database_id); @@ -119,91 +115,14 @@ fn build_dependent_tx_class_impl( }); } - // Accumulate per-collection surrogate sets. The OLLP collection uses the - // predicted surrogates; all other tasks use static key extraction. - let mut doc_surrogates: BTreeMap> = BTreeMap::new(); - // Graph edges (appended implicit-edge deletes) route by from_key(src)/ - // from_key(dst), NOT by collection — mirror `build_static_tx_class`'s edge - // handling so an `EdgeDelete` appended to a dependent txn is classified as - // an `EngineKeySet::Edge` (and dual-homed/locked) rather than misrouted as a - // document write via `surrogate_from_plan`. - let mut edge_pairs: BTreeMap> = BTreeMap::new(); - let mut edge_homes: BTreeMap> = BTreeMap::new(); - - // Seed with the OLLP collection's predicted surrogates. - doc_surrogates - .entry(collection.to_owned()) - .or_default() - .extend_from_slice(predicted_surrogates); - - // Add static surrogates for all non-OLLP tasks. - for task in tasks { - // Edges first: collect surrogate-pair identity + from_key routing homes, - // then skip the doc-surrogate path. EdgePut/EdgeDelete share identity - // fields so both produce an `EngineKeySet::Edge`. - if let PhysicalPlan::Graph( - GraphOp::EdgePut { - collection: edge_coll, - src_id, - dst_id, - src_surrogate, - dst_surrogate, - .. - } - | GraphOp::EdgeDelete { - collection: edge_coll, - src_id, - dst_id, - src_surrogate, - dst_surrogate, - .. - }, - ) = &task.plan - { - edge_pairs - .entry(edge_coll.to_string()) - .or_default() - .push((src_surrogate.as_u32(), dst_surrogate.as_u32())); - let homes = edge_homes.entry(edge_coll.to_string()).or_default(); - homes.push(VShardId::from_key(src_id.as_bytes()).as_u32()); - homes.push(VShardId::from_key(dst_id.as_bytes()).as_u32()); - continue; - } - - let coll = collection_name_from_plan(&task.plan); - if coll.is_empty() || coll == collection { - continue; - } - let surrogate = surrogate_from_plan(&task.plan); - doc_surrogates.entry(coll).or_default().push(surrogate); - } - - let mut write_sets: Vec = doc_surrogates - .into_iter() - .map(|(coll, surrogates)| EngineKeySet::Document { - collection: coll, - surrogates: SortedVec::new(surrogates), - }) - .collect(); - // Emit one Edge keyset per edge collection, with the SAME missing-homes-is- - // hard-error guard `build_static_tx_class` uses: `edge_pairs` and - // `edge_homes` are populated in lockstep, so a missing homes entry is an - // invariant violation, not an empty-participant write. - for (edge_coll, pairs) in edge_pairs { - let homes = edge_homes.remove(&edge_coll).ok_or_else(|| Error::Internal { - detail: format!( - "build_dependent_tx_class invariant violated: no edge_homes for collection {edge_coll}" - ), - })?; - write_sets.push(EngineKeySet::Edge { - collection: edge_coll, - edges: SortedVec::new(pairs), - home_vshards: SortedVec::new(homes), - }); - } - write_sets.sort_by(|a, b| a.collection().cmp(b.collection())); + // Every other task's keys, extracted as `build_static_tx_class` extracts + // them. The OLLP collection's rows are the ones reconnaissance + // predicted, in place of the collection key its predicate write locks. + let mut keys = task_write_keys(tasks)?; + keys.drop_documents(collection); + keys.rows(collection, predicted_surrogates.iter().copied()); - let write_set = ReadWriteSet::new(write_sets); + let write_set = ReadWriteSet::new(keys.into_key_sets()); // A predicate that matched no rows, in a batch whose other tasks write // nothing either, decides an empty write set: there is no state change to // sequence, so no Calvin entry is proposed. `TxClass` rejects an empty @@ -257,7 +176,7 @@ fn build_dependent_tx_class_impl( #[cfg(test)] mod tests { use super::*; - use crate::types::DatabaseId; + use crate::types::{DatabaseId, VShardId}; use nodedb_physical::physical_plan::DocumentOp; fn bulk_delete_task(collection: &str) -> PhysicalTask { @@ -311,7 +230,7 @@ mod tests { fn zero_match_predicate_builds_no_tx_class() { // A predicate matching no rows decides an empty write set: nothing to // sequence, so the builder reports "no transaction" rather than an - // error the retry loop would mistake for predicate drift. + // error the retry loop will mistake for predicate drift. let tasks = vec![bulk_delete_task("users")]; let built = build_single_vshard_dependent_tx_class(&tasks, TenantId::new(1), "users", &[], &[]) diff --git a/nodedb/src/control/planner/calvin/tx_class/mod.rs b/nodedb/src/control/planner/calvin/tx_class/mod.rs index de3c80cb6..a355244f9 100644 --- a/nodedb/src/control/planner/calvin/tx_class/mod.rs +++ b/nodedb/src/control/planner/calvin/tx_class/mod.rs @@ -4,7 +4,8 @@ //! //! Builds the replicated transaction descriptor (`TxClass`) from a physical //! task slice: the per-engine write set (`EngineKeySet` — document / vector -//! surrogates, KV raw keys, graph-edge identity + routing homes) plus the +//! surrogates, KV raw keys, graph-edge identity + routing homes, array tile +//! vShards) plus the //! msgpack-encoded plans. Four builders, split by shape (static vs //! dependent-read) and participant floor (strict multi-vshard vs the //! single-vshard opt-in): @@ -18,7 +19,7 @@ pub mod dependent_builder; pub mod shared; pub mod static_builder; +pub mod write_keys; pub use dependent_builder::{build_dependent_tx_class, build_single_vshard_dependent_tx_class}; -pub(crate) use shared::collection_name_from_plan; pub use static_builder::{build_single_vshard_tx_class, build_static_tx_class}; diff --git a/nodedb/src/control/planner/calvin/tx_class/shared.rs b/nodedb/src/control/planner/calvin/tx_class/shared.rs index fa6373c3f..ea1f1f4ef 100644 --- a/nodedb/src/control/planner/calvin/tx_class/shared.rs +++ b/nodedb/src/control/planner/calvin/tx_class/shared.rs @@ -1,16 +1,14 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Write-set-identity extractors shared by the static and dependent `TxClass` -//! builders. +//! Read-set projections and node lock pairs shared by the static and +//! dependent `TxClass` builders. The write keys live in +//! [`super::write_keys`]. use crate::control::server::shared::session::read_set::{ReadKey, ReadSetEntry}; use nodedb_cluster::calvin::types::{ EngineKeySet, EngineTag, ReadKeyIdent, ReadWriteSet, SortedVec, VersionedReadEntry, VersionedReadSet, }; -use nodedb_physical::physical_plan::{ - ColumnarOp, DocumentOp, GraphOp, KvOp, PhysicalPlan, TimeseriesOp, VectorOp, VectorWriteTargets, -}; /// Map the neutral session read-set into the replicated, LSN-versioned /// [`VersionedReadSet`] carried on the `TxClass`. @@ -41,6 +39,8 @@ pub(super) fn versioned_reads_from(reads: &[ReadSetEntry]) -> VersionedReadSet { }, }, read_lsn: entry.read_version_lsn, + home_vshard: entry.home.map(|home| home.as_u32()), + served_by: entry.home_node, }) .collect(), ) @@ -51,10 +51,11 @@ pub(super) fn versioned_reads_from(reads: &[ReadSetEntry]) -> VersionedReadSet { /// not the LSN-versioned OCC set ([`versioned_reads_from`]). /// /// A [`ReadSetEntry`] carries only `(engine, collection)`, no key identity, -/// so it maps to a COLLECTION-homed [`EngineKeySet`] with an empty key -/// vector — over-approximating participants, which is required and safe for -/// a graph/edge read (no endpoint homes to build an `EngineKeySet::Edge` -/// from): more validation, never a dropped participant. +/// so an unhomed entry maps to a COLLECTION-homed [`EngineKeySet`] with an +/// empty key vector. That over-approximates participants: more validation, +/// never a dropped participant. A homed entry (one vShard of a cross-shard +/// graph read) maps to an `EngineKeySet::Edge` with no edges and its home in +/// `home_vshards`, so the vShard it read participates and validates it. pub(super) fn read_set_from(reads: &[ReadSetEntry]) -> ReadWriteSet { use std::collections::BTreeSet; @@ -65,7 +66,18 @@ pub(super) fn read_set_from(reads: &[ReadSetEntry]) -> ReadWriteSet { let mut vector_colls: BTreeSet = BTreeSet::new(); let mut kv_colls: BTreeSet = BTreeSet::new(); let mut doc_colls: BTreeSet = BTreeSet::new(); + // Homed reads (cross-shard graph reads) participate on their home vShards, + // not on the collection's vShard. They key no edge, so they lock nothing. + let mut homed: std::collections::BTreeMap> = + std::collections::BTreeMap::new(); for entry in reads { + if let Some(home) = entry.home { + homed + .entry(entry.collection.clone()) + .or_default() + .insert(home.as_u32()); + continue; + } if entry.collection.is_empty() { continue; } @@ -110,208 +122,31 @@ pub(super) fn read_set_from(reads: &[ReadSetEntry]) -> ReadWriteSet { surrogates: SortedVec::new(vec![]), }); } - ReadWriteSet::new(sets) -} - -/// Extract `(collection, raw byte keys)` from a KV write plan, or `None` for a -/// KV op with no statically-known point keys (e.g. `BatchPut`). Single-key -/// read-modify-write ops (`Incr`/`IncrFloat`/`Cas`/`GetSet`/`FieldSet`) key on -/// the same `(collection, key)` pair as `Put`/`Insert`, so they sequence -/// identically to the write-admission gate's `kv_point_key`. -pub(super) fn kv_write_keys(op: &KvOp) -> Option<(String, Vec>)> { - match op { - KvOp::Put { - collection, key, .. - } - | KvOp::Insert { - collection, key, .. - } - | KvOp::InsertIfAbsent { - collection, key, .. - } - | KvOp::InsertOnConflictUpdate { - collection, key, .. - } - | KvOp::Incr { - collection, key, .. - } - | KvOp::IncrFloat { - collection, key, .. - } - | KvOp::Cas { - collection, key, .. - } - | KvOp::GetSet { - collection, key, .. - } - | KvOp::FieldSet { - collection, key, .. - } => Some((collection.to_string(), vec![key.clone()])), - KvOp::Delete { - collection, keys, .. - } => Some((collection.to_string(), keys.clone())), - _ => None, - } -} - -/// Extract `(collection, surrogates)` from a Vector write plan, or `None` for a -/// Vector op with no statically-known surrogate identity (e.g. node-id delete). -pub(super) fn vector_write_surrogates(op: &VectorOp) -> Option<(String, Vec)> { - match op { - VectorOp::Insert { - collection, - surrogate, - .. - } - | VectorOp::DeleteBySurrogate { + for (collection, homes) in homed { + sets.push(EngineKeySet::Edge { collection, - surrogate, - .. - } - | VectorOp::DirectInsert { - collection, - surrogate, - .. - } - | VectorOp::DirectInsertIfAbsent { - collection, - surrogate, - .. - } - | VectorOp::DirectUpsert { - collection, - surrogate, - .. - } => Some((collection.to_string(), vec![surrogate.as_u32()])), - VectorOp::BatchInsert { - collection, - surrogates, - .. - } - | VectorOp::DirectDelete { - collection, - targets: VectorWriteTargets::Surrogates(surrogates), - .. - } - | VectorOp::DirectUpdate { - collection, - targets: VectorWriteTargets::Surrogates(surrogates), - .. - } => Some(( - collection.to_string(), - surrogates.iter().map(|s| s.as_u32()).collect(), - )), - _ => None, + edges: SortedVec::new(vec![]), + home_vshards: SortedVec::new(homes.into_iter().collect()), + }); } + ReadWriteSet::new(sets) } -/// Extract the collection name from a write plan. +/// The lock pair of node `node` within one edge collection. /// -/// This name feeds the participant set (hashed by -/// `participating_vshards_in_database`), so an empty name doesn't fail -/// loudly — it hashes to vShard 0, gets enlisted, and aborts far from the -/// real cause ("homes no local write plans or reads"). The document arm is -/// exhaustive for exactly that reason: a new `DocumentOp` is a compile -/// error here, not a silent empty name. -pub(crate) fn collection_name_from_plan(plan: &PhysicalPlan) -> String { - match plan { - PhysicalPlan::Document(op) => document_write_collection(op), - PhysicalPlan::Kv( - KvOp::Put { collection, .. } - | KvOp::Insert { collection, .. } - | KvOp::InsertIfAbsent { collection, .. } - | KvOp::InsertOnConflictUpdate { collection, .. } - | KvOp::Delete { collection, .. } - | KvOp::BatchPut { collection, .. } - | KvOp::Incr { collection, .. } - | KvOp::IncrFloat { collection, .. } - | KvOp::Cas { collection, .. } - | KvOp::GetSet { collection, .. } - | KvOp::FieldSet { collection, .. } - // Predicate DML homes on the collection it names — matching - // `plan_vshard`'s KV routing, which the participant set must - // agree with. - | KvOp::PredicateUpdate { collection, .. } - | KvOp::PredicateDelete { collection, .. }, - ) => collection.to_string(), - PhysicalPlan::Vector( - VectorOp::Insert { collection, .. } - | VectorOp::BatchInsert { collection, .. } - | VectorOp::Delete { collection, .. } - | VectorOp::DeleteBySurrogate { collection, .. } - | VectorOp::DirectTruncate { collection, .. }, - ) => collection.to_string(), - PhysicalPlan::Graph( - GraphOp::EdgePut { collection, .. } | GraphOp::EdgeDelete { collection, .. }, - ) => collection.to_string(), - PhysicalPlan::Timeseries( - TimeseriesOp::Ingest { collection, .. } | TimeseriesOp::Truncate { collection, .. }, - ) - | PhysicalPlan::Columnar(ColumnarOp::Truncate { collection, .. }) => { - collection.to_string() - } - _ => String::new(), - } -} - -/// The collection a DOCUMENT write plan homes on, exhaustively. -fn document_write_collection(op: &DocumentOp) -> String { - match op { - DocumentOp::PointPut { collection, .. } - | DocumentOp::PointInsert { collection, .. } - | DocumentOp::PointDelete { collection, .. } - | DocumentOp::PointUpdate { collection, .. } - | DocumentOp::BatchInsert { collection, .. } - | DocumentOp::Upsert { collection, .. } - | DocumentOp::BulkUpdate { collection, .. } - | DocumentOp::BulkDelete { collection, .. } - | DocumentOp::Truncate { collection, .. } - // Homes on the TARGET collection — same one the routing oracle uses. - | DocumentOp::ApplyBalanceDelta { collection, .. } => collection.to_string(), - DocumentOp::InsertSelect { - target_collection, .. - } => target_collection.to_string(), - // Cross-collection writes: Control-Plane orchestrators resolve them - // into concrete point writes before dispatch, so no raw plan reaches - // this builder; the routing oracle names them `Unroutable`. - DocumentOp::Merge { .. } | DocumentOp::UpdateFromJoin { .. } => String::new(), - // Proposed directly through Raft, so no plan reaches this builder. - DocumentOp::ResolvedWrite { .. } => String::new(), - // Reads and index DDL: caller skips non-`is_write_plan` before here. - DocumentOp::ResolveWrite(_) - | DocumentOp::PointGet { .. } - | DocumentOp::Scan { .. } - | DocumentOp::RangeScan { .. } - | DocumentOp::IndexLookup { .. } - | DocumentOp::IndexedFetch { .. } - | DocumentOp::EstimateCount { .. } - | DocumentOp::MaterializeScan { .. } - | DocumentOp::Register { .. } - | DocumentOp::DropIndex { .. } - | DocumentOp::BackfillIndex { .. } => String::new(), - } -} - -/// Extract a surrogate from a write plan (returns 0 when unavailable). -pub(super) fn surrogate_from_plan(plan: &PhysicalPlan) -> u32 { - match plan { - PhysicalPlan::Document( - DocumentOp::PointPut { surrogate, .. } - | DocumentOp::PointInsert { surrogate, .. } - | DocumentOp::PointDelete { surrogate, .. } - | DocumentOp::PointUpdate { surrogate, .. } - | DocumentOp::Upsert { surrogate, .. } - // The TARGET row's identity — two balance writes onto one row - // serialize; falling through to `0` would lock all of them together. - | DocumentOp::ApplyBalanceDelta { surrogate, .. }, - ) => surrogate.as_u32(), - _ => 0, - } +/// Every edge write takes it for both endpoints, and a node delete's guard +/// takes it for its node, so a guard and every write of an edge on its node +/// run in sequence order on the node's key home. A real edge never locks a +/// pair with a zero destination surrogate, so the pair never aliases an +/// edge. Two node names that hash alike only serialize together. +pub(crate) fn node_lock_pair(node: &str) -> (u32, u32) { + let hash = crate::util::fnv1a_hash(node.as_bytes()); + ((hash ^ (hash >> 32)) as u32, 0) } /// Lockstep proof that the write-admission gate and the Calvin scheduler /// derive IDENTICAL lock keys for the same op — if they diverged, a -/// gate-fenced write and a sequenced txn would lock different keys. +/// gate-fenced write and a sequenced txn will lock different keys. #[cfg(test)] mod lockstep_tests { use super::*; @@ -320,6 +155,7 @@ mod lockstep_tests { use crate::control::server::shared::write_admission::lock_keys::plan_lock_keys; use crate::types::{DatabaseId, TenantId, VShardId}; use nodedb_cluster::calvin::types::EngineKeySet; + use nodedb_physical::physical_plan::{DocumentOp, GraphOp, KvOp, PhysicalPlan}; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use nodedb_types::Surrogate; use std::collections::BTreeSet; @@ -384,6 +220,13 @@ mod lockstep_tests { }); } } + EngineKeySet::Array { collection, .. } => { + keys.insert(LockKey::Surrogate { + collection: Arc::from(collection.as_str()), + surrogate: + crate::control::planner::calvin::tx_class::write_keys::COLLECTION_KEY, + }); + } } } keys @@ -450,33 +293,105 @@ mod lockstep_tests { resolved_sum_targets: Vec::new(), })); } + + /// A bound point delete on the fast path locks its surrogate and its row + /// id, exactly as the scheduler does, so an unbound delete sequenced + /// through Calvin orders against it by the row id. The fence and the + /// keyed order lock still name the row by its surrogate key alone. + #[test] + fn document_point_delete_gate_keys_match_scheduler_keys_with_row_id() { + use crate::control::server::shared::write_admission::lock_keys::plan_row_key; + let plan = PhysicalPlan::Document(DocumentOp::PointDelete { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "docs", + ), + document_id: "d1".to_owned(), + surrogate: Some(Surrogate::new(9)), + pk_bytes: b"d1".to_vec(), + returning: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + resolved_sum_targets: Vec::new(), + }); + let (_, gate_keys) = plan_lock_keys(&plan).expect("a bound point delete is fast-path"); + assert!(gate_keys.contains(&LockKey::Kv { + collection: Arc::from("docs"), + key: Arc::from(b"d1".as_slice()), + })); + assert_eq!( + plan_row_key(&plan), + Some(LockKey::Surrogate { + collection: Arc::from("docs"), + surrogate: 9, + }) + ); + assert_gate_matches_scheduler(plan); + } + + /// A single-home edge write on the fast path locks its edge and both + /// endpoints' node lock pairs, exactly as the scheduler does, so a node + /// delete's guard orders against it either way. + #[test] + fn edge_put_gate_keys_match_scheduler_keys_with_node_locks() { + let src = "n0"; + let dst = (1u32..) + .map(|i| format!("n{i}")) + .find(|name| { + crate::types::VShardId::from_key(name.as_bytes()) + == crate::types::VShardId::from_key(src.as_bytes()) + }) + .expect("some node shares the source's key home"); + let plan = PhysicalPlan::Graph(GraphOp::EdgePut { + collection: nodedb_types::QualifiedCollection::new( + nodedb_types::DatabaseId::DEFAULT, + "g", + ), + src_id: src.to_owned(), + label: "L".to_owned(), + dst_id: dst.clone(), + properties: Vec::new(), + src_surrogate: Surrogate::new(1), + dst_surrogate: Surrogate::new(2), + }); + let (_, gate_keys) = plan_lock_keys(&plan).expect("a single-home edge is fast-path"); + for node in [src, dst.as_str()] { + let (lock_src, lock_dst) = node_lock_pair(node); + assert!(gate_keys.contains(&LockKey::Edge { + collection: Arc::from("g"), + src: lock_src, + dst: lock_dst, + })); + } + assert_gate_matches_scheduler(plan); + } } /// The participant set and the routing oracle must agree about where a plan -/// lives: [`collection_name_from_plan`] feeds the participant list, the -/// scheduler's `plan_vshard` oracle decides who actually gets the plan. A +/// lives: the write keys' collections feed the participant list, and the +/// scheduler's `plan_vshard` oracle decides who gets the plan. A /// disagreement enlists a shard, hands it nothing, and aborts far from the -/// cause. An op missing from the extractor silently hashes to vShard 0. -/// These tests pin the agreement directly. +/// cause. These tests pin the agreement directly. #[cfg(test)] mod routing_agreement_tests { - use super::*; use crate::control::planner::calvin::tx_class::static_builder::build_static_tx_class; + use crate::control::planner::calvin::tx_class::write_keys::{WriteKeys, add_plan_write_keys}; use crate::types::{DatabaseId, TenantId, VShardId}; + use nodedb_cluster::calvin::types::{EngineKeySet, SortedVec}; + use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use nodedb_types::Surrogate; const TENANT: TenantId = TenantId::new(1); const DB: DatabaseId = DatabaseId::DEFAULT; /// The binding's source and its balance target. Asserted to hash apart - /// by [`the_fixture_spans_two_vshards`] — a co-resident pair would never + /// by [`the_fixture_spans_two_vshards`] — a co-resident pair will never /// produce the two-task plan this file is about. const SOURCE: &str = "route_entries"; const TARGET: &str = "route_accounts"; - /// Build a task homed the way PRODUCTION homes it, not by asking - /// [`collection_name_from_plan`] — that would make the agreement true - /// by construction and prove nothing. + /// Build a task homed the way PRODUCTION homes it, not by asking the + /// write keys, which makes the agreement true by construction. fn task(plan: PhysicalPlan, vshard_id: VShardId) -> PhysicalTask { PhysicalTask { tenant_id: TENANT, @@ -537,23 +452,23 @@ mod routing_agreement_tests { ); } - /// A balance write reports the TARGET collection it names — an empty - /// name here enlisted vShard 0 while the plan went to the target's shard. + /// A balance write locks the TARGET row of the TARGET collection it + /// names: not an empty collection on vShard 0, and not the collection + /// key every balance write will share. #[test] - fn a_balance_write_reports_the_collection_it_mutates() { - assert_eq!(collection_name_from_plan(&balance_write()), TARGET); - } - - /// And the TARGET ROW's surrogate; falling through to `0` made every - /// balance write share one lock key. - #[test] - fn a_balance_write_reports_the_target_rows_surrogate() { - assert_eq!(surrogate_from_plan(&balance_write()), 271); + fn a_balance_write_locks_the_target_row() { + let mut keys = WriteKeys::default(); + add_plan_write_keys(&mut keys, &balance_write()).expect("a balance write has keys"); + assert_eq!( + keys.into_key_sets(), + vec![EngineKeySet::Document { + collection: TARGET.to_owned(), + surrogates: SortedVec::new(vec![271]), + }] + ); } - /// The pair enlists exactly the two shards that hold work — no third. - /// Before the extractor named `ApplyBalanceDelta`, this came back as - /// source-shard + vShard 0, and vShard 0 held no plan. + /// The pair enlists exactly the two shards that hold work, no third. #[test] fn the_pair_enlists_only_the_shards_that_hold_work() { let tasks = statement_tasks(); diff --git a/nodedb/src/control/planner/calvin/tx_class/static_builder.rs b/nodedb/src/control/planner/calvin/tx_class/static_builder.rs index bcf7c7ab8..6b18326de 100644 --- a/nodedb/src/control/planner/calvin/tx_class/static_builder.rs +++ b/nodedb/src/control/planner/calvin/tx_class/static_builder.rs @@ -4,18 +4,14 @@ //! known upfront). use crate::Error; -use crate::control::planner::calvin::dispatch::is_write_plan; use crate::control::server::shared::session::read_set::ReadSetEntry; -use crate::types::VShardId; -use nodedb_cluster::calvin::types::{EngineKeySet, ReadWriteSet, SortedVec, TxClass}; -use nodedb_physical::physical_plan::{GraphOp, PhysicalPlan}; +use nodedb_cluster::calvin::types::{ReadWriteSet, TxClass}; +use nodedb_physical::physical_plan::PhysicalPlan; use nodedb_physical::physical_task::PhysicalTask; use nodedb_types::{DatabaseId, TenantId}; -use super::shared::{ - collection_name_from_plan, kv_write_keys, read_set_from, surrogate_from_plan, - vector_write_surrogates, versioned_reads_from, -}; +use super::shared::{read_set_from, versioned_reads_from}; +use super::write_keys::task_write_keys; /// Build a **multi-vshard** `TxClass` from a static write task slice. /// @@ -70,8 +66,6 @@ fn build_static_tx_class_impl( reads: &[ReadSetEntry], allow_single_vshard: bool, ) -> crate::Result { - use std::collections::HashMap; - let database_id = tasks .first() .map_or(DatabaseId::DEFAULT, |task| task.database_id); @@ -82,130 +76,31 @@ fn build_static_tx_class_impl( detail: "Calvin transaction spans multiple databases".to_owned(), }); } - - // Collect surrogates per collection for non-edge write tasks. - let mut doc_surrogates: HashMap> = HashMap::new(); - // Collect edge identity (surrogate pairs) and routing homes - // (from_key of src/dst string keys) per collection for graph edges. - let mut edge_pairs: HashMap> = HashMap::new(); - let mut edge_homes: HashMap> = HashMap::new(); - // KV writes are keyed by raw bytes and Vector writes by surrogate — each - // needs its own EngineKeySet rather than the generic document-surrogate - // bucket (which would mis-key them and break lock-conflict detection). - let mut kv_keys: HashMap>> = HashMap::new(); - let mut vector_surrogates: HashMap> = HashMap::new(); - - for task in tasks { - if !is_write_plan(&task.plan) { - continue; - } - // Graph edges route by from_key(src)/from_key(dst), not by collection. - // EdgePut and EdgeDelete share identity fields so both produce an - // `EngineKeySet::Edge` — a cross-shard delete dual-homes (and locks) - // exactly like the matching insert. - if let PhysicalPlan::Graph( - GraphOp::EdgePut { - collection, - src_id, - dst_id, - src_surrogate, - dst_surrogate, - .. - } - | GraphOp::EdgeDelete { - collection, - src_id, - dst_id, - src_surrogate, - dst_surrogate, - .. - }, - ) = &task.plan - { - edge_pairs - .entry(collection.to_string()) - .or_default() - .push((src_surrogate.as_u32(), dst_surrogate.as_u32())); - let homes = edge_homes.entry(collection.to_string()).or_default(); - homes.push(VShardId::from_key(src_id.as_bytes()).as_u32()); - homes.push(VShardId::from_key(dst_id.as_bytes()).as_u32()); - continue; - } - // KV and Vector writes carry their own key representation. - match &task.plan { - PhysicalPlan::Kv(op) => { - if let Some((coll, keys)) = kv_write_keys(op) { - kv_keys.entry(coll).or_default().extend(keys); - continue; - } - } - PhysicalPlan::Vector(op) => { - if let Some((coll, surrs)) = vector_write_surrogates(op) { - vector_surrogates.entry(coll).or_default().extend(surrs); - continue; - } - } - _ => {} - } - // Document engine (and any other statically-keyed write reaching the - // multishard path): bucket by surrogate. - let collection = collection_name_from_plan(&task.plan); - let surrogate = surrogate_from_plan(&task.plan); - doc_surrogates - .entry(collection) - .or_default() - .push(surrogate); - } - - // Build write set — one EngineKeySet per collection, sorted for - // determinism. - let mut write_sets: Vec = doc_surrogates - .into_iter() - .map(|(collection, surrogates)| EngineKeySet::Document { - collection, - surrogates: SortedVec::new(surrogates), - }) - .collect(); - // Emit one Edge keyset per collection, carrying surrogate-pair identity - // (for locking) and from_key routing homes (for participating vShards). - for (collection, pairs) in edge_pairs { - // `edge_pairs` and `edge_homes` are populated in lockstep in the loop - // above, so a collection in one is always in the other. Treat a missing - // homes entry as a hard error rather than silently emitting an Edge - // keyset with empty `home_vshards` (which would drop Calvin participant - // shards and misroute the cross-shard write with no diagnostic). - let homes = edge_homes.remove(&collection).ok_or_else(|| Error::Internal { + // Every replica resolves a sequenced transaction on its own, and each + // will accept other lines than its peers. A timeseries ingest resolves + // to its rows on the submitting node, before it is sequenced. + if let Some(task) = tasks + .iter() + .find(|task| crate::control::write_resolve::is_unresolved_ingest(&task.plan)) + { + return Err(Error::Internal { detail: format!( - "build_static_tx_class invariant violated: no edge_homes for collection {collection}" + "internal invariant break: a timeseries ingest on '{}' reached the Calvin \ + sequencer unresolved; the submitter resolves it with `resolve_tasks_for_log`", + task.plan.collection().unwrap_or("") ), - })?; - write_sets.push(EngineKeySet::Edge { - collection, - edges: SortedVec::new(pairs), - home_vshards: SortedVec::new(homes), }); } - // Emit one Kv keyset per collection (raw byte keys) and one Vector keyset - // per collection (surrogates), so KV and Vector writes lock on their real - // identity rather than a bogus document surrogate. - for (collection, keys) in kv_keys { - write_sets.push(EngineKeySet::Kv { - collection, - keys: SortedVec::new(keys), - }); - } - for (collection, surrogates) in vector_surrogates { - write_sets.push(EngineKeySet::Vector { - collection, - surrogates: SortedVec::new(surrogates), - }); - } - // Sort by collection name for determinism. - write_sets.sort_by(|a, b| a.collection().cmp(b.collection())); + + // One key set per engine and collection, ordered by collection. Every + // write op is matched by name: an edge locks its pair and both node + // pairs on both homes, a KV write its raw key, a CRDT document its id, + // and a row write its surrogate. + let write_sets = task_write_keys(tasks)?.into_key_sets(); // Read-your-own-write: a SESSION read of a collection this txn also WRITES // must NOT enter the OCC read set. The txn's own staged write advances that - // collection's write floor, so validating the earlier read against it would + // collection's write floor, so validating the earlier read against it will // flag it stale and abort the commit — a false serialization conflict. This // mirrors the written-collection exclusion the single-shard // `si_conflict_abort` path already applies. @@ -216,9 +111,9 @@ fn build_static_tx_class_impl( // value the transaction ships was computed from — a cross-shard // materialized-sum settlement folds a delta from a pre-image of the source // row and sends it to another shard. The source collection is one this - // statement always writes, so the exclusion would drop every such entry and - // `read_set_still_current` would validate nothing: a concurrent write - // between the fold and the apply would commit a total folded from an image + // statement always writes, so the exclusion will drop every such entry and + // `read_set_still_current` will validate nothing: a concurrent write + // between the fold and the apply will commit a total folded from an image // that has moved. The kind is carried on the entry rather than inferred // here, so no entry can be classified by accident of which collection it // names. @@ -284,7 +179,7 @@ fn build_static_tx_class_impl( mod tests { use super::*; use crate::control::server::shared::session::read_set::{EngineTag, ReadKey, ReadOrigin}; - use crate::types::{DatabaseId, KeyRepr, Lsn}; + use crate::types::{DatabaseId, KeyRepr, Lsn, VShardId}; use nodedb_physical::physical_plan::DocumentOp; use nodedb_types::Surrogate; @@ -348,6 +243,8 @@ mod tests { // `versioned_reads_from` propagates; give it the same synthetic LSN. read_version_lsn: Lsn::new(read_lsn), origin, + home: None, + home_node: 0, } } @@ -439,7 +336,7 @@ mod tests { "scan_col must be in read_set" ); - // participating_vshards is now write ∪ read: it contains the two write + // participating_vshards is write ∪ read: it contains the two write // collections' vShards AND both read collections' vShards. let participants: std::collections::BTreeSet = tx .participating_vshards() @@ -530,9 +427,9 @@ mod tests { } /// The other half: an ordinary SESSION read of a collection the transaction - /// writes is STILL excluded. Without this, the fix above could be "passed" - /// by disabling the exclusion, and every transaction that reads a row it - /// then writes would abort on its own staged write. + /// writes is excluded. Without this, disabling the exclusion will pass the + /// test above, and every transaction that reads a row it + /// then writes will abort on its own staged write. #[test] fn a_session_read_of_a_written_collection_is_still_excluded() { let (col_a, col_b) = two_distinct_collections(); diff --git a/nodedb/src/control/planner/calvin/tx_class/write_keys/columnar.rs b/nodedb/src/control/planner/calvin/tx_class/write_keys/columnar.rs new file mode 100644 index 000000000..f815bdb73 --- /dev/null +++ b/nodedb/src/control/planner/calvin/tx_class/write_keys/columnar.rs @@ -0,0 +1,142 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Columnar, timeseries and array write keys. + +#![deny(clippy::wildcard_enum_match_arm)] + +use nodedb_physical::physical_plan::{ArrayOp, ColumnarOp, TimeseriesOp}; + +use super::plan::{not_a_write, unsequenced}; +use super::set::WriteKeys; + +/// Add the write keys of columnar op `op` to `keys`. +pub(super) fn add_columnar_keys(keys: &mut WriteKeys, op: &ColumnarOp) -> crate::Result<()> { + match op { + // Every row it inserts, and the collection key a multi-row write + // holds. + ColumnarOp::Insert { + collection, + surrogates, + .. + } => { + keys.rows(collection.as_str(), surrogates.iter().map(|s| s.as_u32())); + keys.whole_collection(collection.as_str()); + } + // Predicate and primary-key writes carry no row surrogate. + ColumnarOp::Update { collection, .. } + | ColumnarOp::Delete { collection, .. } + | ColumnarOp::ResolvedUpdate { collection, .. } + | ColumnarOp::ResolvedDelete { collection, .. } + | ColumnarOp::Truncate { collection, .. } => keys.whole_collection(collection.as_str()), + ColumnarOp::Scan { .. } + | ColumnarOp::MaterializeScan { .. } + | ColumnarOp::ResolveDml { .. } => { + return Err(not_a_write("a columnar read")); + } + } + Ok(()) +} + +/// Add the write keys of timeseries op `op` to `keys`. +pub(super) fn add_timeseries_keys(keys: &mut WriteKeys, op: &TimeseriesOp) -> crate::Result<()> { + match op { + TimeseriesOp::Ingest { collection, .. } | TimeseriesOp::Truncate { collection, .. } => { + keys.whole_collection(collection.as_str()); + Ok(()) + } + TimeseriesOp::ResolveIngest(_) | TimeseriesOp::Scan { .. } => { + Err(not_a_write("a timeseries read")) + } + } +} + +/// Add the write keys of array op `op` to `keys`. +/// +/// An array is tile-partitioned, and a cell write names the vShard its +/// cells' tiles live on. It locks the whole array on that vShard. +pub(super) fn add_array_keys(keys: &mut WriteKeys, op: &ArrayOp) -> crate::Result<()> { + match op { + ArrayOp::Put { + array_id, + vshard_id, + .. + } + | ArrayOp::Delete { + array_id, + vshard_id, + .. + } => { + let collection = + nodedb_types::QualifiedCollection::new(array_id.database_id, &array_id.name); + keys.array_cells(collection.as_str(), *vshard_id); + Ok(()) + } + // A flush writes every tile the node holds, and a transaction never + // stages one: it runs outside any transaction. + ArrayOp::Flush { .. } => Err(unsequenced( + "an array flush writes every tile of the array on a node, and runs outside a \ + transaction", + )), + ArrayOp::OpenArray { .. } + | ArrayOp::Compact { .. } + | ArrayOp::DropArray { .. } + | ArrayOp::RekeyArray { .. } + | ArrayOp::PurgeArrayDrop { .. } + | ArrayOp::Slice { .. } + | ArrayOp::Project { .. } + | ArrayOp::Aggregate { .. } + | ArrayOp::Elementwise { .. } + | ArrayOp::SurrogateBitmapScan { .. } => Err(not_a_write("an array read or DDL op")), + } +} + +#[cfg(test)] +mod tests { + use nodedb_array::types::ArrayId; + use nodedb_cluster::calvin::types::{EngineKeySet, SortedVec}; + use nodedb_types::{DatabaseId, QualifiedCollection, TenantId}; + + use super::*; + + /// Array cell writes enlist the vShards their tiles live on, one key set + /// per array. A flush is refused. + #[test] + fn array_cell_writes_lock_their_tile_vshards() { + let array_id = ArrayId::new(TenantId::new(1), "grid"); + let mut keys = WriteKeys::default(); + for (vshard_id, delete) in [(40, false), (7, true), (40, true)] { + let op = if delete { + ArrayOp::Delete { + array_id: array_id.clone(), + coords_msgpack: Vec::new(), + wal_lsn: 0, + provenance: None, + vshard_id, + } + } else { + ArrayOp::Put { + array_id: array_id.clone(), + cells_msgpack: Vec::new(), + wal_lsn: 0, + provenance: None, + vshard_id, + } + }; + add_array_keys(&mut keys, &op).expect("a cell write has keys"); + } + let collection = QualifiedCollection::new(DatabaseId::DEFAULT, "grid").to_string(); + assert_eq!( + keys.into_key_sets(), + vec![EngineKeySet::Array { + collection, + vshards: SortedVec::new(vec![7, 40, 40]), + }] + ); + + let flush = ArrayOp::Flush { + array_id, + wal_lsn: 0, + }; + assert!(add_array_keys(&mut WriteKeys::default(), &flush).is_err()); + } +} diff --git a/nodedb/src/control/planner/calvin/tx_class/write_keys/crdt.rs b/nodedb/src/control/planner/calvin/tx_class/write_keys/crdt.rs new file mode 100644 index 000000000..743e5805a --- /dev/null +++ b/nodedb/src/control/planner/calvin/tx_class/write_keys/crdt.rs @@ -0,0 +1,75 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! CRDT write keys. +//! +//! A CRDT document is keyed by its document id, bound or not: a delete +//! planned before a concurrent upsert bound the document locks the key the +//! upsert locks. A write of the whole collection locks the collection key. + +#![deny(clippy::wildcard_enum_match_arm)] + +use nodedb_physical::physical_plan::CrdtOp; + +use super::plan::not_a_write; +use super::set::WriteKeys; + +/// Add the write keys of CRDT op `op` to `keys`. +pub(super) fn add_keys(keys: &mut WriteKeys, op: &CrdtOp) -> crate::Result<()> { + match op { + CrdtOp::Apply { + collection, + document_id, + .. + } + | CrdtOp::ApplyAuthenticated { + collection, + document_id, + .. + } + | CrdtOp::RestoreToVersion { + collection, + document_id, + .. + } + | CrdtOp::ListInsert { + collection, + document_id, + .. + } + | CrdtOp::ListDelete { + collection, + document_id, + .. + } + | CrdtOp::ListMove { + collection, + document_id, + .. + } + | CrdtOp::DocUpsert { + collection, + document_id, + .. + } + | CrdtOp::DocDelete { + collection, + document_id, + .. + } => keys.row_id(collection.as_str(), document_id), + CrdtOp::ImportSnapshot { collection, .. } + | CrdtOp::SetConstraints { collection, .. } + | CrdtOp::DropConstraints { collection, .. } => keys.whole_collection(collection.as_str()), + CrdtOp::Read { .. } + | CrdtOp::ReadConstraints { .. } + | CrdtOp::SetPolicy { .. } + | CrdtOp::GetPolicy { .. } + | CrdtOp::ReadAtVersion { .. } + | CrdtOp::GetVersionVector { .. } + | CrdtOp::ExportDelta { .. } + | CrdtOp::CompactAtVersion { .. } + | CrdtOp::PreviewApply { .. } => { + return Err(not_a_write("a CRDT read, policy or compaction op")); + } + } + Ok(()) +} diff --git a/nodedb/src/control/planner/calvin/tx_class/write_keys/document.rs b/nodedb/src/control/planner/calvin/tx_class/write_keys/document.rs new file mode 100644 index 000000000..ebdd76bc1 --- /dev/null +++ b/nodedb/src/control/planner/calvin/tx_class/write_keys/document.rs @@ -0,0 +1,116 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Document-engine write keys. + +#![deny(clippy::wildcard_enum_match_arm)] + +use nodedb_physical::physical_plan::DocumentOp; + +use super::plan::{not_a_write, unsequenced}; +use super::set::WriteKeys; + +/// Add the write keys of document op `op` to `keys`. +pub(super) fn add_keys(keys: &mut WriteKeys, op: &DocumentOp) -> crate::Result<()> { + match op { + // A point writer locks its row's surrogate and its row id, so it + // orders against a writer that knows either. + DocumentOp::PointPut { + collection, + document_id, + surrogate, + .. + } + | DocumentOp::PointInsert { + collection, + document_id, + surrogate, + .. + } + | DocumentOp::Upsert { + collection, + document_id, + surrogate, + .. + } => { + keys.rows(collection.as_str(), [surrogate.as_u32()]); + keys.row_id(collection.as_str(), document_id); + } + // The TARGET row's identity: two balance writes onto one row + // serialize, and balance writes onto distinct rows do not. Its + // document id is the surrogate's hex, not a key the row binds under. + DocumentOp::ApplyBalanceDelta { + collection, + surrogate, + .. + } => keys.rows(collection.as_str(), [surrogate.as_u32()]), + // A key unbound in its database names no row yet: it takes the + // collection key and its row id. The row id orders it after an + // insert that binds the key, so its rebind at dispatch finds that + // binding on every replica. + DocumentOp::PointDelete { + collection, + document_id, + surrogate, + .. + } + | DocumentOp::PointUpdate { + collection, + document_id, + surrogate, + .. + } => { + match surrogate { + Some(surrogate) => keys.rows(collection.as_str(), [surrogate.as_u32()]), + None => keys.whole_collection(collection.as_str()), + } + keys.row_id(collection.as_str(), document_id); + } + // Every row it inserts, by surrogate and row id, and the collection + // key a multi-row write holds. + DocumentOp::BatchInsert { + collection, + documents, + surrogates, + .. + } => { + keys.rows(collection.as_str(), surrogates.iter().map(|s| s.as_u32())); + for (document_id, _) in documents { + keys.row_id(collection.as_str(), document_id); + } + keys.whole_collection(collection.as_str()); + } + DocumentOp::InsertSelect { + target_collection, .. + } => keys.whole_collection(target_collection.as_str()), + DocumentOp::BulkUpdate { collection, .. } + | DocumentOp::BulkDelete { collection, .. } + | DocumentOp::Truncate { collection, .. } => keys.whole_collection(collection.as_str()), + // Control-Plane orchestrators resolve these into concrete point + // writes before dispatch; the routing oracle names them unroutable. + DocumentOp::Merge { .. } | DocumentOp::UpdateFromJoin { .. } => { + return Err(unsequenced( + "a cross-collection document write has no enforced source/target co-location", + )); + } + DocumentOp::ResolvedWrite { .. } => { + return Err(unsequenced( + "a resolved governed document write is proposed by the write-resolve \ + orchestrator", + )); + } + DocumentOp::ResolveWrite(_) + | DocumentOp::PointGet { .. } + | DocumentOp::Scan { .. } + | DocumentOp::RangeScan { .. } + | DocumentOp::IndexLookup { .. } + | DocumentOp::IndexedFetch { .. } + | DocumentOp::EstimateCount { .. } + | DocumentOp::MaterializeScan { .. } + | DocumentOp::Register { .. } + | DocumentOp::DropIndex { .. } + | DocumentOp::BackfillIndex { .. } => { + return Err(not_a_write("a document read or index DDL op")); + } + } + Ok(()) +} diff --git a/nodedb/src/control/planner/calvin/tx_class/write_keys/graph.rs b/nodedb/src/control/planner/calvin/tx_class/write_keys/graph.rs new file mode 100644 index 000000000..2c66de50d --- /dev/null +++ b/nodedb/src/control/planner/calvin/tx_class/write_keys/graph.rs @@ -0,0 +1,115 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Graph write keys. +//! +//! Every edge write locks its edge's surrogate pair and both endpoints' +//! node lock pairs, on both endpoint homes. A batch does so for each of its +//! edges. A node delete's guard locks its node's pair on the node's key +//! home, so the guard and every write of an edge on its node run in +//! sequence order there. + +#![deny(clippy::wildcard_enum_match_arm)] + +use nodedb_physical::physical_plan::{BatchEdge, GraphOp}; + +use super::plan::{not_a_write, unsequenced}; +use super::set::{WriteKeys, row_id_key}; + +/// The lock namespace of node-label writes. Labels belong to a node, not to +/// a collection, so their node lock pairs never alias an edge collection's. +pub const NODE_LABEL_LOCKS: &str = "\0__graph_node_labels__"; + +/// Add the write keys of graph op `op` to `keys`. +pub(super) fn add_keys(keys: &mut WriteKeys, op: &GraphOp) -> crate::Result<()> { + match op { + GraphOp::EdgePut { + collection, + src_id, + dst_id, + src_surrogate, + dst_surrogate, + .. + } + | GraphOp::EdgeDelete { + collection, + src_id, + dst_id, + src_surrogate, + dst_surrogate, + .. + } => keys.edge( + collection.as_str(), + src_id, + dst_id, + (src_surrogate.as_u32(), dst_surrogate.as_u32()), + ), + GraphOp::EdgePutBatch { edges } | GraphOp::EdgeDeleteBatch { edges } => { + if edges.is_empty() { + return Err(unsequenced("an edge batch carries no edges")); + } + for edge in edges { + add_batch_edge(keys, edge); + } + } + GraphOp::SetNodeLabels { node_id, .. } | GraphOp::RemoveNodeLabels { node_id, .. } => { + keys.node(NODE_LABEL_LOCKS, node_id); + } + GraphOp::NodeEdgeGuard { + collection, + node_id, + .. + } => keys.node(collection.as_str(), node_id), + // A TRUNCATE's edge share locks no edge: it records a cut at its + // transaction's ordinal, which hides every version applied below it + // whatever order this vShard flushes the edge writes in. + GraphOp::TruncateEdges { collection, vshard } => keys.home(collection.as_str(), *vshard), + // A CRDT delete's presence guard locks every document it names, + // stored or not, by the key every writer of that document locks, + // and the collection key every collection-wide write locks. So it + // runs after every lower writer of those documents flushed. + GraphOp::NodePresenceGuard { + collection, + vshard, + present, + absent, + } => { + keys.home(collection.as_str(), *vshard); + keys.kv_keys( + collection.as_str(), + present.iter().chain(absent.iter()).map(|id| row_id_key(id)), + ); + keys.whole_collection(collection.as_str()); + } + GraphOp::ResolveEdgeDelete(_) + | GraphOp::Hop { .. } + | GraphOp::Neighbors { .. } + | GraphOp::NeighborsMulti { .. } + | GraphOp::Path { .. } + | GraphOp::Subgraph { .. } + | GraphOp::RagFusion { .. } + | GraphOp::Algo { .. } + | GraphOp::Match { .. } + | GraphOp::MatchContinuation { .. } + | GraphOp::MatchVarLenResume { .. } + | GraphOp::BspSuperstep(_) + | GraphOp::WccSuperstep(_) + | GraphOp::TemporalNeighbors { .. } + | GraphOp::TemporalAlgorithm { .. } + | GraphOp::Stats { .. } + | GraphOp::NodePresenceRead { .. } => { + return Err(not_a_write("a graph read")); + } + } + Ok(()) +} + +/// Lock one edge of a batch in its own collection, as a single edge write +/// does. +fn add_batch_edge(keys: &mut WriteKeys, edge: &BatchEdge) { + keys.edge( + edge.collection.as_str(), + &edge.src_id, + &edge.dst_id, + (edge.src_surrogate.as_u32(), edge.dst_surrogate.as_u32()), + ); +} diff --git a/nodedb/src/control/planner/calvin/tx_class/write_keys/kv.rs b/nodedb/src/control/planner/calvin/tx_class/write_keys/kv.rs new file mode 100644 index 000000000..9e5974052 --- /dev/null +++ b/nodedb/src/control/planner/calvin/tx_class/write_keys/kv.rs @@ -0,0 +1,108 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Key-Value-engine write keys. +//! +//! A single-key op locks the same `(collection, key)` the write-admission +//! gate's `kv_point_key` locks, so a fast-path write and a sequenced one +//! serialize on one key. + +#![deny(clippy::wildcard_enum_match_arm)] + +use nodedb_physical::physical_plan::KvOp; + +use super::plan::{not_a_write, unsequenced}; +use super::set::WriteKeys; + +/// Add the write keys of KV op `op` to `keys`. +pub(super) fn add_keys(keys: &mut WriteKeys, op: &KvOp) -> crate::Result<()> { + match op { + KvOp::Put { + collection, key, .. + } + | KvOp::Insert { + collection, key, .. + } + | KvOp::InsertIfAbsent { + collection, key, .. + } + | KvOp::InsertOnConflictUpdate { + collection, key, .. + } + | KvOp::Incr { + collection, key, .. + } + | KvOp::IncrFloat { + collection, key, .. + } + | KvOp::Cas { + collection, key, .. + } + | KvOp::GetSet { + collection, key, .. + } + | KvOp::FieldSet { + collection, key, .. + } + | KvOp::Expire { + collection, key, .. + } + | KvOp::Persist { + collection, key, .. + } => keys.kv_keys(collection.as_str(), [key.clone()]), + KvOp::Delete { + collection, + keys: deleted, + .. + } => keys.kv_keys(collection.as_str(), deleted.iter().cloned()), + // Both keys the transfer moves value between. + KvOp::Transfer { + collection, + source_key, + dest_key, + .. + } => keys.kv_keys(collection.as_str(), [source_key.clone(), dest_key.clone()]), + // Every key it writes, and the collection key a multi-row write + // holds. + KvOp::BatchPut { + collection, + entries, + .. + } => { + keys.kv_keys(collection.as_str(), entries.iter().map(|(k, _)| k.clone())); + keys.whole_collection(collection.as_str()); + } + KvOp::Truncate { collection, .. } + | KvOp::PredicateUpdate { collection, .. } + | KvOp::PredicateDelete { collection, .. } => keys.whole_collection(collection.as_str()), + KvOp::TransferItem { .. } => { + return Err(unsequenced( + "a cross-collection KV transfer has no enforced source/target co-location", + )); + } + KvOp::ResolvedWrite { .. } => { + return Err(unsequenced( + "a resolved governed KV write is proposed by the write-resolve orchestrator", + )); + } + KvOp::ResolveWrite(_) + | KvOp::Get { .. } + | KvOp::Scan { .. } + | KvOp::GetTtl { .. } + | KvOp::BatchGet { .. } + | KvOp::FieldGet { .. } + | KvOp::MaterializeScan { .. } + | KvOp::RegisterIndex { .. } + | KvOp::DropIndex { .. } + | KvOp::RegisterSortedIndex { .. } + | KvOp::DropSortedIndex { .. } + | KvOp::SortedIndexRank { .. } + | KvOp::SortedIndexTopK { .. } + | KvOp::SortedIndexRange { .. } + | KvOp::SortedIndexCount { .. } + | KvOp::SortedIndexScore { .. } + | KvOp::SortedIndexTxnRead { .. } => { + return Err(not_a_write("a KV read or index DDL op")); + } + } + Ok(()) +} diff --git a/nodedb/src/control/planner/calvin/tx_class/write_keys/mod.rs b/nodedb/src/control/planner/calvin/tx_class/write_keys/mod.rs new file mode 100644 index 000000000..21df48d04 --- /dev/null +++ b/nodedb/src/control/planner/calvin/tx_class/write_keys/mod.rs @@ -0,0 +1,17 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Calvin write keys of physical write plans, exhaustive over every write +//! op of every engine. + +pub mod columnar; +pub mod crdt; +pub mod document; +pub mod graph; +pub mod kv; +pub mod plan; +pub mod restore; +pub mod set; +pub mod vector; + +pub use plan::{add_plan_write_keys, task_write_keys}; +pub use set::{COLLECTION_KEY, WriteKeys, row_id_key}; diff --git a/nodedb/src/control/planner/calvin/tx_class/write_keys/plan.rs b/nodedb/src/control/planner/calvin/tx_class/write_keys/plan.rs new file mode 100644 index 000000000..e02396acd --- /dev/null +++ b/nodedb/src/control/planner/calvin/tx_class/write_keys/plan.rs @@ -0,0 +1,486 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The write keys of one physical plan, by engine family. + +#![deny(clippy::wildcard_enum_match_arm)] + +use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; +use nodedb_physical::physical_task::PhysicalTask; + +use super::set::WriteKeys; +use super::{columnar, crdt, document, graph, kv, restore, vector}; +use crate::control::planner::calvin::write_class::is_write_plan; + +/// The write keys of every write task among `tasks`. A task that writes +/// nothing adds none. +pub fn task_write_keys(tasks: &[PhysicalTask]) -> crate::Result { + let mut keys = WriteKeys::default(); + for task in tasks.iter().filter(|task| is_write_plan(&task.plan)) { + add_plan_write_keys(&mut keys, &task.plan)?; + } + Ok(keys) +} + +/// Add the Calvin write keys of `plan` to `keys`. +/// +/// Every write op of every engine is matched by name. A write with row +/// identity locks its rows. A write without one locks +/// [`super::COLLECTION_KEY`] of its collection. A graph write locks its +/// edges and nodes on their key homes. +/// +/// A write no single transaction can sequence, and a plan that writes +/// nothing, is refused with a typed error. The caller never builds a +/// transaction that the scheduler will refuse at dispatch. +pub fn add_plan_write_keys(keys: &mut WriteKeys, plan: &PhysicalPlan) -> crate::Result<()> { + match plan { + PhysicalPlan::Document(op) => document::add_keys(keys, op), + PhysicalPlan::Kv(op) => kv::add_keys(keys, op), + PhysicalPlan::Vector(op) => vector::add_keys(keys, op), + PhysicalPlan::Graph(op) => graph::add_keys(keys, op), + PhysicalPlan::Crdt(op) => crdt::add_keys(keys, op), + PhysicalPlan::Columnar(op) => columnar::add_columnar_keys(keys, op), + PhysicalPlan::Timeseries(op) => columnar::add_timeseries_keys(keys, op), + PhysicalPlan::Array(op) => columnar::add_array_keys(keys, op), + PhysicalPlan::Meta(MetaOp::RestoreRedo(batch)) => restore::add_keys(keys, batch), + PhysicalPlan::Text(_) + | PhysicalPlan::Spatial(_) + | PhysicalPlan::Query(_) + | PhysicalPlan::Meta(_) + | PhysicalPlan::ClusterArray(_) + | PhysicalPlan::ClusterEvent(_) => Err(not_a_write( + "a text, spatial, query, meta or cluster-fanned plan", + )), + } +} + +/// A plan in a Calvin write set that writes nothing. `what` names it. +pub(super) fn not_a_write(what: &str) -> crate::Error { + crate::Error::Internal { + detail: format!( + "internal invariant break: {what} writes nothing, yet it reached the Calvin \ + write-key extraction; the builders skip every plan `is_write_plan` refuses" + ), + } +} + +/// A write that no Calvin transaction sequences, with the reason. +pub(super) fn unsequenced(reason: &str) -> crate::Error { + crate::Error::BadRequest { + detail: format!("this write cannot run as a Calvin transaction: {reason}"), + } +} + +/// Two transactions, the lower one sequenced first, each taking the lock +/// keys the scheduler expands from the `TxClass` the builders produce. A +/// guard is sound only when the higher transaction waits for the lower one. +#[cfg(test)] +mod tests { + use std::collections::BTreeSet; + + use nodedb_cluster::calvin::types::SequencedTxn; + use nodedb_physical::physical_plan::{ + BatchEdge, ColumnarOp, CrdtOp, CrdtWriteVerb, DocumentOp, GraphOp, KvOp, VectorOp, + }; + use nodedb_physical::physical_task::PostSetOp; + use nodedb_types::{CollectionKey, QualifiedCollection, Surrogate}; + + use super::*; + use crate::control::cluster::calvin::scheduler::driver::helpers::expand_rw_set; + use crate::control::cluster::calvin::scheduler::{AcquireOutcome, LockKey, LockManager, TxnId}; + use crate::control::planner::calvin::tx_class::build_single_vshard_tx_class; + use crate::types::{DatabaseId, RecordHomes, TenantId, VShardId}; + + const EPOCH: u64 = 9; + const CRDT: &str = "crdt_nodes"; + const GRAPH: &str = "g"; + + fn qualified(name: &str) -> QualifiedCollection { + QualifiedCollection::new(DatabaseId::DEFAULT, name) + } + + fn task(plan: PhysicalPlan) -> PhysicalTask { + PhysicalTask { + tenant_id: TenantId::new(1), + vshard_id: VShardId::new(0), + database_id: DatabaseId::DEFAULT, + plan, + post_set_op: PostSetOp::None, + txn_id: None, + } + } + + /// The lock keys the scheduler takes for a transaction of `plans`. + fn locks(plans: Vec, position: u32) -> BTreeSet { + let tasks: Vec = plans.into_iter().map(task).collect(); + let tx_class = build_single_vshard_tx_class(&tasks, TenantId::new(1), &[]) + .expect("a valid transaction class"); + expand_rw_set(&SequencedTxn { + epoch: EPOCH, + position, + tx_class, + epoch_system_ms: 1_700_000_000_000, + epoch_vshard_txn_count: 2, + lock_owner: None, + }) + } + + /// Acquire the lower transaction's keys, then the higher one's, in + /// sequence order. `true` when the higher one waits; the lower one's + /// release then hands it every key. + fn higher_waits(lower: Vec, higher: Vec) -> bool { + let (lower_id, higher_id) = (TxnId::new(EPOCH, 0), TxnId::new(EPOCH, 1)); + let mut table = LockManager::new(); + assert!(matches!( + table.acquire(lower_id, locks(lower, 0)), + AcquireOutcome::Ready + )); + let waits = matches!( + table.acquire(higher_id, locks(higher, 1)), + AcquireOutcome::Blocked + ); + if waits { + assert_eq!(table.release(lower_id), vec![higher_id]); + } + waits + } + + fn crdt_upsert(id: &str) -> PhysicalPlan { + PhysicalPlan::Crdt(CrdtOp::DocUpsert { + collection: qualified(CRDT), + document_id: id.to_owned(), + fields_json: "{}".to_owned(), + surrogate: Surrogate::new(41), + partial: false, + verb: CrdtWriteVerb::Insert, + returning: None, + rls_filters: Vec::new(), + }) + } + + fn crdt_delete(id: &str, surrogate: Option) -> PhysicalPlan { + PhysicalPlan::Crdt(CrdtOp::DocDelete { + collection: qualified(CRDT), + document_id: id.to_owned(), + surrogate, + returning: None, + rls_filters: Vec::new(), + }) + } + + fn presence_guard(present: &[&str], absent: &[&str]) -> PhysicalPlan { + PhysicalPlan::Graph(GraphOp::NodePresenceGuard { + collection: qualified(CRDT), + vshard: CollectionKey::from_bare(DatabaseId::DEFAULT, CRDT) + .vshard() + .as_u32(), + present: present.iter().map(|id| (*id).to_owned()).collect(), + absent: absent.iter().map(|id| (*id).to_owned()).collect(), + }) + } + + fn batch_edge(collection: &str, src: &str, dst: &str) -> BatchEdge { + BatchEdge { + collection: qualified(collection), + src_id: src.to_owned(), + label: "L".to_owned(), + dst_id: dst.to_owned(), + src_surrogate: Surrogate::new(1), + dst_surrogate: Surrogate::new(2), + } + } + + fn node_guard(node: &str) -> PhysicalPlan { + PhysicalPlan::Graph(GraphOp::NodeEdgeGuard { + collection: qualified(GRAPH), + node_id: node.to_owned(), + expected: Vec::new(), + }) + } + + /// An upsert binds and stores `x` at a lower position. A delete planned + /// while `x` was unbound names it absent and carries an unbound delete. + /// Its guard must run after the upsert flushed, see `x` stored, and + /// retry; before, it ran first and the delete was lost. + #[test] + fn a_crdt_delete_of_an_unbound_document_waits_for_its_upsert() { + assert!(higher_waits( + vec![crdt_upsert("x")], + vec![presence_guard(&[], &["x"]), crdt_delete("x", None)], + )); + // The guard alone orders against the upsert, whatever the deletes + // beside it lock. + assert!(higher_waits( + vec![crdt_upsert("x")], + vec![presence_guard(&[], &["x"])] + )); + assert!(higher_waits( + vec![crdt_upsert("x")], + vec![presence_guard(&["x"], &[])] + )); + } + + /// Every writer of one CRDT document locks one key, bound or not, and + /// writers of distinct documents do not wait on each other. + #[test] + fn crdt_writers_of_one_document_serialize_bound_or_not() { + assert!(higher_waits( + vec![crdt_delete("x", None)], + vec![crdt_upsert("x")] + )); + assert!(higher_waits( + vec![crdt_upsert("x")], + vec![crdt_delete("x", Some(Surrogate::new(41)))], + )); + assert!(!higher_waits( + vec![crdt_upsert("x")], + vec![crdt_upsert("y")] + )); + } + + /// A collection-wide write can store or remove any document, so a + /// presence guard waits for it. + #[test] + fn a_presence_guard_waits_for_a_collection_wide_write() { + let truncate = PhysicalPlan::Document(DocumentOp::Truncate { + collection: qualified(CRDT), + restart_identity: false, + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }); + assert!(higher_waits( + vec![truncate], + vec![presence_guard(&[], &["x"])] + )); + } + + const DOCS: &str = "doc_rows"; + + fn doc_insert(id: &str, surrogate: u32) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::PointInsert { + collection: qualified(DOCS), + document_id: id.to_owned(), + value: Vec::new(), + if_absent: false, + surrogate: Surrogate::new(surrogate), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }) + } + + fn doc_delete(id: &str, surrogate: Option) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::PointDelete { + collection: qualified(DOCS), + document_id: id.to_owned(), + surrogate: surrogate.map(Surrogate::new), + pk_bytes: id.as_bytes().to_vec(), + returning: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + resolved_sum_targets: Vec::new(), + }) + } + + fn doc_update(id: &str, surrogate: Option) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::PointUpdate { + collection: qualified(DOCS), + document_id: id.to_owned(), + surrogate: surrogate.map(Surrogate::new), + pk_bytes: id.as_bytes().to_vec(), + updates: Vec::new(), + returning: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + resolved_sum_targets: Vec::new(), + declared_primary_key: None, + }) + } + + /// An insert binds and stores `x` at a lower position. A delete or + /// update planned while `x` was unbound must wait for it: its rebind at + /// dispatch then finds the binding on every replica. Before, it locked + /// only the collection key and can dispatch first on one replica and + /// second on another. + #[test] + fn an_unbound_document_delete_or_update_waits_for_an_insert_of_its_id() { + assert!(higher_waits( + vec![doc_insert("x", 17)], + vec![doc_delete("x", None)] + )); + assert!(higher_waits( + vec![doc_insert("x", 17)], + vec![doc_update("x", None)] + )); + let batch = PhysicalPlan::Document(DocumentOp::BatchInsert { + collection: qualified(DOCS), + documents: vec![("w".to_owned(), Vec::new()), ("x".to_owned(), Vec::new())], + surrogates: vec![Surrogate::new(16), Surrogate::new(17)], + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + deferred_sum_targets: Vec::new(), + }); + assert!(higher_waits(vec![batch], vec![doc_delete("x", None)])); + } + + /// Every point writer of one document id serializes, bound or not, and + /// writers of distinct ids do not wait on each other. + #[test] + fn document_writers_of_one_id_serialize_bound_or_not() { + assert!(higher_waits( + vec![doc_delete("x", None)], + vec![doc_insert("x", 17)] + )); + assert!(higher_waits( + vec![doc_update("x", Some(17))], + vec![doc_delete("x", Some(17))] + )); + assert!(!higher_waits( + vec![doc_insert("x", 17)], + vec![doc_insert("y", 18)] + )); + assert!(!higher_waits( + vec![doc_delete("x", Some(17))], + vec![doc_delete("y", Some(18))] + )); + } + + /// A batch edge write locks both endpoints' node pairs, so it orders + /// against a node delete's guard either way round. Before, it locked + /// only a document key of an empty collection name. + #[test] + fn an_edge_batch_orders_against_a_node_guard() { + for batch in [ + GraphOp::EdgePutBatch { + edges: vec![batch_edge(GRAPH, "a", "b")], + }, + GraphOp::EdgeDeleteBatch { + edges: vec![batch_edge(GRAPH, "b", "a")], + }, + ] { + assert!(higher_waits( + vec![PhysicalPlan::Graph(batch.clone())], + vec![node_guard("a")] + )); + assert!(higher_waits( + vec![node_guard("a")], + vec![PhysicalPlan::Graph(batch)] + )); + } + assert!(!higher_waits( + vec![PhysicalPlan::Graph(GraphOp::EdgePutBatch { + edges: vec![batch_edge(GRAPH, "c", "d")], + })], + vec![node_guard("a")] + )); + } + + /// A batch locks each edge as the single edge write does, in each edge's + /// own collection, and enlists every endpoint home the routing oracle + /// sends it to. + #[test] + fn an_edge_batch_keys_each_edge_on_its_homes() { + let edges = vec![batch_edge(GRAPH, "a", "b"), batch_edge("h", "c", "d")]; + let batch = locks( + vec![PhysicalPlan::Graph(GraphOp::EdgePutBatch { + edges: edges.clone(), + })], + 0, + ); + let singles = locks( + edges + .iter() + .map(|edge| { + PhysicalPlan::Graph(GraphOp::EdgePut { + collection: edge.collection.clone(), + src_id: edge.src_id.clone(), + label: edge.label.clone(), + dst_id: edge.dst_id.clone(), + properties: Vec::new(), + src_surrogate: edge.src_surrogate, + dst_surrogate: edge.dst_surrogate, + }) + }) + .collect(), + 0, + ); + assert_eq!(batch, singles); + + let tasks = vec![task(PhysicalPlan::Graph(GraphOp::EdgePutBatch { + edges: edges.clone(), + }))]; + let tx_class = + build_single_vshard_tx_class(&tasks, TenantId::new(1), &[]).expect("tx class"); + let mut homes: Vec = edges + .iter() + .flat_map(|edge| RecordHomes::edge(&edge.src_id, &edge.dst_id).iter()) + .collect(); + homes.sort_by_key(|vshard| vshard.as_u32()); + homes.dedup(); + assert_eq!(tx_class.participating_vshards(), homes.as_slice()); + } + + /// No write op keys an empty collection name: each one names the + /// collection it writes, or a node's label namespace. + #[test] + fn no_write_op_locks_an_unnamed_collection() { + let plans = [ + PhysicalPlan::Kv(KvOp::Expire { + collection: qualified("kv"), + key: b"k".to_vec(), + ttl_ms: 5, + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + }), + PhysicalPlan::Kv(KvOp::Persist { + collection: qualified("kv"), + key: b"k".to_vec(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + }), + PhysicalPlan::Columnar(ColumnarOp::Delete { + collection: qualified("col"), + filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + }), + PhysicalPlan::Vector(VectorOp::SparseDelete { + collection: qualified("vec"), + field_name: "f".to_owned(), + doc_id: "d".to_owned(), + }), + PhysicalPlan::Graph(GraphOp::SetNodeLabels { + node_id: "n".to_owned(), + labels: vec!["L".to_owned()], + }), + ]; + for plan in plans { + let mut keys = WriteKeys::default(); + add_plan_write_keys(&mut keys, &plan).expect("a write op has keys"); + let sets = keys.into_key_sets(); + assert!(!sets.is_empty(), "{plan:?} locks nothing"); + assert!( + sets.iter().all(|set| !set.collection().is_empty()), + "{plan:?} keys an unnamed collection: {sets:?}" + ); + } + } + + /// A write no Calvin transaction can sequence is refused by name, never + /// keyed onto some collection's vShard. + #[test] + fn an_unroutable_write_is_refused() { + let plan = PhysicalPlan::Kv(KvOp::TransferItem { + source_collection: qualified("a"), + dest_collection: qualified("b"), + item_key: b"k".to_vec(), + dest_key: b"k".to_vec(), + surrogate: Surrogate::new(5), + source_rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + dest_rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + }); + let mut keys = WriteKeys::default(); + assert!(matches!( + add_plan_write_keys(&mut keys, &plan), + Err(crate::Error::BadRequest { .. }) + )); + } +} diff --git a/nodedb/src/control/planner/calvin/tx_class/write_keys/restore.rs b/nodedb/src/control/planner/calvin/tx_class/write_keys/restore.rs new file mode 100644 index 000000000..f5ddf966b --- /dev/null +++ b/nodedb/src/control/planner/calvin/tx_class/write_keys/restore.rs @@ -0,0 +1,34 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Write keys of a RESTORE batch. +//! +//! A restored row locks what every point writer of the row locks: its +//! surrogate and its row id. A restored edge version locks what every edge +//! write locks: its surrogate pair and both endpoints' node lock pairs, on +//! both endpoint homes. So a restored write orders against every other +//! writer of the same row or edge, and against a node delete's guard. + +use nodedb_physical::physical_plan::RestoredRedo; + +use super::plan::unsequenced; +use super::set::WriteKeys; + +/// Add the write keys of the RESTORE batch `batch` to `keys`. +pub(super) fn add_keys(keys: &mut WriteKeys, batch: &RestoredRedo) -> crate::Result<()> { + if batch.rows.is_empty() && batch.edges.is_empty() { + return Err(unsequenced("a RESTORE batch carries no row and no edge")); + } + for row in &batch.rows { + keys.rows(&row.collection, [row.surrogate]); + keys.row_id(&row.collection, &row.document_id); + } + for edge in &batch.edges { + keys.edge( + &edge.collection, + &edge.src_id, + &edge.dst_id, + (edge.src_surrogate, edge.dst_surrogate), + ); + } + Ok(()) +} diff --git a/nodedb/src/control/planner/calvin/tx_class/write_keys/set.rs b/nodedb/src/control/planner/calvin/tx_class/write_keys/set.rs new file mode 100644 index 000000000..bed524f56 --- /dev/null +++ b/nodedb/src/control/planner/calvin/tx_class/write_keys/set.rs @@ -0,0 +1,178 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The accumulated Calvin write keys of a task slice. + +use std::collections::BTreeMap; + +use nodedb_cluster::calvin::types::{EngineKeySet, SortedVec}; + +use super::super::shared::node_lock_pair; +use crate::types::{RecordHomes, VShardId}; + +/// The row surrogate a write with no single row identity locks: a +/// predicate, bulk, truncate or multi-row write, or a point write whose key +/// is unbound. Every such write of a collection serializes on it. +pub const COLLECTION_KEY: u32 = 0; + +/// The lock key of the row `document_id` names, bound or not: the bytes its +/// surrogate binds under (`bind_plan`). +/// +/// Every point writer of a document row, Document or CRDT, and every +/// presence guard that names it, locks this key beside any surrogate key. +/// So a delete planned while the key was unbound waits for an insert that +/// binds it, and the scheduler's dispatch-time rebind of that delete runs +/// after the insert's binding landed, alike on every replica. +pub fn row_id_key(document_id: &str) -> Vec { + document_id.as_bytes().to_vec() +} + +/// The lock pairs and routing homes of one edge collection. +#[derive(Debug, Default)] +struct EdgeKeys { + pairs: Vec<(u32, u32)>, + homes: Vec, +} + +/// Write keys by engine and collection. Every map is ordered, so the key +/// sets it yields are identical on every node that builds them. +#[derive(Debug, Default)] +pub struct WriteKeys { + documents: BTreeMap>, + vectors: BTreeMap>, + kv: BTreeMap>>, + edges: BTreeMap, + /// Array collection to the vShards its written cells live on. + arrays: BTreeMap>, +} + +impl WriteKeys { + /// Lock the document rows `surrogates` of `collection`. An empty list + /// still enlists the collection's vShard. + pub fn rows(&mut self, collection: &str, surrogates: impl IntoIterator) { + self.documents + .entry(collection.to_owned()) + .or_default() + .extend(surrogates); + } + + /// Lock [`COLLECTION_KEY`] of `collection`. + pub fn whole_collection(&mut self, collection: &str) { + self.rows(collection, [COLLECTION_KEY]); + } + + /// Lock the vector rows `surrogates` of `collection`. An empty list + /// still enlists the collection's vShard. + pub fn vector_rows(&mut self, collection: &str, surrogates: impl IntoIterator) { + self.vectors + .entry(collection.to_owned()) + .or_default() + .extend(surrogates); + } + + /// Lock the raw keys `keys` of `collection`. An empty list still enlists + /// the collection's vShard. + pub fn kv_keys(&mut self, collection: &str, keys: impl IntoIterator>) { + self.kv + .entry(collection.to_owned()) + .or_default() + .extend(keys); + } + + /// Lock row `document_id` of `collection` by [`row_id_key`]. + pub fn row_id(&mut self, collection: &str, document_id: &str) { + self.kv_keys(collection, [row_id_key(document_id)]); + } + + /// Lock the edge `src_id -> dst_id` of `collection` by its surrogate + /// pair, and both endpoints' node lock pairs, on both endpoint homes. + pub fn edge(&mut self, collection: &str, src_id: &str, dst_id: &str, pair: (u32, u32)) { + let keys = self.edges.entry(collection.to_owned()).or_default(); + keys.pairs + .extend([pair, node_lock_pair(src_id), node_lock_pair(dst_id)]); + keys.homes.extend( + RecordHomes::edge(src_id, dst_id) + .iter() + .map(VShardId::as_u32), + ); + } + + /// Lock node `node_id` of `collection` on the node's key home. + pub fn node(&mut self, collection: &str, node_id: &str) { + let keys = self.edges.entry(collection.to_owned()).or_default(); + keys.pairs.push(node_lock_pair(node_id)); + keys.homes + .push(VShardId::from_key(node_id.as_bytes()).as_u32()); + } + + /// Enlist `vshard` for `collection` without locking a key. + pub fn home(&mut self, collection: &str, vshard: u32) { + self.edges + .entry(collection.to_owned()) + .or_default() + .homes + .push(vshard); + } + + /// Lock the whole array `collection` on `vshard`, the vShard the written + /// cells' tiles live on, and enlist that vShard. + pub fn array_cells(&mut self, collection: &str, vshard: u32) { + self.arrays + .entry(collection.to_owned()) + .or_default() + .push(vshard); + } + + /// Drop every document row key of `collection`. A dependent + /// transaction replaces them with the rows its reconnaissance predicted. + pub fn drop_documents(&mut self, collection: &str) { + self.documents.remove(collection); + } + + /// The key sets, one per engine and collection, ordered by collection. + pub fn into_key_sets(self) -> Vec { + let mut sets: Vec = Vec::new(); + sets.extend( + self.documents + .into_iter() + .map(|(collection, rows)| EngineKeySet::Document { + collection, + surrogates: SortedVec::new(rows), + }), + ); + sets.extend( + self.vectors + .into_iter() + .map(|(collection, rows)| EngineKeySet::Vector { + collection, + surrogates: SortedVec::new(rows), + }), + ); + sets.extend( + self.kv + .into_iter() + .map(|(collection, keys)| EngineKeySet::Kv { + collection, + keys: SortedVec::new(keys), + }), + ); + sets.extend( + self.edges + .into_iter() + .map(|(collection, keys)| EngineKeySet::Edge { + collection, + edges: SortedVec::new(keys.pairs), + home_vshards: SortedVec::new(keys.homes), + }), + ); + sets.extend( + self.arrays + .into_iter() + .map(|(collection, vshards)| EngineKeySet::Array { + collection, + vshards: SortedVec::new(vshards), + }), + ); + sets.sort_by(|a, b| a.collection().cmp(b.collection())); + sets + } +} diff --git a/nodedb/src/control/planner/calvin/tx_class/write_keys/vector.rs b/nodedb/src/control/planner/calvin/tx_class/write_keys/vector.rs new file mode 100644 index 000000000..e3fb7ccde --- /dev/null +++ b/nodedb/src/control/planner/calvin/tx_class/write_keys/vector.rs @@ -0,0 +1,105 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Vector-engine write keys. +//! +//! A single-row op locks the same `(collection, surrogate)` the +//! write-admission gate's `vector_point_key` locks. + +#![deny(clippy::wildcard_enum_match_arm)] + +use nodedb_physical::physical_plan::{VectorOp, VectorWriteTargets}; + +use super::plan::{not_a_write, unsequenced}; +use super::set::WriteKeys; + +/// Add the write keys of vector op `op` to `keys`. +pub(super) fn add_keys(keys: &mut WriteKeys, op: &VectorOp) -> crate::Result<()> { + match op { + VectorOp::Insert { + collection, + surrogate, + .. + } + | VectorOp::DirectInsert { + collection, + surrogate, + .. + } + | VectorOp::DirectInsertIfAbsent { + collection, + surrogate, + .. + } + | VectorOp::DirectUpsert { + collection, + surrogate, + .. + } => keys.vector_rows(collection.as_str(), [surrogate.as_u32()]), + // The document row a multi-vector field belongs to. + VectorOp::MultiVectorInsert { + collection, + document_surrogate, + .. + } + | VectorOp::MultiVectorDelete { + collection, + document_surrogate, + .. + } => keys.vector_rows(collection.as_str(), [document_surrogate.as_u32()]), + // A key unbound in its database names no row: it takes the + // collection key. + VectorOp::DeleteBySurrogate { + collection, + surrogate, + .. + } => match surrogate { + Some(surrogate) => keys.vector_rows(collection.as_str(), [surrogate.as_u32()]), + None => keys.whole_collection(collection.as_str()), + }, + VectorOp::BatchInsert { + collection, + surrogates, + .. + } => keys.vector_rows(collection.as_str(), surrogates.iter().map(|s| s.as_u32())), + VectorOp::DirectDelete { + collection, + targets, + .. + } + | VectorOp::DirectUpdate { + collection, + targets, + .. + } => match targets { + VectorWriteTargets::Surrogates(surrogates) => { + keys.vector_rows(collection.as_str(), surrogates.iter().map(|s| s.as_u32())); + } + VectorWriteTargets::Predicate(_) => keys.whole_collection(collection.as_str()), + }, + // No row surrogate on the plan: a node-id delete, a sparse entry + // keyed by its document id, and a truncate. + VectorOp::Delete { collection, .. } + | VectorOp::SparseInsert { collection, .. } + | VectorOp::SparseDelete { collection, .. } + | VectorOp::DirectTruncate { collection, .. } => keys.whole_collection(collection.as_str()), + VectorOp::ResolvedDirectWrite { .. } => { + return Err(unsequenced( + "a resolved governed vector write is proposed by the write-resolve orchestrator", + )); + } + VectorOp::ResolveDirectWrite(_) + | VectorOp::Search { .. } + | VectorOp::MultiSearch { .. } + | VectorOp::SetParams { .. } + | VectorOp::DropIndex { .. } + | VectorOp::QueryStats { .. } + | VectorOp::Seal { .. } + | VectorOp::CompactIndex { .. } + | VectorOp::Rebuild { .. } + | VectorOp::SparseSearch { .. } + | VectorOp::MultiVectorScoreSearch { .. } => { + return Err(not_a_write("a vector read or index DDL op")); + } + } + Ok(()) +} diff --git a/nodedb/src/control/planner/calvin/write_class.rs b/nodedb/src/control/planner/calvin/write_class.rs index 26fcb832b..8e8072c7b 100644 --- a/nodedb/src/control/planner/calvin/write_class.rs +++ b/nodedb/src/control/planner/calvin/write_class.rs @@ -7,19 +7,21 @@ //! decide write-key-set membership. Mirrors `plan_vshard`: every op in the //! eight write-capable engines is matched explicitly `true`/`false` (no //! wildcard), so a new op variant is a compile error here. Text/Spatial/ -//! Query/Meta stay one blanket `false` arm each (`NotAWrite` in -//! `plan_vshard`), still exhaustive over `PhysicalPlan`. +//! Query stay one blanket `false` arm each (`NotAWrite` in `plan_vshard`), +//! still exhaustive over `PhysicalPlan`. Meta is `false` except a RESTORE +//! batch, which routes to the vShard it names. //! //! Does NOT delegate to `plan_is_write` (`Permission::Write`): several //! `Permission::Write` variants carry no vshard to lock in `plan_vshard` //! (index-metadata ops, cross-collection writes like `Merge`/`TransferItem`, -//! Text/Spatial write ops, most `MetaOp` writes) and would misclassify as +//! Text/Spatial write ops, most `MetaOp` writes) and will misclassify as //! Calvin writes, turning a routing gap into an aborted transaction. #![deny(clippy::wildcard_enum_match_arm)] use nodedb_physical::physical_plan::{ - ArrayOp, ColumnarOp, CrdtOp, DocumentOp, GraphOp, KvOp, PhysicalPlan, TimeseriesOp, VectorOp, + ArrayOp, ColumnarOp, CrdtOp, DocumentOp, GraphOp, KvOp, MetaOp, PhysicalPlan, TimeseriesOp, + VectorOp, }; fn document_is_write(op: &DocumentOp) -> bool { @@ -42,7 +44,7 @@ fn document_is_write(op: &DocumentOp) -> bool { | DocumentOp::ApplyBalanceDelta { .. } // Mutates the rows its mutation list names, like any other write. | DocumentOp::ResolvedWrite { .. } => true, - // Read-only: reports what the wrapped write would apply, mutates nothing. + // Read-only: reports what the wrapped write will apply, mutates nothing. DocumentOp::ResolveWrite(_) | DocumentOp::PointGet { .. } | DocumentOp::Scan { .. } @@ -68,7 +70,7 @@ fn document_is_write(op: &DocumentOp) -> bool { /// Both are real writes that must enter Calvin's write-key set — what they /// must NOT do is answer the client: a derived participant's response /// describes a row the statement never named, so shaping `CommandComplete` -/// from it would report the wrong count. Named once here rather than as an +/// from it will report the wrong count. Named once here rather than as an /// inline negation, after the balance write (modelled on the implicit graph /// edge) failed to inherit an ad hoc `!matches!` check and raced the source /// write to deposit the statement's response. @@ -147,7 +149,7 @@ fn kv_is_write(op: &KvOp) -> bool { | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } | KvOp::SortedIndexTxnRead { .. } - // Read-only: reports what a governed write would apply, mutates + // Read-only: reports what a governed write will apply, mutates // nothing, and is `NotAWrite` in `plan_vshard`. | KvOp::ResolveWrite(_) // `Permission::Write` but `NotAWrite` in `plan_vshard` — index @@ -160,7 +162,7 @@ fn kv_is_write(op: &KvOp) -> bool { // (cross-collection write, no enforced co-location). | KvOp::TransferItem { .. } // `Permission::Write` but `Unroutable` in `plan_vshard` — its - // mutations may span two collections, so no single vshard to lock on. + // mutations can span two collections, so no single vshard to lock on. | KvOp::ResolvedWrite { .. } => false, } } @@ -183,7 +185,7 @@ fn vector_is_write(op: &VectorOp) -> bool { | VectorOp::DirectUpdate { .. } // Mutates the rows its mutation list names, like any other write. | VectorOp::ResolvedDirectWrite { .. } => true, - // Read-only: reports what the wrapped write would apply, mutates nothing. + // Read-only: reports what the wrapped write will apply, mutates nothing. VectorOp::ResolveDirectWrite(_) | VectorOp::Search { .. } | VectorOp::MultiSearch { .. } @@ -208,7 +210,12 @@ fn graph_is_write(op: &GraphOp) -> bool { | GraphOp::EdgeDelete { .. } | GraphOp::EdgeDeleteBatch { .. } | GraphOp::SetNodeLabels { .. } - | GraphOp::RemoveNodeLabels { .. } => true, + | GraphOp::RemoveNodeLabels { .. } + | GraphOp::TruncateEdges { .. } => true, + // A node delete's guards write nothing, but they take part in the + // delete's transaction on their vShards, so they route and sequence + // like writes. + GraphOp::NodeEdgeGuard { .. } | GraphOp::NodePresenceGuard { .. } => true, // The resolve pass writes nothing; the delete it decides is proposed // separately by the write-resolve orchestrator. GraphOp::ResolveEdgeDelete(_) @@ -226,7 +233,8 @@ fn graph_is_write(op: &GraphOp) -> bool { | GraphOp::WccSuperstep(_) | GraphOp::TemporalNeighbors { .. } | GraphOp::TemporalAlgorithm { .. } - | GraphOp::Stats { .. } => false, + | GraphOp::Stats { .. } + | GraphOp::NodePresenceRead { .. } => false, } } @@ -290,7 +298,7 @@ fn array_is_write(op: &ArrayOp) -> bool { ArrayOp::OpenArray { .. } | ArrayOp::Compact { .. } | ArrayOp::DropArray { .. } - | ArrayOp::RestoreArrayDrop { .. } + | ArrayOp::RekeyArray { .. } | ArrayOp::PurgeArrayDrop { .. } | ArrayOp::Slice { .. } | ArrayOp::Project { .. } @@ -315,7 +323,10 @@ pub fn is_write_plan(plan: &PhysicalPlan) -> bool { PhysicalPlan::Columnar(op) => columnar_is_write(op), PhysicalPlan::Crdt(op) => crdt_is_write(op), PhysicalPlan::Array(op) => array_is_write(op), - // Reads, scans, queries, meta, spatial, text: none of these + // A RESTORE batch writes the rows and edges it names, on the vShard + // it names. + PhysicalPlan::Meta(MetaOp::RestoreRedo(_)) => true, + // Reads, scans, queries, other meta, spatial, text: none of these // families carry a Calvin-lockable write in `plan_vshard`. PhysicalPlan::Spatial(_) | PhysicalPlan::Text(_) @@ -353,7 +364,7 @@ mod tests { delta: Vec::new(), peer_id: 0, mutation_id: 0, - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), provenance: None, constraint_version_required: 0, expected_frontier_digest: None, @@ -672,7 +683,7 @@ mod tests { /// is. /// /// Both are appended by the Control Plane alongside a statement they do not - /// appear in, and neither may own that statement's applied response: the + /// appear in, and neither can own that statement's applied response: the /// `CommandComplete` tag is shaped from ONE deposited response, so a derived /// participant winning the deposit hands the user's `INSERT` a count that /// belongs to a row the statement never named. @@ -682,7 +693,7 @@ mod tests { } /// The user's own write is not, so it remains the participant that deposits. - /// Without this the fix would leave every cross-shard statement with no applied + /// Without a depositing participant, a cross-shard statement has no applied /// response at all. #[test] fn the_users_own_write_is_not_a_derived_side_effect() { diff --git a/nodedb/src/control/planner/catalog_adapter/adapter.rs b/nodedb/src/control/planner/catalog_adapter/adapter.rs index 6a80c1489..c6e0485ae 100644 --- a/nodedb/src/control/planner/catalog_adapter/adapter.rs +++ b/nodedb/src/control/planner/catalog_adapter/adapter.rs @@ -28,10 +28,10 @@ pub struct OriginCatalog { pub(super) database_id: DatabaseId, pub(super) retention_policy_registry: Option>, - /// Array catalog handle. When `None`, `lookup_array` returns - /// `None` for every name — used by sub-planners that don't own - /// array state. - pub(super) array_catalog: Option, + /// The node's array catalog. Every adapter resolves array names against + /// it, so a sub-planner plans array DML exactly as a client statement + /// does. + pub(super) array_catalog: crate::control::array_catalog::ArrayCatalogHandle, /// Optional reference to the host's drain tracker. When /// present, `get_collection` checks for an active drain /// on each descriptor it reads and returns @@ -77,6 +77,7 @@ impl OriginCatalog { /// `QueryLeaseScope`. pub fn new( credentials: Arc, + array_catalog: crate::control::array_catalog::ArrayCatalogHandle, tenant_id: u64, database_id: DatabaseId, retention_policy_registry: Option< @@ -90,7 +91,7 @@ impl OriginCatalog { retention_policy_registry, drain_tracker: None, recorded_versions: Mutex::new(DescriptorVersionSet::new()), - array_catalog: None, + array_catalog, sequence_registry: None, session_sequences: None, } @@ -116,7 +117,7 @@ impl OriginCatalog { retention_policy_registry, drain_tracker: Some(Arc::clone(&shared.lease_drain)), recorded_versions: Mutex::new(DescriptorVersionSet::new()), - array_catalog: Some(shared.array_catalog.clone()), + array_catalog: shared.array_catalog.clone(), sequence_registry: Some(Arc::clone(&shared.sequence_registry)), session_sequences: None, } diff --git a/nodedb/src/control/planner/catalog_adapter/sql_catalog_impl.rs b/nodedb/src/control/planner/catalog_adapter/sql_catalog_impl.rs index 4bac4b2e6..403437863 100644 --- a/nodedb/src/control/planner/catalog_adapter/sql_catalog_impl.rs +++ b/nodedb/src/control/planner/catalog_adapter/sql_catalog_impl.rs @@ -66,7 +66,7 @@ impl SqlCatalog for OriginCatalog { // pre-upgrade row the GC sweeper has not yet adopted a time // for — see `event::collection_gc::policy::PurgeDecision`, // which never purges such a row). Report `u64::MAX` rather - // than `0 + retention`, which would falsely tell the caller + // than `0 + retention`, which will falsely tell the caller // the row is already past its window. let retention_expires_at_ns = if stored.deactivated_at_ns == 0 { u64::MAX @@ -106,18 +106,18 @@ impl SqlCatalog for OriginCatalog { } // Drain observation: if a DDL is currently draining - // this descriptor at the version we just read, return + // this descriptor at the version we read, return // `RetryableSchemaChanged` so the pgwire handler's // retry loop re-plans. Without this check the planner - // would compile a plan against a version that's about + // will compile a plan against a version that's about // to be retired, and only the post-plan lease acquisition - // would notice — or (worse) it would succeed on first - // holder because the drain finished just before the + // will notice — or (worse) it will succeed on first + // holder because the drain finished right before the // refcount check. Catching it here keeps the common case // cheap; the acquisition sits inside the same retry unit // for the drain that starts after this read. // - // Leases themselves are NOT acquired here anymore. + // Leases themselves are NOT acquired here. // The handler calls // `SharedState::acquire_plan_lease_scope` after // planning finishes (or after a cache hit returns a @@ -125,12 +125,16 @@ impl SqlCatalog for OriginCatalog { // refcounts, performs a single raft acquire per // descriptor (on first-holder), and returns a // `QueryLeaseScope` the handler holds through execute. - if let Some(drain) = &self.drain_tracker - && drain.is_draining(&descriptor_id, version) - { - return Err(SqlCatalogError::RetryableSchemaChanged { - descriptor: format!("collection {name}"), - }); + if let Some(drain) = &self.drain_tracker { + let owners = drain.draining_owners(&descriptor_id, version); + if !owners.is_empty() { + return Err(SqlCatalogError::RetryableSchemaChanged { + descriptor: format!( + "collection {name}: {}", + crate::control::lease::drain_owner_list(&owners) + ), + }); + } } let (engine, columns, primary_key) = convert_collection_type(&stored); @@ -198,7 +202,7 @@ impl SqlCatalog for OriginCatalog { .flatten() .filter(|stored| stored.is_active)?; - // A folded regclass literal is a schema dependency just like a scan of + // A folded regclass literal is a schema dependency like a scan of // the relation. Record it so dropping or altering the target invalidates // a cached physical plan that embeds its OID. let descriptor_id = DescriptorId::new( @@ -264,9 +268,13 @@ impl SqlCatalog for OriginCatalog { ArrayAttrAst, ArrayAttrType, ArrayDimAst, ArrayDimType, ArrayDomainBound, }; - let handle = self.array_catalog.as_ref()?; let entry = { - let cat = handle.read().ok()?; + // A poisoned lock still holds a consistent mirror: every writer + // replaces whole entries. Every other reader recovers it too. + let cat = self + .array_catalog + .read() + .unwrap_or_else(|poisoned| poisoned.into_inner()); cat.lookup_by_name_in_database( nodedb_types::TenantId::new(self.tenant_id), self.database_id, diff --git a/nodedb/src/control/planner/context/catalog_inputs.rs b/nodedb/src/control/planner/context/catalog_inputs.rs index 38004fe09..d7b863b6d 100644 --- a/nodedb/src/control/planner/context/catalog_inputs.rs +++ b/nodedb/src/control/planner/context/catalog_inputs.rs @@ -13,6 +13,9 @@ use crate::control::security::credential::CredentialStore; #[derive(Clone)] pub(super) struct CatalogInputs { pub(super) credentials: Arc, + /// The node's array catalog, which every adapter resolves array names + /// against. + pub(super) array_catalog: crate::control::array_catalog::ArrayCatalogHandle, pub(super) shared: Option>, pub(super) retention_policy_registry: Option>, @@ -36,6 +39,7 @@ impl CatalogInputs { } else { super::super::catalog_adapter::OriginCatalog::new( Arc::clone(&self.credentials), + self.array_catalog.clone(), tenant_id, database_id, self.retention_policy_registry.clone(), diff --git a/nodedb/src/control/planner/context/mod.rs b/nodedb/src/control/planner/context/mod.rs index 24e8fa68a..90d9a9229 100644 --- a/nodedb/src/control/planner/context/mod.rs +++ b/nodedb/src/control/planner/context/mod.rs @@ -12,5 +12,5 @@ pub mod security; pub mod system_security; pub use query::{PlanSqlWithRlsParams, QueryContext, SYSTEM_FUNCTION_NAMES}; -pub use security::PlanSecurityContext; +pub use security::{PermissionTreeSource, PlanSecurityContext}; pub use system_security::SystemPlanSecurity; diff --git a/nodedb/src/control/planner/context/query/context.rs b/nodedb/src/control/planner/context/query/context.rs index f9fdc30db..ce16fe3cc 100644 --- a/nodedb/src/control/planner/context/query/context.rs +++ b/nodedb/src/control/planner/context/query/context.rs @@ -4,8 +4,6 @@ use std::sync::Arc; -use crate::control::security::credential::CredentialStore; - use super::super::catalog_inputs::CatalogInputs; /// Query context for the Control Plane. @@ -18,14 +16,14 @@ use super::super::catalog_inputs::CatalogInputs; /// The catalog adapter is **constructed per-plan**, not cached on /// the context, because the adapter's `recorded_versions` field /// is per-plan state. Holding a shared adapter across concurrent -/// plans would interleave their recorded descriptor sets and +/// plans will interleave their recorded descriptor sets and /// poison the plan-cache keys. The context stores the inputs /// needed to construct an adapter (credentials, optional /// `Weak` for lease integration, tenant id, /// retention registry) and builds a fresh one on every planning /// call. pub struct QueryContext { - pub(super) catalog_inputs: Option, + pub(super) catalog_inputs: CatalogInputs, /// Retention policy registry for auto-tier routing. pub(super) retention_registry: Option>, @@ -35,11 +33,9 @@ pub struct QueryContext { pub(super) array_catalog: Option, /// WAL allocator — required by array DML for `wal_lsn` allocation. pub(super) wal: Option>, - /// Surrogate assigner — `Some` when the planner has access to - /// `SharedState` (production path); `None` only for legacy - /// `QueryContext::new()` test fixtures that never lower to - /// surrogate-bearing variants. - pub(super) surrogate_assigner: Option>, + /// The node's surrogate assigner. Every write the planner lowers binds + /// its rows through it. + pub(super) surrogate_assigner: Arc, /// Cluster mode flag — `true` when the node has a live cluster /// topology. Passed into `ConvertContext` so array converters can /// emit `ClusterArray` variants instead of local `Array` variants. @@ -108,23 +104,46 @@ pub struct QueryContext { } impl QueryContext { - /// Create a new query context without catalog integration. - pub fn new() -> Self { + /// Create a query context from `SharedState` without lease + /// integration. Used by internal sub-planners (neutral DDL + /// readback queries — check constraints, type guards, ANALYZE, + /// COPY TO, materialized view refresh — plus procedural DML, + /// event trigger dispatch, and graph scatter-gather hops) that + /// run inside a handler whose outer query already acquired + /// leases. Re-acquiring via a sub-planner is redundant — + /// the lease store's fast path returns instantly anyway, + /// but going through the sub-planner without a direct + /// `Arc` reference will require threading one + /// through every call site. + pub fn for_state(state: &crate::control::state::SharedState) -> Self { + let retention = Some(Arc::clone(&state.retention_policy_registry)); Self { - catalog_inputs: None, - retention_registry: None, - array_catalog: None, + catalog_inputs: CatalogInputs { + credentials: Arc::clone(&state.credentials), + array_catalog: state.array_catalog.clone(), + shared: None, + retention_policy_registry: retention.clone(), + }, + retention_registry: retention, + // `SharedState` owns the array catalog from construction, so every + // context can plan array DML. + array_catalog: Some(state.array_catalog.clone()), wal: None, - surrogate_assigner: None, - cluster_enabled: false, - bitemporal_retention_registry: None, + surrogate_assigner: Arc::clone(&state.surrogate_assigner), + cluster_enabled: state.cluster_topology.is_some(), + bitemporal_retention_registry: Some(Arc::clone(&state.bitemporal_retention_registry)), + // max_vector_dim starts at 0 (unlimited); connection handlers + // call set_max_vector_dim before each planning call. max_vector_dim: std::sync::atomic::AtomicU32::new(0), force_shuffle_join: std::sync::atomic::AtomicBool::new(false), shuffle_num_parts: std::sync::atomic::AtomicU32::new(0), force_shuffle_agg: std::sync::atomic::AtomicBool::new(false), shuffle_agg_num_parts: std::sync::atomic::AtomicU32::new(0), + // Seeded from the node's configured tuning default. Connection + // handlers re-resolve it per request (session override OR this + // default) via `set_broadcast_threshold_bytes`. broadcast_threshold_bytes: std::sync::atomic::AtomicUsize::new( - default_broadcast_threshold_bytes(), + state.tuning.cluster_transport.broadcast_threshold_bytes, ), shuffle_agg_threshold: std::sync::atomic::AtomicUsize::new( super::tuning::DEFAULT_SHUFFLE_AGG_THRESHOLD, @@ -133,40 +152,6 @@ impl QueryContext { } } - /// Create a query context from `SharedState` without lease - /// integration. Used by internal sub-planners (neutral DDL - /// readback queries — check constraints, type guards, ANALYZE, - /// COPY TO, materialized view refresh — plus procedural DML, - /// event trigger dispatch, and graph scatter-gather hops) that - /// run inside a handler whose outer query already acquired - /// leases. Re-acquiring via a sub-planner would be redundant — - /// the lease store's fast path would return instantly anyway, - /// but going through the sub-planner without a direct - /// `Arc` reference would require threading one - /// through every call site. - pub fn for_state(state: &crate::control::state::SharedState) -> Self { - let mut ctx = Self::with_catalog( - Arc::clone(&state.credentials), - Some(Arc::clone(&state.retention_policy_registry)), - ); - ctx.surrogate_assigner = Some(Arc::clone(&state.surrogate_assigner)); - ctx.cluster_enabled = state.cluster_topology.is_some(); - ctx.bitemporal_retention_registry = Some(Arc::clone(&state.bitemporal_retention_registry)); - // max_vector_dim starts at 0 (unlimited); connection handlers call - // set_max_vector_dim before each planning call. - ctx.max_vector_dim - .store(0, std::sync::atomic::Ordering::Relaxed); - // Seed the broadcast-vs-shuffle byte threshold from the node's - // configured tuning default. Connection handlers re-resolve it per - // request (session override OR this default) via - // `set_broadcast_threshold_bytes`. - ctx.broadcast_threshold_bytes.store( - state.tuning.cluster_transport.broadcast_threshold_bytes, - std::sync::atomic::Ordering::Relaxed, - ); - ctx - } - /// Create a query context with descriptor lease integration. /// Used by the top-level pgwire dispatch so every user /// query's plan acquires descriptor leases before execution. @@ -175,15 +160,16 @@ impl QueryContext { pub fn for_state_with_lease(state: &Arc) -> Self { let retention = Some(Arc::clone(&state.retention_policy_registry)); Self { - catalog_inputs: Some(CatalogInputs { + catalog_inputs: CatalogInputs { credentials: Arc::clone(&state.credentials), + array_catalog: state.array_catalog.clone(), shared: Some(Arc::downgrade(state)), retention_policy_registry: retention.clone(), - }), + }, retention_registry: retention, array_catalog: Some(state.array_catalog.clone()), wal: Some(Arc::clone(&state.wal)), - surrogate_assigner: Some(Arc::clone(&state.surrogate_assigner)), + surrogate_assigner: Arc::clone(&state.surrogate_assigner), cluster_enabled: state.cluster_topology.is_some(), bitemporal_retention_registry: Some(Arc::clone(&state.bitemporal_retention_registry)), // max_vector_dim is tenant-specific; callers supply it via @@ -204,44 +190,6 @@ impl QueryContext { session_sequences: std::sync::Mutex::new(None), } } - - /// Create a query context with catalog integration but no - /// lease acquisition. Used by `for_state` and by callers - /// that construct a context without an `Arc`. - pub fn with_catalog( - credentials: Arc, - retention_policy_registry: Option< - Arc, - >, - ) -> Self { - let catalog_inputs = Some(CatalogInputs { - credentials, - shared: None, - retention_policy_registry: retention_policy_registry.clone(), - }); - - Self { - catalog_inputs, - retention_registry: retention_policy_registry, - array_catalog: None, - wal: None, - surrogate_assigner: None, - cluster_enabled: false, - bitemporal_retention_registry: None, - max_vector_dim: std::sync::atomic::AtomicU32::new(0), - force_shuffle_join: std::sync::atomic::AtomicBool::new(false), - shuffle_num_parts: std::sync::atomic::AtomicU32::new(0), - force_shuffle_agg: std::sync::atomic::AtomicBool::new(false), - shuffle_agg_num_parts: std::sync::atomic::AtomicU32::new(0), - broadcast_threshold_bytes: std::sync::atomic::AtomicUsize::new( - default_broadcast_threshold_bytes(), - ), - shuffle_agg_threshold: std::sync::atomic::AtomicUsize::new( - super::tuning::DEFAULT_SHUFFLE_AGG_THRESHOLD, - ), - session_sequences: std::sync::Mutex::new(None), - } - } } impl QueryContext { @@ -267,16 +215,3 @@ impl QueryContext { .clone() } } - -/// The node-default broadcast threshold (bytes) for fixtures that have no -/// `SharedState` tuning to read. Sourced from `ClusterTransportTuning::default()` -/// so the planner default and the config default never drift. -pub(super) fn default_broadcast_threshold_bytes() -> usize { - nodedb_types::config::tuning::ClusterTransportTuning::default().broadcast_threshold_bytes -} - -impl Default for QueryContext { - fn default() -> Self { - Self::new() - } -} diff --git a/nodedb/src/control/planner/context/query/convert_plans.rs b/nodedb/src/control/planner/context/query/convert_plans.rs new file mode 100644 index 000000000..25ddc71cd --- /dev/null +++ b/nodedb/src/control/planner/context/query/convert_plans.rs @@ -0,0 +1,60 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Conversion of planned SQL to physical tasks for [`QueryContext`], with +//! every surrogate key resolved at its home first. + +use std::sync::Arc; +use std::sync::atomic::Ordering; + +use nodedb_physical::physical_task::PhysicalTask; +use nodedb_sql::types::SqlPlan; + +use super::QueryContext; +use crate::control::planner::sql_plan_convert::{ConvertContext, PlanningPurpose, convert_bound}; + +impl QueryContext { + /// The conversion context for one plan batch, read from this context's + /// wiring and per-request knobs. + fn convert_context( + &self, + purpose: PlanningPurpose, + database_id: crate::types::DatabaseId, + tenant_id: crate::types::TenantId, + ) -> ConvertContext { + ConvertContext { + purpose, + retention_registry: self.retention_registry.clone(), + array_catalog: self.array_catalog.clone(), + credentials: Some(Arc::clone(&self.catalog_inputs.credentials)), + wal: self.wal.clone(), + surrogate_assigner: self.surrogate_assigner.clone(), + cluster_enabled: self.cluster_enabled, + bitemporal_retention_registry: self.bitemporal_retention_registry.clone(), + max_vector_dim: self.max_vector_dim.load(Ordering::Relaxed), + force_shuffle_join: self.force_shuffle_join.load(Ordering::Relaxed), + shuffle_num_parts: self.shuffle_num_parts.load(Ordering::Relaxed) as usize, + force_shuffle_agg: self.force_shuffle_agg.load(Ordering::Relaxed), + shuffle_agg_num_parts: self.shuffle_agg_num_parts.load(Ordering::Relaxed) as usize, + broadcast_threshold_bytes: self.broadcast_threshold_bytes.load(Ordering::Relaxed), + shuffle_agg_threshold: self.shuffle_agg_threshold.load(Ordering::Relaxed), + database_id, + tenant_id, + prefetched: Default::default(), + } + } + + /// Convert `plans` to physical tasks. Every surrogate the conversion uses + /// is drawn or answered at its collection home first, awaited, so the + /// synchronous conversion finds each answer in hand and never parks a + /// runtime worker. + pub(super) async fn convert_prefetched( + &self, + plans: &[SqlPlan], + purpose: PlanningPurpose, + database_id: crate::types::DatabaseId, + tenant_id: crate::types::TenantId, + ) -> crate::Result> { + let mut ctx = self.convert_context(purpose, database_id, tenant_id); + convert_bound(plans, tenant_id, &mut ctx).await + } +} diff --git a/nodedb/src/control/planner/context/query/mod.rs b/nodedb/src/control/planner/context/query/mod.rs index 9927f5e9a..b834f5ec1 100644 --- a/nodedb/src/control/planner/context/query/mod.rs +++ b/nodedb/src/control/planner/context/query/mod.rs @@ -1,6 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 mod context; +mod convert_plans; mod functions; mod planning; mod tuning; diff --git a/nodedb/src/control/planner/context/query/planning.rs b/nodedb/src/control/planner/context/query/planning.rs index 0078b683a..77aac1272 100644 --- a/nodedb/src/control/planner/context/query/planning.rs +++ b/nodedb/src/control/planner/context/query/planning.rs @@ -77,7 +77,7 @@ impl QueryContext { /// `returning_items` is the raw text after a DML `RETURNING` keyword. It /// resolves here against the planned target: the announced output schema /// carries the projection, and every task carries the Data-Plane spec. - fn plan_with_nodedb_sql_for_purpose( + async fn plan_with_nodedb_sql_for_purpose( &self, sql: &str, tenant_id: crate::types::TenantId, @@ -90,24 +90,18 @@ impl QueryContext { crate::control::planner::descriptor_set::DescriptorVersionSet, nodedb_sql::types::PlanCacheEligibility, )> { - let inputs = match &self.catalog_inputs { - Some(i) => i, - None => { - return Err(crate::Error::PlanError { - detail: "no catalog available for SQL planning".into(), - }); - } - }; + let inputs = &self.catalog_inputs; // Fresh adapter per plan call: the adapter's // `recorded_versions` field is per-plan state, and // two concurrent plans through a shared QueryContext - // would otherwise interleave their recorded sets. + // will otherwise interleave their recorded sets. let catalog = if purpose == PlanningPurpose::Metadata { // Metadata requests intentionally do not participate in descriptor // lease admission; they only need a stable catalog snapshot for // authorization and response shaping. crate::control::planner::catalog_adapter::OriginCatalog::new( Arc::clone(&inputs.credentials), + inputs.array_catalog.clone(), tenant_id.as_u64(), database_id, inputs.retention_policy_registry.clone(), @@ -138,43 +132,6 @@ impl QueryContext { .collect::>() .map_err(|error| map_plan_error(error, tenant_id))?; let version_set = catalog.take_recorded_versions(); - let ctx = crate::control::planner::sql_plan_convert::ConvertContext { - purpose, - retention_registry: self.retention_registry.clone(), - array_catalog: self.array_catalog.clone(), - credentials: self - .catalog_inputs - .as_ref() - .map(|i| Arc::clone(&i.credentials)), - wal: self.wal.clone(), - surrogate_assigner: self.surrogate_assigner.clone(), - cluster_enabled: self.cluster_enabled, - bitemporal_retention_registry: self.bitemporal_retention_registry.clone(), - max_vector_dim: self - .max_vector_dim - .load(std::sync::atomic::Ordering::Relaxed), - force_shuffle_join: self - .force_shuffle_join - .load(std::sync::atomic::Ordering::Relaxed), - shuffle_num_parts: self - .shuffle_num_parts - .load(std::sync::atomic::Ordering::Relaxed) as usize, - force_shuffle_agg: self - .force_shuffle_agg - .load(std::sync::atomic::Ordering::Relaxed), - shuffle_agg_num_parts: self - .shuffle_agg_num_parts - .load(std::sync::atomic::Ordering::Relaxed) - as usize, - broadcast_threshold_bytes: self - .broadcast_threshold_bytes - .load(std::sync::atomic::Ordering::Relaxed), - shuffle_agg_threshold: self - .shuffle_agg_threshold - .load(std::sync::atomic::Ordering::Relaxed), - database_id, - tenant_id, - }; let (output_schema, returning) = resolve_returning_and_output_schema( &plans, returning_items, @@ -184,8 +141,9 @@ impl QueryContext { )?; let cache_eligibility = crate::control::planner::sql_plan_convert::batch_cache_eligibility(&plans); - let mut tasks = - crate::control::planner::sql_plan_convert::convert(&plans, tenant_id, &ctx)?; + let mut tasks = self + .convert_prefetched(&plans, purpose, database_id, tenant_id) + .await?; // Attached before the tasks leave: RLS injection, caching, expansion, // and dispatch all read the plan after this point. if let Some(clause) = &returning { @@ -308,23 +266,22 @@ impl QueryContext { nodedb_sql::types::PlanCacheEligibility, )> { let (mut tasks, output_schema, mut version_set, cache_eligibility) = self - .plan_with_nodedb_sql_for_purpose( - sql, - tenant_id, - database_id, - purpose, - returning_items, - )?; + .plan_with_nodedb_sql_for_purpose(sql, tenant_id, database_id, purpose, returning_items) + .await?; // Versions read BEFORE injection, never after: injection reads live // policy/grant state under its own lock, and a mutation racing in - // between would otherwise let a post-injection read stamp a version + // between will otherwise let a post-injection read stamp a version // newer than what was actually filtered against, making a plan built // from stale state compare as fresh forever. A pre-injection read // only ever under-states freshness (an extra harmless replan), never // over-states it. - let permission_tree_version = sec - .permission_cache + // + // The permission tree is read here, after planning's last await, so + // no lock is held across a request to a surrogate's collection home. + let permission_view = sec.permission_tree.read().await; + let permission_cache = permission_view.as_deref(); + let permission_tree_version = permission_cache .map(|c| c.tenant_version(tenant_id.as_u64())) .unwrap_or(0); let rls_version = sec.rls_store.tenant_version(tenant_id.as_u64()); @@ -341,7 +298,7 @@ impl QueryContext { )?; // Inject permission tree filters (hierarchical ACL). - if let Some(cache) = sec.permission_cache { + if let Some(cache) = permission_cache { crate::control::planner::rls_injection::inject_permission_tree( &mut tasks, cache, sec.auth, )?; @@ -397,14 +354,7 @@ impl QueryContext { OutputSchema, crate::control::planner::descriptor_set::DescriptorVersionSet, )> { - let inputs = match &self.catalog_inputs { - Some(i) => i, - None => { - return Err(crate::Error::PlanError { - detail: "no catalog available for SQL planning".into(), - }); - } - }; + let inputs = &self.catalog_inputs; // Fresh adapter per plan call: same rationale as // `plan_with_nodedb_sql_for_purpose`. Its recorded version set is returned to the // caller so parameterized plans participate in descriptor admission. @@ -431,43 +381,6 @@ impl QueryContext { }) .collect::>() .map_err(|error| map_plan_error(error, tenant_id))?; - let ctx = crate::control::planner::sql_plan_convert::ConvertContext { - purpose: PlanningPurpose::Execute, - retention_registry: self.retention_registry.clone(), - array_catalog: self.array_catalog.clone(), - credentials: self - .catalog_inputs - .as_ref() - .map(|i| Arc::clone(&i.credentials)), - wal: self.wal.clone(), - surrogate_assigner: self.surrogate_assigner.clone(), - cluster_enabled: self.cluster_enabled, - bitemporal_retention_registry: self.bitemporal_retention_registry.clone(), - max_vector_dim: self - .max_vector_dim - .load(std::sync::atomic::Ordering::Relaxed), - force_shuffle_join: self - .force_shuffle_join - .load(std::sync::atomic::Ordering::Relaxed), - shuffle_num_parts: self - .shuffle_num_parts - .load(std::sync::atomic::Ordering::Relaxed) as usize, - force_shuffle_agg: self - .force_shuffle_agg - .load(std::sync::atomic::Ordering::Relaxed), - shuffle_agg_num_parts: self - .shuffle_agg_num_parts - .load(std::sync::atomic::Ordering::Relaxed) - as usize, - broadcast_threshold_bytes: self - .broadcast_threshold_bytes - .load(std::sync::atomic::Ordering::Relaxed), - shuffle_agg_threshold: self - .shuffle_agg_threshold - .load(std::sync::atomic::Ordering::Relaxed), - database_id, - tenant_id, - }; let (output_schema, returning) = resolve_returning_and_output_schema( &plans, returning_items, @@ -475,16 +388,19 @@ impl QueryContext { database_id, tenant_id, )?; - let mut tasks = - crate::control::planner::sql_plan_convert::convert(&plans, tenant_id, &ctx)?; + let mut tasks = self + .convert_prefetched(&plans, PlanningPurpose::Execute, database_id, tenant_id) + .await?; if let Some(clause) = &returning { attach_returning_spec(&mut tasks, &clause.spec)?; } // Versions read BEFORE injection — see the comment on the sibling - // planning path in this file for why a post-injection read is unsafe. - let permission_tree_version = sec - .permission_cache + // planning path in this file for why a post-injection read is unsafe, + // and why the permission tree is read only after the last await. + let permission_view = sec.permission_tree.read().await; + let permission_cache = permission_view.as_deref(); + let permission_tree_version = permission_cache .map(|c| c.tenant_version(tenant_id.as_u64())) .unwrap_or(0); let rls_version = sec.rls_store.tenant_version(tenant_id.as_u64()); @@ -500,7 +416,7 @@ impl QueryContext { sec.redaction_store, )?; - if let Some(cache) = sec.permission_cache { + if let Some(cache) = permission_cache { crate::control::planner::rls_injection::inject_permission_tree( &mut tasks, cache, sec.auth, )?; diff --git a/nodedb/src/control/planner/context/security.rs b/nodedb/src/control/planner/context/security.rs index d8629df1b..dc0eed6f0 100644 --- a/nodedb/src/control/planner/context/security.rs +++ b/nodedb/src/control/planner/context/security.rs @@ -18,7 +18,38 @@ pub struct PlanSecurityContext<'a> { pub redaction_store: &'a crate::control::security::redaction::RedactionStore, pub permissions: &'a PermissionStore, pub roles: &'a RoleStore, - /// Permission tree cache for hierarchical ACL injection. - /// `None` = skip permission tree filtering (e.g., internal queries). - pub permission_cache: Option<&'a crate::control::security::permission_tree::PermissionCache>, + /// Where planning reads the permission tree cache for hierarchical ACL + /// injection. + pub permission_tree: PermissionTreeSource<'a>, +} + +/// Where planning reads the permission tree cache for hierarchical ACL +/// injection. +#[derive(Clone, Copy)] +pub enum PermissionTreeSource<'a> { + /// Skip permission tree filtering (e.g., internal queries). + None, + /// The node's live permission tree cache. The caller runs the + /// authorization fence first (`auth_fence::admit_permission_view`). + /// Planning takes the read lock only after its last await, so no lock is + /// held across a request to a surrogate's collection home. + Live(&'a tokio::sync::RwLock), +} + +impl<'a> PermissionTreeSource<'a> { + /// The permission tree cache to inject from, read-locked. `None` when + /// this source skips permission tree filtering. + pub async fn read( + self, + ) -> Option< + tokio::sync::RwLockReadGuard< + 'a, + crate::control::security::permission_tree::PermissionCache, + >, + > { + match self { + Self::None => None, + Self::Live(cache) => Some(cache.read().await), + } + } } diff --git a/nodedb/src/control/planner/context/system_security.rs b/nodedb/src/control/planner/context/system_security.rs index c05cfb164..54498cef1 100644 --- a/nodedb/src/control/planner/context/system_security.rs +++ b/nodedb/src/control/planner/context/system_security.rs @@ -29,7 +29,7 @@ use crate::control::security::identity::{AuthenticatedIdentity, DatabaseSet}; use crate::control::state::SharedState; use crate::types::TenantId; -use super::security::PlanSecurityContext; +use super::security::{PermissionTreeSource, PlanSecurityContext}; /// Owns the identity and auth context a server-owned plan borrows. /// @@ -62,9 +62,9 @@ impl SystemPlanSecurity { /// Borrow it as a planning security context over `state`'s policy stores. /// - /// `permission_cache` is `None`: hierarchical ACL filtering narrows a + /// `permission_tree` is `None`: hierarchical ACL filtering narrows a /// requester's view of a collection, and this context has no requester - /// whose view could be narrowed. + /// whose view can be narrowed. pub fn context<'a>(&'a self, state: &'a SharedState) -> PlanSecurityContext<'a> { PlanSecurityContext { identity: &self.identity, @@ -73,7 +73,7 @@ impl SystemPlanSecurity { redaction_store: &state.redaction, permissions: &state.permissions, roles: &state.roles, - permission_cache: None, + permission_tree: PermissionTreeSource::None, } } } diff --git a/nodedb/src/control/planner/implicit_edges/catalog.rs b/nodedb/src/control/planner/implicit_edges/catalog.rs index 4691149a2..b964f46ed 100644 --- a/nodedb/src/control/planner/implicit_edges/catalog.rs +++ b/nodedb/src/control/planner/implicit_edges/catalog.rs @@ -21,20 +21,22 @@ use crate::types::{DatabaseId, TenantId}; /// /// The flag is committed via the REPLICATED metadata path /// (`propose_catalog_entry` → `CatalogEntry::PutCollection`), exactly like -/// CREATE/ALTER COLLECTION. A bare local `put_collection` would only update the -/// proposing node's catalog, so a DELETE coordinated on a different node would -/// not observe the flag and would skip implicit-edge cleanup — the bug this -/// routing gate exists to prevent. The `LocalOnly` single-node path -/// bypasses the applier, so it writes through locally (mirrors the DDL handlers). +/// CREATE/ALTER COLLECTION. A bare local `put_collection` will only update the +/// proposing node's catalog, so a DELETE coordinated on a different node will +/// not observe the flag and will skip implicit-edge cleanup. The replicated +/// path prevents that. pub async fn mark_collection_edge_bearing( state: &SharedState, database_id: DatabaseId, tenant_id: TenantId, collection: &str, ) -> crate::Result<()> { + // Callers pass the plan's database-qualified name or the bare DDL name. + // The catalog keys collections by the bare name. + let bare = + crate::control::target_identity::naming::bare_collection_name(database_id, collection); let catalog = state.credentials.catalog(); - let Some(mut coll) = catalog.get_collection(database_id, tenant_id.as_u64(), collection)? - else { + let Some(mut coll) = catalog.get_collection(database_id, tenant_id.as_u64(), &bare)? else { // Collection row absent — don't fail the write over flag bookkeeping. return Ok(()); }; @@ -44,12 +46,7 @@ pub async fn mark_collection_edge_bearing( } coll.has_implicit_edges = true; - let entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(coll.clone())); - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry)?; - if outcome.needs_local_apply() { - // Single-node path: the metadata applier's post-apply hook is bypassed, - // so write through to the local catalog directly. - catalog.put_collection(database_id, &coll)?; - } + let entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(coll)); + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry).await?; Ok(()) } diff --git a/nodedb/src/control/planner/implicit_edges/insert.rs b/nodedb/src/control/planner/implicit_edges/insert.rs index c7d3e614f..97abc4416 100644 --- a/nodedb/src/control/planner/implicit_edges/insert.rs +++ b/nodedb/src/control/planner/implicit_edges/insert.rs @@ -10,7 +10,7 @@ use super::catalog::mark_collection_edge_bearing; use super::extract::{extract_edge, weight_properties}; use crate::control::server::surrogate_exchange::assign_surrogate_routed; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; +use crate::types::{DatabaseId, RecordHomes, TenantId, TraceId}; /// Scan the current document-write tasks for `_from` / `_to` documents and /// append a `GraphOp::EdgePut` task per implicit edge. @@ -25,7 +25,7 @@ use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; /// is derived from a `DocumentOp` write on the same collection, that write was /// already decided against the collection's write policy before this runs, and /// a denial there fails the statement before any mirror is derived. Deciding -/// the mirror as well would refuse every governed document insert on the +/// the mirror as well will refuse every governed document insert on the /// strength of its own edge, whose property object holds a weight and none of /// the columns a policy names. pub async fn append_implicit_edge_tasks( @@ -79,17 +79,17 @@ pub async fn append_implicit_edge_tasks( } for edge in edges { - let vsrc = VShardId::from_key(edge.src.as_bytes()); - let vdst = VShardId::from_key(edge.dst.as_bytes()); + // The write routes to the source endpoint's home. Both endpoints' + // surrogates come from the collection home, where every key of the + // collection is minted. + let vsrc = RecordHomes::edge(&edge.src, &edge.dst).owner(); // `edge.collection` is the plan's database-qualified name. let key = nodedb_types::CollectionKey::from_qualified_str(database_id, &edge.collection)?; let src_surrogate = - assign_surrogate_routed(state, vsrc, key, tenant_id, edge.src.as_bytes(), trace_id) - .await?; + assign_surrogate_routed(state, key, tenant_id, edge.src.as_bytes(), trace_id).await?; let dst_surrogate = - assign_surrogate_routed(state, vdst, key, tenant_id, edge.dst.as_bytes(), trace_id) - .await?; + assign_surrogate_routed(state, key, tenant_id, edge.dst.as_bytes(), trace_id).await?; let properties = match edge.weight { Some(w) => weight_properties(w), diff --git a/nodedb/src/control/planner/implicit_edges/routed.rs b/nodedb/src/control/planner/implicit_edges/routed.rs index b77fb13e4..7f04b3025 100644 --- a/nodedb/src/control/planner/implicit_edges/routed.rs +++ b/nodedb/src/control/planner/implicit_edges/routed.rs @@ -13,7 +13,7 @@ use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use super::extract::weight_properties; use crate::control::server::surrogate_exchange::assign_surrogate_routed; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; +use crate::types::{DatabaseId, RecordHomes, TenantId, TraceId}; /// Shared routing context for a single implicit-edge task (delete or put): /// the endpoint tenancy/collection identity plus the two endpoint keys. @@ -51,15 +51,17 @@ pub(super) async fn push_edge_delete( src, dst, } = ctx; - let vsrc = VShardId::from_key(src.as_bytes()); - let vdst = VShardId::from_key(dst.as_bytes()); + // The write routes to the source endpoint's home. Both endpoints' + // surrogates come from the collection home, where every key of the + // collection is minted. + let vsrc = RecordHomes::edge(src, dst).owner(); // `collection` is the plan's database-qualified name. let key = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?; let src_surrogate = - assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), trace_id).await?; + assign_surrogate_routed(state, key, tenant_id, src.as_bytes(), trace_id).await?; let dst_surrogate = - assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), trace_id).await?; + assign_surrogate_routed(state, key, tenant_id, dst.as_bytes(), trace_id).await?; out.push(PhysicalTask { tenant_id, @@ -74,7 +76,7 @@ pub(super) async fn push_edge_delete( dst_surrogate, // A mirrored edge is reconciliation of the document write that owns // it, and the policy on that same collection decided that write - // before this task was derived. Gating the mirror as well would + // before this task was derived. Gating the mirror as well will // refuse a document write the policy already admitted. The // identity that decided it is live and known here, which is what // separates this from a follower or replay path. @@ -92,7 +94,7 @@ pub(super) async fn push_edge_delete( /// from `weight` via the SAME [`weight_properties`] helper the INSERT path uses, /// so INSERT and UPDATE produce byte-identical properties for equal weight /// (`None` → empty properties → CSR unit weight). Endpoint surrogates are -/// resolved get-or-create (a new endpoint may not exist yet), homed on +/// resolved get-or-create (a new endpoint can be absent yet), homed on /// `from_key(src)`. pub(super) async fn push_edge_put( ctx: EdgeRouteCtx<'_>, @@ -114,15 +116,17 @@ pub(super) async fn push_edge_put( None => Vec::new(), }; - let vsrc = VShardId::from_key(src.as_bytes()); - let vdst = VShardId::from_key(dst.as_bytes()); + // The write routes to the source endpoint's home. Both endpoints' + // surrogates come from the collection home, where every key of the + // collection is minted. + let vsrc = RecordHomes::edge(src, dst).owner(); // `collection` is the plan's database-qualified name. let key = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?; let src_surrogate = - assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), trace_id).await?; + assign_surrogate_routed(state, key, tenant_id, src.as_bytes(), trace_id).await?; let dst_surrogate = - assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), trace_id).await?; + assign_surrogate_routed(state, key, tenant_id, dst.as_bytes(), trace_id).await?; out.push(PhysicalTask { tenant_id, diff --git a/nodedb/src/control/planner/materialized_sum/predicate.rs b/nodedb/src/control/planner/materialized_sum/predicate.rs index 767198aef..4b448843c 100644 --- a/nodedb/src/control/planner/materialized_sum/predicate.rs +++ b/nodedb/src/control/planner/materialized_sum/predicate.rs @@ -9,7 +9,7 @@ //! runs, since nearly every collection drives no binding. The Data-Plane //! leader then re-verifies the join-key set against the rows it actually //! matched, returning `OllpRetryRequired` before writing on any drift — a -//! silent divergence would leave a stored total that disagrees with +//! silent divergence will leave a stored total that disagrees with //! `SUM(...)` over the source rows. use std::sync::Arc; @@ -97,7 +97,7 @@ pub(super) async fn resolve_predicate_sum_targets( ) .await?; - // Folded from the SAME scan the resolution came from: a second scan would + // Folded from the SAME scan the resolution came from: a second scan will // see a different snapshot, and two snapshots is two totals. let images = predicate_images(&scope, &read.rows)?; let input = SettleInput { @@ -108,6 +108,7 @@ pub(super) async fn resolve_predicate_sum_targets( // that JOINS the match set after the scan has to invalidate it too. source_row: None, read_version_lsn: read.read_version_lsn, + served_by: read.served_by, }; let settlement = settle_cross_shard_images(&bindings, &input, &resolved, txn_id, tenant_id, database_id)?; @@ -175,7 +176,7 @@ async fn resolve_scanned_rows( /// by predicate. `UpdateFromJoin` is deliberately absent: which target rows it /// matches depends on the SOURCE collection's rows, which are only shipped by /// its Control-Plane orchestrator — so it resolves there, from the RESOLVE -/// pass's own classification, rather than from a predicate-only scan that would +/// pass's own classification, rather than from a predicate-only scan that will /// over-approximate the match set. fn predicate_scope(op: &DocumentOp) -> Option { match op { diff --git a/nodedb/src/control/planner/materialized_sum/recon.rs b/nodedb/src/control/planner/materialized_sum/recon.rs index ceaabb30f..4e741e9ca 100644 --- a/nodedb/src/control/planner/materialized_sum/recon.rs +++ b/nodedb/src/control/planner/materialized_sum/recon.rs @@ -16,20 +16,18 @@ //! one way to read a source row at plan time rather than two that can disagree //! about where the collection lives. //! -//! Like the OLLP pre-execution scan, the read is routed through the gateway when -//! one is wired: a bare local dispatch on a coordinator that does not host the -//! collection's vShard returns nothing, which would silently under-resolve and -//! leave the write with no target to address. +//! Like the OLLP pre-execution scan, the read is routed through the gateway: a +//! bare local dispatch on a coordinator that does not host the collection's +//! vShard returns nothing, which will silently under-resolve and leave the +//! write with no target to address. //! //! # Plane discipline //! -//! Runs on the coordinator's Control Plane (Tokio). The scan crosses the SPSC -//! bridge (or the gateway) exactly as a `SELECT` does — no storage I/O and no -//! io_uring here. +//! Runs on the coordinator's Control Plane (Tokio). The scan goes through the +//! gateway exactly as a `SELECT` does — no storage I/O and no io_uring here. use nodedb_types::{Surrogate, TenantId}; -use crate::control::server::dispatch_utils::dispatch_to_data_plane; use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, TraceId}; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; @@ -40,7 +38,7 @@ use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; /// The version travels with the rows because a delta settled from them is only /// as good as the images it was folded from: the caller stamps it onto a /// read-set entry so the Calvin OCC check aborts the statement if the source -/// rows moved between this read and the apply. Rows without their version would +/// rows moved between this read and the apply. Rows without their version will /// be a silently stale total. pub(crate) struct ReconRead { /// The decoded rows. @@ -48,6 +46,10 @@ pub(crate) struct ReconRead { /// The source collection's write floor at read time — the comparand /// cross-shard OCC validation checks the read against. pub read_version_lsn: Lsn, + /// The node that served the read. `read_version_lsn` is a position in + /// that node's WAL, so the commit vote compares it only on that node. + /// `0` when no one node is known to have served it. + pub served_by: u64, } /// Scan `collection` for the rows `filters` matches, returning each row's full @@ -55,8 +57,8 @@ pub(crate) struct ReconRead { /// /// Whole documents rather than a projection: the join column of every binding /// the collection drives has to be readable, and so does every column an -/// expression assignment to a join column evaluates over. A projection would -/// have to enumerate all of them and would silently drop a value the assignment +/// expression assignment to a join column evaluates over. A projection will +/// have to enumerate all of them and will silently drop a value the assignment /// depends on. /// /// Empty `filters` means "no WHERE clause" — every row, which is what `TRUNCATE` @@ -91,6 +93,7 @@ pub(in crate::control::planner) async fn recon_scan_rows( Ok(ReconRead { rows, read_version_lsn: read.read_version_lsn, + served_by: read.served_by, }) } @@ -119,11 +122,11 @@ pub(crate) async fn recon_point_row( let get_plan = PhysicalPlan::Document(DocumentOp::PointGet { collection: nodedb_types::QualifiedCollection::from_stored(collection.to_owned()), document_id: document_id.to_owned(), - surrogate, + surrogate: Some(surrogate), pk_bytes: document_id.as_bytes().to_vec(), // No RLS filters, for the same reason the recon scan carries none: this - // read decides which TOTAL a write moves, not what a principal may see. - // Filtering it would let a row the caller cannot read leave its + // read decides which TOTAL a write moves, not what a principal can see. + // Filtering it will let a row the caller cannot read leave its // contribution stranded on a target forever. rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, @@ -140,16 +143,20 @@ pub(crate) async fn recon_point_row( .find(|payload| !payload.is_empty()) .and_then(|payload| nodedb_types::json_from_msgpack(payload.as_slice()).ok()), read_version_lsn: read.read_version_lsn, + served_by: read.served_by, }) } -/// Run one read plan against `collection`, through the gateway when one is -/// wired and over the SPSC bridge otherwise, returning the raw payloads. +/// Run one read plan through the gateway, returning the raw payloads. /// /// A bare local dispatch on a coordinator that does not host the collection's -/// vShard returns nothing, which would silently under-resolve and leave the -/// write with no target to address — so the gateway is preferred whenever it -/// exists, for every shape of plan-time read alike. +/// vShard returns nothing, which will silently under-resolve and leave the +/// write with no target to address — so every plan-time read routes through +/// the gateway. +/// +/// The gateway notes the node that served each vShard it read. The read +/// validates on `collection`'s vShard, so that vShard's note names the node +/// whose WAL numbers `read_version_lsn`. async fn execute_read( state: &SharedState, tenant_id: TenantId, @@ -157,50 +164,29 @@ async fn execute_read( collection: &str, plan: PhysicalPlan, ) -> crate::Result>>> { - if let Some(gateway) = state.gateway.get() { - let gw_ctx = crate::control::gateway::core::QueryContext { - tenant_id, - trace_id: TraceId::ZERO, - database_id, - txn_id: None, - }; - // A shard verdict keeps its own typed error. - let (payloads, _watermarks, read_version_lsn) = gateway - .execute_internal_with_watermarks(&gw_ctx, plan) - .await?; - return Ok(ReconRead { - rows: payloads, - read_version_lsn, - }); - } - - let vshard_id = - nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); - let response = dispatch_to_data_plane( - state, + let validation_vshard = + nodedb_types::CollectionKey::from_qualified_str(database_id, collection)? + .vshard() + .as_u32(); + let gateway = state.installed_gateway()?; + let gw_ctx = crate::control::gateway::core::QueryContext { tenant_id, + trace_id: TraceId::ZERO, database_id, - vshard_id, - plan, - TraceId::ZERO, - ) - .await?; - read_from_response(&response) -} - -/// The rows of a local read response. -/// -/// A shard verdict keeps its own typed error, so a read the statement's -/// deadline cut short reports the deadline rather than a storage fault. -/// `reject_data_plane_error` passes only a `NotFound` refusal. Its payload is -/// empty, so it reads as no rows, the answer the gateway path gives. -fn read_from_response( - response: &crate::bridge::envelope::Response, -) -> crate::Result>>> { - crate::control::local_dispatch::reject_data_plane_error(response)?; + txn_id: None, + linearizable: true, + }; + // A shard verdict keeps its own typed error. + let (payloads, _watermarks, read_version_lsn) = gateway + .execute_internal_with_watermarks(&gw_ctx, plan) + .await?; Ok(ReconRead { - read_version_lsn: response.read_version_lsn, - rows: vec![response.payload.to_vec()], + rows: payloads, + read_version_lsn, + served_by: crate::control::server::shared::session::read_set::serving_node( + state, + validation_vshard, + ), }) } @@ -218,47 +204,3 @@ fn decode_rows(payload: &[u8]) -> Vec { .filter_map(|(_, body)| nodedb_types::json_from_msgpack(&body).ok()) .collect() } - -#[cfg(test)] -mod tests { - use super::*; - use crate::bridge::envelope::{ErrorCode, Payload, Response, Status}; - use crate::types::RequestId; - - fn refusal(code: ErrorCode) -> Response { - Response { - request_id: RequestId::new(1), - status: Status::Error, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: Lsn::ZERO, - error_code: Some(Box::new(code)), - read_set_valid: None, - read_version_lsn: Lsn::ZERO, - write_set: Vec::new(), - } - } - - /// A refused read keeps its code, never a storage error. - #[test] - fn a_refused_read_keeps_its_code() { - let code = ErrorCode::Unsupported { - detail: "not on this engine".into(), - }; - match read_from_response(&refusal(code.clone())) { - Err(crate::Error::DataPlane(kept)) => assert_eq!(kept, code), - Err(other) => panic!("expected the typed refusal, got {other:?}"), - Ok(_) => panic!("a refused read must fail"), - } - } - - /// A `NotFound` refusal reads as one empty payload, which decodes to no - /// rows. - #[test] - fn a_not_found_read_has_no_rows() { - let read = read_from_response(&refusal(ErrorCode::NotFound)) - .expect("a NotFound refusal reads as no rows"); - assert!(read.rows.iter().all(|payload| payload.is_empty())); - } -} diff --git a/nodedb/src/control/planner/materialized_sum/resolve.rs b/nodedb/src/control/planner/materialized_sum/resolve.rs index 24019d3bf..81eb0f526 100644 --- a/nodedb/src/control/planner/materialized_sum/resolve.rs +++ b/nodedb/src/control/planner/materialized_sum/resolve.rs @@ -20,12 +20,12 @@ use super::stored::stored_row_scope; use crate::control::server::shared::session::read_set::ReadSetEntry; use crate::control::server::surrogate_exchange::lookup_surrogate_routed; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; +use crate::types::{DatabaseId, TenantId, TraceId}; /// Resolve the materialized-sum target rows for every document write in /// `tasks`, storing the result in that op's `resolved_sum_targets`. /// -/// Three join-value sources feed resolution, and one op may draw on more +/// Three join-value sources feed resolution, and one op can draw on more /// than one: row BODIES read straight off each op; the PREDICATE /// `BulkUpdate`/`BulkDelete`/`TRUNCATE` name rows by (via a recon scan); /// and the STORED row a point write rewrites/removes (`PointDelete`/ @@ -116,7 +116,7 @@ pub async fn resolve_materialized_sum_targets( // A point write that rewrites a stored row reads it here — for the // join values it addresses AND for the pre-image a cross-shard // delta is folded from. One read, one snapshot: settling from a - // second read would total a different one. + // second read will total a different one. match &stored { None => (resolved.into_vec(), None), Some(scope) => { @@ -136,6 +136,7 @@ pub async fn resolve_materialized_sum_targets( images: &images.images, source_row: Some(scope.surrogate), read_version_lsn: images.read_version_lsn, + served_by: images.served_by, }; let settlement = settle_cross_shard_images( &bindings, @@ -146,7 +147,7 @@ pub async fn resolve_materialized_sum_targets( database_id, )?; // The resolution the source op keeps is the one the source - // core may still apply. Removing the shipped values IS the + // core can still apply. Removing the shipped values IS the // deferral: what is left resolved is exactly what no // sibling task carries. omit_shipped( @@ -283,8 +284,8 @@ fn set_resolved(op: &mut DocumentOp, resolved: Vec) { /// `MERGE` and `UPDATE ... FROM` all resolve their rows on the Control Plane and /// re-issue concrete work (a `BatchInsert` page, an APPLY pass, a write pass) /// through `dispatch_local`, which never passes through -/// [`resolve_materialized_sum_targets`]. Without this they would ship an empty -/// resolution and the Data-Plane fold would have no target to address. +/// [`resolve_materialized_sum_targets`]. Without this they will ship an empty +/// resolution and the Data-Plane fold will have no target to address. /// /// `source_collection` is the db-qualified name as it appears on the plan. /// Returns an empty vec — and issues no lookup at all — when the collection @@ -353,10 +354,8 @@ pub(super) async fn lookup_join_value( database_id: DatabaseId, trace_id: TraceId, ) -> crate::Result { - let vshard = VShardId::from_key(join_value.as_bytes()); lookup_surrogate_routed( state, - vshard, nodedb_types::CollectionKey::from_bare(database_id, &binding.target_collection), tenant_id, join_value.as_bytes(), @@ -375,7 +374,7 @@ pub(super) async fn lookup_join_value( /// One entry per DISTINCT `(target collection, join value)` PAIR: a batch that /// touches the same target row many times resolves it once, while two bindings /// that share a join column and name different targets each get their own entry -/// — deduping on the value alone would resolve the first and silently hand its +/// — deduping on the value alone will resolve the first and silently hand its /// target row to the second. async fn resolve_bodies( state: &SharedState, @@ -424,6 +423,7 @@ mod tests { use crate::bridge::dispatch::Dispatcher; use crate::control::security::catalog::{MaterializedSumDef, StoredCollection}; + use crate::types::VShardId; use crate::wal::WalManager; const TENANT: TenantId = TenantId::new(7); @@ -444,7 +444,7 @@ mod tests { /// `balance` materialized sum on `accounts`, joined on `account_id`. fn declare_binding(state: &SharedState) { let catalog = state.credentials.catalog(); - let mut target = StoredCollection::new(TENANT.as_u64(), "accounts", "tester"); + let mut target = StoredCollection::stamped_for_test(TENANT.as_u64(), "accounts", "tester"); target.materialized_sums.push(MaterializedSumDef { target_collection: "accounts".to_string(), target_column: "balance".to_string(), @@ -455,7 +455,7 @@ mod tests { catalog .put_collection(DB, &target) .expect("persist target collection"); - let source = StoredCollection::new(TENANT.as_u64(), "entries", "tester"); + let source = StoredCollection::stamped_for_test(TENANT.as_u64(), "entries", "tester"); catalog .put_collection(DB, &source) .expect("persist source collection"); @@ -466,7 +466,8 @@ mod tests { /// join column into a DIFFERENT target collection. fn declare_second_binding(state: &SharedState) { let catalog = state.credentials.catalog(); - let mut target = StoredCollection::new(TENANT.as_u64(), "audit_totals", "tester"); + let mut target = + StoredCollection::stamped_for_test(TENANT.as_u64(), "audit_totals", "tester"); target.materialized_sums.push(MaterializedSumDef { target_collection: "audit_totals".to_string(), target_column: "balance".to_string(), @@ -540,6 +541,7 @@ mod tests { TENANT, b"acc-1", ) + .await .expect("bind target row"); let mut tasks = vec![insert_task("entries", body("acc-1"))]; @@ -577,6 +579,7 @@ mod tests { TENANT, b"acc-1", ) + .await .expect("bind accounts row"); let audit_row = state .surrogate_assigner @@ -585,6 +588,7 @@ mod tests { TENANT, b"acc-1", ) + .await .expect("bind audit_totals row"); assert_ne!( accounts_row, audit_row, @@ -617,8 +621,8 @@ mod tests { } /// A join key naming no target row fails the statement with a typed error - /// that says which collection, column, and value could not be resolved. - /// Skipping the row would leave the stored balance short of the sum + /// that says which collection, column, and value cannot be resolved. + /// Skipping the row will leave the stored balance short of the sum /// `VERIFY_BALANCE` recomputes over every source row. #[tokio::test] async fn unresolvable_join_key_fails_with_a_typed_error() { @@ -688,6 +692,7 @@ mod tests { TENANT, b"acc-1", ) + .await .expect("bind target row"); let mut tasks = vec![PhysicalTask { diff --git a/nodedb/src/control/planner/materialized_sum/settle.rs b/nodedb/src/control/planner/materialized_sum/settle.rs index ab09823b0..c6c744cde 100644 --- a/nodedb/src/control/planner/materialized_sum/settle.rs +++ b/nodedb/src/control/planner/materialized_sum/settle.rs @@ -17,7 +17,7 @@ //! routed point read for the point shapes and [`super::recon`] scans the //! predicate for the bulk ones, both so the join values can be resolved at all. //! Those images are folded HERE, in the same pass, from the same read: a second -//! pass would fold a different snapshot, and two snapshots is two totals. +//! pass will fold a different snapshot, and two snapshots is two totals. //! //! # Deferral is the ABSENCE of a resolution //! @@ -31,7 +31,7 @@ //! homed on the target's vShard and dual-homed with the source write through //! Calvin. //! -//! Nothing is added to the source op to say so. A marker field would have to be +//! Nothing is added to the source op to say so. A marker field will have to be //! spelled on seven more plan variants and seven more replicated-write //! variants, and every one of them is a place the marker and the appended task //! can disagree. The resolution IS the marker: the plane-neutral @@ -91,6 +91,9 @@ pub(super) struct SettleInput<'a> { pub source_row: Option, /// Version the images were read at. pub read_version_lsn: Lsn, + /// The node that served the image read. `read_version_lsn` is a position + /// in its WAL, so only that node's commit vote can find the read current. + pub served_by: u64, } /// What settling produced. @@ -102,7 +105,7 @@ pub(super) struct Settlement { /// — that removal is the whole of the deferral signal. /// /// Keyed on the PAIR, never on the value: two bindings of one source can - /// share a join column, and shipping one binding's value would otherwise + /// share a join column, and shipping one binding's value will otherwise /// strip the other binding's resolution and silently drop its delta. pub shipped: Vec, /// Read-set entries covering the images the deltas were folded from. @@ -132,7 +135,7 @@ impl Settlement { /// [`MaterializedSumTargetNotFound`](crate::Error::MaterializedSumTargetNotFound): /// the row addresses a target that does not exist, and failing the statement is /// what the resolution pass itself does with the same finding. Shipping nothing -/// instead would leave the stored total short of the `SUM(...)` over the source +/// instead will leave the stored total short of the `SUM(...)` over the source /// rows. pub(super) fn settle_cross_shard_images( bindings: &[MaterializedSumBinding], @@ -160,8 +163,8 @@ pub(super) fn settle_cross_shard_images( } for (join_value, delta) in crate::query::coalesce_binding_deltas(folded) { // A zero net delta leaves the stored total unchanged, so the - // read-modify-write on the target would rewrite the row - // byte-for-byte. Shipping a task for it would also make an + // read-modify-write on the target will rewrite the row + // byte-for-byte. Shipping a task for it will also make an // otherwise single-shard statement multi-shard for nothing. The // join value is still recorded as shipped: the source core must not // apply it either, and applying nothing is what it does when the @@ -203,7 +206,7 @@ pub(super) fn settle_cross_shard_images( /// Everything one appended balance task needs. pub(super) struct BalanceTaskSpec<'a> { /// Inherited from the source write: the balance belongs to the same - /// statement, and a task that lost the transaction would commit on its own. + /// statement, and a task that lost the transaction will commit on its own. pub txn_id: Option, pub database_id: DatabaseId, pub tenant_id: TenantId, @@ -274,10 +277,12 @@ fn image_read_entry( // A DERIVATION read, never a read-your-own-write. The image was read // from committed base state before this transaction existed, and the // delta shipped to the target rests entirely on it; the statement writes - // the source collection too, so an entry marked `Session` here would be - // dropped by the own-write exclusion and the fold would never be + // the source collection too, so an entry marked `Session` here will be + // dropped by the own-write exclusion and the fold will never be // validated against a concurrent writer. origin: ReadOrigin::PlanDerivation, + home: None, + home_node: input.served_by, } } @@ -288,8 +293,8 @@ fn image_read_entry( /// The removal is the deferral signal, so it has to be exact in both /// directions: a pair left behind is applied twice, and a pair removed that a /// co-resident binding still addresses is a delta dropped on the floor. Matching -/// on the join value alone would do both at once when a source drives two -/// bindings that share a join column — the cross-shard one's shipment would +/// on the join value alone will do both at once when a source drives two +/// bindings that share a join column — the cross-shard one's shipment will /// strip the co-resident one's entry. pub(super) fn omit_shipped( resolved: &mut Vec, @@ -393,6 +398,7 @@ mod tests { images: &images, source_row: Some(Surrogate::new(11)), read_version_lsn: Lsn::new(42), + served_by: 7, }; let settlement = settle_cross_shard_images( &[binding(&target)], @@ -428,6 +434,7 @@ mod tests { images: &images, source_row: Some(Surrogate::new(11)), read_version_lsn: Lsn::new(42), + served_by: 7, }; let settlement = settle_cross_shard_images( &[binding(&target)], @@ -468,6 +475,7 @@ mod tests { images: &images, source_row: Some(Surrogate::new(11)), read_version_lsn: Lsn::new(42), + served_by: 7, }; let settlement = settle_cross_shard_images( &[binding(&target)], @@ -482,7 +490,7 @@ mod tests { } /// A net-zero UPDATE ships no task — and STILL defers, because the source - /// core applying its own non-zero halves would double-count them. + /// core applying its own non-zero halves will double-count them. #[test] fn a_net_zero_update_defers_without_shipping() { let (source, target) = cross_shard_pair(); @@ -492,6 +500,7 @@ mod tests { images: &images, source_row: Some(Surrogate::new(11)), read_version_lsn: Lsn::new(42), + served_by: 7, }; let settlement = settle_cross_shard_images( &[binding(&target)], @@ -520,6 +529,7 @@ mod tests { images: &images, source_row: Some(Surrogate::new(11)), read_version_lsn: Lsn::new(42), + served_by: 7, }; // A binding whose target IS the source collection is co-resident by // construction, whatever the hash function does. @@ -549,6 +559,7 @@ mod tests { images: &images, source_row: Some(Surrogate::new(11)), read_version_lsn: Lsn::new(42), + served_by: 7, }; let settlement = settle_cross_shard_images( &[binding(&target)], @@ -568,6 +579,11 @@ mod tests { } ); assert_eq!(settlement.reads[0].read_version_lsn, Lsn::new(42)); + assert_eq!( + settlement.reads[0].home_node, 7, + "the entry names the node that served the image read; `0` makes \ + every commit vote treat the read as changed" + ); } /// A predicate-shaped settlement observes the whole collection, so its @@ -582,6 +598,7 @@ mod tests { images: &images, source_row: None, read_version_lsn: Lsn::new(42), + served_by: 7, }; let settlement = settle_cross_shard_images( &[binding(&target)], @@ -616,8 +633,8 @@ mod tests { /// resolution of a SIBLING binding that reads the same join column into a /// different target. /// - /// Keyed on the join value alone, the shipment below would remove both - /// entries and the co-resident target's delta would be dropped: no error, + /// Keyed on the join value alone, the shipment below will remove both + /// entries and the co-resident target's delta will be dropped: no error, /// a total silently short by the row's whole value. #[test] fn shipping_one_target_leaves_a_sibling_targets_resolution_intact() { diff --git a/nodedb/src/control/planner/materialized_sum/stored.rs b/nodedb/src/control/planner/materialized_sum/stored.rs index 84ede6e42..138abac09 100644 --- a/nodedb/src/control/planner/materialized_sum/stored.rs +++ b/nodedb/src/control/planner/materialized_sum/stored.rs @@ -11,7 +11,7 @@ //! resolves from the plan; the one the row is leaving is readable only from //! the stored image. //! -//! [`recon_point_row`] is the only plan-time source read: two readers could +//! [`recon_point_row`] is the only plan-time source read: two readers can //! disagree about where a collection lives, and the Data Plane never //! resolves this itself (the pk→surrogate map is catalog state, off-limits //! cross-plane). [`binding_join_keys`](crate::query::binding_join_keys) @@ -73,12 +73,14 @@ pub(super) struct StoredRowScope<'a> { /// What reading the stored row produced: the write's pre-/post-image pairs /// and the version they were read at — from the SAME read that resolved -/// the join values, so a re-read would fold a different snapshot. +/// the join values, so a re-read will fold a different snapshot. pub(super) struct StoredImages { /// One pair per row this write touches — at most one, for a point shape. pub images: Vec<(Option, Option)>, /// The source collection's write floor at read time. pub read_version_lsn: Lsn, + /// The node that served the read, whose WAL numbers `read_version_lsn`. + pub served_by: u64, } /// The stored row an op rewrites or removes, or `None` for every op that @@ -89,16 +91,17 @@ pub(super) struct StoredImages { /// by construction, so there's no pre-image to read. pub(super) fn stored_row_scope(op: &DocumentOp) -> Option> { match op { + // A key unbound in its database names no stored row. DocumentOp::PointUpdate { collection, document_id, surrogate, updates, .. - } => Some(StoredRowScope { + } => surrogate.map(|surrogate| StoredRowScope { collection: collection.as_str(), document_id: document_id.as_str(), - surrogate: *surrogate, + surrogate, updates: updates.as_slice(), post_image: PostImage::Assigned, }), @@ -109,10 +112,10 @@ pub(super) fn stored_row_scope(op: &DocumentOp) -> Option> { document_id, surrogate, .. - } => Some(StoredRowScope { + } => surrogate.map(|surrogate| StoredRowScope { collection: collection.as_str(), document_id: document_id.as_str(), - surrogate: *surrogate, + surrogate, updates: &[], post_image: PostImage::Removed, }), @@ -207,6 +210,7 @@ pub(super) async fn extend_with_stored_row( let outcome = StoredImages { images: images_of(scope, read.rows.as_ref())?, read_version_lsn: read.read_version_lsn, + served_by: read.served_by, }; if read.rows.is_none() { @@ -220,7 +224,7 @@ pub(super) async fn extend_with_stored_row( // read off the images the deltas will be folded from rather than re-derived // from the assignments: the two shapes that carry a body form their // post-image from the body as well as from the assignments, and a - // re-derivation that saw only the assignments would resolve a target the + // re-derivation that saw only the assignments will resolve a target the // fold never addresses — or miss the one it does. let mut rows: Vec = Vec::with_capacity(outcome.images.len() * 2); for (old, new) in &outcome.images { @@ -320,7 +324,7 @@ mod tests { DocumentOp::PointDelete { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "entries"), document_id: "e1".to_string(), - surrogate: ROW, + surrogate: Some(ROW), pk_bytes: b"e1".to_vec(), returning: None, rls_filters: Vec::new(), @@ -333,7 +337,7 @@ mod tests { DocumentOp::PointUpdate { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "entries"), document_id: "e1".to_string(), - surrogate: ROW, + surrogate: Some(ROW), pk_bytes: b"e1".to_vec(), updates: assignments(), returning: None, diff --git a/nodedb/src/control/planner/period_lock/lookup.rs b/nodedb/src/control/planner/period_lock/lookup.rs index 297243e3a..e9d207c24 100644 --- a/nodedb/src/control/planner/period_lock/lookup.rs +++ b/nodedb/src/control/planner/period_lock/lookup.rs @@ -4,7 +4,7 @@ use crate::control::server::surrogate_exchange::lookup_surrogate_routed; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; +use crate::types::{DatabaseId, TenantId, TraceId}; /// Request scope shared by every per-row period-lock resolution call within /// one [`resolve_period_lock_targets`](super::resolve::resolve_period_lock_targets) @@ -31,10 +31,8 @@ pub(super) async fn lookup_period_surrogate( database_id: DatabaseId, trace_id: TraceId, ) -> crate::Result> { - let vshard = VShardId::from_key(period_key.as_bytes()); lookup_surrogate_routed( state, - vshard, nodedb_types::CollectionKey::from_bare(database_id, ref_table), tenant_id, period_key.as_bytes(), diff --git a/nodedb/src/control/planner/period_lock/resolve.rs b/nodedb/src/control/planner/period_lock/resolve.rs index 2a14c1722..bcc7f2856 100644 --- a/nodedb/src/control/planner/period_lock/resolve.rs +++ b/nodedb/src/control/planner/period_lock/resolve.rs @@ -112,7 +112,7 @@ pub async fn resolve_period_lock_targets( } // `PointUpdate` carries field assignments rather than a whole row, and - // it may rewrite the period column itself — both images need their own + // it can rewrite the period column itself — both images need their own // resolution, distinctly from the single-value path below. if let DocumentOp::PointUpdate { document_id, @@ -122,11 +122,15 @@ pub async fn resolve_period_lock_targets( .. } = op { + // A key unbound in its database names no row to rewrite. + let Some(surrogate) = *surrogate else { + continue; + }; let resolved = resolve_update_period_values( &scope, &collection, document_id, - *surrogate, + surrogate, updates, &def, ) @@ -242,7 +246,7 @@ pub async fn resolve_period_lock_targets_for_bodies( /// The gate an orchestrator checks FIRST, alongside /// [`source_drives_bindings`](crate::control::planner::materialized_sum::source_drives_bindings): /// a target with neither a materialized-sum binding nor a period lock skips -/// the RESOLVE round trip its statement would otherwise pay for nothing. +/// the RESOLVE round trip its statement will otherwise pay for nothing. pub fn target_declares_period_lock( state: &SharedState, target_collection: &str, diff --git a/nodedb/src/control/planner/period_lock/singular.rs b/nodedb/src/control/planner/period_lock/singular.rs index e7ddc91c2..d75baa496 100644 --- a/nodedb/src/control/planner/period_lock/singular.rs +++ b/nodedb/src/control/planner/period_lock/singular.rs @@ -37,6 +37,10 @@ pub(super) async fn singular_period_value( surrogate, .. } => { + // A key unbound in its database names no row to remove. + let Some(surrogate) = surrogate else { + return Ok(None); + }; let read = recon_point_row( state, tenant_id, diff --git a/nodedb/src/control/planner/procedural/executor/core/applied.rs b/nodedb/src/control/planner/procedural/executor/core/applied.rs new file mode 100644 index 000000000..4f98e986d --- /dev/null +++ b/nodedb/src/control/planner/procedural/executor/core/applied.rs @@ -0,0 +1,47 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The key a body's commit records, so a body fired again applies once. +//! +//! A cross-shard receiver commits the key of the request it applies. A +//! trigger body fired from the Event Plane commits a key of its own: its +//! source event's replicated identity with the body's origin tag. Every +//! replica records the key as the commit applies. A later owner that fires +//! the same event again finds the key and does not run the body. + +use crate::wal::CrossShardAppliedKey; + +use super::state::StatementExecutor; + +impl StatementExecutor<'_> { + /// The key this body's commit records, with the vShard whose redo record + /// carries it. `None` for a body that commits no key. + pub(super) fn commit_key(&self) -> Option<(CrossShardAppliedKey, u32)> { + if let Some(applied) = &self.applied_key { + return Some(applied.clone()); + } + let origin = self.cross_shard_origin.as_ref()?; + let body = self.body.as_ref()?; + let key = CrossShardAppliedKey { + source_vshard: origin.source_vshard, + source_lsn: origin.source_lsn, + source_sequence: origin.source_sequence, + origin: format!("{}/local", body.origin_tag(self.database_id)), + }; + Some((key, origin.source_vshard)) + } + + /// Whether this body's commit already applied: a body fired from the + /// Event Plane whose key an earlier firing of the same event recorded. + pub fn already_applied(&self) -> crate::Result { + if self.applied_key.is_some() { + return Ok(false); + } + let Some((key, _)) = self.commit_key() else { + return Ok(false); + }; + match self.state.cross_shard_dedup.get() { + Some(dedup) => dedup.is_applied(&key), + None => Ok(false), + } + } +} diff --git a/nodedb/src/control/planner/procedural/executor/core/block.rs b/nodedb/src/control/planner/procedural/executor/core/block.rs index 499f85b6c..b73521b40 100644 --- a/nodedb/src/control/planner/procedural/executor/core/block.rs +++ b/nodedb/src/control/planner/procedural/executor/core/block.rs @@ -1,6 +1,7 @@ //! Block-level execution: runs a procedural block's statements with exception handling. use crate::control::planner::procedural::ast::{ExceptionHandler, ProceduralBlock, Statement}; +use crate::control::server::shared::session::conn_scope::scoped_system_txn; use super::super::bindings::RowBindings; use super::super::exception::exception_matches; @@ -8,41 +9,55 @@ use super::super::fuel::ExecutionBudget; use super::StatementExecutor; impl<'a> StatementExecutor<'a> { + /// Run `block` under the trigger budget. See [`Self::execute_block_with_budget`]. pub async fn execute_block( &self, block: &ProceduralBlock, bindings: &RowBindings, ) -> crate::Result<()> { let mut budget = ExecutionBudget::trigger_default(); - self.execute_block_with_exceptions( - &block.statements, - &block.exception_handlers, - bindings, - &mut budget, - ) - .await + self.execute_block_with_budget(block, bindings, &mut budget) + .await } + /// Run `block` and commit what it staged as one transaction. On error + /// nothing it staged since its last COMMIT is applied. + /// + /// The block runs in its own connection slots, so its DDL buffers with + /// its writes and never into a client transaction it runs inside. A body + /// joined to its statement's transaction runs in that transaction's + /// slots instead. pub async fn execute_block_with_budget( &self, block: &ProceduralBlock, bindings: &RowBindings, budget: &mut ExecutionBudget, ) -> crate::Result<()> { - let result = self - .execute_block_with_exceptions( - &block.statements, - &block.exception_handlers, - bindings, - budget, - ) - .await; + let run = async { + let result = self + .execute_block_with_exceptions( + &block.statements, + &block.exception_handlers, + bindings, + budget, + ) + .await; - if result.is_ok() { - self.flush_transaction_buffer().await?; + match result { + Ok(()) => self.flush_transaction_buffer().await, + Err(error) => { + self.discard_transaction_buffer().await; + Err(error) + } + } + }; + // A joined body buffers its DDL into its statement's transaction, + // which runs in the connection's slots. + if self.joined.is_some() { + run.await + } else { + scoped_system_txn(run).await } - - result } async fn execute_block_with_exceptions( @@ -57,10 +72,7 @@ impl<'a> StatementExecutor<'a> { if let Err(ref err) = result && !handlers.is_empty() { - if let Some(ref tx_ctx) = self.tx_ctx { - let mut guard = tx_ctx.lock().unwrap_or_else(|p| p.into_inner()); - guard.rollback(); - } + self.discard_transaction_buffer().await; let err_str = err.to_string(); for handler in handlers { diff --git a/nodedb/src/control/planner/procedural/executor/core/body.rs b/nodedb/src/control/planner/procedural/executor/core/body.rs new file mode 100644 index 000000000..243f7e9f2 --- /dev/null +++ b/nodedb/src/control/planner/procedural/executor/core/body.rs @@ -0,0 +1,70 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Server-run bodies that execute as exactly one transaction. + +/// A body the server runs on its own: its statements commit together on +/// success and are discarded on error, so a retry never repeats a part that +/// already applied. COMMIT and ROLLBACK are refused inside one. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum AtomicBody { + /// A trigger body, named by its trigger. + Trigger { name: String }, + /// A scheduled job body, named by its schedule. + ScheduledJob { name: String }, + /// An alert's history write, named by its alert. + Alert { name: String }, + /// Trigger DML shipped from another node and applied on this one. The + /// body plans each statement with every write it derives, and they all + /// commit in its one transaction, whichever vShards they span. + CrossShardApply, +} + +impl AtomicBody { + pub fn trigger(name: &str) -> Self { + Self::Trigger { + name: name.to_owned(), + } + } + + pub fn scheduled_job(name: &str) -> Self { + Self::ScheduledJob { + name: name.to_owned(), + } + } + + pub fn alert(name: &str) -> Self { + Self::Alert { + name: name.to_owned(), + } + } + + /// Identity of this body within one source write. The cross-shard + /// receiver deduplicates on it together with the source write's position. + pub(super) fn origin_tag(&self, database_id: crate::types::DatabaseId) -> String { + let database = database_id.as_u64(); + match self { + Self::Trigger { name } => format!("trigger/{database}/{name}"), + Self::ScheduledJob { name } => format!("schedule/{database}/{name}"), + Self::Alert { name } => format!("alert/{database}/{name}"), + Self::CrossShardApply => format!("cross-shard/{database}"), + } + } + + /// The error for a transaction-control statement inside this body. + pub(super) fn refuse_transaction_control(&self, statement: &str) -> crate::Error { + crate::Error::NotInTransactionBlock { + statement: format!("{statement} in {self}"), + } + } +} + +impl std::fmt::Display for AtomicBody { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::Trigger { name } => write!(f, "the body of trigger '{name}'"), + Self::ScheduledJob { name } => write!(f, "the body of schedule '{name}'"), + Self::Alert { name } => write!(f, "the history write of alert '{name}'"), + Self::CrossShardApply => f.write_str("a cross-shard trigger write"), + } + } +} diff --git a/nodedb/src/control/planner/procedural/executor/core/dispatch.rs b/nodedb/src/control/planner/procedural/executor/core/dispatch.rs index 9712c88e3..8759385a2 100644 --- a/nodedb/src/control/planner/procedural/executor/core/dispatch.rs +++ b/nodedb/src/control/planner/procedural/executor/core/dispatch.rs @@ -2,14 +2,26 @@ //! DML dispatch and transaction control for the statement executor. -use super::super::transaction::ProcedureTransactionCtx; +use super::super::transaction::{ProcedureTransactionCtx, RemoteWrite}; use super::StatementExecutor; +use super::route::StatementRoute; use super::sql_literal_concat::fold_literal_string_concat; use crate::control::planner::procedural::ast::SqlExpr; use crate::control::planner::procedural::executor::bindings::RowBindings; use crate::control::planner::procedural::executor::eval; -use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; -use crate::types::TraceId; +use crate::control::system_txn::{OpenSystemTxn, SystemTxnStatement}; + +/// Whether a body statement goes to the DDL router before the planner. +/// `INSERT` and `UPSERT` go to the planner: the router's object-literal and +/// `UPSERT INTO` forms fire the target's triggers again. +fn routes_through_ddl_router(sql: &str) -> bool { + let head = sql.trim_start(); + let starts_with = |word: &str| { + head.get(..word.len()) + .is_some_and(|prefix| prefix.eq_ignore_ascii_case(word)) + }; + !(starts_with("INSERT") || starts_with("UPSERT")) +} impl<'a> StatementExecutor<'a> { // ── ASSIGN handling ───────────────────────────────────────────────── @@ -49,29 +61,55 @@ impl<'a> StatementExecutor<'a> { pub(super) async fn execute_sql(&self, sql: &str, bindings: &RowBindings) -> crate::Result<()> { let bound_sql = fold_literal_string_concat(&bindings.substitute(sql)); - // First, attempt unified dispatch for NodeDB SQL extensions (PUBLISH TO, - // topic/consumer-group DDL, etc.). If the SQL is not an extension, - // `dispatch_sql` returns None and we fall through to plan_sql with - // transaction-buffer semantics preserved exactly as before. - if let Some(_outcome) = crate::control::sql_dispatch::dispatch_sql_in_database( - self.state, - &self.identity_for_dispatch(), - self.database_id, - &bound_sql, - ) - .await? - { - return Ok(()); + // A NodeDB SQL extension (`PUBLISH TO`) is checked now and sent only + // after the transaction commits, so a rolled-back body publishes + // nothing. + if crate::control::sql_dispatch::is_sql_extension(&bound_sql) { + // A shipped block carries only the DML its origin routed here: + // the origin publishes from its own commit. + if matches!(self.body, Some(super::AtomicBody::CrossShardApply)) { + return Err(crate::Error::BadRequest { + detail: "PUBLISH is not accepted in a cross-shard trigger write".into(), + }); + } + let publish = crate::control::sql_dispatch::prepare_publish( + self.state, + &self.identity_for_dispatch(), + self.database_id, + &bound_sql, + )?; + let mut guard = self.tx_ctx.lock().unwrap_or_else(|p| p.into_inner()); + return guard.buffer_publish(publish); } - // Not a NodeDB extension: plan with descriptor versions so admission - // occurs before this internal path buffers, WAL-appends, or dispatches. - // Stored procedures are trusted internal execution, so this deliberately - // does not add user authorization beyond their existing semantics. + // DDL and the other router-owned statements run on the open + // transaction's session, as in a client transaction: DDL buffers and + // commits with the body's writes. + if routes_through_ddl_router(bound_sql.as_str()) { + let mut txn = self.txn.lock().await; + let open = self.open_txn(&mut txn).await?; + let txn_ctx = open.txn_ctx()?; + if let Some(result) = crate::control::server::shared::ddl::dispatch( + self.state, + &self.identity, + &bound_sql, + self.database_id, + &txn_ctx, + ) + .await + { + return result.map(drop).map_err(crate::Error::from); + } + } + + // Plan with descriptor versions so admission occurs before this + // internal path stages anything. Stored procedures are trusted + // internal execution, so this deliberately does not add user + // authorization beyond their existing semantics. // // Planning and lease admission run as ONE retried unit: the lease that // pins the planned descriptor version is acquired after the catalog - // read, so a descriptor drain starting in between would otherwise fail + // read, so a descriptor drain starting in between will otherwise fail // the whole procedure. Re-planning is pure, and admission fails closed // before granting anything, so an absorbed attempt reads nothing. let ctx = crate::control::planner::context::QueryContext::for_state(self.state); @@ -80,183 +118,113 @@ impl<'a> StatementExecutor<'a> { // A stored-procedure body is server-defined code, not a client // statement, and runs SECURITY DEFINER exactly as a trigger body does — // so it plans as the system rather than under the invoker's scope. The - // context is built once and borrowed by the retry closure, which may run + // context is built once and borrowed by the retry closure, which can run // it several times. let security = crate::control::planner::context::SystemPlanSecurity::new( self.tenant_id, "_system_procedure", ); let security = &security; - let (tasks, lease_scope) = - crate::control::server::shared::retry::retry_on_schema_change(move || async move { - let (tasks, _output_schema, versions, _) = ctx - .plan_sql_with_rls_and_versions( - bound_sql, - self.tenant_id, - self.database_id, - &security.context(self.state), - None, - ) - .await?; - let lease_scope = self.state.acquire_plan_lease_scope(&versions)?; - Ok::<_, crate::Error>((tasks, lease_scope)) - }) + // The derived writes (implicit edges, materialized sums, period + // locks) join the statement as they do for a client statement. The + // plan never carries a Calvin OLLP prediction or a resolved write: + // the planner emits neither, so every task takes the staging gate. + let (tasks, planned, lease_scope, sum_target_reads) = + crate::control::server::shared::retry::retry_on_schema_change( + &self.state.lease_drain, + move || async move { + let (mut tasks, _output_schema, versions, _) = ctx + .plan_sql_with_rls_and_versions( + bound_sql, + self.tenant_id, + self.database_id, + &security.context(self.state), + None, + ) + .await?; + let planned = tasks.len(); + let sum_target_reads = + crate::control::server::shared::plan_admission::append_derived_tasks( + self.state, + &mut tasks, + self.tenant_id, + self.database_id, + crate::types::TraceId::ZERO, + ) + .await?; + let lease_scope = self.state.acquire_plan_lease_scope(&versions).await?; + Ok::<_, crate::Error>((tasks, planned, lease_scope, sum_target_reads)) + }, + ) .await?; - - if let Some(ref tx_ctx) = self.tx_ctx { - let mut guard = tx_ctx.lock().unwrap_or_else(|p| p.into_inner()); - guard.buffer_statement(tasks, lease_scope); - } else { - // Keep the scope through every route decision, WAL append, and - // Data-Plane dispatch; errors also release it via Drop. - let _lease_scope = lease_scope; - for task in tasks { - // Cross-shard trigger origination: when this executor carries a - // source-write origin (Event-Plane AFTER-trigger fire path) AND - // the node is clustered, a task whose target collection is homed - // on a remote node must be dispatched to that node via the - // cross-shard event subsystem — NOT written to the local core - // (the historical silent mis-write). Stored procedures and - // normal client SQL carry no origin, so `route` is `None` and - // they always take the unchanged local path below. - if let Some(origin) = self.cross_shard_origin.as_ref() { - let route = { - let routing_guard = self - .state - .cluster_routing - .as_ref() - .map(|rw| rw.read().unwrap_or_else(|p| p.into_inner())); - routing_guard.as_deref().map(|routing| { - crate::control::gateway::router::resolve_decision( - task.vshard_id.as_u32(), - self.state.node_id, - Some(routing), - None, - ) - }) - }; - - match route { - // Single-node (no routing table) or this node owns the - // target vShard: fall through to the local write path. - None | Some(crate::control::gateway::RouteDecision::Local) => {} - Some(crate::control::gateway::RouteDecision::Remote { - node_id, .. - }) => { - self.enqueue_cross_shard_write( - node_id, - origin, - task.vshard_id.as_u32(), - bound_sql, - )?; - continue; - } - Some(crate::control::gateway::RouteDecision::LeaderUnknown { - vshard_id, - }) => { - return Err(crate::Error::NotLeader { - vshard_id: crate::types::VShardId::new(vshard_id as u32), - leader_node: 0, - leader_addr: String::new(), - }); - } - Some(crate::control::gateway::RouteDecision::Broadcast { .. }) => { - // `resolve_decision` resolves a single vShard and - // never returns Broadcast; treat as an invariant - // violation rather than silently mis-routing. - return Err(crate::Error::Internal { - detail: "cross-shard trigger: resolve_decision returned \ - Broadcast for a single vShard" - .into(), - }); - } - } - } - - // The window opens before the append and closes from the - // write's outcome inside the funnel. - let owner = RecordOwner { - tenant_id: task.tenant_id, - database_id: task.database_id, - vshard_id: task.vshard_id, - }; - let minted = MintedRecords::open(&self.state.outcome_floor); - let outcome = - match minted.append_plan(&self.state.wal, owner, &task.plan, self.event_source) - { - Ok(outcome) => outcome, - Err(error) => { - // Any record appended before the error never reaches - // a core. - minted.cancel(&self.state.wal, owner, 0).await?; - return Err(error); - } - }; - - crate::control::server::dispatch_utils::dispatch_trusted_internal_write_to_data_plane( - self.state, - crate::control::server::dispatch_utils::WriteDispatch { - tenant_id: task.tenant_id, - database_id: task.database_id, - vshard_id: task.vshard_id, - plan: task.plan, - trace_id: TraceId::ZERO, - event_source: self.event_source, - txn_id: None, - wal_lsn: outcome.lsn, - resolved_now_ms: outcome.resolved_now_ms, - minted: Some(minted), - }, - ) - .await?; + // A lease this node lost ends the procedure with a retryable error + // before the statement stages. + lease_scope.check_not_revoked()?; + + // A statement led by another node ships whole and waits for the local + // COMMIT. That node plans it again and derives every write it makes, + // on every vShard, so the statement and its derived writes commit in + // one transaction there, or not at all. Every other statement stages + // into the open transaction now, so the next statement sees its + // writes. A derived write on another shard stages on that shard's + // leader, so placement reads the statement's own tasks only. + match self.statement_route(tasks.get(..planned).unwrap_or(&tasks))? { + StatementRoute::Remote { vshard_id, .. } => { + let mut guard = self.tx_ctx.lock().unwrap_or_else(|p| p.into_inner()); + guard.buffer_remote(RemoteWrite { + target_vshard: vshard_id, + sql: bound_sql.to_string(), + }) } + StatementRoute::Local => self.stage_tasks(tasks, sum_target_reads, lease_scope).await, } - - Ok(()) } - /// Enqueue a trigger-originated write for delivery to the vShard's owning - /// node via the cross-shard event dispatcher. - /// - /// Event-Plane safe: this only performs a bounded in-memory push (the - /// dispatcher's per-target queue). The durable write happens on the target - /// node's `CrossShardReceiver`, which WAL-appends and dispatches there. No - /// storage I/O or remote DML executes inline here. - fn enqueue_cross_shard_write( + /// Stage `tasks` and record `reads` in the open transaction, beginning it + /// when none is open. + async fn stage_tasks( &self, - target_node: u64, - origin: &super::CrossShardOrigin, - target_vshard: u32, - bound_sql: &str, + tasks: Vec, + reads: Vec, + lease_scope: crate::control::lease::QueryLeaseScope, ) -> crate::Result<()> { - let request = crate::event::cross_shard::types::CrossShardWriteRequest { - sql: bound_sql.to_string(), - tenant_id: self.tenant_id.as_u64(), - database_id: self.database_id.as_u64(), - source_vshard: origin.source_vshard, - source_lsn: origin.source_lsn, - source_sequence: origin.source_sequence, - cascade_depth: self.cascade_depth(), - source_collection: origin.source_collection.clone(), - target_vshard, - }; + let mut txn = self.txn.lock().await; + let open = self.open_txn(&mut txn).await?; + open.record_reads(reads)?; + open.stage(SystemTxnStatement { + tasks, + lease_scope: std::sync::Arc::new(lease_scope), + }) + .await + .map_err(crate::Error::from) + } - let dispatcher = - self.state - .cross_shard_dispatcher - .as_ref() - .ok_or(crate::Error::Dispatch { - detail: "cross-shard dispatcher not initialised for trigger origination" - .to_string(), - })?; - - if !dispatcher.enqueue(target_node, request) { - return Err(crate::Error::Dispatch { - detail: format!("cross-shard send queue full for target node {target_node}"), - }); + /// The open transaction, begun now when none is open. A joined body + /// joins its statement's transaction instead. + async fn open_txn<'g>( + &self, + txn: &'g mut Option>, + ) -> crate::Result<&'g OpenSystemTxn<'a>> { + if txn.is_none() { + let open = match self.joined { + Some(ctx) => { + OpenSystemTxn::join(self.state, self.identity.clone(), ctx, self.tenant_id) + .await? + } + None => { + let mut open = + OpenSystemTxn::begin(self.state, self.identity.clone(), self.event_source)?; + if let Some((key, target_vshard)) = self.commit_key() { + open.set_applied_key(key, target_vshard); + } + open + } + }; + *txn = Some(open); } - - Ok(()) + txn.as_ref().ok_or(crate::Error::Internal { + detail: "the procedural transaction did not open".into(), + }) } /// Return the procedural session's identity for use when dispatching SQL extensions. @@ -267,119 +235,150 @@ impl<'a> StatementExecutor<'a> { // ── Transaction control ───────────────────────────────────────────── pub(super) async fn execute_commit(&self) -> crate::Result<()> { + self.refuse_in_atomic_body("COMMIT")?; self.flush_transaction_buffer().await } - pub(super) fn execute_rollback(&self) -> crate::Result<()> { - self.with_tx_ctx("ROLLBACK", |ctx| { - ctx.rollback(); - Ok(()) - }) + pub(super) async fn execute_rollback(&self) -> crate::Result<()> { + self.refuse_in_atomic_body("ROLLBACK")?; + self.discard_transaction_buffer().await; + Ok(()) } - pub(super) fn execute_savepoint(&self, name: &str) -> crate::Result<()> { - self.with_tx_ctx("SAVEPOINT", |ctx| { + pub(super) async fn execute_savepoint(&self, name: &str) -> crate::Result<()> { + let mut txn = self.txn.lock().await; + self.open_txn(&mut txn) + .await? + .savepoint(self.tenant_id, name) + .await?; + self.with_tx_ctx(|ctx| { ctx.savepoint(name); Ok(()) }) } - pub(super) fn execute_rollback_to(&self, name: &str) -> crate::Result<()> { - self.with_tx_ctx("ROLLBACK TO", |ctx| ctx.rollback_to(name)) + pub(super) async fn execute_rollback_to(&self, name: &str) -> crate::Result<()> { + self.with_tx_ctx(|ctx| ctx.rollback_to(name))?; + let txn = self.txn.lock().await; + match txn.as_ref() { + Some(open) => open.rollback_to(self.tenant_id, name).await, + None => Err(crate::Error::BadRequest { + detail: format!("savepoint '{name}' does not exist"), + }), + } + } + + pub(super) async fn execute_release_savepoint(&self, name: &str) -> crate::Result<()> { + self.with_tx_ctx(|ctx| ctx.release_savepoint(name))?; + let txn = self.txn.lock().await; + match txn.as_ref() { + Some(open) => open.release(name), + None => Err(crate::Error::BadRequest { + detail: format!("savepoint '{name}' does not exist"), + }), + } } - pub(super) fn execute_release_savepoint(&self, name: &str) -> crate::Result<()> { - self.with_tx_ctx("RELEASE SAVEPOINT", |ctx| ctx.release_savepoint(name)) + /// A server-run body commits once, at its end, so it refuses statements + /// that end its transaction early. + fn refuse_in_atomic_body(&self, statement: &str) -> crate::Result<()> { + match self.body { + Some(ref body) => Err(body.refuse_transaction_control(statement)), + None => Ok(()), + } } fn with_tx_ctx( &self, - stmt_name: &str, f: impl FnOnce(&mut ProcedureTransactionCtx) -> crate::Result<()>, ) -> crate::Result<()> { - match self.tx_ctx { - Some(ref tx_ctx) => { - let mut guard = tx_ctx.lock().unwrap_or_else(|p| p.into_inner()); - f(&mut guard) - } - None => Err(crate::Error::BadRequest { - detail: format!("{stmt_name} is only valid inside stored procedures"), - }), + let mut guard = self.tx_ctx.lock().unwrap_or_else(|p| p.into_inner()); + f(&mut guard) + } + + /// Roll back the open transaction and drop its held effects. + pub(super) async fn discard_transaction_buffer(&self) { + { + let mut guard = self.tx_ctx.lock().unwrap_or_else(|p| p.into_inner()); + guard.rollback(); + } + let open = self.txn.lock().await.take(); + if let Some(open) = open { + open.rollback().await; } } - /// Commit the procedure transaction buffer as one system transaction. + /// Commit the open transaction: every staged and buffered write lands in + /// one redo record that installs all of them or none. Restart replay + /// installs that same record. Each statement's descriptor leases stay on + /// the tasks it buffered until COMMIT has checked them. /// - /// Every statement's tasks stage through the same path a client - /// transaction takes, and COMMIT resolves them into one redo record that - /// installs all of them or none. Restart replay installs that same - /// record. Each statement's descriptor leases stay on the tasks it - /// buffered until COMMIT has checked them. + /// Held publishes commit in a redo record: a joined body's in its + /// statement's, every other body's in this one, so a publish and its + /// transaction commit together. Held cross-node writes are queued once + /// the commit succeeds, and a queueing error leaves a durable retry + /// record. All of them drop with the commit when it fails. pub(super) async fn flush_transaction_buffer(&self) -> crate::Result<()> { - let statements = if let Some(ref tx_ctx) = self.tx_ctx { - let mut guard = tx_ctx.lock().unwrap_or_else(|p| p.into_inner()); - guard.take_statements() - } else { - return Ok(()); + let mut effects = { + let mut guard = self.tx_ctx.lock().unwrap_or_else(|p| p.into_inner()); + guard.take_effects() }; - if statements.iter().all(|(tasks, _)| tasks.is_empty()) { - return Ok(()); + if let Some(ctx) = self.joined { + let open = self.txn.lock().await.take(); + if let Some(open) = open { + open.commit().await.map_err(crate::Error::from)?; + } + return self.defer_to_statement(ctx, effects); } - let statements = statements - .into_iter() - .map( - |(tasks, lease_scope)| crate::control::system_txn::SystemTxnStatement { - tasks, - lease_scope: std::sync::Arc::new(lease_scope), - }, - ) - .collect(); - crate::control::system_txn::run_statements_atomically( - self.state, - &self.identity_for_dispatch(), - statements, - self.event_source, - ) - .await - .map_err(crate::Error::from) + // The body's messages and its cross-node requests commit in its redo + // record, with its writes. + let mut publishes = self.redo_publishes(std::mem::take(&mut effects.publishes)); + publishes.extend(self.outbox_messages(std::mem::take(&mut effects.remote))?); + if !publishes.is_empty() { + let mut txn = self.txn.lock().await; + self.open_txn(&mut txn).await?.hold_publishes(publishes)?; + } + let open = self.txn.lock().await.take(); + if let Some(open) = open { + open.commit().await.map_err(crate::Error::from)?; + } + Ok(()) } } -/// Deterministic coverage for the cross-shard trigger ORIGINATION logic -/// (the `execute_sql` routing branch above). A full-cluster e2e test cannot -/// cover this: that harness cannot place different vShards' Raft leadership -/// on different nodes, so a same-node "remote" route never arises there. -/// Here the routing table is built directly, so both `Local` and `Remote` -/// decisions are reachable without a cluster. +/// A trigger body's local writes on a one-node cluster: staging, rollback, +/// transaction control, and committed publishes. The cluster applies its +/// Raft entries through a fake Data-Plane core, so tests observe what is +/// staged, WAL-appended and queued at each statement. /// -/// The send/receive path (dispatcher retry/DLQ/HWM-dedup, wire -/// serialization, receiver apply) is covered separately by -/// `nodedb/tests/event_cross_shard.rs` and -/// `nodedb/src/event/cross_shard/dispatcher.rs`'s own unit tests; this -/// module only proves the origination gate in `execute_sql`. +/// The cross-node half of origination runs on the multi-node cluster harness +/// (`trigger_cross_shard_origination` and `trigger_body_atomic_cross_node` +/// in the cluster test suite), where a real remote node exists. The +/// send/receive path (dispatcher retry/DLQ, dedup, wire serialization, +/// receiver apply) is covered by `nodedb/tests/inproc/cases/event_cross_shard.rs` +/// and the `event::cross_shard` unit tests. Read-your-own-writes against a +/// real Data Plane is covered by `nodedb/tests/wire/cases/procedural_txn_visibility.rs`. #[cfg(test)] -mod cross_shard_origination_tests { - use std::sync::{Arc, RwLock}; +mod origination_tests { + use std::sync::atomic::{AtomicBool, Ordering}; + use std::sync::{Arc, Mutex}; + use std::time::Duration; - use nodedb_cluster::RoutingTable; + use nodedb_physical::physical_plan::MetaOp; use nodedb_types::DatabaseId; + use crate::bridge::dispatch::{BridgeResponse, CoreChannelDataSide}; + use crate::bridge::envelope::{ErrorCode, Payload, PhysicalPlan, Response, Status}; + use crate::control::cluster::test_one_node::{OneNodeCluster, boot_with_core}; use crate::control::planner::procedural::executor::bindings::RowBindings; use crate::control::planner::procedural::executor::core::{ - CrossShardOrigin, StatementExecutor, + AtomicBody, CrossShardOrigin, StatementExecutor, }; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::neutral::collection::create::handler::create_collection; use crate::control::server::shared::ddl::neutral::collection::create::request::CreateCollectionRequest; use crate::control::state::SharedState; - use crate::event::cross_shard::{CrossShardDispatcher, CrossShardMetrics}; - use crate::types::TenantId; - use crate::wal::WalManager; - - /// This node's id in every fixture below. - const LOCAL_NODE: u64 = 1; - /// The other cluster member every fixture routes remote writes to. - const REMOTE_NODE: u64 = 2; + use crate::types::{Lsn, TenantId}; fn test_identity() -> AuthenticatedIdentity { AuthenticatedIdentity::new_internal_service( @@ -393,101 +392,165 @@ mod cross_shard_origination_tests { ) } - /// Build a `SharedState` wired for cross-shard trigger origination: a - /// 2-group routing table (data group 1 led by `LOCAL_NODE`, data group 2 - /// led by `REMOTE_NODE`) plus a live `CrossShardDispatcher`. Under - /// `RoutingTable::uniform`, even vShards map to group 1 (local) and odd - /// vShards map to group 2 (remote). Also creates `coll_name` as a - /// `document_strict` collection with `id TEXT PRIMARY KEY` so `INSERT` - /// plans against it. - async fn build_state_with_collection( - dir: &tempfile::TempDir, - coll_name: &str, - ) -> Arc { - let wal_path = dir.path().join("test.wal"); - let wal = Arc::new(WalManager::open_for_testing(&wal_path).unwrap()); - let (dispatcher, _data_sides) = crate::bridge::dispatch::Dispatcher::new(1, 16); - let mut state = SharedState::new(dispatcher, wal).unwrap(); - - { - let s = Arc::get_mut(&mut state) - .expect("sole owner: no clone has been taken yet in this fixture"); - s.node_id = LOCAL_NODE; - let routing = RoutingTable::uniform(2, &[LOCAL_NODE, REMOTE_NODE], 1); - s.cluster_routing = Some(Arc::new(RwLock::new(routing))); - s.cross_shard_dispatcher = Some(Arc::new(CrossShardDispatcher::new( - LOCAL_NODE, - Arc::new(CrossShardMetrics::new()), - ))); - } - - let identity = test_identity(); + /// Create `name` as a `document_strict` collection with + /// `id TEXT PRIMARY KEY, val INT` in `database_id`. + async fn create_strict(state: &Arc, name: &str, database_id: DatabaseId) { let columns = vec![ ("id".to_string(), "TEXT PRIMARY KEY".to_string()), ("val".to_string(), "INT".to_string()), ]; let req = CreateCollectionRequest { - name: coll_name, + name, engine: Some("document_strict"), columns: &columns, options: &[], flags: &[], balanced_raw: None, }; - create_collection(&state, &identity, &req, DatabaseId::DEFAULT) + create_collection(state, &test_identity(), &req, database_id) .await - .unwrap_or_else(|e| panic!("create_collection({coll_name}) failed: {e:?}")); + .unwrap_or_else(|e| panic!("create_collection({name}) failed: {e:?}")); + } - state + /// How the fake core answers everything that is not a staged write. + #[derive(Clone, Copy)] + enum OtherRequests { + Succeed, + Fail, } - /// Build an unclustered state with a collection in an explicit database. - /// This keeps procedural DML buffered while the test verifies planning - /// scope, and lets procedural PUBLISH complete locally. - async fn build_unrouted_state_with_collection( - dir: &tempfile::TempDir, - coll_name: &str, - database_id: DatabaseId, - ) -> Arc { - let wal_path = dir.path().join("test-unrouted.wal"); - let wal = Arc::new(WalManager::open_for_testing(&wal_path).unwrap()); - let (dispatcher, _data_sides) = crate::bridge::dispatch::Dispatcher::new(1, 16); - let state = SharedState::new(dispatcher, wal).unwrap(); - let identity = test_identity(); - let columns = vec![ - ("id".to_string(), "TEXT PRIMARY KEY".to_string()), - ("val".to_string(), "INT".to_string()), - ]; - let req = CreateCollectionRequest { - name: coll_name, - engine: Some("document_strict"), - columns: &columns, - options: &[], - flags: &[], - balanced_raw: None, - }; - create_collection(&state, &identity, &req, database_id) - .await - .unwrap_or_else(|e| panic!("create_collection({coll_name}) failed: {e:?}")); + /// A fake Data-Plane core the one-node cluster applies its Raft entries + /// through. A staged write answers with one affected row. It records the + /// staged writes and overlay releases it sees, in order. + #[derive(Clone)] + struct FakeCore { + seen: Arc>>, + fail_other: Arc, + } + + impl FakeCore { + fn new() -> Self { + Self { + seen: Arc::new(Mutex::new(Vec::new())), + fail_other: Arc::new(AtomicBool::new(false)), + } + } + + fn seen(&self) -> Vec<&'static str> { + self.seen.lock().unwrap_or_else(|p| p.into_inner()).clone() + } + + /// Answer every later request that is not a staged write as `other`. + fn answer_other(&self, other: OtherRequests) { + self.fail_other + .store(matches!(other, OtherRequests::Fail), Ordering::Relaxed); + } + + fn spawn( + &self, + state: Arc, + side: CoreChannelDataSide, + ) -> tokio::task::JoinHandle<()> { + tokio::spawn(answer(self.clone(), state, side)) + } + } + + async fn answer(core: FakeCore, state: Arc, mut side: CoreChannelDataSide) { + loop { + while let Ok(request) = side.request_rx.try_pop() { + let request = request.inner; + let kind = match &request.plan { + PhysicalPlan::Meta(MetaOp::StageWrite { .. }) => Some("stage"), + PhysicalPlan::Meta(MetaOp::DropTxnOverlay { .. }) => Some("drop_overlay"), + _ => None, + }; + if let Some(kind) = kind { + core.seen + .lock() + .unwrap_or_else(|p| p.into_inner()) + .push(kind); + } + let (status, payload, error_code) = match kind { + Some("stage") => ( + Status::Ok, + crate::data::executor::response_codec::encode_count("affected", 1) + .expect("count payload"), + None, + ), + _ if core.fail_other.load(Ordering::Relaxed) => ( + Status::Error, + Vec::new(), + Some(Box::new(ErrorCode::Internal { + detail: "fake core refuses".into(), + })), + ), + _ => (Status::Ok, Vec::new(), None), + }; + let response = Response { + request_id: request.request_id, + status, + attempt: 1, + partial: false, + payload: Payload::from_vec(payload), + watermark_lsn: Lsn::ZERO, + error_code, + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + }; + side.response_tx + .try_push(BridgeResponse { inner: response }) + .expect("fake core response queue has capacity"); + } + state.poll_and_route_responses(); + tokio::task::yield_now().await; + } + } + + /// A one-node cluster whose Data Plane is `core`, with every name in + /// `collections` created in the default database. + async fn node_with_collections(core: &FakeCore, collections: &[&str]) -> OneNodeCluster { + let spawner = core.clone(); + let cluster = boot_with_core(|_| {}, move |state, side| spawner.spawn(state, side)).await; + for name in collections { + create_strict(&cluster.state, name, DatabaseId::DEFAULT).await; + } + cluster + } + + /// The WAL's data records at or above `from`: row writes and committed + /// transaction redo. + fn data_records_since(state: &SharedState, from: Lsn) -> usize { + use nodedb_wal::record::RecordType; + state.wal.sync().expect("sync wal"); state + .wal + .replay() + .expect("read wal") + .into_iter() + .filter(|record| record.header.lsn >= from.as_u64()) + .filter(|record| { + matches!( + RecordType::from_raw(record.logical_record_type()), + Some(RecordType::Put | RecordType::Delete | RecordType::TransactionRedo) + ) + }) + .count() } - /// Procedural PUBLISH and DML must both retain the executor's explicit - /// database instead of falling back to `DatabaseId::DEFAULT`. - #[tokio::test] - async fn procedural_publish_and_dml_use_explicit_non_default_database() { - let dir = tempfile::tempdir().unwrap(); - let database_id = DatabaseId::new(9); - let state = build_unrouted_state_with_collection(&dir, "scoped_orders", database_id).await; + /// Register `name` as a durable topic in `database_id`. + fn register_topic(state: &SharedState, name: &str, database_id: DatabaseId) { let topic = crate::event::topic::TopicDef { tenant_id: 1, - name: "scoped_events".into(), + name: name.into(), retention: crate::event::cdc::stream_def::RetentionConfig::default(), owner: "cross_shard_origin_test".into(), created_at: 0, database_id, last_sequence: 0, last_lsn: 0, + last_epoch: 0, + modification_hlc: nodedb_types::Hlc::ZERO, }; // A topic exists only once it is durable: PUBLISH revalidates the // catalog row under the lifecycle lock before it accepts a message, @@ -498,16 +561,68 @@ mod cross_shard_origination_tests { .put_ep_topic(&topic) .expect("persist topic"); state.ep_topic_registry.register(topic); + } + + /// A trigger-body executor carrying a source-write origin. + fn trigger_executor(state: &SharedState) -> StatementExecutor<'_> { + StatementExecutor::with_source( + state, + test_identity(), + TenantId::new(1), + 0, + crate::event::EventSource::Trigger, + ) + .with_atomic_body(AtomicBody::trigger("probe")) + .with_cross_shard_origin(CrossShardOrigin { + source_lsn: 100, + source_sequence: 7, + source_vshard: 999, + source_collection: "src_probe".to_string(), + }) + } + + fn parse(sql: &str) -> crate::control::planner::procedural::ast::ProceduralBlock { + crate::control::planner::procedural::parse_block(sql) + .unwrap_or_else(|e| panic!("parse {sql}: {e}")) + } + + /// Writes this node's cross-shard dispatcher holds for another node. A + /// one-node cluster homes every collection here, so it never holds one. + fn pending(state: &SharedState) -> usize { + state + .cross_shard_dispatcher + .as_ref() + .expect("the cluster wiring installs the dispatcher") + .total_pending() + } + + /// PUBLISH and DML both keep the executor's explicit database instead of + /// falling back to `DatabaseId::DEFAULT`. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn procedural_publish_and_dml_use_explicit_non_default_database() { + let core = FakeCore::new(); + let cluster = node_with_collections(&core, &[]).await; + let state = &cluster.state; + let database_id = DatabaseId::new(9); + let mut database = crate::control::security::catalog::DatabaseDescriptor::default_db(); + database.id = database_id; + database.name = "scoped_database".into(); + state + .credentials + .catalog() + .put_database(&database) + .expect("add the scoped database"); + create_strict(state, "scoped_orders", database_id).await; + register_topic(state, "scoped_events", database_id); let executor = StatementExecutor::with_source_in_database( - &state, + state, test_identity(), TenantId::new(1), database_id, 0, crate::event::EventSource::User, - ) - .with_transaction_context(); + ); executor .execute_sql( @@ -522,148 +637,297 @@ mod cross_shard_origination_tests { &RowBindings::empty(), ) .await - .expect("DML must plan against the executor database"); + .expect("DML must plan and stage against the executor database"); + executor.discard_transaction_buffer().await; + drop(executor); + cluster.shutdown().await; } - /// Find a `{prefix}_` collection name whose vShard is homed on - /// `REMOTE_NODE` under the routing table `build_state_with_collection` - /// installs (odd vShard → data group 2 → `REMOTE_NODE`). - fn remote_homed_name(prefix: &str) -> String { - for i in 0..4096u32 { - let name = format!("{prefix}_{i}"); - let vshard = nodedb_cluster::routing::vshard_for_collection( - nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name), - ); - if vshard % 2 == 1 { - return name; - } - } - panic!("could not find a remote-homed collection name for prefix {prefix}"); + /// A local write stages into the overlay at its statement, before the next + /// statement plans. Nothing reaches the WAL until COMMIT. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn local_write_stages_at_its_statement() { + let core = FakeCore::new(); + let cluster = node_with_collections(&core, &["cs_stage_local"]).await; + let state = &cluster.state; + let lsn_before = state.wal.next_lsn(); + let executor = trigger_executor(state); + + executor + .execute_sql( + "INSERT INTO cs_stage_local (id, val) VALUES ('local', 1)", + &RowBindings::empty(), + ) + .await + .expect("the local write stages"); + + assert_eq!( + core.seen(), + vec!["stage"], + "staged before the statement returns" + ); + assert_eq!( + data_records_since(state, lsn_before), + 0, + "nothing is WAL-appended" + ); + assert_eq!(pending(state), 0, "a local write is never queued remotely"); + executor.discard_transaction_buffer().await; + drop(executor); + cluster.shutdown().await; } - /// Find a `{prefix}_` collection name whose vShard is homed on - /// `LOCAL_NODE` (even vShard → data group 1 → `LOCAL_NODE`). - fn local_homed_name(prefix: &str) -> String { - for i in 0..4096u32 { - let name = format!("{prefix}_{i}"); - let vshard = nodedb_cluster::routing::vshard_for_collection( - nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &name), + /// An executor dropped with its transaction open (its future cancelled) + /// releases the staging overlay on a spawned rollback. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_dropped_open_transaction_releases_its_overlay() { + let core = FakeCore::new(); + // The spawned rollback takes its owned handle to the state through + // the node's gateway. + let cluster = node_with_collections(&core, &["cs_drop_local"]).await; + let state = &cluster.state; + let tgt = "cs_drop_local"; + + let executor = trigger_executor(state); + executor + .execute_sql( + &format!("INSERT INTO {tgt} (id, val) VALUES ('dropped', 1)"), + &RowBindings::empty(), + ) + .await + .expect("the local write stages"); + drop(executor); + + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + while !core.seen().contains(&"drop_overlay") { + assert!( + tokio::time::Instant::now() < deadline, + "the dropped transaction never released its overlay: {:?}", + core.seen() ); - if vshard.is_multiple_of(2) { - return name; - } + tokio::time::sleep(Duration::from_millis(10)).await; } - panic!("could not find a local-homed collection name for prefix {prefix}"); - } - - /// WHEN `cross_shard_origin` is set AND the write's target vShard - /// resolves to a REMOTE node, THEN `execute_sql` enqueues a - /// `CrossShardWriteRequest` to that node's dispatcher queue instead of - /// taking the local write path. Non-vacuous: reverting the `Some(origin)` - /// branch in `execute_sql` back to always-local would make this test - /// enqueue nothing and fail the `total_pending() == 1` assertion. - #[tokio::test] - async fn trigger_write_to_remote_homed_collection_enqueues_cross_shard() { - let dir = tempfile::tempdir().unwrap(); - let tgt = remote_homed_name("cs_origin_remote"); - let state = build_state_with_collection(&dir, &tgt).await; - - let executor = StatementExecutor::with_source( - &state, - test_identity(), - TenantId::new(1), - 0, - crate::event::EventSource::Trigger, - ) - .with_cross_shard_origin(CrossShardOrigin { - source_lsn: 100, - source_sequence: 7, - source_vshard: 999, - source_collection: "src_probe".to_string(), - }); + cluster.shutdown().await; + } - let sql = format!("INSERT INTO {tgt} (id, val) VALUES ('fired', 1)"); + /// A body whose local COMMIT fails reports the refusal and queues + /// nothing for another node. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_refused_local_commit_fails_the_body() { + let core = FakeCore::new(); + let cluster = node_with_collections(&core, &["cs_wait_local"]).await; + let state = &cluster.state; + let executor = trigger_executor(state); executor - .execute_sql(&sql, &RowBindings::empty()) + .execute_sql( + "INSERT INTO cs_wait_local (id, val) VALUES ('l', 1)", + &RowBindings::empty(), + ) .await - .expect("execute_sql should enqueue the remote write, not fail"); + .expect("the local write stages"); + + core.answer_other(OtherRequests::Fail); + let committed = + tokio::time::timeout(Duration::from_secs(5), executor.flush_transaction_buffer()) + .await + .expect("the refused COMMIT returns"); + core.answer_other(OtherRequests::Succeed); + assert!(committed.is_err(), "the fake core refuses the COMMIT"); + assert_eq!(pending(state), 0, "a refused body queues nothing"); + drop(executor); + cluster.shutdown().await; + } - let dispatcher = state - .cross_shard_dispatcher - .as_ref() - .expect("dispatcher configured by build_state_with_collection"); - assert_eq!( - dispatcher.total_pending(), - 1, - "exactly one cross-shard write must be enqueued" + /// A body whose second statement fails WAL-appends nothing and rolls its + /// staged write back out of the overlay. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn failed_body_leaves_no_local_write() { + let core = FakeCore::new(); + let cluster = node_with_collections(&core, &["cs_fail_local"]).await; + let state = &cluster.state; + let lsn_before = state.wal.next_lsn(); + + let executor = trigger_executor(state); + let block = parse( + "BEGIN INSERT INTO cs_fail_local (id, val) VALUES ('a', 1); \ + INSERT INTO cs_fail_missing (id, val) VALUES ('b', 2); END", ); - assert_eq!( - dispatcher.active_targets(), - vec![REMOTE_NODE], - "the write must be enqueued to the target's owning node" + let result = tokio::time::timeout( + Duration::from_secs(5), + executor.execute_block(&block, &RowBindings::empty()), + ) + .await + .expect("a failed body returns"); + assert!( + result.is_err(), + "the second statement's collection is missing" ); - let pending = dispatcher.peek_pending(REMOTE_NODE); - assert_eq!(pending.len(), 1); - let req = &pending[0]; - assert_eq!(req.sql, sql); - assert_eq!(req.source_lsn, 100); - assert_eq!(req.source_sequence, 7); - assert_eq!(req.source_vshard, 999); - assert_eq!(req.source_collection, "src_probe"); - assert_eq!(req.cascade_depth, 0); assert_eq!( - req.target_vshard, - nodedb_cluster::routing::vshard_for_collection(nodedb_types::CollectionKey::from_bare( - DatabaseId::DEFAULT, - &tgt, - )) + data_records_since(state, lsn_before), + 0, + "nothing is WAL-appended" + ); + let seen = core.seen(); + assert_eq!( + seen.first(), + Some(&"stage"), + "the first write staged: {seen:?}" + ); + assert!( + seen.contains(&"drop_overlay"), + "the failed body releases its overlay: {seen:?}" ); + assert_eq!(pending(state), 0); + drop(executor); + cluster.shutdown().await; } - /// Companion gate assertion: with the SAME `cross_shard_origin` set, a - /// write whose target vShard resolves to `Local` (this node owns it) - /// must NOT be enqueued to the cross-shard dispatcher — proving the gate - /// is routing-driven, not "always enqueue when origin is set". No Data - /// Plane core is running to drain the SPSC bridge in this fixture, so the - /// local path's `dispatch_write_to_data_plane` await never resolves on - /// its own; bounding it with a timeout is enough to observe that the - /// cross-shard dispatcher was never touched before that await blocks. - #[tokio::test] - async fn trigger_write_to_locally_homed_collection_never_enqueues_cross_shard() { - let dir = tempfile::tempdir().unwrap(); - let tgt = local_homed_name("cs_origin_local"); - let state = build_state_with_collection(&dir, &tgt).await; - - let executor = StatementExecutor::with_source( - &state, + /// COMMIT and ROLLBACK are refused inside a trigger body. SAVEPOINT is + /// allowed, and a stored procedure still accepts COMMIT. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn transaction_control_is_refused_in_trigger_bodies() { + let core = FakeCore::new(); + let cluster = node_with_collections(&core, &[]).await; + let state = &cluster.state; + + for statement in ["COMMIT", "ROLLBACK"] { + let result = trigger_executor(state) + .execute_block( + &parse(&format!("BEGIN {statement}; END")), + &RowBindings::empty(), + ) + .await; + assert!( + matches!(result, Err(crate::Error::NotInTransactionBlock { .. })), + "{statement} in a trigger body must be refused, got {result:?}" + ); + } + + let savepoints = trigger_executor(state); + savepoints + .execute_savepoint("sp1") + .await + .expect("SAVEPOINT is allowed in a trigger body"); + savepoints + .execute_rollback_to("sp1") + .await + .expect("ROLLBACK TO is allowed in a trigger body"); + savepoints + .execute_release_savepoint("sp1") + .await + .expect("RELEASE SAVEPOINT is allowed in a trigger body"); + savepoints.discard_transaction_buffer().await; + drop(savepoints); + + StatementExecutor::with_source( + state, test_identity(), TenantId::new(1), 0, - crate::event::EventSource::Trigger, + crate::event::EventSource::User, ) - .with_cross_shard_origin(CrossShardOrigin { - source_lsn: 1, - source_sequence: 1, - source_vshard: 0, - source_collection: "src_probe".to_string(), - }); + .execute_block(&parse("BEGIN COMMIT; END"), &RowBindings::empty()) + .await + .expect("a stored procedure accepts COMMIT"); + cluster.shutdown().await; + } - let sql = format!("INSERT INTO {tgt} (id, val) VALUES ('local', 1)"); - let _ = tokio::time::timeout( - std::time::Duration::from_millis(200), - executor.execute_sql(&sql, &RowBindings::empty()), + /// A shipped block never carries PUBLISH: its origin publishes from its + /// own commit. One that does is refused before anything applies, so the + /// receiver never holds a publish it owes after its commit. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn a_shipped_block_refuses_publish() { + let core = FakeCore::new(); + let cluster = node_with_collections(&core, &["shipped_rows"]).await; + let state = &cluster.state; + register_topic(state, "shipped_events", DatabaseId::DEFAULT); + + let result = StatementExecutor::with_source( + state, + test_identity(), + TenantId::new(1), + 1, + crate::event::EventSource::Trigger, + ) + .with_atomic_body(AtomicBody::CrossShardApply) + .execute_block( + &parse("BEGIN PUBLISH TO shipped_events 'x'; END"), + &RowBindings::empty(), ) .await; + assert!( + matches!(result, Err(crate::Error::BadRequest { .. })), + "{result:?}" + ); + cluster.shutdown().await; + } - let dispatcher = state - .cross_shard_dispatcher - .as_ref() - .expect("dispatcher configured by build_state_with_collection"); - assert_eq!( - dispatcher.total_pending(), - 0, - "a Local-routed write must never be enqueued to the cross-shard dispatcher" + /// A PUBLISH in a body that fails is never sent. The same PUBLISH in a + /// body that succeeds is sent once. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn publish_in_a_failed_body_is_never_sent() { + let core = FakeCore::new(); + // One node homes every collection, so the body commits on this + // node's own apply. + let cluster = node_with_collections(&core, &["body_rows"]).await; + let state = &cluster.state; + register_topic(state, "body_events", DatabaseId::DEFAULT); + let mut messages = state + .ep_topic_registry + .sender(DatabaseId::DEFAULT, 1, "body_events") + .expect("topic sender") + .subscribe(); + + let failed = trigger_executor(state) + .execute_block( + &parse("BEGIN PUBLISH TO body_events 'rolled back'; RAISE EXCEPTION 'boom'; END"), + &RowBindings::empty(), + ) + .await; + assert!(failed.is_err(), "the body raises"); + assert!( + committed_publishes(state).is_empty(), + "a rolled-back body commits no message" + ); + + trigger_executor(state) + .execute_block( + &parse("BEGIN PUBLISH TO body_events 'committed'; END"), + &RowBindings::empty(), + ) + .await + .expect("the body commits"); + let committed = committed_publishes(state); + assert_eq!(committed.len(), 1, "the message commits once"); + assert_eq!(committed[0].topic, "body_events"); + assert_eq!(committed[0].payload, "committed"); + assert!( + messages.try_recv().is_err(), + "the body sends nothing itself: the Event Plane delivers the committed message" ); + drop(messages); + cluster.shutdown().await; + } + + /// Every `PUBLISH TO` message the node's WAL holds in a committed redo + /// record, in WAL order. + fn committed_publishes(state: &SharedState) -> Vec { + state.wal.sync().expect("sync wal"); + state + .wal + .replay() + .expect("read wal") + .into_iter() + .filter(|record| { + nodedb_wal::record::RecordType::from_raw(record.logical_record_type()) + == Some(nodedb_wal::record::RecordType::TransactionRedo) + }) + .flat_map(|record| { + crate::wal::RedoRecord::from_bytes(&record.payload) + .expect("decode redo record") + .publishes + }) + .collect() } } diff --git a/nodedb/src/control/planner/procedural/executor/core/mod.rs b/nodedb/src/control/planner/procedural/executor/core/mod.rs index 351f2eb69..ec6023d13 100644 --- a/nodedb/src/control/planner/procedural/executor/core/mod.rs +++ b/nodedb/src/control/planner/procedural/executor/core/mod.rs @@ -8,13 +8,24 @@ //! - `statement`: single-statement dispatch //! - `control_flow`: IF/WHILE/LOOP/FOR execution //! - `dispatch`: DML dispatch, ASSIGN, RETURN, transaction control +//! - `body`: server-run bodies that execute as one transaction +//! - `route`: local-or-remote placement of a trigger statement +//! - `outbox`: committing cross-node writes and PUBLISH as redo-record +//! messages +//! - `applied`: the key a body's commit records, so a body fired again +//! applies once +mod applied; mod block; +mod body; mod control_flow; mod dispatch; +mod outbox; +mod route; pub mod sql_literal_concat; mod state; mod statement; +pub use body::AtomicBody; pub(super) use state::Flow; pub use state::{CrossShardOrigin, MAX_CASCADE_DEPTH, StatementExecutor}; diff --git a/nodedb/src/control/planner/procedural/executor/core/outbox.rs b/nodedb/src/control/planner/procedural/executor/core/outbox.rs new file mode 100644 index 000000000..0d186ad79 --- /dev/null +++ b/nodedb/src/control/planner/procedural/executor/core/outbox.rs @@ -0,0 +1,163 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A transaction's post-commit effects: cross-node trigger writes and +//! `PUBLISH TO`. +//! +//! Every `PUBLISH TO` message commits in its transaction's redo record: a +//! joined body's in its statement's record, every other body's and a stored +//! procedure's in its own. The Event Plane delivers each from there (see +//! `crate::event::topic::committed`). +//! +//! Cross-node writes commit in the body's redo record as outbox messages, and +//! the Event Plane delivers them from there (see +//! `crate::event::topic::committed::outbox`). One request holds every +//! statement the body sent to other nodes, applied by its receiver as one +//! transaction, all or none. The request's dedup key is the source event's replicated +//! identity `(source_vshard, source_lsn, source_sequence)` plus the body's +//! tag, so re-sending it after a lost reply or a leader change applies it +//! once. + +use super::super::transaction::{PostCommitEffects, RemoteWrite}; +use super::StatementExecutor; +use crate::control::server::shared::session::DmlTxnCtx; +use crate::control::sql_dispatch::PreparedPublish; +use crate::event::cross_shard::types::CrossShardWriteRequest; + +/// The owner a stored procedure's messages name: a procedure runs with no +/// body. +const UNNAMED_PUBLISH_OWNER: &str = "procedure"; + +/// Wrap statements as one procedural block for the receiver's parser. +fn block_sql(statements: &[String]) -> String { + let mut sql = String::from("BEGIN\n"); + for statement in statements { + sql.push_str(statement.trim().trim_end_matches(';').trim_end()); + sql.push_str(";\n"); + } + sql.push_str("END;"); + sql +} + +impl StatementExecutor<'_> { + /// One request holding every remote write of a committed body, addressed + /// to the first write's vShard. `None` for a body with no remote write. + /// + /// The receiver applies the block as one transaction, staging each write + /// on its vShard's leader, and commits it all or none: through Calvin + /// when it spans vShards. The request's key is the source event's + /// replicated identity with the body's tag, so a retry after a leader + /// change finds it applied. + fn cross_shard_request( + &self, + writes: Vec, + ) -> crate::Result> { + let Some(target_vshard) = writes.first().map(|write| write.target_vshard) else { + return Ok(None); + }; + let origin = self + .cross_shard_origin + .as_ref() + .ok_or(crate::Error::Internal { + detail: "a body without a cross-shard origin held remote writes".into(), + })?; + let origin_tag = self + .body + .as_ref() + .map(|body| body.origin_tag(self.database_id)) + .unwrap_or_default(); + + let statements: Vec = writes.into_iter().map(|write| write.sql).collect(); + Ok(Some(CrossShardWriteRequest { + sql: block_sql(&statements), + tenant_id: self.tenant_id.as_u64(), + database_id: self.database_id.as_u64(), + source_vshard: origin.source_vshard, + source_lsn: origin.source_lsn, + source_sequence: origin.source_sequence, + origin: format!("{origin_tag}/remote"), + cascade_depth: self.cascade_depth(), + source_collection: origin.source_collection.clone(), + target_vshard, + })) + } + + /// The outbox messages that carry the cross-node writes `remote` of this + /// transaction. They commit in its redo record with its writes and its + /// applied key, so a committed body never loses its request (see + /// `crate::event::topic::committed::outbox`). + pub(super) fn outbox_messages( + &self, + remote: Vec, + ) -> crate::Result> { + let owner = self.body.as_ref().map_or_else( + || UNNAMED_PUBLISH_OWNER.to_owned(), + |body| body.origin_tag(self.database_id), + ); + match self.cross_shard_request(remote)? { + Some(request) => Ok(vec![ + crate::event::topic::committed::outbox::outbox_message(&owner, &request)?, + ]), + None => Ok(Vec::new()), + } + } + + /// Hand a joined body's effects to its statement's transaction: its + /// publishes commit in that transaction's redo record, and drop when it + /// rolls back. A joined body carries no cross-shard origin, so every + /// statement it runs stages here and it holds no remote write. + pub(super) fn defer_to_statement( + &self, + ctx: &DmlTxnCtx<'_>, + effects: PostCommitEffects, + ) -> crate::Result<()> { + if !effects.remote.is_empty() { + return Err(crate::Error::Internal { + detail: "a trigger body joined to its statement held a cross-node write".into(), + }); + } + let publishes = self.redo_publishes(effects.publishes); + ctx.sessions.buffer_publishes(ctx.session_id, publishes); + Ok(()) + } + + /// `publishes` as a redo record carries them, each naming this body as + /// its owner, or [`UNNAMED_PUBLISH_OWNER`] with no body. + pub(super) fn redo_publishes( + &self, + publishes: Vec, + ) -> Vec { + let owner = self.body.as_ref().map_or_else( + || UNNAMED_PUBLISH_OWNER.to_owned(), + |body| body.origin_tag(self.database_id), + ); + publishes + .into_iter() + .map(|publish| crate::wal::RedoPublish { + owner: owner.clone(), + database_id: publish.database_id, + tenant_id: publish.tenant_id, + topic: publish.topic, + payload: publish.payload, + metadata_floor: publish.metadata_floor, + position: None, + }) + .collect() + } +} + +#[cfg(test)] +mod tests { + use super::block_sql; + + #[test] + fn statements_become_one_block() { + let sql = block_sql(&[ + "INSERT INTO a (id) VALUES ('x')".into(), + " INSERT INTO b (id) VALUES ('y'); ".into(), + ]); + assert_eq!( + sql, + "BEGIN\nINSERT INTO a (id) VALUES ('x');\nINSERT INTO b (id) VALUES ('y');\nEND;" + ); + } +} diff --git a/nodedb/src/control/planner/procedural/executor/core/route.rs b/nodedb/src/control/planner/procedural/executor/core/route.rs new file mode 100644 index 000000000..ad0dfd40f --- /dev/null +++ b/nodedb/src/control/planner/procedural/executor/core/route.rs @@ -0,0 +1,119 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Placement of one planned trigger statement: this node, or one remote node. + +use nodedb_physical::physical_task::PhysicalTask; + +use super::StatementExecutor; +use crate::control::gateway::RouteDecision; +use crate::control::gateway::router::resolve_decision; + +/// Where one statement's tasks execute. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum StatementRoute { + /// Every task is led by this node. It joins the body's local transaction. + Local, + /// Every task writes `vshard_id`, led by `node_id`. The statement is + /// re-planned there, with every write it derives. + Remote { node_id: u64, vshard_id: u32 }, +} + +impl StatementRoute { + /// A local statement can span local vShards. A remote one names exactly + /// one vShard: a request addresses one vShard, and its dedup key rides + /// that vShard's redo record. + fn same_target(self, other: Self) -> bool { + match (self, other) { + (Self::Local, Self::Local) => true, + (Self::Remote { .. }, Self::Remote { .. }) => self == other, + _ => false, + } + } +} + +impl StatementExecutor<'_> { + /// Resolve where `tasks` execute. Only a body carrying a cross-shard + /// origin routes remotely; every other body runs all of its tasks here. + /// + /// A statement split across nodes, or across remote vShards, is refused. + /// It is shipped as SQL, so sending it whole to one node will write the + /// other node's rows there, and a request addresses one vShard. + pub(super) fn statement_route(&self, tasks: &[PhysicalTask]) -> crate::Result { + if self.cross_shard_origin.is_none() { + return Ok(StatementRoute::Local); + } + let Some(routing) = self.state.cluster_routing.as_ref() else { + return Ok(StatementRoute::Local); + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + + let mut route: Option = None; + for task in tasks { + let vshard_id = task.vshard_id.as_u32(); + let next = match resolve_decision(vshard_id, self.state.node_id, Some(&*routing), None) + { + RouteDecision::Local => StatementRoute::Local, + RouteDecision::Remote { node_id, .. } => { + StatementRoute::Remote { node_id, vshard_id } + } + RouteDecision::LeaderUnknown { .. } => { + return Err(crate::Error::NotLeader { + vshard_id: task.vshard_id, + leader_node: 0, + leader_addr: String::new(), + leader_term: 0, + }); + } + RouteDecision::Broadcast { .. } => { + return Err(crate::Error::Internal { + detail: "cross-shard trigger: resolve_decision returned Broadcast \ + for a single vShard" + .into(), + }); + } + }; + route = match route { + None => Some(next), + Some(prev) if prev.same_target(next) => Some(prev), + Some(_) => { + return Err(crate::Error::BadRequest { + detail: "a trigger statement writes vShards led by different nodes, \ + or several vShards on another node; write each collection \ + in its own statement" + .into(), + }); + } + }; + } + Ok(route.unwrap_or(StatementRoute::Local)) + } +} + +#[cfg(test)] +mod tests { + use super::StatementRoute; + + #[test] + fn a_remote_placement_names_one_vshard() { + let a = StatementRoute::Remote { + node_id: 2, + vshard_id: 1, + }; + let b = StatementRoute::Remote { + node_id: 2, + vshard_id: 5, + }; + let c = StatementRoute::Remote { + node_id: 3, + vshard_id: 1, + }; + assert!(a.same_target(a)); + assert!( + !a.same_target(b), + "two vShards on one node are two requests" + ); + assert!(!a.same_target(c)); + assert!(!a.same_target(StatementRoute::Local)); + assert!(StatementRoute::Local.same_target(StatementRoute::Local)); + } +} diff --git a/nodedb/src/control/planner/procedural/executor/core/state.rs b/nodedb/src/control/planner/procedural/executor/core/state.rs index fa20841a8..46a1bffb4 100644 --- a/nodedb/src/control/planner/procedural/executor/core/state.rs +++ b/nodedb/src/control/planner/procedural/executor/core/state.rs @@ -4,29 +4,37 @@ use std::collections::HashMap; use std::sync::{Arc, Mutex}; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::shared::session::DmlTxnCtx; use crate::control::state::SharedState; +use crate::control::system_txn::OpenSystemTxn; use crate::types::{DatabaseId, TenantId}; +use crate::wal::CrossShardAppliedKey; use super::super::transaction::ProcedureTransactionCtx; +use super::body::AtomicBody; /// Maximum trigger cascade depth (trigger A fires trigger B fires trigger A). pub const MAX_CASCADE_DEPTH: u32 = 16; /// Source-write context propagated into a trigger body's executor so that DML -/// targeting a remote-homed collection is dispatched to the owning node via the -/// cross-shard event subsystem instead of being silently mis-written to the -/// local Data-Plane core. +/// targeting a remote-homed collection is sent to the owning node through the +/// cross-shard event subsystem, after the body's local commit succeeds. /// /// Populated ONLY by the Event-Plane AFTER-trigger fire path (from the source /// `WriteEvent`). Stored procedures and normal client SQL leave it `None`; its /// presence is the gate that enables cross-shard routing in `execute_sql`. +/// +/// The receiver deduplicates on `(source_vshard, source_lsn, source_sequence)` +/// plus the emitting body and target vShard, so a resend of one body's writes +/// applies once. #[derive(Debug, Clone)] pub struct CrossShardOrigin { - /// LSN of the source write that fired the trigger (target-side HWM dedup key). + /// The source event's replicated identity: its position's index, which + /// every replica shares (`event::trigger::lane::action_identity`). pub source_lsn: u64, - /// Sequence number of the source write (monotonic per core/collection). + /// The source event's position sequence within its write. pub source_sequence: u64, - /// vShard that owns the source collection (dedup key on the target). + /// vShard that owns the source collection. pub source_vshard: u32, /// Collection whose write fired the trigger. pub source_collection: String, @@ -43,11 +51,26 @@ pub struct StatementExecutor<'a> { pub(super) event_source: crate::event::EventSource, /// Arc required (not RefCell) because execute_statement returns `+ Send` futures. pub(super) new_mutations: Arc>>, - pub(super) tx_ctx: Option>>, + /// The open transaction every statement stages into, begun by the first + /// statement after the previous COMMIT or ROLLBACK. + pub(super) txn: tokio::sync::Mutex>>, + /// Effects held until the open transaction commits. + pub(super) tx_ctx: Arc>, + /// `Some` for a server-run body, which refuses COMMIT and ROLLBACK and + /// commits once at its end. `None` for a stored procedure. + pub(super) body: Option, pub(super) out_values: Arc>>, /// Cross-shard origin context; `Some` only in the Event-Plane trigger fire /// path. Gates remote-write dispatch in `execute_sql`. pub(super) cross_shard_origin: Option, + /// The cross-shard request this body applies, with the vShard it + /// addresses. Its commit records the key in the same redo record as that + /// vShard's writes. + pub(super) applied_key: Option<(CrossShardAppliedKey, u32)>, + /// The triggering statement's transaction, for a BEFORE, INSTEAD OF or + /// SYNC AFTER body. The body's writes stage into it and commit with the + /// statement. + pub(super) joined: Option<&'a DmlTxnCtx<'a>>, } /// Control flow signal from statement execution. @@ -111,28 +134,48 @@ impl<'a> StatementExecutor<'a> { cascade_depth, event_source, new_mutations: Arc::new(Mutex::new(HashMap::new())), - tx_ctx: None, + txn: tokio::sync::Mutex::new(None), + tx_ctx: Arc::new(Mutex::new(ProcedureTransactionCtx::new())), + body: None, out_values: Arc::new(Mutex::new(HashMap::new())), cross_shard_origin: None, + applied_key: None, + joined: None, } } - /// Enable procedure transaction context for COMMIT/ROLLBACK/SAVEPOINT. - pub fn with_transaction_context(mut self) -> Self { - self.tx_ctx = Some(Arc::new(Mutex::new(ProcedureTransactionCtx::new()))); + /// Run as a server-run body: one transaction, committed at the end of + /// the block, with COMMIT and ROLLBACK refused. + pub fn with_atomic_body(mut self, body: AtomicBody) -> Self { + self.body = Some(body); self } /// Attach cross-shard origin context (Event-Plane AFTER-trigger fire path). /// - /// When set, `execute_sql` route-resolves every write task: a task homed on - /// a remote node is dispatched to that node via the cross-shard dispatcher - /// instead of being written to the local core. + /// When set, `execute_sql` route-resolves every statement: one led by a + /// remote node is held and sent there once the body's local commit + /// succeeds. pub fn with_cross_shard_origin(mut self, origin: CrossShardOrigin) -> Self { self.cross_shard_origin = Some(origin); self } + /// Apply a cross-shard request addressed to `target_vshard`: the commit + /// writes `key` into that vShard's redo record, so the key is recorded + /// exactly when the writes are. + pub fn with_applied_key(mut self, key: CrossShardAppliedKey, target_vshard: u32) -> Self { + self.applied_key = Some((key, target_vshard)); + self + } + + /// Join the triggering statement's transaction `ctx`: the body's writes + /// stage into it, and the statement's COMMIT commits them. + pub fn joined_into(mut self, ctx: &'a DmlTxnCtx<'a>) -> Self { + self.joined = Some(ctx); + self + } + pub fn take_new_mutations(&self) -> HashMap { let mut guard = self.new_mutations.lock().unwrap_or_else(|p| p.into_inner()); std::mem::take(&mut *guard) diff --git a/nodedb/src/control/planner/procedural/executor/core/statement.rs b/nodedb/src/control/planner/procedural/executor/core/statement.rs index 16810c2ab..337f031dd 100644 --- a/nodedb/src/control/planner/procedural/executor/core/statement.rs +++ b/nodedb/src/control/planner/procedural/executor/core/statement.rs @@ -91,19 +91,19 @@ impl<'a> StatementExecutor<'a> { Ok(Flow::Continue) } Statement::Rollback => { - self.execute_rollback()?; + self.execute_rollback().await?; Ok(Flow::Continue) } Statement::Savepoint { name } => { - self.execute_savepoint(name)?; + self.execute_savepoint(name).await?; Ok(Flow::Continue) } Statement::RollbackTo { name } => { - self.execute_rollback_to(name)?; + self.execute_rollback_to(name).await?; Ok(Flow::Continue) } Statement::ReleaseSavepoint { name } => { - self.execute_release_savepoint(name)?; + self.execute_release_savepoint(name).await?; Ok(Flow::Continue) } } diff --git a/nodedb/src/control/planner/procedural/executor/eval.rs b/nodedb/src/control/planner/procedural/executor/eval.rs index b4ae52bda..d59e46977 100644 --- a/nodedb/src/control/planner/procedural/executor/eval.rs +++ b/nodedb/src/control/planner/procedural/executor/eval.rs @@ -19,7 +19,7 @@ use crate::types::TenantId; /// a separate, sqlparser-based tree-walk evaluator with no dependency on /// `nodedb_query` (whose own `EvalError::DivisionByZero`, from the /// DataFusion-backed row-expression evaluator, covers a different code path -/// entirely). Every other historically `None`-folding case (unknown +/// entirely). Every other `None`-folding case (unknown /// identifiers, unsupported operators, overflow, non-finite float results) /// is unaffected and keeps folding to `Ok(None)` — see `try_eval_constant`, /// `evaluate_to_value`. @@ -125,15 +125,19 @@ pub async fn evaluate_to_value( /// /// Returns `Ok(None)` for anything this mini-evaluator doesn't handle /// (unparseable input, unknown identifiers, unsupported operators, -/// overflow) — callers fold that to `Value::Null`, matching historical -/// behavior. Returns `Err(ConstEvalError::DivisionByZero)` only when +/// overflow) — callers fold that to `Value::Null`. Returns `Err(ConstEvalError::DivisionByZero)` only when /// evaluation reaches a `/` or `%` with a zero divisor. fn try_eval_constant(sql: &str) -> Result, ConstEvalError> { let trimmed = sql.trim(); - let expr_str = if trimmed.to_uppercase().starts_with("SELECT ") { - trimmed[7..].trim() - } else { - trimmed + // An expression keeps its source spelling, so any whitespace can follow + // a leading `SELECT`. + let expr_str = match (trimmed.get(..6), trimmed.get(6..)) { + (Some(head), Some(rest)) + if head.eq_ignore_ascii_case("SELECT") && rest.starts_with(char::is_whitespace) => + { + rest.trim() + } + _ => trimmed, }; let dialect = sqlparser::dialect::PostgreSqlDialect {}; diff --git a/nodedb/src/control/planner/procedural/executor/transaction.rs b/nodedb/src/control/planner/procedural/executor/transaction.rs index ae2ffad7f..67816c2a2 100644 --- a/nodedb/src/control/planner/procedural/executor/transaction.rs +++ b/nodedb/src/control/planner/procedural/executor/transaction.rs @@ -1,158 +1,143 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Procedure transaction context: buffers DML tasks for COMMIT/ROLLBACK/SAVEPOINT. +//! Post-commit effects of one procedural transaction. //! -//! Stored procedures can execute COMMIT mid-body to finalize buffered DML, -//! ROLLBACK to discard it, and SAVEPOINT/ROLLBACK TO for partial rollback. -//! -//! Triggers do NOT use this — they dispatch DML immediately. +//! A body's DML stages into an open system transaction at the statement +//! (see `control::system_txn::OpenSystemTxn`). Two effects cannot stage as +//! writes: trigger writes homed on another node, and `PUBLISH TO`. Both are +//! held here and commit as messages in the transaction's redo record, which +//! the Event Plane delivers once the COMMIT succeeds. ROLLBACK drops them, +//! and a savepoint rewinds them with the staged writes. + +use crate::control::sql_dispatch::PreparedPublish; + +/// Upper bound on the post-commit effects one transaction holds. +pub const MAX_POST_COMMIT_EFFECTS: usize = 1024; + +/// One trigger statement homed on another node, held until the local commit. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct RemoteWrite { + /// vShard the statement writes. The outbox sends the request to the + /// vShard's leader at delivery. + pub target_vshard: u32, + /// The bound statement text, re-planned on the target node. + pub sql: String, +} -use nodedb_physical::physical_task::PhysicalTask; +/// Effects a committed transaction still owes, in statement order. +#[derive(Debug, Default)] +pub struct PostCommitEffects { + pub remote: Vec, + pub publishes: Vec, +} -use crate::control::lease::QueryLeaseScope; +impl PostCommitEffects { + pub fn is_empty(&self) -> bool { + self.remote.is_empty() && self.publishes.is_empty() + } +} -/// Lease scope retained for one planned statement in a procedure transaction. -/// -/// `task_start` lets savepoint rollback drop scopes for every statement it -/// discards without requiring individual tasks to own a scope. -struct BufferedStatementScope { - task_start: usize, - scope: QueryLeaseScope, +/// Positions a savepoint rewinds the held effects to. +struct Savepoint { + name: String, + remote: usize, + publishes: usize, } -/// Buffered transaction context for stored procedure execution. -/// -/// DML statements inside a procedure body are collected here until -/// an explicit COMMIT flushes them as one system transaction, or ROLLBACK -/// discards them. An implicit COMMIT occurs at the end of the procedure. +/// Post-commit effects held for the open transaction. #[derive(Default)] pub struct ProcedureTransactionCtx { - /// Buffered DML tasks awaiting COMMIT. - buffer: Vec, - /// Descriptor lease scopes, one per planned statement, retained until its - /// tasks have finished COMMIT dispatch or the statement is discarded. - statement_scopes: Vec, - /// Savepoint stack: (name, buffer_position_at_savepoint_time). - savepoints: Vec<(String, usize)>, + effects: PostCommitEffects, + /// Savepoint stack, innermost last. + savepoints: Vec, } -impl ProcedureTransactionCtx { - pub fn new() -> Self { - Self { - buffer: Vec::new(), - statement_scopes: Vec::new(), - savepoints: Vec::new(), - } - } - - /// Buffer all tasks planned for one statement and retain its descriptor - /// leases until those tasks are dispatched at COMMIT. - pub fn buffer_statement(&mut self, tasks: Vec, scope: QueryLeaseScope) { - let task_start = self.buffer.len(); - self.buffer.extend(tasks); - self.statement_scopes - .push(BufferedStatementScope { task_start, scope }); +fn over_limit() -> crate::Error { + crate::Error::BadRequest { + detail: format!( + "a procedural transaction holds at most {MAX_POST_COMMIT_EFFECTS} cross-node \ + writes and PUBLISH statements; split the body or the source statement" + ), } +} - /// Buffer a task without a lease scope. Kept as a task-only test helper. - pub fn buffer_task(&mut self, task: PhysicalTask) { - self.buffer_task_with_empty_scope(task); +impl ProcedureTransactionCtx { + pub fn new() -> Self { + Self::default() } - fn buffer_task_with_empty_scope(&mut self, task: PhysicalTask) { - self.buffer_statement(vec![task], QueryLeaseScope::empty()); + fn held(&self) -> usize { + self.effects.remote.len() + self.effects.publishes.len() } - /// Take all buffered tasks and their owned scopes (on COMMIT). The caller - /// must keep the scopes alive until every WAL append and dispatch finishes. - pub fn take_buffered(&mut self) -> (Vec, Vec) { - self.savepoints.clear(); - let scopes = std::mem::take(&mut self.statement_scopes) - .into_iter() - .map(|statement| statement.scope) - .collect(); - (std::mem::take(&mut self.buffer), scopes) + /// Hold a remote-homed write until the local COMMIT. + pub fn buffer_remote(&mut self, write: RemoteWrite) -> crate::Result<()> { + if self.held() >= MAX_POST_COMMIT_EFFECTS { + return Err(over_limit()); + } + self.effects.remote.push(write); + Ok(()) } - /// Take every buffered statement with its lease scope, in buffer order - /// (on COMMIT). Clears the savepoint stack. - pub fn take_statements(&mut self) -> Vec<(Vec, QueryLeaseScope)> { - self.savepoints.clear(); - let mut tasks = std::mem::take(&mut self.buffer); - let scopes = std::mem::take(&mut self.statement_scopes); - let mut statements = Vec::with_capacity(scopes.len() + 1); - // Each statement owns the tasks from its start to the next start. - for statement in scopes.into_iter().rev() { - let own = tasks.split_off(statement.task_start.min(tasks.len())); - statements.push((own, statement.scope)); - } - // Tasks buffered ahead of the first statement carry no lease. - if !tasks.is_empty() { - statements.push((tasks, QueryLeaseScope::empty())); + /// Hold a checked `PUBLISH TO` until the local COMMIT. + pub fn buffer_publish(&mut self, publish: PreparedPublish) -> crate::Result<()> { + if self.held() >= MAX_POST_COMMIT_EFFECTS { + return Err(over_limit()); } - statements.reverse(); - statements + self.effects.publishes.push(publish); + Ok(()) } - /// Take tasks only. Kept for existing task-oriented tests; it intentionally - /// drops the associated scopes when the returned tasks are taken. - pub fn take_buffered_tasks(&mut self) -> Vec { - self.take_buffered().0 + /// Take the held effects (at COMMIT). Clears the savepoint stack. + pub fn take_effects(&mut self) -> PostCommitEffects { + self.savepoints.clear(); + std::mem::take(&mut self.effects) } - /// Discard all buffered tasks and their descriptor lease scopes (on - /// ROLLBACK). Clears the savepoint stack. + /// Drop the held effects (at ROLLBACK). Clears the savepoint stack. pub fn rollback(&mut self) { - self.buffer.clear(); - self.statement_scopes.clear(); + self.effects = PostCommitEffects::default(); self.savepoints.clear(); } - /// Record a savepoint at the current buffer position. + /// Record a savepoint at the current positions. pub fn savepoint(&mut self, name: &str) { - let pos = self.buffer.len(); - // Remove any existing savepoint with the same name (redefine). - self.savepoints.retain(|(n, _)| n != name); - self.savepoints.push((name.to_string(), pos)); + // A redefined name moves to the new position. + self.savepoints.retain(|sp| sp.name != name); + self.savepoints.push(Savepoint { + name: name.to_string(), + remote: self.effects.remote.len(), + publishes: self.effects.publishes.len(), + }); } - /// Rollback to a named savepoint: discard tasks buffered after it. - pub fn rollback_to(&mut self, name: &str) -> crate::Result<()> { - let pos = self - .savepoints - .iter() - .rev() - .find(|(n, _)| n == name) - .map(|(_, p)| *p); + /// Whether a savepoint named `name` exists. + pub fn has_savepoint(&self, name: &str) -> bool { + self.savepoints.iter().any(|sp| sp.name == name) + } - match pos { - Some(p) => { - self.buffer.truncate(p); - // Savepoints occur between statements, so every discarded - // statement starts at or after the saved task position. - self.statement_scopes - .retain(|statement| statement.task_start < p); - // Remove savepoints created after this one. - if let Some(idx) = self.savepoints.iter().position(|(n, _)| n == name) { - self.savepoints.truncate(idx + 1); - } - Ok(()) - } - None => Err(crate::Error::BadRequest { + /// Drop the effects held after a savepoint, and every later savepoint. + pub fn rollback_to(&mut self, name: &str) -> crate::Result<()> { + let Some(idx) = self.savepoints.iter().rposition(|sp| sp.name == name) else { + return Err(crate::Error::BadRequest { detail: format!("savepoint '{name}' does not exist"), - }), - } + }); + }; + let (remote, publishes) = (self.savepoints[idx].remote, self.savepoints[idx].publishes); + self.effects.remote.truncate(remote); + self.effects.publishes.truncate(publishes); + self.savepoints.truncate(idx + 1); + Ok(()) } - /// Release a savepoint without rolling back (keeps buffered tasks). + /// Release a savepoint without rolling back. pub fn release_savepoint(&mut self, name: &str) -> crate::Result<()> { - let existed = self.savepoints.iter().any(|(n, _)| n == name); - if !existed { + if !self.has_savepoint(name) { return Err(crate::Error::BadRequest { detail: format!("savepoint '{name}' does not exist"), }); } - self.savepoints.retain(|(n, _)| n != name); + self.savepoints.retain(|sp| sp.name != name); Ok(()) } } @@ -160,155 +145,87 @@ impl ProcedureTransactionCtx { #[cfg(test)] mod tests { use super::*; - use crate::bridge::envelope::PhysicalPlan; - use crate::types::{TenantId, VShardId}; - use nodedb_physical::physical_plan::DocumentOp; - use nodedb_physical::physical_task::PostSetOp; - fn dummy_task(id: &str) -> PhysicalTask { - PhysicalTask { - tenant_id: TenantId::new(1), - vshard_id: VShardId::new(0), - database_id: crate::types::DatabaseId::DEFAULT, - plan: PhysicalPlan::Document(DocumentOp::PointPut { - collection: nodedb_types::QualifiedCollection::new( - crate::types::DatabaseId::DEFAULT, - "test", - ), - document_id: id.into(), - value: vec![], - surrogate: nodedb_types::Surrogate::ZERO, - pk_bytes: Vec::new(), - returning: None, - rls_filters: Vec::new(), - resolved_sum_targets: Vec::new(), - }), - post_set_op: PostSetOp::None, - txn_id: None, + fn remote(sql: &str) -> RemoteWrite { + RemoteWrite { + target_vshard: 7, + sql: sql.into(), } } - #[test] - fn buffer_and_take() { - let mut ctx = ProcedureTransactionCtx::new(); - ctx.buffer_task(dummy_task("a")); - ctx.buffer_task(dummy_task("b")); - let tasks = ctx.take_buffered_tasks(); - assert_eq!(tasks.len(), 2); - assert!(ctx.take_buffered_tasks().is_empty()); + fn publish(payload: &str) -> PreparedPublish { + PreparedPublish { + database_id: 1, + tenant_id: 1, + topic: "events".into(), + payload: payload.into(), + metadata_floor: 0, + } } #[test] - fn rollback_clears_buffer() { + fn rollback_drops_every_effect() { let mut ctx = ProcedureTransactionCtx::new(); - ctx.buffer_task(dummy_task("a")); + ctx.buffer_remote(remote("a")).unwrap(); + ctx.buffer_publish(publish("p")).unwrap(); ctx.rollback(); - assert!(ctx.take_buffered_tasks().is_empty()); + assert!(ctx.take_effects().is_empty()); } #[test] - fn commit_takes_each_statement_with_its_own_tasks() { + fn rollback_to_savepoint_truncates_effects() { let mut ctx = ProcedureTransactionCtx::new(); - ctx.buffer_statement( - vec![dummy_task("a"), dummy_task("b")], - QueryLeaseScope::empty(), - ); - ctx.buffer_statement(vec![dummy_task("c")], QueryLeaseScope::empty()); - let statements = ctx.take_statements(); - let sizes: Vec = statements.iter().map(|(tasks, _)| tasks.len()).collect(); - assert_eq!(sizes, vec![2, 1]); - assert!(ctx.take_statements().is_empty()); - } - - #[test] - fn rollback_and_commit_take_owned_statement_scopes() { - let mut ctx = ProcedureTransactionCtx::new(); - ctx.buffer_statement(vec![dummy_task("a")], QueryLeaseScope::empty()); - assert_eq!(ctx.statement_scopes.len(), 1); - - ctx.rollback(); - assert!(ctx.statement_scopes.is_empty()); + ctx.buffer_remote(remote("a")).unwrap(); + ctx.buffer_publish(publish("kept")).unwrap(); + ctx.savepoint("sp1"); + ctx.buffer_remote(remote("b")).unwrap(); + ctx.buffer_publish(publish("dropped")).unwrap(); - ctx.buffer_statement(vec![dummy_task("b")], QueryLeaseScope::empty()); - let (tasks, scopes) = ctx.take_buffered(); - assert_eq!(tasks.len(), 1); - assert_eq!(scopes.len(), 1); - assert!(ctx.statement_scopes.is_empty()); + ctx.rollback_to("sp1").unwrap(); + let effects = ctx.take_effects(); + assert_eq!(effects.remote, vec![remote("a")]); + assert_eq!(effects.publishes, vec![publish("kept")]); } #[test] - fn savepoint_and_rollback_to() { + fn nested_savepoints_rewind_in_order() { let mut ctx = ProcedureTransactionCtx::new(); - ctx.buffer_task(dummy_task("a")); + ctx.buffer_remote(remote("a")).unwrap(); ctx.savepoint("sp1"); - ctx.buffer_task(dummy_task("b")); - ctx.buffer_task(dummy_task("c")); + ctx.buffer_remote(remote("b")).unwrap(); + ctx.savepoint("sp2"); + ctx.buffer_remote(remote("c")).unwrap(); + ctx.rollback_to("sp2").unwrap(); ctx.rollback_to("sp1").unwrap(); - assert_eq!(ctx.statement_scopes.len(), 1); - let tasks = ctx.take_buffered_tasks(); - assert_eq!(tasks.len(), 1); // Only "a" remains + assert_eq!(ctx.take_effects().remote, vec![remote("a")]); } #[test] - fn rollback_to_nonexistent_fails() { + fn released_savepoint_keeps_effects() { let mut ctx = ProcedureTransactionCtx::new(); - assert!(ctx.rollback_to("nope").is_err()); - } - - #[test] - fn release_savepoint() { - let mut ctx = ProcedureTransactionCtx::new(); - ctx.buffer_task(dummy_task("a")); ctx.savepoint("sp1"); - ctx.buffer_task(dummy_task("b")); + ctx.buffer_remote(remote("a")).unwrap(); ctx.release_savepoint("sp1").unwrap(); - - // Rollback to released savepoint should fail. assert!(ctx.rollback_to("sp1").is_err()); - - // But tasks are still there. - let tasks = ctx.take_buffered_tasks(); - assert_eq!(tasks.len(), 2); + assert_eq!(ctx.take_effects().remote.len(), 1); } #[test] - fn nested_savepoints() { - let mut ctx = ProcedureTransactionCtx::new(); - ctx.buffer_task(dummy_task("a")); - ctx.savepoint("sp1"); - ctx.buffer_task(dummy_task("b")); - ctx.savepoint("sp2"); - ctx.buffer_task(dummy_task("c")); - - // Rollback to sp2: discard "c" only. - ctx.rollback_to("sp2").unwrap(); - assert_eq!(ctx.buffer.len(), 2); // a + b - - // Rollback to sp1: discard "b". - ctx.rollback_to("sp1").unwrap(); - assert_eq!(ctx.buffer.len(), 1); // a only - - let tasks = ctx.take_buffered_tasks(); - assert_eq!(tasks.len(), 1); - } - - #[test] - fn release_nonexistent_fails() { + fn unknown_savepoint_is_refused() { let mut ctx = ProcedureTransactionCtx::new(); + assert!(ctx.rollback_to("nope").is_err()); assert!(ctx.release_savepoint("nope").is_err()); } #[test] - fn savepoint_redefine() { + fn effects_are_bounded() { let mut ctx = ProcedureTransactionCtx::new(); - ctx.buffer_task(dummy_task("a")); - ctx.savepoint("sp1"); - ctx.buffer_task(dummy_task("b")); - ctx.savepoint("sp1"); // Redefine — now at position 2 - ctx.buffer_task(dummy_task("c")); - - ctx.rollback_to("sp1").unwrap(); - assert_eq!(ctx.buffer.len(), 2); // a + b (redefined position) + for _ in 0..MAX_POST_COMMIT_EFFECTS { + ctx.buffer_remote(remote("w")).unwrap(); + } + assert!(ctx.buffer_remote(remote("over")).is_err()); + assert!(ctx.buffer_publish(publish("over")).is_err()); + assert_eq!(ctx.take_effects().remote.len(), MAX_POST_COMMIT_EFFECTS); } } diff --git a/nodedb/src/control/planner/procedural/parser/block.rs b/nodedb/src/control/planner/procedural/parser/block.rs index 295a2c355..b71684f94 100644 --- a/nodedb/src/control/planner/procedural/parser/block.rs +++ b/nodedb/src/control/planner/procedural/parser/block.rs @@ -4,7 +4,7 @@ use super::super::ast::*; use super::super::error::ProceduralError; -use super::super::tokenizer::Token; +use super::super::tokenizer::{Token, TokenStream}; use super::exception; use super::statements; use super::utils::*; @@ -37,12 +37,17 @@ pub fn parse_block(input: &str) -> Result { /// Parse a sequence of statements until we hit END, ELSE, ELSIF, EXCEPTION, or end of tokens. pub(crate) fn parse_statements( - tokens: &[Token], + tokens: &TokenStream<'_>, pos: &mut usize, ) -> Result, ProceduralError> { let mut stmts = Vec::new(); while *pos < tokens.len() { + // A comment between statements belongs to none of them. + if tokens[*pos].is_comment() { + *pos += 1; + continue; + } match tokens.get(*pos) { Some( Token::End diff --git a/nodedb/src/control/planner/procedural/parser/exception.rs b/nodedb/src/control/planner/procedural/parser/exception.rs index eea36fd96..d90f8da9d 100644 --- a/nodedb/src/control/planner/procedural/parser/exception.rs +++ b/nodedb/src/control/planner/procedural/parser/exception.rs @@ -4,7 +4,7 @@ use super::super::ast::*; use super::super::error::ProceduralError; -use super::super::tokenizer::Token; +use super::super::tokenizer::{Token, TokenStream}; use super::statements::parse_statement; /// Parse `EXCEPTION WHEN THEN [WHEN ...] ...` @@ -12,7 +12,7 @@ use super::statements::parse_statement; /// Called when the parser encounters Token::Exception inside a BEGIN block. /// Parses one or more WHEN handlers until END is reached. pub(super) fn parse_exception_handlers( - tokens: &[Token], + tokens: &TokenStream<'_>, pos: &mut usize, ) -> Result, ProceduralError> { *pos += 1; // skip EXCEPTION @@ -77,11 +77,16 @@ fn parse_exception_condition( } fn parse_exception_body( - tokens: &[Token], + tokens: &TokenStream<'_>, pos: &mut usize, ) -> Result, ProceduralError> { let mut stmts = Vec::new(); while *pos < tokens.len() { + // A comment between statements belongs to none of them. + if tokens[*pos].is_comment() { + *pos += 1; + continue; + } match tokens.get(*pos) { Some(Token::End) => break, Some(Token::Ident(w)) if w.to_uppercase() == "WHEN" => break, diff --git a/nodedb/src/control/planner/procedural/parser/statements.rs b/nodedb/src/control/planner/procedural/parser/statements.rs index 3d1667c35..d4687238d 100644 --- a/nodedb/src/control/planner/procedural/parser/statements.rs +++ b/nodedb/src/control/planner/procedural/parser/statements.rs @@ -4,12 +4,12 @@ use super::super::ast::*; use super::super::error::ProceduralError; -use super::super::tokenizer::Token; +use super::super::tokenizer::{Token, TokenStream}; use super::utils::*; /// Parse a single statement. pub(super) fn parse_statement( - tokens: &[Token], + tokens: &TokenStream<'_>, pos: &mut usize, ) -> Result { match tokens.get(*pos) { @@ -58,7 +58,7 @@ pub(super) fn parse_statement( } /// `DECLARE name TYPE [:= default];` -fn parse_declare(tokens: &[Token], pos: &mut usize) -> Result { +fn parse_declare(tokens: &TokenStream<'_>, pos: &mut usize) -> Result { *pos += 1; let name = expect_ident(tokens, pos)?; let data_type = expect_ident(tokens, pos)?; @@ -106,7 +106,7 @@ fn is_assignment_ahead(tokens: &[Token], pos: usize) -> bool { } /// `name := expr;` or `NEW.field := expr;` -fn parse_assign(tokens: &[Token], pos: &mut usize) -> Result { +fn parse_assign(tokens: &TokenStream<'_>, pos: &mut usize) -> Result { // Collect dotted target: `name` or `NEW.field` let mut target = expect_ident(tokens, pos)?; while let Some(Token::Ident(dot)) = tokens.get(*pos) { @@ -125,7 +125,7 @@ fn parse_assign(tokens: &[Token], pos: &mut usize) -> Result Result { +fn parse_if(tokens: &TokenStream<'_>, pos: &mut usize) -> Result { *pos += 1; let condition = collect_sql_until(tokens, pos, &[Token::Then])?; expect_token(tokens, pos, &Token::Then)?; @@ -162,7 +162,7 @@ fn parse_if(tokens: &[Token], pos: &mut usize) -> Result Result { +fn parse_while(tokens: &TokenStream<'_>, pos: &mut usize) -> Result { *pos += 1; let condition = collect_sql_until(tokens, pos, &[Token::Loop])?; expect_token(tokens, pos, &Token::Loop)?; @@ -173,7 +173,7 @@ fn parse_while(tokens: &[Token], pos: &mut usize) -> Result Result { +fn parse_for(tokens: &TokenStream<'_>, pos: &mut usize) -> Result { *pos += 1; let var = expect_ident(tokens, pos)?; expect_token(tokens, pos, &Token::In)?; @@ -200,7 +200,7 @@ fn parse_for(tokens: &[Token], pos: &mut usize) -> Result Result { +fn parse_loop(tokens: &TokenStream<'_>, pos: &mut usize) -> Result { *pos += 1; let body = super::parse_statements(tokens, pos)?; expect_token(tokens, pos, &Token::EndLoop)?; @@ -209,7 +209,7 @@ fn parse_loop(tokens: &[Token], pos: &mut usize) -> Result Result { +fn parse_return(tokens: &TokenStream<'_>, pos: &mut usize) -> Result { *pos += 1; let expr = collect_sql_until(tokens, pos, &[Token::Semicolon])?; skip_if(tokens, pos, &Token::Semicolon); @@ -217,7 +217,10 @@ fn parse_return(tokens: &[Token], pos: &mut usize) -> Result Result { +fn parse_return_query( + tokens: &TokenStream<'_>, + pos: &mut usize, +) -> Result { *pos += 1; let query = collect_raw_sql_until(tokens, pos, &[Token::Semicolon]); skip_if(tokens, pos, &Token::Semicolon); @@ -225,7 +228,7 @@ fn parse_return_query(tokens: &[Token], pos: &mut usize) -> Result Result { +fn parse_raise(tokens: &TokenStream<'_>, pos: &mut usize) -> Result { *pos += 1; let level = match tokens.get(*pos) { Some(Token::Notice) => { @@ -251,7 +254,7 @@ fn parse_raise(tokens: &[Token], pos: &mut usize) -> Result Result { +fn parse_sql(tokens: &TokenStream<'_>, pos: &mut usize) -> Result { let sql = collect_raw_sql_until(tokens, pos, &[Token::Semicolon]); skip_if(tokens, pos, &Token::Semicolon); Ok(Statement::Sql { sql }) diff --git a/nodedb/src/control/planner/procedural/parser/utils.rs b/nodedb/src/control/planner/procedural/parser/utils.rs index 36064207f..86fff683c 100644 --- a/nodedb/src/control/planner/procedural/parser/utils.rs +++ b/nodedb/src/control/planner/procedural/parser/utils.rs @@ -3,7 +3,7 @@ //! Token utility functions for the procedural SQL parser. use super::super::error::ProceduralError; -use super::super::tokenizer::Token; +use super::super::tokenizer::{Token, TokenStream}; /// Check if a token matches a pattern token (ignoring content for parameterized variants). pub(super) fn token_matches(token: &Token, pattern: &Token) -> bool { @@ -53,52 +53,9 @@ pub(super) fn skip_if(tokens: &[Token], pos: &mut usize, token: &Token) { } } -/// Convert a token back to its SQL text representation. -pub(super) fn token_to_sql(token: &Token) -> String { - match token { - Token::Ident(s) => s.clone(), - Token::StringLit(s) => format!("'{}'", s.replace('\'', "''")), - Token::NumberLit(s) => s.clone(), - Token::SqlFragment(s) => s.clone(), - Token::Semicolon => ";".into(), - Token::Assign => ":=".into(), - Token::DotDot => "..".into(), - Token::In => "IN".into(), - Token::Reverse => "REVERSE".into(), - Token::If => "IF".into(), - Token::Then => "THEN".into(), - Token::Else => "ELSE".into(), - Token::End => "END".into(), - Token::Begin => "BEGIN".into(), - Token::Loop => "LOOP".into(), - Token::Return => "RETURN".into(), - Token::Insert => "INSERT".into(), - Token::Update => "UPDATE".into(), - Token::Delete => "DELETE".into(), - Token::To => "TO".into(), - Token::Savepoint => "SAVEPOINT".into(), - Token::Release => "RELEASE".into(), - Token::Commit => "COMMIT".into(), - Token::Rollback => "ROLLBACK".into(), - Token::Declare => "DECLARE".into(), - Token::Raise => "RAISE".into(), - Token::Notice => "NOTICE".into(), - Token::Warning => "WARNING".into(), - Token::Exception => "EXCEPTION".into(), - Token::Break => "BREAK".into(), - Token::Continue => "CONTINUE".into(), - Token::While => "WHILE".into(), - Token::For => "FOR".into(), - Token::Elsif => "ELSIF".into(), - Token::ReturnQuery => "RETURN QUERY".into(), - Token::EndIf => "END IF".into(), - Token::EndLoop => "END LOOP".into(), - } -} - /// Collect tokens as a SQL expression until one of the terminator tokens is found. pub(super) fn collect_sql_until( - tokens: &[Token], + tokens: &TokenStream<'_>, pos: &mut usize, terminators: &[Token], ) -> Result { @@ -112,19 +69,61 @@ pub(super) fn collect_sql_until( Ok(super::super::ast::SqlExpr::new(sql)) } -/// Collect tokens as raw SQL text until a terminator is found. +/// Collect tokens until a terminator is found, and return the source text +/// they cover, spelled as written. +/// +/// The text ends at the last token that is not a comment. A trailing line +/// comment will otherwise swallow the `;` a caller appends to the text. pub(super) fn collect_raw_sql_until( - tokens: &[Token], + tokens: &TokenStream<'_>, pos: &mut usize, terminators: &[Token], ) -> String { - let mut parts = Vec::new(); + let first = *pos; + let mut last_code: Option = None; while *pos < tokens.len() { - if terminators.iter().any(|t| token_matches(&tokens[*pos], t)) { + let token = &tokens[*pos]; + if terminators.iter().any(|t| token_matches(token, t)) { break; } - parts.push(token_to_sql(&tokens[*pos])); + if !token.is_comment() { + last_code = Some(*pos); + } *pos += 1; } - parts.join(" ").trim().to_string() + last_code + .and_then(|last| tokens.source_of(first, last)) + .map(|sql| sql.trim().to_string()) + .unwrap_or_default() +} + +#[cfg(test)] +mod tests { + use super::super::super::tokenizer::tokenize; + use super::*; + + #[test] + fn collected_sql_keeps_its_source_spelling() { + let tokens = tokenize( + "INSERT INTO t (id, v) VALUES ('it''s', 1.5e3::float) WHERE a>=1 AND b||c <> d;", + ) + .expect("tokenize"); + let mut pos = 0; + assert_eq!( + collect_raw_sql_until(&tokens, &mut pos, &[Token::Semicolon]), + "INSERT INTO t (id, v) VALUES ('it''s', 1.5e3::float) WHERE a>=1 AND b||c <> d" + ); + assert_eq!(tokens.get(pos), Some(&Token::Semicolon)); + } + + #[test] + fn collected_sql_ends_at_its_last_code_token() { + let tokens = + tokenize("DELETE FROM t /* keep */ WHERE id = 1 -- trailing\n;").expect("tokenize"); + let mut pos = 0; + assert_eq!( + collect_raw_sql_until(&tokens, &mut pos, &[Token::Semicolon]), + "DELETE FROM t /* keep */ WHERE id = 1" + ); + } } diff --git a/nodedb/src/control/planner/procedural/tokenizer.rs b/nodedb/src/control/planner/procedural/tokenizer.rs index 17df4d2c0..cb92666ad 100644 --- a/nodedb/src/control/planner/procedural/tokenizer.rs +++ b/nodedb/src/control/planner/procedural/tokenizer.rs @@ -73,23 +73,73 @@ pub enum Token { NumberLit(String), } +impl Token { + /// Whether this token is a `--` or `/* */` comment. + pub fn is_comment(&self) -> bool { + matches!(self, Token::SqlFragment(text) if text.starts_with("--") || text.starts_with("/*")) + } +} + +/// The tokens of one source text, each with its byte span in that text. +/// +/// A parser reads the tokens through `Deref`. It takes the SQL a statement or +/// expression covers from the source through [`TokenStream::source_of`], so +/// the SQL keeps its original spelling. Rejoining the tokens will split +/// multi-character operators (`>=`, `||`, `::`) and re-space every literal. +#[derive(Debug)] +pub struct TokenStream<'a> { + source: &'a str, + tokens: Vec, + spans: Vec>, +} + +impl<'a> TokenStream<'a> { + /// The source text from the start of token `first` to the end of token + /// `last`, both included. `None` when either index is out of range or + /// `last` precedes `first`. + pub fn source_of(&self, first: usize, last: usize) -> Option<&'a str> { + let start = self.spans.get(first)?.start; + let end = self.spans.get(last)?.end; + self.source.get(start..end) + } +} + +impl std::ops::Deref for TokenStream<'_> { + type Target = [Token]; + + fn deref(&self) -> &[Token] { + &self.tokens + } +} + /// Tokenize procedural SQL text into a token stream. /// /// The tokenizer is keyword-aware: it recognizes procedural keywords and /// captures everything else as `SqlFragment` tokens. String literals are -/// preserved (not split on keywords inside strings). -pub fn tokenize(input: &str) -> Result, ProceduralError> { +/// preserved (not split on keywords inside strings). Each token keeps its +/// byte span in `input`. +pub fn tokenize(input: &str) -> Result, ProceduralError> { let mut tokens = Vec::new(); + let mut spans: Vec> = Vec::new(); let bytes = input.as_bytes(); let len = bytes.len(); let mut i = 0; + let mut token_start = 0; while i < len { + // Each branch below pushes at most one token and moves `i` past it, + // so the token an iteration pushed spans `token_start..i` at the top + // of the next iteration. + if spans.len() < tokens.len() { + spans.push(token_start..i); + } + // Skip whitespace. if bytes[i].is_ascii_whitespace() { i += 1; continue; } + token_start = i; // Opaque SQL regions must not surface procedural keywords. if bytes[i] == b'"' { @@ -239,8 +289,15 @@ pub fn tokenize(input: &str) -> Result, ProceduralError> { tokens.push(Token::Ident(ch.to_string())); i += ch.len_utf8(); } + if spans.len() < tokens.len() { + spans.push(token_start..i); + } - Ok(tokens) + Ok(TokenStream { + source: input, + tokens, + spans, + }) } /// Read a single-quoted string literal, handling escaped quotes (''). @@ -485,6 +542,20 @@ mod tests { assert!(tokens.contains(&Token::EndLoop)); } + #[test] + fn each_token_spans_its_source_text() { + let input = "IF x>=1 THEN RETURN 'it''s'::text; END IF;"; + let tokens = tokenize(input).expect("tokenize"); + assert_eq!(tokens.source_of(1, 4), Some("x>=1")); + let semicolon = tokens + .iter() + .position(|token| *token == Token::Semicolon) + .expect("a semicolon"); + assert_eq!(tokens.source_of(7, semicolon - 1), Some("'it''s'::text")); + assert_eq!(tokens.source_of(0, tokens.len() - 1), Some(input)); + assert_eq!(tokens.source_of(0, tokens.len()), None); + } + #[test] fn tokenize_dml_detected() { let tokens = tokenize("INSERT INTO users VALUES (1);").unwrap(); diff --git a/nodedb/src/control/planner/redaction_refusal/graph.rs b/nodedb/src/control/planner/redaction_refusal/graph.rs index f0ee4460c..f3daa108a 100644 --- a/nodedb/src/control/planner/redaction_refusal/graph.rs +++ b/nodedb/src/control/planner/redaction_refusal/graph.rs @@ -49,6 +49,12 @@ pub(super) fn refuse_graph_op(op: &GraphOp, ctx: &RefusalCtx<'_>) -> crate::Resu | GraphOp::WccSuperstep(_) | GraphOp::SetNodeLabels { .. } | GraphOp::RemoveNodeLabels { .. } + | GraphOp::NodeEdgeGuard { .. } + | GraphOp::NodePresenceGuard { .. } + | GraphOp::TruncateEdges { .. } + // A delete's planner reads which ids are stored. It returns no + // column to the client. + | GraphOp::NodePresenceRead { .. } | GraphOp::Stats { .. } => Ok(()), } } @@ -85,8 +91,8 @@ pub(super) fn refuse_match(ctx: &RefusalCtx<'_>, query: &[u8]) -> crate::Result< /// Refuse a pattern match already known to be scoped (or not) to `collection`. /// /// Shares the fail-closed fallback with [`refuse_match`] for a caller that -/// already holds the decoded `MatchQuery` and would otherwise have to -/// re-serialize it just to decode it back here. +/// already holds the decoded `MatchQuery` and will otherwise have to +/// re-serialize it only to decode it back here. pub(super) fn refuse_match_scoped( ctx: &RefusalCtx<'_>, collection: Option<&str>, diff --git a/nodedb/src/control/planner/rls_injection/array.rs b/nodedb/src/control/planner/rls_injection/array.rs index 61066c1e7..0f2887565 100644 --- a/nodedb/src/control/planner/rls_injection/array.rs +++ b/nodedb/src/control/planner/rls_injection/array.rs @@ -35,7 +35,7 @@ pub(super) fn inject_array(_ctx: &RlsCtx<'_>, op: &ArrayOp) -> crate::Result<()> | ArrayOp::Flush { .. } | ArrayOp::Compact { .. } | ArrayOp::DropArray { .. } - | ArrayOp::RestoreArrayDrop { .. } + | ArrayOp::RekeyArray { .. } | ArrayOp::PurgeArrayDrop { .. } => Ok(()), } } @@ -55,12 +55,13 @@ pub(super) fn inject_cluster_array(_ctx: &RlsCtx<'_>, op: &ClusterArrayOp) -> cr /// Exhaustive over [`ClusterEventOp`]. pub(super) fn inject_cluster_event(_ctx: &RlsCtx<'_>, op: &ClusterEventOp) -> crate::Result<()> { match op { - // No-op: a stream consume is addressed by `(stream, group)` and a - // topic publish by topic name — neither names a collection this pass - // could resolve a policy against. Access to a stream or topic is - // authorized on the stream/topic object itself. + // No-op: a stream consume is addressed by `(stream, group)`, which + // names no collection this pass can resolve a policy against. + // Access to a stream is authorized on the stream object itself. ClusterEventOp::ConsumeStream { .. } - | ClusterEventOp::PublishTopic { .. } - | ClusterEventOp::TenantWriteMarks { .. } => Ok(()), + | ClusterEventOp::TenantWriteMarks { .. } + | ClusterEventOp::SurrogateBinds { .. } + | ClusterEventOp::MetadataApplied { .. } + | ClusterEventOp::SurrogateHolders { .. } => Ok(()), } } diff --git a/nodedb/src/control/planner/rls_injection/columnar.rs b/nodedb/src/control/planner/rls_injection/columnar.rs index 9bbeb1d83..16e0807fb 100644 --- a/nodedb/src/control/planner/rls_injection/columnar.rs +++ b/nodedb/src/control/planner/rls_injection/columnar.rs @@ -97,7 +97,7 @@ pub(super) fn inject_timeseries(ctx: &RlsCtx<'_>, op: &mut TimeseriesOp) -> crat // Recurse: the resolve pass carries the ingest it is about to decide, // and that ingest's own slots are the ones the policy fills. - TimeseriesOp::ResolveIngest(inner) => inject_timeseries(ctx, inner), + TimeseriesOp::ResolveIngest(inner) => inject_timeseries(ctx, &mut inner.ingest), // Refuse: removes every row without reading one, so no image // exists to evaluate against. Mirrors `KvOp::Truncate`. @@ -481,7 +481,7 @@ mod tests { "places", ), field: "geom".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: Some(nodedb_types::Surrogate::new(1)), provenance: None, }); assert_write_refused(inject(&mut plan, &store), "places"); diff --git a/nodedb/src/control/planner/rls_injection/crdt.rs b/nodedb/src/control/planner/rls_injection/crdt.rs index 7f80490f5..d2799e281 100644 --- a/nodedb/src/control/planner/rls_injection/crdt.rs +++ b/nodedb/src/control/planner/rls_injection/crdt.rs @@ -33,8 +33,8 @@ pub(super) fn inject_crdt(ctx: &RlsCtx<'_>, op: &mut CrdtOp) -> crate::Result<() match op { // Refuse: all four return stored row content — the current state, a // historical state, the oplog deltas those states were built from, or - // the state a delta would produce — and none has a slot the policy - // could occupy. + // the state a delta will produce — and none has a slot the policy + // can occupy. CrdtOp::Read { collection, .. } | CrdtOp::ReadAtVersion { collection, .. } | CrdtOp::ExportDelta { collection, .. } @@ -144,7 +144,7 @@ mod tests { "notes", ), document_id: "d1".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: Some(nodedb_types::Surrogate::new(1)), returning: None, rls_filters: Vec::new(), }); @@ -172,7 +172,7 @@ mod tests { ), document_id: "d1".into(), fields_json: "{}".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), partial: false, verb: nodedb_physical::physical_plan::CrdtWriteVerb::Insert, returning: None, diff --git a/nodedb/src/control/planner/rls_injection/document.rs b/nodedb/src/control/planner/rls_injection/document.rs index e4e03f771..625f481ef 100644 --- a/nodedb/src/control/planner/rls_injection/document.rs +++ b/nodedb/src/control/planner/rls_injection/document.rs @@ -101,7 +101,7 @@ pub(super) fn inject_document(ctx: &RlsCtx<'_>, op: &mut DocumentOp) -> crate::R ), // Refuse: streams raw triples with no filter slot — every stored - // body would be copied regardless of policy. + // body will be copied regardless of policy. DocumentOp::MaterializeScan { collection, .. } => ctx.refuse_if_policy( collection, "the materializing scan streams raw stored bodies through a cursor payload that \ @@ -172,7 +172,7 @@ pub(super) fn inject_document(ctx: &RlsCtx<'_>, op: &mut DocumentOp) -> crate::R ), // Refuse: removes every row without reading one, so no image the - // policy could be evaluated against exists. + // policy can be evaluated against exists. DocumentOp::Truncate { collection, .. } => ctx.refuse_if_write_policy( collection, "a truncate removes every row without reading one, so no row image is available", @@ -188,7 +188,7 @@ pub(super) fn inject_document(ctx: &RlsCtx<'_>, op: &mut DocumentOp) -> crate::R // decided when the SOURCE row it derives from was admitted. DocumentOp::ApplyBalanceDelta { .. } => Ok(()), - // No-op: already decided by the resolve pass; re-injecting would + // No-op: already decided by the resolve pass; re-injecting will // replace a verdict with a predicate no applying node can decide. DocumentOp::ResolvedWrite { .. } => Ok(()), @@ -227,7 +227,7 @@ mod tests { document_id: "d1".into(), value: body(owner_id), if_absent: false, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), returning: None, rls_filters: Vec::new(), resolved_sum_targets: Vec::new(), @@ -242,7 +242,7 @@ mod tests { collection, ), document_id: "d1".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: None, pk_bytes: Vec::new(), updates: Vec::new(), returning: None, @@ -289,7 +289,7 @@ mod tests { } /// A batch fails whole when any one of its rows violates the policy: a - /// silently dropped row would report a write that never happened. + /// silently dropped row will report a write that never happened. #[test] fn batch_insert_is_rejected_when_any_row_violates_the_policy() { let store = store_with_write_policy("orders"); @@ -411,7 +411,7 @@ mod tests { document_id: "d1".into(), value: body("42"), on_conflict_updates: Vec::new(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), @@ -422,7 +422,7 @@ mod tests { } /// A `RETURNING` on an insert ships rows back, so a read-only policy must - /// land in the insert's post-filter slot. Leaving it empty would return + /// land in the insert's post-filter slot. Leaving it empty will return /// rows the same principal's `SELECT` hides. #[test] fn insert_receives_the_read_policy_filter() { @@ -459,7 +459,7 @@ mod tests { } /// A truncate removes every row without reading one, so there is no image - /// the policy could decide. + /// the policy can decide. #[test] fn truncate_is_refused_under_a_write_policy() { let store = store_with_write_policy("orders"); diff --git a/nodedb/src/control/planner/rls_injection/graph.rs b/nodedb/src/control/planner/rls_injection/graph.rs index 1b9a1f4e4..5fcc2b269 100644 --- a/nodedb/src/control/planner/rls_injection/graph.rs +++ b/nodedb/src/control/planner/rls_injection/graph.rs @@ -7,7 +7,7 @@ //! fusion returns fused rows. Each refuses while a read policy restricts //! the collection, rather than return rows the policy hides — including //! algorithm/stats/RAG-fusion shapes the redaction pass permits, since RLS -//! restricts the row set itself, not just column values. +//! restricts the row set itself, not only column values. use nodedb_physical::physical_plan::GraphOp; @@ -43,7 +43,7 @@ pub(super) fn inject_graph(ctx: &RlsCtx<'_>, op: &mut GraphOp) -> crate::Result< } // Refuse: returns bindings with no filter slot; its own `WHERE` - // could probe a hidden row's field one predicate at a time. + // can probe a hidden row's field one predicate at a time. GraphOp::Match { query, .. } | GraphOp::MatchContinuation { query, .. } | GraphOp::MatchVarLenResume { query, .. } => refuse_match(ctx, query), @@ -68,7 +68,7 @@ pub(super) fn inject_graph(ctx: &RlsCtx<'_>, op: &mut GraphOp) -> crate::Result< ), // Refuse: the fusion envelope has no filter slot and no sub-plan - // to recurse into — hidden rows would be ranked and returned. + // to recurse into — hidden rows will be ranked and returned. GraphOp::RagFusion { collection, .. } => ctx.refuse_if_policy( collection, "fusion returns ranked document rows through a fused response shape that carries no \ @@ -121,6 +121,15 @@ pub(super) fn inject_graph(ctx: &RlsCtx<'_>, op: &mut GraphOp) -> crate::Result< "a node-label write is keyed on a node id that names no collection, and it carries \ no row body for the policy to be evaluated against", ), + + // Admit: these derive from a document delete or TRUNCATE the policy + // on the same collection already decided. The guards write nothing, + // and each edge a TRUNCATE removes goes with the rows it removes. + // A delete's planner reads which ids are stored, for that delete. + GraphOp::NodeEdgeGuard { .. } + | GraphOp::NodePresenceGuard { .. } + | GraphOp::TruncateEdges { .. } + | GraphOp::NodePresenceRead { .. } => Ok(()), } } @@ -172,6 +181,7 @@ mod tests { mode: None, personalization_vector: None, }, + stage: nodedb_physical::physical_plan::AlgoStage::Local, }) } @@ -208,7 +218,7 @@ mod tests { assert!(inject(&mut plan, &store).is_ok()); } - /// An unscoped match may traverse anything the tenant holds, so it falls + /// An unscoped match can traverse anything the tenant holds, so it falls /// back to the tenant-wide question. #[test] fn unscoped_match_falls_back_to_the_tenant_wide_question() { @@ -241,8 +251,8 @@ mod tests { label: "knows".into(), dst_id: "b".into(), properties: properties.as_bytes().to_vec(), - src_surrogate: nodedb_types::Surrogate::ZERO, - dst_surrogate: nodedb_types::Surrogate::ZERO, + src_surrogate: nodedb_types::Surrogate::new(1), + dst_surrogate: nodedb_types::Surrogate::new(2), }) } @@ -255,8 +265,8 @@ mod tests { src_id: "a".into(), label: "knows".into(), dst_id: "b".into(), - src_surrogate: nodedb_types::Surrogate::ZERO, - dst_surrogate: nodedb_types::Surrogate::ZERO, + src_surrogate: nodedb_types::Surrogate::new(1), + dst_surrogate: nodedb_types::Surrogate::new(2), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), }) } diff --git a/nodedb/src/control/planner/rls_injection/kv.rs b/nodedb/src/control/planner/rls_injection/kv.rs index 43c642e35..b3cfb4880 100644 --- a/nodedb/src/control/planner/rls_injection/kv.rs +++ b/nodedb/src/control/planner/rls_injection/kv.rs @@ -95,7 +95,7 @@ pub(super) fn inject_kv(ctx: &RlsCtx<'_>, op: &mut KvOp) -> crate::Result<()> { ctx.set_post_filters(collection, rls_filters) } - // Admit every entry: a silently dropped row would report a write + // Admit every entry: a silently dropped row will report a write // that never happened. KvOp::BatchPut { collection, @@ -232,7 +232,7 @@ pub(super) fn inject_kv(ctx: &RlsCtx<'_>, op: &mut KvOp) -> crate::Result<()> { KvOp::ResolveWrite(inner) => inject_kv(ctx, inner), // No-op: already decided before this write was proposed; - // re-injecting would replace a verdict with an unevaluable predicate. + // re-injecting will replace a verdict with an unevaluable predicate. KvOp::ResolvedWrite { .. } => Ok(()), // No-op: index DDL writes no user row, so no row policy restricts it. @@ -278,7 +278,7 @@ mod tests { key: b"k1".to_vec(), value: body(owner_id), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), returning: None, rls_filters: Vec::new(), provenance: None, @@ -339,7 +339,7 @@ mod tests { } /// A single-column `value` write stores one opaque scalar: it carries no - /// field the predicate could name, so it fails closed rather than being + /// field the predicate can name, so it fails closed rather than being /// waved through as "not a document". #[test] fn an_opaque_scalar_value_is_rejected_under_a_write_policy() { @@ -352,7 +352,7 @@ mod tests { key: b"k1".to_vec(), value: b"v1".to_vec(), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), returning: None, rls_filters: Vec::new(), provenance: None, @@ -364,7 +364,7 @@ mod tests { } /// A `RETURNING` on a KV write ships rows back, so a read-only policy must - /// land in the write's post-filter slot. Leaving it empty would return rows + /// land in the write's post-filter slot. Leaving it empty will return rows /// the same principal's `SELECT` hides. #[test] fn a_kv_write_receives_the_read_policy_filter() { @@ -443,7 +443,7 @@ mod tests { ), key: b"k1".to_vec(), updates: Vec::new(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), if_present: false, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, @@ -475,7 +475,7 @@ mod tests { collection: collection(), key: b"k1".to_vec(), updates: Vec::new(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), if_present: false, rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, @@ -528,7 +528,7 @@ mod tests { key: b"k1".to_vec(), delta: 1, ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), shape: nodedb_physical::physical_plan::KvCounterShape::Raw, }); @@ -557,7 +557,7 @@ mod tests { ), key: b"k1".to_vec(), new_value: body("42"), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), rls_filters: Vec::new(), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), }); @@ -597,7 +597,7 @@ mod tests { ), item_key: b"i1".to_vec(), dest_key: b"d1".to_vec(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), source_rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), dest_rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), }); @@ -622,7 +622,7 @@ mod tests { } /// A truncate removes every row without reading one, so there is no image - /// the policy could decide. + /// the policy can decide. #[test] fn truncate_is_refused_under_a_write_policy() { let store = store_with_write_policy("sessions"); @@ -682,7 +682,7 @@ mod tests { /// A sorted-index read names no collection, so a read policy anywhere in /// the tenant refuses it: its ranked keys come from stored rows and carry - /// no filter slot the policy could be applied through. + /// no filter slot the policy can be applied through. #[test] fn sorted_index_read_is_refused_under_a_read_policy() { let store = store_with_read_policy("scores"); diff --git a/nodedb/src/control/planner/rls_injection/meta.rs b/nodedb/src/control/planner/rls_injection/meta.rs index 555323526..1506be6b1 100644 --- a/nodedb/src/control/planner/rls_injection/meta.rs +++ b/nodedb/src/control/planner/rls_injection/meta.rs @@ -23,6 +23,15 @@ pub(super) fn inject_meta(ctx: &RlsCtx<'_>, op: &mut MetaOp) -> crate::Result<() ) } + // Refuse: hash-chain verification recomputes every link from every + // stored row. A walk over the rows the policy leaves visible is not + // the chain, and its verdict will expose rows the policy hides. + MetaOp::VerifyHashChain { collection } => ctx.refuse_if_policy( + collection, + "hash-chain verification reads every stored row, which the row filter cannot be \ + evaluated against", + ), + // Refuse: a byte-size estimate is derived from every stored row of the // collection, including the ones the policy hides, and carries no row // to filter. `name` is a bare collection name — qualify it against @@ -67,7 +76,7 @@ pub(super) fn inject_meta(ctx: &RlsCtx<'_>, op: &mut MetaOp) -> crate::Result<() // No-op: durability, cancellation, snapshot install, purge, retention, // index and synonym maintenance, transaction-overlay bookkeeping, and // continuous-aggregate administration. None of these returns stored - // rows to a caller, and none writes a user row a policy predicate could + // rows to a caller, and none writes a user row a policy predicate can // be evaluated against — they are authorized by the permission check // that precedes this pass. MetaOp::WalAppend { .. } @@ -95,14 +104,15 @@ pub(super) fn inject_meta(ctx: &RlsCtx<'_>, op: &mut MetaOp) -> crate::Result<() | MetaOp::RebuildIndex { .. } | MetaOp::PutSynonymGroup { .. } | MetaOp::DeleteSynonymGroup { .. } - | MetaOp::RenameCollection { .. } | MetaOp::DropTxnOverlay { .. } | MetaOp::MarkSavepoint { .. } | MetaOp::RollbackToSavepoint { .. } | MetaOp::CalvinFlush { .. } | MetaOp::CalvinDrop { .. } | MetaOp::CalvinResolve { .. } - | MetaOp::ApplyTransactionRedo { .. } => Ok(()), + | MetaOp::ApplyTransactionRedo { .. } + | MetaOp::RestoreRedo(_) + | MetaOp::HomeVersions { .. } => Ok(()), } } @@ -138,6 +148,8 @@ mod tests { let mut plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: 1, cut_watermark: None, + cut_capture: None, + arrays: false, }); assert!(matches!( inject(&mut plan, &store), diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/array.rs b/nodedb/src/control/planner/rls_injection/permission_tree/array.rs index dc2fba05d..92582515c 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/array.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/array.rs @@ -33,7 +33,7 @@ pub(super) fn apply_array(_ctx: &PermCtx<'_>, op: &ArrayOp) -> crate::Result<()> | ArrayOp::Flush { .. } | ArrayOp::Compact { .. } | ArrayOp::DropArray { .. } - | ArrayOp::RestoreArrayDrop { .. } + | ArrayOp::RekeyArray { .. } | ArrayOp::PurgeArrayDrop { .. } => Ok(()), } } @@ -53,12 +53,14 @@ pub(super) fn apply_cluster_array(_ctx: &PermCtx<'_>, op: &ClusterArrayOp) -> cr /// Exhaustive over [`ClusterEventOp`]. pub(super) fn apply_cluster_event(_ctx: &PermCtx<'_>, op: &ClusterEventOp) -> crate::Result<()> { match op { - // No-op: a stream consume is addressed by `(stream, group)` and a - // topic publish by topic name — neither names a collection this pass - // could resolve a tree definition against. Access to a stream or topic - // is authorized on the stream/topic object itself. + // No-op: a stream consume is addressed by `(stream, group)`, which + // names no collection this pass can resolve a tree definition + // against. Access to a stream is authorized on the stream object + // itself. ClusterEventOp::ConsumeStream { .. } - | ClusterEventOp::PublishTopic { .. } - | ClusterEventOp::TenantWriteMarks { .. } => Ok(()), + | ClusterEventOp::TenantWriteMarks { .. } + | ClusterEventOp::SurrogateBinds { .. } + | ClusterEventOp::MetadataApplied { .. } + | ClusterEventOp::SurrogateHolders { .. } => Ok(()), } } diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/columnar.rs b/nodedb/src/control/planner/rls_injection/permission_tree/columnar.rs index 2d978456d..1db69bc3a 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/columnar.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/columnar.rs @@ -88,7 +88,7 @@ pub(super) fn apply_timeseries(ctx: &PermCtx<'_>, op: &mut TimeseriesOp) -> crat // Recurse: the resolve pass stands in for the ingest it wraps, so it // needs the same write level on the same collection. - TimeseriesOp::ResolveIngest(inner) => apply_timeseries(ctx, inner), + TimeseriesOp::ResolveIngest(inner) => apply_timeseries(ctx, &mut inner.ingest), // Delete level, blanket: a truncate removes rows it never // enumerates, so there is no predicate to narrow. @@ -125,8 +125,8 @@ mod tests { }; use crate::bridge::envelope::PhysicalPlan; - /// A timeseries scan over a governed collection was previously unlisted - /// and returned every series. It is now narrowed to the readable subtree. + /// A timeseries scan over a governed collection narrows to the readable + /// subtree instead of returning every series. #[test] fn timeseries_scan_is_narrowed_to_the_readable_subtree() { let cache = cache_with_tree("metrics"); diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/context.rs b/nodedb/src/control/planner/rls_injection/permission_tree/context.rs index 78fd7ae81..9dd87d065 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/context.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/context.rs @@ -19,7 +19,9 @@ use crate::bridge::scan_filter::{FilterOp, ScanFilter}; use crate::control::security::auth_context::AuthContext; use crate::control::security::permission_tree::resolver::accessible_resources; -use crate::control::security::permission_tree::{PermissionCache, PermissionTreeDef}; +use crate::control::security::permission_tree::{ + PermissionCache, PermissionTreeDef, TreeKey, TreeScope, +}; use crate::types::TenantId; use super::super::filters::merge_filters; @@ -43,7 +45,8 @@ impl PermTreeLevel { } } -/// Permission cache plus the requester's tenant and authenticated identity. +/// Permission cache plus the requester's database, tenant, and authenticated +/// identity. /// /// A superuser bypasses permission-tree filtering entirely, so [`PermCtx::new`] /// returns `None` for one and no walk runs at all — no arm has to restate the @@ -53,8 +56,9 @@ pub(super) struct PermCtx<'a> { cache: &'a PermissionCache, tenant_id: u64, auth: &'a AuthContext, - /// Database bare op-carried collection names (e.g. - /// `AlgoParams.collection`) must be qualified against before a lookup. + /// The database the task runs in. Every tree lookup resolves in it, and + /// bare op-carried collection names (e.g. `AlgoParams.collection`) are + /// qualified against it. pub(super) database_id: nodedb_types::DatabaseId, } @@ -84,7 +88,7 @@ impl<'a> PermCtx<'a> { /// this operation produces or acts on: a storage-pushdown `filters` field /// or a dedicated post-fetch `rls_filters` field. The bytes are merged /// rather than replaced because this pass runs after RLS injection, which - /// may already own the slot. + /// can already own the slot. /// /// An identity with no accessible resource yields `IN ()`, which matches /// nothing — the caller sees an empty result rather than an error, exactly @@ -95,10 +99,10 @@ impl<'a> PermCtx<'a> { level: PermTreeLevel, slot: &mut Vec, ) -> crate::Result<()> { - let Some(def) = self.cache.get_tree_def(self.tenant_id, collection.as_str()) else { + let Some((scope, def)) = self.tree_def(collection)? else { return Ok(()); }; - let accessible = self.accessible(def, level); + let accessible = self.accessible(scope, def, level); let in_filter = ScanFilter { field: def.resource_column.clone(), op: FilterOp::In, @@ -131,10 +135,10 @@ impl<'a> PermCtx<'a> { collection: &nodedb_types::QualifiedCollection, level: PermTreeLevel, ) -> crate::Result<()> { - let Some(def) = self.cache.get_tree_def(self.tenant_id, collection.as_str()) else { + let Some((scope, def)) = self.tree_def(collection)? else { return Ok(()); }; - if self.accessible(def, level).is_empty() { + if self.accessible(scope, def, level).is_empty() { let required = level.required_in(def); return Err(crate::Error::RejectedAuthz { tenant_id: TenantId::new(self.tenant_id), @@ -156,12 +160,7 @@ impl<'a> PermCtx<'a> { collection: &nodedb_types::QualifiedCollection, why: &str, ) -> crate::Result<()> { - if collection.as_str().is_empty() - || self - .cache - .get_tree_def(self.tenant_id, collection.as_str()) - .is_none() - { + if collection.as_str().is_empty() || self.tree_def(collection)?.is_none() { return Ok(()); } Err(crate::Error::PlanError { @@ -171,12 +170,15 @@ impl<'a> PermCtx<'a> { }) } - /// Refuse when any collection in the tenant carries a permission tree. + /// Refuse when any collection of the tenant, in any database, carries a + /// permission tree. /// /// Used only where the plan does not name the collection it reads, so the /// narrow per-collection question cannot be asked and the plan cannot be - /// shown to avoid a governed collection. Mirrors the RLS pass's - /// tenant-wide fallback for the same shapes. + /// shown to avoid a governed collection. A plan without a collection is + /// not shown to stay inside its database either, so the question spans + /// every database. Mirrors the RLS pass's tenant-wide fallback for the + /// same shapes. pub(super) fn refuse_if_any_tree(&self, why: &str) -> crate::Result<()> { if !self.cache.has_tree_defs_for_tenant(self.tenant_id) { return Ok(()); @@ -189,12 +191,35 @@ impl<'a> PermCtx<'a> { }) } + /// The tree governing `collection`, a name qualified for the task's + /// database, and the scope its hierarchy and grants live in. + /// + /// A name not qualified for that database cannot be resolved. While the + /// tenant has a tree that is an error, since the name can be governed. + fn tree_def( + &self, + collection: &nodedb_types::QualifiedCollection, + ) -> crate::Result> { + let key = + match TreeKey::from_qualified(self.database_id, self.tenant_id, collection.as_str()) { + Ok(key) => key, + Err(_) if !self.cache.has_tree_defs_for_tenant(self.tenant_id) => return Ok(None), + Err(e) => return Err(e), + }; + Ok(self.cache.get_tree_def(&key).map(|def| (key.scope, def))) + } + /// Resource ids this identity holds at least `level` on. - fn accessible(&self, def: &PermissionTreeDef, level: PermTreeLevel) -> Vec { + fn accessible( + &self, + scope: TreeScope, + def: &PermissionTreeDef, + level: PermTreeLevel, + ) -> Vec { accessible_resources( self.cache, def, - self.tenant_id, + scope, &self.auth.id, &self.auth.roles, level.required_in(def), diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/document.rs b/nodedb/src/control/planner/rls_injection/permission_tree/document.rs index 260e7f154..e5dab825d 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/document.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/document.rs @@ -62,7 +62,7 @@ pub(super) fn apply_document(ctx: &PermCtx<'_>, op: &mut DocumentOp) -> crate::R ), // Refuse: streams raw triples with no filter slot — every body - // would copy regardless of the tree. + // will copy regardless of the tree. DocumentOp::MaterializeScan { collection, .. } => ctx.refuse_if_tree( collection, "the materializing scan streams raw stored bodies through a cursor payload that \ @@ -195,7 +195,7 @@ mod tests { "docs", ), document_id: "d1".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: None, pk_bytes: Vec::new(), rls_filters: Vec::new(), system_time: Default::default(), diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/graph.rs b/nodedb/src/control/planner/rls_injection/permission_tree/graph.rs index 34e0f0459..532014cea 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/graph.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/graph.rs @@ -43,7 +43,7 @@ pub(super) fn apply_graph(ctx: &PermCtx<'_>, op: &GraphOp) -> crate::Result<()> } // Refuse: returns bindings with no filter slot; its own `WHERE` - // could probe a hidden row's field one predicate at a time. + // can probe a hidden row's field one predicate at a time. GraphOp::Match { query, .. } | GraphOp::MatchContinuation { query, .. } | GraphOp::MatchVarLenResume { query, .. } => refuse_match(ctx, query), @@ -116,6 +116,12 @@ pub(super) fn apply_graph(ctx: &PermCtx<'_>, op: &GraphOp) -> crate::Result<()> // No-op: node labels are keyed by node id alone and name no // collection, so this pass has no tree definition to resolve. GraphOp::SetNodeLabels { .. } | GraphOp::RemoveNodeLabels { .. } => Ok(()), + // No-op: these derive from a document delete or TRUNCATE this pass + // already decided on the same collection. + GraphOp::NodeEdgeGuard { .. } + | GraphOp::NodePresenceGuard { .. } + | GraphOp::TruncateEdges { .. } + | GraphOp::NodePresenceRead { .. } => Ok(()), } } @@ -203,6 +209,7 @@ mod tests { mode: None, personalization_vector: None, }, + stage: nodedb_physical::physical_plan::AlgoStage::Local, }); assert_refused(apply(&mut plan, &cache), "docs"); } diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs b/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs index 26d6746aa..9d52694a0 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/meta.rs @@ -24,6 +24,14 @@ pub(super) fn apply_meta(ctx: &PermCtx<'_>, op: &mut MetaOp) -> crate::Result<() ) } + // Refuse: hash-chain verification recomputes every link from every + // stored row, including the ones outside the subtree. + MetaOp::VerifyHashChain { collection } => ctx.refuse_if_tree( + collection, + "hash-chain verification reads every stored row, which the subtree filter cannot be \ + evaluated against", + ), + // Refuse: a byte-size estimate is derived from every stored row of the // collection, including the ones outside the subtree, and carries no // resource column to filter on. `name` is a bare collection name — @@ -95,14 +103,15 @@ pub(super) fn apply_meta(ctx: &PermCtx<'_>, op: &mut MetaOp) -> crate::Result<() | MetaOp::RebuildIndex { .. } | MetaOp::PutSynonymGroup { .. } | MetaOp::DeleteSynonymGroup { .. } - | MetaOp::RenameCollection { .. } | MetaOp::DropTxnOverlay { .. } | MetaOp::MarkSavepoint { .. } | MetaOp::RollbackToSavepoint { .. } | MetaOp::CalvinFlush { .. } | MetaOp::CalvinDrop { .. } | MetaOp::CalvinResolve { .. } - | MetaOp::ApplyTransactionRedo { .. } => Ok(()), + | MetaOp::ApplyTransactionRedo { .. } + | MetaOp::RestoreRedo(_) + | MetaOp::HomeVersions { .. } => Ok(()), } } @@ -138,6 +147,8 @@ mod tests { let mut plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: 1, cut_watermark: None, + cut_capture: None, + arrays: false, }); assert!(matches!( apply(&mut plan, &cache), diff --git a/nodedb/src/control/planner/rls_injection/permission_tree/plan.rs b/nodedb/src/control/planner/rls_injection/permission_tree/plan.rs index 0ada83e52..ebd0fb7a0 100644 --- a/nodedb/src/control/planner/rls_injection/permission_tree/plan.rs +++ b/nodedb/src/control/planner/rls_injection/permission_tree/plan.rs @@ -88,8 +88,8 @@ pub(super) fn walk(ctx: &PermCtx<'_>, plan: &mut PhysicalPlan) -> crate::Result< pub(super) mod test_support { use crate::control::security::auth_context::AuthContext; use crate::control::security::permission_tree::types::{PermissionGrant, PermissionTreeDef}; - use crate::control::security::permission_tree::{PermissionCache, resolver}; - use crate::types::TenantId; + use crate::control::security::permission_tree::{PermissionCache, TreeScope, resolver}; + use crate::types::{DatabaseId, TenantId}; pub(in crate::control::planner::rls_injection::permission_tree) const TENANT: u64 = 1; @@ -97,7 +97,8 @@ pub(super) mod test_support { /// spells it: `AuthContext::id` is the numeric user id rendered as text. const ALICE: &str = "42"; - /// A cache holding a tree on `collection` over three resources. + /// A cache holding a tree on the default database's `collection` over + /// three resources. /// /// `doc_a` is granted to alice at `owner`, so she clears read, write, and /// delete. `doc_b` is granted at `viewer`, so she clears read only. @@ -105,10 +106,18 @@ pub(super) mod test_support { pub(in crate::control::planner::rls_injection::permission_tree) fn cache_with_tree( collection: &str, ) -> PermissionCache { + cache_with_tree_in(DatabaseId::DEFAULT, collection) + } + + /// [`cache_with_tree`] with the tree and its grants in `database_id`. + pub(in crate::control::planner::rls_injection::permission_tree) fn cache_with_tree_in( + database_id: DatabaseId, + collection: &str, + ) -> PermissionCache { + let scope = TreeScope::new(database_id, TENANT); let mut cache = PermissionCache::new(); cache.register_tree_def( - TENANT, - collection, + scope.collection(collection), PermissionTreeDef { resource_column: "doc_id".into(), graph_index: "resource_tree".into(), @@ -128,7 +137,7 @@ pub(super) mod test_support { ("doc_c", "someone_else", "owner"), ] { cache.put_grant( - TENANT, + scope, &PermissionGrant { resource_id: resource.into(), grantee: grantee.into(), @@ -190,7 +199,7 @@ pub(super) mod test_support { cache: &PermissionCache, auth: &AuthContext, ) -> crate::Result<()> { - match super::PermCtx::new(cache, TENANT, auth, nodedb_types::DatabaseId::DEFAULT) { + match super::PermCtx::new(cache, TENANT, auth, DatabaseId::DEFAULT) { Some(ctx) => super::walk(&ctx, plan), None => Ok(()), } @@ -242,7 +251,7 @@ pub(super) mod test_support { } } - /// The resource ids alice may read, sorted for comparison. + /// The resource ids alice can read, sorted for comparison. pub(in crate::control::planner::rls_injection::permission_tree) fn readable() -> Vec { vec!["doc_a".to_owned(), "doc_b".to_owned()] } @@ -261,12 +270,15 @@ pub(super) mod test_support { #[test] fn fixture_separates_the_levels() { let cache = cache_with_tree("docs"); - let def = cache.get_tree_def(TENANT, "docs").expect("tree def"); + let scope = TreeScope::new(DatabaseId::DEFAULT, TENANT); + let def = cache + .get_tree_def(&scope.collection("docs")) + .expect("tree def"); let auth = regular_auth(); let delete = resolver::accessible_resources( &cache, def, - TENANT, + scope, &auth.id, &auth.roles, &def.delete_level, @@ -281,18 +293,22 @@ mod tests { ColumnarOp, DocumentOp, ExchangeMode, ExchangeOp, QueryOp, }; + use nodedb_physical::physical_task::PhysicalTask; + use super::test_support::{ - apply, apply_as, apply_without_tree, cache_with_tree, injected_resources, readable, sorted, - superuser_auth, + TENANT, apply, apply_as, apply_without_tree, cache_with_tree, cache_with_tree_in, + injected_resources, readable, regular_auth, sorted, superuser_auth, }; use crate::bridge::envelope::PhysicalPlan; + use crate::types::{DatabaseId, TenantId, VShardId}; fn columnar_scan(collection: &str) -> PhysicalPlan { + columnar_scan_in(DatabaseId::DEFAULT, collection) + } + + fn columnar_scan_in(database_id: DatabaseId, collection: &str) -> PhysicalPlan { PhysicalPlan::Columnar(ColumnarOp::Scan { - collection: nodedb_types::QualifiedCollection::new( - nodedb_types::DatabaseId::DEFAULT, - collection, - ), + collection: nodedb_types::QualifiedCollection::new(database_id, collection), projection: Vec::new(), limit: 0, filters: Vec::new(), @@ -345,9 +361,8 @@ mod tests { } } - /// A columnar scan was not listed before this pass became exhaustive, so - /// it returned every row of a governed collection. It is now narrowed to - /// the readable subtree. + /// A columnar scan over a governed collection narrows to the readable + /// subtree instead of returning every row. #[test] fn columnar_scan_is_narrowed_to_the_readable_subtree() { let cache = cache_with_tree("events"); @@ -429,4 +444,64 @@ mod tests { other => panic!("plan shape changed: {other:?}"), } } + + fn task_in(database_id: DatabaseId, plan: PhysicalPlan) -> PhysicalTask { + PhysicalTask { + tenant_id: TenantId::new(TENANT), + vshard_id: VShardId::new(0), + database_id, + plan, + post_set_op: nodedb_physical::physical_task::PostSetOp::None, + txn_id: None, + } + } + + /// The pass resolves each task's tree in that task's database: a tree in + /// one database filters a scan there, and leaves the same collection name + /// in the default database and in another database untouched. + #[test] + fn injection_resolves_trees_in_the_task_database() { + let db1 = DatabaseId::new(7); + let db2 = DatabaseId::new(8); + let cache = cache_with_tree_in(db1, "events"); + let mut tasks = vec![ + task_in(db1, columnar_scan_in(db1, "events")), + task_in(db2, columnar_scan_in(db2, "events")), + task_in(DatabaseId::DEFAULT, columnar_scan("events")), + ]; + let untouched_db2 = tasks[1].plan.clone(); + let untouched_default = tasks[2].plan.clone(); + + super::inject_permission_tree(&mut tasks, &cache, ®ular_auth()).expect("inject"); + + assert_eq!(scan_subtree(&tasks[0].plan), readable()); + assert_eq!(tasks[1].plan, untouched_db2); + assert_eq!(tasks[2].plan, untouched_default); + } + + /// A default-database tree never filters a named database's collection + /// of the same name. + #[test] + fn a_default_database_tree_does_not_apply_in_another_database() { + let db = DatabaseId::new(7); + let cache = cache_with_tree("events"); + let mut tasks = vec![task_in(db, columnar_scan_in(db, "events"))]; + let before = tasks[0].plan.clone(); + + super::inject_permission_tree(&mut tasks, &cache, ®ular_auth()).expect("inject"); + + assert_eq!(tasks[0].plan, before); + } + + /// A collection name not qualified for the task's database is refused + /// rather than resolved in another database. + #[test] + fn a_name_qualified_for_another_database_is_refused() { + let cache = cache_with_tree_in(DatabaseId::new(7), "events"); + let mut tasks = vec![task_in( + DatabaseId::new(7), + columnar_scan_in(DatabaseId::new(8), "events"), + )]; + assert!(super::inject_permission_tree(&mut tasks, &cache, ®ular_auth()).is_err()); + } } diff --git a/nodedb/src/control/planner/rls_injection/plan.rs b/nodedb/src/control/planner/rls_injection/plan.rs index 671884b88..a97fd4d09 100644 --- a/nodedb/src/control/planner/rls_injection/plan.rs +++ b/nodedb/src/control/planner/rls_injection/plan.rs @@ -330,6 +330,7 @@ mod tests { options: Default::default(), bm25_query: None, bm25_field: None, + stage: nodedb_physical::physical_plan::RagStage::Local, }) } diff --git a/nodedb/src/control/planner/rls_injection/text.rs b/nodedb/src/control/planner/rls_injection/text.rs index 6cfa7ce2b..2ae88ae98 100644 --- a/nodedb/src/control/planner/rls_injection/text.rs +++ b/nodedb/src/control/planner/rls_injection/text.rs @@ -11,7 +11,7 @@ use super::context::RlsCtx; pub(super) fn inject_text(ctx: &RlsCtx<'_>, op: &mut TextOp) -> crate::Result<()> { match op { // Inject: the policy lands in the post-score / post-fusion slot the - // handler applies before the ranked hits are returned. The result may + // handler applies before the ranked hits are returned. The result can // hold fewer than `top_k` rows, which is the intended effect. TextOp::Search { collection, @@ -72,7 +72,7 @@ mod tests { nodedb_types::DatabaseId::DEFAULT, collection, ), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), text: "hello".into(), provenance: None, }) diff --git a/nodedb/src/control/planner/rls_injection/vector.rs b/nodedb/src/control/planner/rls_injection/vector.rs index 0bd93d9f6..a29c52007 100644 --- a/nodedb/src/control/planner/rls_injection/vector.rs +++ b/nodedb/src/control/planner/rls_injection/vector.rs @@ -36,7 +36,7 @@ pub(super) fn inject_vector(ctx: &RlsCtx<'_>, op: &mut VectorOp) -> crate::Resul } => ctx.set_post_filters(collection, rls_filters), // Refuse: both return scored document identities with no filter slot, - // so the rows a policy hides would still be ranked and returned. + // so the rows a policy hides will still be ranked and returned. VectorOp::SparseSearch { collection, .. } | VectorOp::MultiVectorScoreSearch { collection, .. } => ctx.refuse_if_policy( collection, @@ -53,7 +53,7 @@ pub(super) fn inject_vector(ctx: &RlsCtx<'_>, op: &mut VectorOp) -> crate::Resul ), // Admit: a vector-primary collection stores the row here and nowhere - // else — no companion document write would gate it — and this op + // else — no companion document write will gate it — and this op // carries `payload`, the MessagePack image of every non-vector column // the statement supplied, so the policy decides that image directly. // @@ -85,7 +85,7 @@ pub(super) fn inject_vector(ctx: &RlsCtx<'_>, op: &mut VectorOp) -> crate::Resul // The read filter is independent of that write admission. A // `RETURNING` clause on this op ships the stored row back, and that // output is a read, so a row a read-only policy hides must not - // become visible just because the statement wrote it. A collection + // become visible only because the statement wrote it. A collection // can carry a `FOR SELECT` policy and no write policy at all, in // which case the write is unrestricted and only the returned row // set shrinks. @@ -135,7 +135,7 @@ pub(super) fn inject_vector(ctx: &RlsCtx<'_>, op: &mut VectorOp) -> crate::Resul // Refuse: these carry an embedding, a surrogate, or an opaque document // id — never the row body a policy predicate names — so no image is // available for the write policy to be evaluated against. A vector - // entry is a claim about a row, and admitting it unchecked would let an + // entry is a claim about a row, and admitting it unchecked will let an // identity the policy restricts make a hidden row reachable by search. // // `Insert` is also reachable from the document insert path, where the @@ -162,7 +162,7 @@ pub(super) fn inject_vector(ctx: &RlsCtx<'_>, op: &mut VectorOp) -> crate::Resul "a truncate removes every row without reading one, so no row image is available", ), - // No-op: already decided by the resolve pass; re-injecting would + // No-op: already decided by the resolve pass; re-injecting will // replace a verdict with a predicate no applying node can decide. VectorOp::ResolvedDirectWrite { .. } => Ok(()), @@ -197,7 +197,7 @@ mod tests { vector: vec![0.0], dim: 1, field_name: String::new(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), pk_bytes: None, provenance: None, }) @@ -305,7 +305,7 @@ mod tests { collection, ), field: "emb".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), pk_bytes: Vec::new(), vector: vec![0.1, 0.2], payload, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_alter_convert.rs b/nodedb/src/control/planner/sql_plan_convert/array_alter_convert.rs index c742523c1..56914c57f 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_alter_convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_alter_convert.rs @@ -3,8 +3,8 @@ //! `SqlPlan::AlterArray` → `PhysicalTask` lowering. //! //! Conversion only reads the current catalog entry and validates the requested -//! change. The authorized dispatch boundary persists the new entry and updates -//! the runtime retention mirror immediately around execution. +//! change. The front door proposes the updated entry as a replicated +//! `PutArray` (`array_catalog::ddl`). use crate::bridge::envelope::PhysicalPlan; use crate::control::array_catalog::ArrayCatalogEntry; @@ -66,20 +66,6 @@ pub(super) fn convert_alter_array( })?; } - let updated = ArrayCatalogEntry { - audit_retain_ms: new_retain, - minimum_audit_retain_ms: if minimum_audit_retain_ms.is_some() { - Some(new_min) - } else { - current.minimum_audit_retain_ms - }, - ..current.clone() - }; - - // `updated` is intentionally not installed here. It is reconstructed and - // durably installed by the authorized dispatch boundary. - let _updated = updated; - let vshard = ctx.collection_key(name).vshard(); Ok(vec![PhysicalTask { tenant_id, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_convert/ddl.rs b/nodedb/src/control/planner/sql_plan_convert/array_convert/ddl.rs index b60359d56..2b79939a0 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_convert/ddl.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_convert/ddl.rs @@ -93,7 +93,7 @@ pub(in super::super) fn convert_create_array( // reject a negative, so today every value arrives inside `i64`; this is // checked rather than cast so that a future entry point which skips that // validation fails here instead of wrapping to a negative retention, which - // would put the horizon in the future and purge every version the array has. + // will put the horizon in the future and purge every version the array has. let audit_retain_ms = audit_retain_ms .map(i64::try_from) .transpose() @@ -105,8 +105,8 @@ pub(in super::super) fn convert_create_array( ), })?; - // 3. Build DDL metadata only. Catalog and retention mutations happen at - // the authorized dispatch boundary, never while converting a SQL plan. + // 3. Build DDL metadata only. The catalog changes through the replicated + // `PutArray` entry the front door proposes, never while converting. let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); let entry = ArrayCatalogEntry { array_id: aid.clone(), @@ -117,6 +117,8 @@ pub(in super::super) fn convert_create_array( prefix_bits, audit_retain_ms, minimum_audit_retain_ms, + modification_hlc: nodedb_types::Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, }; { let cat = array_catalog.read().map_err(|_| crate::Error::PlanError { @@ -129,8 +131,8 @@ pub(in super::super) fn convert_create_array( } } - // 4. Emit OpenArray so the authorized execution boundary can durably - // register it immediately before opening the engine side. + // 4. Emit OpenArray. The front door hands it to the replicated array + // catalog (`array_catalog::ddl`), which opens it on every node. let vshard = ctx.collection_key(name).vshard(); Ok(vec![PhysicalTask { tenant_id, diff --git a/nodedb/src/control/planner/sql_plan_convert/array_convert/dml.rs b/nodedb/src/control/planner/sql_plan_convert/array_convert/dml.rs index f29870c52..46bf76048 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_convert/dml.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_convert/dml.rs @@ -8,6 +8,7 @@ use nodedb_array::types::ArrayId; use nodedb_sql::types_array::{ArrayCoordLiteral, ArrayInsertRow}; use crate::bridge::envelope::PhysicalPlan; +use crate::control::array_catalog::ArrayCatalogEntry; use crate::engine::array::wal::{ArrayDeleteCell, ArrayPutCell}; use crate::types::TenantId; use nodedb_physical::physical_plan::{ArrayOp, ClusterArrayOp}; @@ -16,12 +17,13 @@ use super::super::convert::ConvertContext; use super::helpers::{coerce_attrs, coerce_coords}; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; -pub(in super::super) fn convert_insert_array( +/// The catalog entry and decoded schema of the array `name` an +/// `INSERT INTO ARRAY` writes. +fn insert_target( name: &str, - rows: &[ArrayInsertRow], tenant_id: TenantId, ctx: &ConvertContext, -) -> crate::Result> { +) -> crate::Result<(ArrayCatalogEntry, ArraySchema)> { let array_catalog = ctx .array_catalog .as_ref() @@ -42,6 +44,30 @@ pub(in super::super) fn convert_insert_array( format: "msgpack".into(), detail: format!("array schema decode: {e}"), })?; + Ok((entry, schema)) +} + +/// The identity key of every cell an `INSERT INTO ARRAY` writes, in row +/// order. +pub(in super::super) fn insert_array_cell_pks( + name: &str, + rows: &[ArrayInsertRow], + tenant_id: TenantId, + ctx: &ConvertContext, +) -> crate::Result>> { + let (_, schema) = insert_target(name, tenant_id, ctx)?; + rows.iter() + .map(|row| array_coord_pk(&row.coords, &schema)) + .collect() +} + +pub(in super::super) fn convert_insert_array( + name: &str, + rows: &[ArrayInsertRow], + tenant_id: TenantId, + ctx: &ConvertContext, +) -> crate::Result> { + let (entry, schema) = insert_target(name, tenant_id, ctx)?; let aid = ArrayId::in_database(tenant_id, ctx.database_id, name); let vshard = ctx.collection_key(name).vshard(); @@ -137,10 +163,12 @@ pub(in super::super) fn convert_insert_array( // Stamped by the write funnel with the LSN of the redo record it // appends for this write, so the live tile version equals the one // replay rebuilds from that record's header. The planner must not - // allocate one: it appends nothing, so any LSN it picked would name + // allocate one: it appends nothing, so any LSN it picked will name // a record that does not exist. wal_lsn: 0, provenance: None, + // Single node: the task's vShard holds every tile. + vshard_id: vshard.as_u32(), }), post_set_op: PostSetOp::None, txn_id: None, @@ -245,8 +273,19 @@ pub(in super::super) fn convert_delete_array( // Stamped by the write funnel — see `convert_insert_array`. wal_lsn: 0, provenance: None, + // Single node: the task's vShard holds every tile. + vshard_id: vshard.as_u32(), }), post_set_op: PostSetOp::None, txn_id: None, }]) } + +/// The identity key of an array cell: its coerced coordinate, msgpack-encoded. +fn array_coord_pk(coords: &[ArrayCoordLiteral], schema: &ArraySchema) -> crate::Result> { + let coord = coerce_coords(coords, schema)?; + zerompk::to_msgpack_vec(&coord).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("array coord pk encode: {e}"), + }) +} diff --git a/nodedb/src/control/planner/sql_plan_convert/array_convert/mod.rs b/nodedb/src/control/planner/sql_plan_convert/array_convert/mod.rs index 6e9017da0..08333b733 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_convert/mod.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_convert/mod.rs @@ -5,4 +5,4 @@ mod dml; mod helpers; pub(super) use ddl::{CreateArrayArgs, convert_create_array, convert_drop_array}; -pub(super) use dml::{convert_delete_array, convert_insert_array}; +pub(super) use dml::{convert_delete_array, convert_insert_array, insert_array_cell_pks}; diff --git a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/aggregate.rs b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/aggregate.rs index 7b6da2551..fea8d2ab4 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/aggregate.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/aggregate.rs @@ -134,6 +134,8 @@ mod tests { prefix_bits: 8, audit_retain_ms: None, minimum_audit_retain_ms: None, + modification_hlc: nodedb_types::Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, }) .expect("register"); } @@ -143,7 +145,8 @@ mod tests { array_catalog: Some(handle), credentials: None, wal: None, - surrogate_assigner: None, + surrogate_assigner: + crate::control::planner::sql_plan_convert::test_support::test_assigner(), cluster_enabled, bitemporal_retention_registry: None, max_vector_dim: 0, @@ -153,6 +156,7 @@ mod tests { shuffle_agg_num_parts: 0, broadcast_threshold_bytes: 8 * 1024 * 1024, shuffle_agg_threshold: 10_000, + prefetched: Default::default(), database_id: crate::types::DatabaseId::DEFAULT, tenant_id: crate::types::TenantId::new(0), } diff --git a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/slice.rs b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/slice.rs index 0f9bd496a..7daa52d12 100644 --- a/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/slice.rs +++ b/nodedb/src/control/planner/sql_plan_convert/array_fn_convert/slice.rs @@ -111,13 +111,13 @@ pub(crate) fn convert_slice( /// endpoints. Generates all 2^num_dims corner coordinates of the bounding /// box and computes the Hilbert prefix for each. Returns a single /// `(min_prefix, max_prefix)` range that conservatively covers all cells -/// that could match the slice. +/// that can match the slice. /// /// Falls back to an empty vec (unbounded = contact all shards) if the -/// schema has more than 16 dimensions (2^16 corners would be excessive) +/// schema has more than 16 dimensions (2^16 corners will be excessive) /// or if any Hilbert encoding fails. /// -/// The result may over-estimate the shard set — the shard-side slice +/// The result can over-estimate the shard set — the shard-side slice /// filter is always applied and never produces false positives. fn compute_slice_hilbert_ranges( schema: &ArraySchema, @@ -234,6 +234,8 @@ mod tests { prefix_bits: 8, audit_retain_ms: None, minimum_audit_retain_ms: None, + modification_hlc: nodedb_types::Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, }; { let mut cat = handle.write().expect("write lock"); @@ -245,7 +247,8 @@ mod tests { array_catalog: Some(handle.clone()), credentials: None, wal: None, - surrogate_assigner: None, + surrogate_assigner: + crate::control::planner::sql_plan_convert::test_support::test_assigner(), cluster_enabled, bitemporal_retention_registry: None, max_vector_dim: 0, @@ -255,6 +258,7 @@ mod tests { shuffle_agg_num_parts: 0, broadcast_threshold_bytes: 8 * 1024 * 1024, shuffle_agg_threshold: 10_000, + prefetched: Default::default(), database_id: crate::types::DatabaseId::DEFAULT, tenant_id: crate::types::TenantId::new(0), }; diff --git a/nodedb/src/control/planner/sql_plan_convert/body.rs b/nodedb/src/control/planner/sql_plan_convert/body.rs index 56ed45ad3..e4690ac37 100644 --- a/nodedb/src/control/planner/sql_plan_convert/body.rs +++ b/nodedb/src/control/planner/sql_plan_convert/body.rs @@ -170,7 +170,8 @@ mod tests { array_catalog: None, credentials: None, wal: None, - surrogate_assigner: None, + surrogate_assigner: + crate::control::planner::sql_plan_convert::test_support::test_assigner(), cluster_enabled: false, bitemporal_retention_registry: None, max_vector_dim: 0, @@ -180,6 +181,7 @@ mod tests { shuffle_agg_num_parts: 0, broadcast_threshold_bytes: 8 * 1024 * 1024, shuffle_agg_threshold: 10_000, + prefetched: Default::default(), database_id: crate::types::DatabaseId::DEFAULT, tenant_id: crate::types::TenantId::new(0), } diff --git a/nodedb/src/control/planner/sql_plan_convert/convert.rs b/nodedb/src/control/planner/sql_plan_convert/convert.rs index 40930d641..cfc4b97b0 100644 --- a/nodedb/src/control/planner/sql_plan_convert/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/convert.rs @@ -49,7 +49,7 @@ pub enum PlanningPurpose { /// Conversion context holding optional references needed during plan conversion. pub struct ConvertContext { - /// Execution conversion may allocate identities and apply converter-owned + /// Execution conversion can allocate identities and apply converter-owned /// catalog changes. Metadata conversion is strictly side-effect-free. pub purpose: PlanningPurpose, pub retention_registry: Option>, @@ -65,10 +65,8 @@ pub struct ConvertContext { /// CP-side surrogate assigner — bound to the same `Arc` held on /// `SharedState`. Threaded into INSERT/UPSERT/KV-INSERT converters /// to bind `(collection, pk_bytes)` → `Surrogate` before the op - /// crosses the SPSC bridge. `None` only for converters used by - /// sub-planners that never lower to the surrogate-bearing variants - /// (e.g. CREATE/DROP/ARRAY paths). - pub surrogate_assigner: Option>, + /// crosses the SPSC bridge. + pub surrogate_assigner: Arc, /// `true` when the node is running in cluster mode with a live /// topology. Array DML/query converters emit `ClusterArray` variants /// when this flag is set; single-node mode emits local `Array` variants. @@ -129,6 +127,11 @@ pub struct ConvertContext { /// `DEFAULT_SHUFFLE_AGG_THRESHOLD`; overridable per-session via /// `nodedb.shuffle_agg_threshold` for operator control and test determinism. pub shuffle_agg_threshold: usize, + /// The surrogate answers for this batch's keys, resolved async before + /// conversion (`convert_bound`). Conversion reads them and never draws a + /// surrogate or asks a key's home. A key with no answer is recorded here + /// as a miss for the bind step to resolve. + pub prefetched: super::surrogate_prefetch::PrefetchedSurrogates, } impl ConvertContext { @@ -142,48 +145,72 @@ impl ConvertContext { nodedb_types::CollectionKey::from_bare(self.database_id, bare) } - /// Resolve an existing surrogate without creating a mapping while planning - /// metadata. Execute planning retains the allocating assignment behavior. + /// The surrogate a write that creates its row plans `pk_bytes` with. + /// + /// The bind step's answer or this node's catalog binding answers. A key + /// neither answers is recorded as a miss and plans with the + /// `Surrogate::ZERO` placeholder: the bind step binds it and converts the + /// plan again. A metadata plan binds nothing, so it reads the existing + /// binding and renders an unbound key as the placeholder. pub fn surrogate_for_pk( &self, key: nodedb_types::CollectionKey<'_>, pk_bytes: &[u8], ) -> crate::Result { - let Some(assigner) = self.surrogate_assigner.as_ref() else { - return Ok(nodedb_types::Surrogate::ZERO); - }; if self.is_metadata() { - return Ok(assigner - .lookup(key, self.tenant_id, pk_bytes)? + return Ok(self + .surrogate_for_existing_pk(key, pk_bytes)? .unwrap_or(nodedb_types::Surrogate::ZERO)); } - assigner.assign(key, self.tenant_id, pk_bytes) + if let Some(bound) = self.prefetched.bound(key, pk_bytes) { + return Ok(bound); + } + if let Some(bound) = self + .surrogate_assigner + .lookup_bound(key, self.tenant_id, pk_bytes)? + { + return Ok(bound); + } + self.prefetched.record_bind_miss(key, pk_bytes); + Ok(nodedb_types::Surrogate::ZERO) } - /// Resolve an EXISTING pk → surrogate binding read-only, yielding - /// `Surrogate::ZERO` when the key is unbound. Used by writes that mutate - /// rows they never create (PK UPDATE / DELETE): allocating there would mint - /// a node-local phantom binding for a key no replica agrees on. + /// Resolve an EXISTING pk → surrogate binding read-only, or `None` when + /// the key is unbound in this database. Used by reads and by writes that + /// mutate rows they never create (PK UPDATE / DELETE): allocating there + /// will mint a node-local phantom binding for a key no replica agrees on. + /// + /// A single node's catalog miss is the answer. A cluster node records a + /// key neither the bind step nor its catalog answers as a miss, and plans + /// it as unbound until the bind step asks the key's home. pub fn surrogate_for_existing_pk( &self, key: nodedb_types::CollectionKey<'_>, pk_bytes: &[u8], - ) -> crate::Result { - let Some(assigner) = self.surrogate_assigner.as_ref() else { - return Ok(nodedb_types::Surrogate::ZERO); - }; - Ok(assigner - .lookup(key, self.tenant_id, pk_bytes)? - .unwrap_or(nodedb_types::Surrogate::ZERO)) + ) -> crate::Result> { + if let Some(answer) = self.prefetched.get(key, pk_bytes) { + return Ok(answer); + } + let assigner = &self.surrogate_assigner; + if let Some(bound) = assigner.lookup_bound(key, self.tenant_id, pk_bytes)? { + return Ok(Some(bound)); + } + if assigner.resolves_at_home()? { + self.prefetched.record_lookup_miss(key, pk_bytes); + } + Ok(None) } - /// Allocate a new surrogate and its bound identity string, only while - /// producing executable work. + /// The next fresh surrogate and its bound identity string the bind step + /// drew for `key`, only while producing executable work. /// - /// A metadata plan, or a plan with no wired assigner, returns a - /// `Surrogate::ZERO` placeholder paired with its rendered identity, via + /// A metadata plan, and an execute plan whose drawn identities ran out, + /// get a `Surrogate::ZERO` placeholder paired with its rendered identity, + /// via /// [`RowIdentity::for_surrogate`](crate::engine::document::store::RowIdentity::for_surrogate), - /// the same type the allocator renders through. + /// the same type the allocator renders through. The execute plan also + /// records the shortfall as a miss, so the bind step draws one more + /// identity and converts the plan again. pub fn fresh_surrogate( &self, key: nodedb_types::CollectionKey<'_>, @@ -198,9 +225,12 @@ impl ConvertContext { if self.is_metadata() { return placeholder(); } - match self.surrogate_assigner.as_ref() { - Some(assigner) => assigner.assign_fresh(key, self.tenant_id), - None => placeholder(), + match self.prefetched.take_fresh(key) { + Some(fresh) => Ok(fresh), + None => { + self.prefetched.record_fresh_miss(key); + placeholder() + } } } @@ -213,65 +243,75 @@ impl ConvertContext { } Ok(()) } +} - /// Build the deployment-neutral subset shared with `nodedb-physical`'s - /// converter helpers. Cheap: 3 `Copy` fields + an `Arc` clone. - pub fn shared(&self) -> nodedb_physical::SharedConvertContext { - nodedb_physical::SharedConvertContext { - database_id: self.database_id, - max_vector_dim: self.max_vector_dim, - cluster_enabled: self.cluster_enabled, - surrogate_assigner: self - .surrogate_assigner - .as_ref() - .map(|a| a.clone() as std::sync::Arc), - } +/// Convert a list of SqlPlans to PhysicalTasks with the surrogate answers +/// `ctx` already holds. A pure planning pass: it draws nothing and asks no +/// home. A key it finds no answer for fails the conversion. Execution +/// planning converts through +/// [`convert_bound`](super::surrogate_prefetch::convert_bound), which awaits +/// every answer first. +pub fn convert( + plans: &[SqlPlan], + tenant_id: TenantId, + ctx: &ConvertContext, +) -> crate::Result> { + let mut tasks = Vec::new(); + for plan in plans { + tasks.extend(convert_plan(plan, tenant_id, ctx)?); + } + let misses = ctx.prefetched.take_misses(); + if !misses.is_empty() { + return Err(crate::Error::PlanError { + detail: format!( + "surrogate keys of {} have no answer; convert through `convert_bound`, which \ + resolves them first", + misses.collection_names() + ), + }); } + Ok(tasks) } -/// Convert a list of SqlPlans to PhysicalTasks. +/// Convert one SqlPlan to PhysicalTasks. /// /// After each task is produced, any top-level read plan that is a sharded /// source is wrapped in `Exchange{Gather}` so the coordinator knows to fan /// it to all Data Plane cores and merge the results. Non-sharded plans /// (point gets, writes, constant `ProviderScan`s, coordinator-local joins) /// are left unwrapped. -pub fn convert( - plans: &[SqlPlan], +pub(super) fn convert_plan( + plan: &SqlPlan, tenant_id: TenantId, ctx: &ConvertContext, ) -> crate::Result> { - let mut tasks = Vec::new(); - for plan in plans { - let mut one = convert_one(plan, tenant_id, ctx)?; - for task in &mut one { - if task.plan.is_sharded_source() { - let as_aggregate = matches!( - &task.plan, - PhysicalPlan::Query(QueryOp::Aggregate { .. }) - | PhysicalPlan::Query(QueryOp::PartialAggregate { .. }) - ); - // Move the plan out, wrap it in Exchange{Gather}, put it back. - let sentinel = PhysicalPlan::Query(QueryOp::ProviderScan { - provider: None, - rows: Vec::new(), - filters: Vec::new(), - projection: Vec::new(), - computed_columns: Vec::new(), - window_functions: Vec::new(), - sort_keys: Vec::new(), - limit: None, - offset: 0, - distinct: false, - }); - let inner = std::mem::replace(&mut task.plan, sentinel); - task.plan = PhysicalPlan::Query(QueryOp::Exchange(ExchangeOp { - child: Box::new(inner), - mode: ExchangeMode::Gather { as_aggregate }, - })); - } + let mut tasks = convert_one(plan, tenant_id, ctx)?; + for task in &mut tasks { + if task.plan.is_sharded_source() { + let as_aggregate = matches!( + &task.plan, + PhysicalPlan::Query(QueryOp::Aggregate { .. }) + | PhysicalPlan::Query(QueryOp::PartialAggregate { .. }) + ); + // Move the plan out, wrap it in Exchange{Gather}, put it back. + let sentinel = PhysicalPlan::Query(QueryOp::ProviderScan { + provider: None, + rows: Vec::new(), + filters: Vec::new(), + projection: Vec::new(), + computed_columns: Vec::new(), + window_functions: Vec::new(), + sort_keys: Vec::new(), + limit: None, + offset: 0, + distinct: false, + }); + let inner = std::mem::replace(&mut task.plan, sentinel); + task.plan = PhysicalPlan::Query(QueryOp::Exchange(ExchangeOp { + child: Box::new(inner), + mode: ExchangeMode::Gather { as_aggregate }, + })); } - tasks.extend(one); } Ok(tasks) } @@ -303,7 +343,7 @@ mod tests { array_catalog: None, credentials: None, wal: None, - surrogate_assigner: Some(assigner), + surrogate_assigner: assigner, cluster_enabled: false, bitemporal_retention_registry: None, max_vector_dim: 0, @@ -315,6 +355,7 @@ mod tests { shuffle_agg_num_parts: 0, broadcast_threshold_bytes: 0, shuffle_agg_threshold: 0, + prefetched: Default::default(), } } @@ -350,7 +391,7 @@ mod tests { ); assert_eq!( assigner - .lookup( + .lookup_bound( nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), TenantId::new(1), b"new-user", @@ -359,25 +400,50 @@ mod tests { None ); assert_eq!(registry.read().expect("registry").current_hwm(), 0); + assert!(metadata.prefetched.take_misses().is_empty()); + } + /// Execute conversion never draws: an unanswered key plans with the + /// placeholder and is recorded for the bind step, and the registry stays + /// untouched. + #[test] + fn execute_conversion_records_unanswered_keys_instead_of_drawing() { + let dir = tempfile::tempdir().expect("tempdir"); + let credentials = Arc::new( + CredentialStore::open(&dir.path().join("system.redb")).expect("credential store"), + ); + let registry = Arc::new(RwLock::new(SurrogateRegistry::new())); + let wal: Arc = Arc::new(NoopWalAppender); + let assigner = Arc::new(SurrogateAssigner::new( + Arc::clone(®istry), + credentials, + wal, + )); let execute = context(PlanningPurpose::Execute, Arc::clone(&assigner)); - let allocated = execute - .surrogate_for_pk(execute.collection_key("users"), b"new-user") - .unwrap(); - assert_ne!(allocated.as_u32(), 0); + let users = execute.collection_key("users"); + assert_eq!( - assigner - .lookup( - nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "users"), - TenantId::new(1), - b"new-user", - ) - .unwrap(), - Some(allocated) + execute + .surrogate_for_pk(users, b"new-user") + .unwrap() + .as_u32(), + 0 ); + assert_eq!(execute.fresh_surrogate(users).unwrap().0.as_u32(), 0); + // A single node's catalog miss is the answer: no lookup is recorded. assert_eq!( - registry.read().expect("registry").current_hwm(), - allocated.as_u32() + execute.surrogate_for_existing_pk(users, b"ghost").unwrap(), + None ); + assert_eq!(registry.read().expect("registry").current_hwm(), 0); + + let misses = execute.prefetched.take_misses(); + assert_eq!(misses.len(), 2); + let (key, recorded) = misses.iter().next().expect("one collection"); + assert_eq!(key.name(), "users"); + assert!(recorded.binds.contains(b"new-user".as_slice())); + assert!(recorded.lookups.is_empty()); + assert_eq!(recorded.fresh, 1); + assert!(execute.prefetched.take_misses().is_empty()); } } diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs index d59c80c02..2b8db680d 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/convert.rs @@ -113,7 +113,7 @@ pub(in super::super::super) fn convert_insert( row, )?; // One page for the whole statement: the rows of a balanced - // INSERT are judged together, so they may not be split across + // INSERT are judged together, so they must not be split across // one task — one boundary — per row. if is_balanced { balanced_documents.push((doc_id, value_bytes)); @@ -216,7 +216,7 @@ pub(in super::super::super) fn convert_insert( // Both slots are filled by later passes over the built plan — // `inject_returning_spec` from the statement's RETURNING list, // and the row-level-security injector from the collection's read - // policy. Filling either here would duplicate a decision that + // policy. Filling either here will duplicate a decision that // has one owner. returning: None, rls_filters: Vec::new(), @@ -248,16 +248,16 @@ mod tests { CredentialStore::open(&dir.path().join("system.redb")).expect("open credential store"); { let catalog = store.catalog(); - let mut edges = StoredCollection::new(0, "edges", "owner"); + let mut edges = StoredCollection::stamped_for_test(0, "edges", "owner"); edges.has_implicit_edges = true; catalog .put_collection(crate::types::DatabaseId::DEFAULT, &edges) .expect("put edges collection"); - let plain = StoredCollection::new(0, "plain", "owner"); + let plain = StoredCollection::stamped_for_test(0, "plain", "owner"); catalog .put_collection(crate::types::DatabaseId::DEFAULT, &plain) .expect("put plain collection"); - let mut crdt_coll = StoredCollection::new(0, "crdt_coll", "owner"); + let mut crdt_coll = StoredCollection::stamped_for_test(0, "crdt_coll", "owner"); crdt_coll.crdt = true; catalog .put_collection(crate::types::DatabaseId::DEFAULT, &crdt_coll) @@ -270,7 +270,8 @@ mod tests { array_catalog: None, credentials: Some(Arc::new(store)), wal: None, - surrogate_assigner: None, + surrogate_assigner: + crate::control::planner::sql_plan_convert::test_support::test_assigner(), cluster_enabled: false, bitemporal_retention_registry: None, max_vector_dim: 0, @@ -280,6 +281,7 @@ mod tests { shuffle_agg_num_parts: 0, broadcast_threshold_bytes: 8 * 1024 * 1024, shuffle_agg_threshold: 10_000, + prefetched: Default::default(), database_id: crate::types::DatabaseId::DEFAULT, tenant_id: crate::types::TenantId::new(0), }; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs index 9bf70f12c..d03f21a87 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/identity.rs @@ -38,7 +38,7 @@ pub(in super::super::super) fn declared_primary_key_name( let Some(credentials) = ctx.credentials.as_ref() else { return Ok(None); }; - // `collection` may be bare or db-qualified. The catalog keys collections + // `collection` can be bare or db-qualified. The catalog keys collections // by the bare name. let bare = crate::control::target_identity::naming::bare_collection_name(ctx.database_id, collection); @@ -100,6 +100,45 @@ pub(in super::super) fn resolve_doc_identity_with_declared( } } +/// The keys that content-address the surrogates of `rows`, the rows that +/// name one, in row order. An auto-`_rowid` collection has none: each of its +/// rows mints a fresh surrogate. +pub(in super::super) fn doc_identity_keys( + primary_key: &str, + declared: Option<&str>, + rows: &[impl AsRef<[(String, SqlValue)]>], +) -> Vec { + if is_auto_rowid_pk(primary_key) { + return Vec::new(); + } + let mint_key = declared.unwrap_or(primary_key); + rows.iter() + .filter_map(|row| match extract_doc_id(row.as_ref(), mint_key) { + DocId::Present(id) => Some(id), + DocId::ExplicitNull | DocId::Absent => None, + }) + .collect() +} + +/// How many of `rows` mint a fresh surrogate: every row of an auto-`_rowid` +/// collection, and a row that names no key when no primary key is declared. +/// A row that names no declared key mints nothing: its conversion refuses it. +pub(in super::super) fn fresh_identity_count( + primary_key: &str, + declared: Option<&str>, + rows: &[impl AsRef<[(String, SqlValue)]>], +) -> usize { + if is_auto_rowid_pk(primary_key) { + return rows.len(); + } + if declared.is_some() { + return 0; + } + rows.iter() + .filter(|row| !matches!(extract_doc_id(row.as_ref(), primary_key), DocId::Present(_))) + .count() +} + pub(in super::super) fn assign_for_pk( ctx: &ConvertContext, key: CollectionKey<'_>, diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/insert/mod.rs b/nodedb/src/control/planner/sql_plan_convert/dml/insert/mod.rs index acbc2dc53..3442dd412 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/insert/mod.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/insert/mod.rs @@ -9,7 +9,8 @@ pub(crate) use schema::{DEFAULT_IDENTITY_COLUMN, build_columnar_schema}; pub(in super::super) use identity::declared_primary_key_name; pub(super) use identity::{ - assign_for_pk, columnar_row_surrogates, is_auto_rowid_pk, resolve_doc_identity_with_declared, + assign_for_pk, columnar_row_surrogates, doc_identity_keys, fresh_identity_count, + is_auto_rowid_pk, resolve_doc_identity_with_declared, }; pub(in super::super) use convert::{ConvertInsertArgs, convert_insert}; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/mod.rs b/nodedb/src/control/planner/sql_plan_convert/dml/mod.rs index 675ef2f79..286e78e92 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/mod.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/mod.rs @@ -5,6 +5,7 @@ mod crdt_gate; mod insert; mod kv_insert; mod merge; +mod surrogate_keys; mod update_delete; mod upsert; mod vector_primary; @@ -13,6 +14,7 @@ pub(super) use insert::{ConvertInsertArgs, convert_insert, declared_primary_key_ pub(crate) use insert::{DEFAULT_IDENTITY_COLUMN, build_columnar_schema}; pub(super) use kv_insert::convert_kv_insert; pub(super) use merge::{ConvertMergeArgs, convert_merge}; +pub(super) use surrogate_keys::plan_key_batches; pub(super) use update_delete::{ UpdateFromParams, UpdateParams, convert_delete, convert_update, convert_update_from, }; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/surrogate_keys.rs b/nodedb/src/control/planner/sql_plan_convert/dml/surrogate_keys.rs new file mode 100644 index 000000000..5eebac5aa --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/dml/surrogate_keys.rs @@ -0,0 +1,448 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The surrogate keys a batch of write plans resolves while it converts, read +//! off the plans before conversion. Each batch derives its keys the way the +//! plan's converter does, so conversion finds every key already answered. +//! +//! A key this module does not derive is still correct: conversion records it +//! as a miss, and the bind step resolves it and converts the plan again. + +use nodedb_sql::types::{EngineType, SqlPlan, SqlValue}; + +use super::super::convert::ConvertContext; +use super::super::value::{sql_value_to_bytes, sql_value_to_string}; +use super::crdt_gate::document_collection_is_crdt; +use super::insert::{ + declared_primary_key_name, doc_identity_keys, fresh_identity_count, is_auto_rowid_pk, +}; + +/// The keys of one collection that one plan resolves. +pub(in super::super) struct KeyBatch<'p> { + /// The collection name as the plan names it. The converter builds the + /// collection key from this same name. + pub collection: &'p str, + pub pks: Vec>, + /// Whether conversion binds an absent key (a write that creates the row) + /// or only reads the key's binding. + pub binds: bool, + /// How many rows conversion mints a fresh surrogate for. + pub fresh: usize, +} + +/// The key batches of every plan in `plans`, in plan order. A plan whose +/// conversion resolves no key yields none. +/// +/// A plan whose keys cannot be read yet yields none either: an +/// `INSERT INTO ARRAY` whose array an earlier statement of the batch +/// creates, say. Its conversion reports the real error or records its keys. +pub(in super::super) fn plan_key_batches<'p>( + plans: &'p [SqlPlan], + ctx: &ConvertContext, +) -> Vec> { + let mut batches = Vec::new(); + for plan in plans { + collect_key_batches(plan, ctx, &mut batches); + } + batches +} + +/// The key batches of `plan` and of every plan it converts with it: a CTE's +/// definitions and outer statement (`WITH ... INSERT`), and the inputs of a +/// join, set operation, aggregate or subquery. +fn collect_key_batches<'p>( + plan: &'p SqlPlan, + ctx: &ConvertContext, + batches: &mut Vec>, +) { + match plan { + SqlPlan::Cte { definitions, outer } => { + for (_, definition) in definitions { + collect_key_batches(definition, ctx, batches); + } + collect_key_batches(outer, ctx, batches); + } + SqlPlan::Join { left, right, .. } + | SqlPlan::Intersect { left, right, .. } + | SqlPlan::Except { left, right, .. } => { + collect_key_batches(left, ctx, batches); + collect_key_batches(right, ctx, batches); + } + SqlPlan::Union { inputs, .. } => { + for input in inputs { + collect_key_batches(input, ctx, batches); + } + } + SqlPlan::Aggregate { input, .. } | SqlPlan::Subquery { input, .. } => { + collect_key_batches(input, ctx, batches); + } + _ => { + if let Ok(Some(batch)) = plan_key_batch(plan, ctx) { + batches.push(batch); + } + } + } +} + +fn plan_key_batch<'p>( + plan: &'p SqlPlan, + ctx: &ConvertContext, +) -> crate::Result>> { + let batch = match plan { + SqlPlan::Insert { + collection, + rows, + primary_key: Some(primary_key), + .. + } + | SqlPlan::Upsert { + collection, + rows, + primary_key: Some(primary_key), + .. + } => row_identity_batch(ctx, collection, primary_key, rows)?, + SqlPlan::VectorPrimaryInsert { + collection, + rows, + primary_key: Some(primary_key), + .. + } => { + let rows: Vec> = rows + .iter() + .map(|row| { + row.payload_fields + .iter() + .map(|(k, v)| (k.clone(), v.clone())) + .collect() + }) + .collect(); + row_identity_batch(ctx, collection, primary_key, &rows)? + } + SqlPlan::KvInsert { + collection, + entries, + .. + } => KeyBatch { + collection, + pks: entries + .iter() + .filter(|(key, _)| !matches!(key, SqlValue::Null)) + .filter_map(|(key, _)| sql_value_to_bytes(key).ok()) + .collect(), + binds: true, + fresh: 0, + }, + SqlPlan::Update { + collection, + engine, + target_keys, + .. + } => match engine { + // A KV UPDATE resolves each key through the binding path. + EngineType::KeyValue => KeyBatch { + collection, + pks: kv_keys(target_keys), + binds: true, + fresh: 0, + }, + EngineType::Timeseries | EngineType::Columnar | EngineType::Spatial => { + return Ok(None); + } + // A CRDT UPDATE creates an absent key. A document UPDATE only + // reads the binding. A document UPDATE naming several keys + // converts to one predicate write, which resolves no key. + _ => { + let crdt = !target_keys.is_empty() && document_collection_is_crdt(ctx, collection)?; + if !crdt && target_keys.len() > 1 { + return Ok(None); + } + KeyBatch { + collection, + pks: document_keys(target_keys), + binds: crdt, + fresh: 0, + } + } + }, + // A document point read resolves its key's binding read-only. + SqlPlan::PointGet { + collection, + engine: EngineType::DocumentSchemaless | EngineType::DocumentStrict, + key_value, + .. + } => KeyBatch { + collection, + pks: document_keys(std::slice::from_ref(key_value)), + binds: false, + fresh: 0, + }, + SqlPlan::Delete { + collection, + engine, + target_keys, + .. + } => match engine { + EngineType::KeyValue + | EngineType::Timeseries + | EngineType::Columnar + | EngineType::Spatial => return Ok(None), + _ => KeyBatch { + collection, + pks: document_keys(target_keys), + binds: false, + fresh: 0, + }, + }, + SqlPlan::VectorPrimaryDelete { + collection, + target_keys, + .. + } + | SqlPlan::VectorPrimaryUpdate { + collection, + target_keys, + .. + } => KeyBatch { + collection, + pks: document_keys(target_keys), + binds: false, + fresh: 0, + }, + // Every timeseries row takes a fresh identity. + SqlPlan::TimeseriesIngest { + collection, rows, .. + } => KeyBatch { + collection, + pks: Vec::new(), + binds: true, + fresh: rows.len(), + }, + SqlPlan::InsertArray { name, rows } => KeyBatch { + collection: name, + pks: super::super::array_convert::insert_array_cell_pks( + name, + rows, + ctx.tenant_id, + ctx, + )?, + binds: true, + fresh: 0, + }, + _ => return Ok(None), + }; + Ok((!batch.pks.is_empty() || batch.fresh > 0).then_some(batch)) +} + +/// The content keys of rows whose identity comes from the collection's +/// primary key, found as the insert converters find them. +fn row_identity_batch<'p>( + ctx: &ConvertContext, + collection: &'p str, + primary_key: &str, + rows: &[Vec<(String, SqlValue)>], +) -> crate::Result> { + let declared = if is_auto_rowid_pk(primary_key) { + None + } else { + declared_primary_key_name(ctx, collection)? + }; + Ok(KeyBatch { + collection, + pks: doc_identity_keys(primary_key, declared.as_deref(), rows) + .into_iter() + .map(String::into_bytes) + .collect(), + binds: true, + fresh: fresh_identity_count(primary_key, declared.as_deref(), rows), + }) +} + +/// A document key's bytes: its rendered string. +fn document_keys(target_keys: &[SqlValue]) -> Vec> { + target_keys + .iter() + .map(|key| sql_value_to_string(key).into_bytes()) + .collect() +} + +/// A KV key's bytes. A key that has no byte form is left to conversion, which +/// reports it. +fn kv_keys(target_keys: &[SqlValue]) -> Vec> { + target_keys + .iter() + .filter_map(|key| sql_value_to_bytes(key).ok()) + .collect() +} + +#[cfg(test)] +mod tests { + use nodedb_sql::types::{EngineType, SqlPlan, SqlValue, WriteRoute}; + + use super::plan_key_batches; + use crate::control::planner::sql_plan_convert::{ConvertContext, PlanningPurpose}; + + fn ctx() -> ConvertContext { + ConvertContext { + purpose: PlanningPurpose::Execute, + retention_registry: None, + array_catalog: None, + credentials: None, + wal: None, + surrogate_assigner: + crate::control::planner::sql_plan_convert::test_support::test_assigner(), + cluster_enabled: true, + bitemporal_retention_registry: None, + max_vector_dim: 0, + force_shuffle_join: false, + shuffle_num_parts: 0, + force_shuffle_agg: false, + shuffle_agg_num_parts: 0, + broadcast_threshold_bytes: 8 * 1024 * 1024, + shuffle_agg_threshold: 10_000, + prefetched: Default::default(), + database_id: crate::types::DatabaseId::DEFAULT, + tenant_id: crate::types::TenantId::new(0), + } + } + + fn row(id: Option<&str>) -> Vec<(String, SqlValue)> { + let mut row = vec![("name".to_string(), SqlValue::String("n".to_string()))]; + if let Some(id) = id { + row.push(("id".to_string(), SqlValue::String(id.to_string()))); + } + row + } + + fn insert(collection: &str, primary_key: &str, rows: Vec>) -> SqlPlan { + SqlPlan::Insert { + collection: collection.to_string(), + engine: EngineType::DocumentSchemaless, + route: WriteRoute::Document, + rows, + volatile_defaults: false, + if_absent: false, + column_schema: Vec::new(), + primary_key: Some(primary_key.to_string()), + } + } + + fn keys(pks: &[Vec]) -> Vec<&[u8]> { + pks.iter().map(Vec::as_slice).collect() + } + + #[test] + fn insert_binds_named_keys_and_counts_fresh_rows() { + let ctx = ctx(); + let plans = vec![insert( + "users", + "id", + vec![row(Some("a")), row(Some("b")), row(None)], + )]; + let batches = plan_key_batches(&plans, &ctx); + assert_eq!(batches.len(), 1); + assert_eq!(batches[0].collection, "users"); + assert_eq!(keys(&batches[0].pks), vec![&b"a"[..], &b"b"[..]]); + assert!(batches[0].binds); + assert_eq!(batches[0].fresh, 1); + } + + #[test] + fn auto_rowid_insert_draws_one_fresh_identity_per_row() { + let ctx = ctx(); + let plans = vec![insert("events", "_rowid", vec![row(Some("a")), row(None)])]; + let batches = plan_key_batches(&plans, &ctx); + assert_eq!(batches.len(), 1); + assert!(batches[0].pks.is_empty()); + assert_eq!(batches[0].fresh, 2); + } + + #[test] + fn document_delete_reads_and_kv_update_binds() { + let ctx = ctx(); + let plans = vec![ + SqlPlan::Delete { + collection: "users".to_string(), + engine: EngineType::DocumentSchemaless, + filters: Vec::new(), + target_keys: vec![SqlValue::String("x".to_string())], + }, + SqlPlan::Update { + collection: "cache".to_string(), + engine: EngineType::KeyValue, + assignments: Vec::new(), + filters: Vec::new(), + target_keys: vec![SqlValue::String("k".to_string())], + returning: false, + }, + SqlPlan::Delete { + collection: "cache".to_string(), + engine: EngineType::KeyValue, + filters: Vec::new(), + target_keys: vec![SqlValue::String("k".to_string())], + }, + ]; + let batches = plan_key_batches(&plans, &ctx); + assert_eq!(batches.len(), 2); + assert_eq!(batches[0].collection, "users"); + assert!(!batches[0].binds); + assert_eq!(keys(&batches[0].pks), vec![&b"x"[..]]); + assert_eq!(batches[1].collection, "cache"); + assert!(batches[1].binds); + assert_eq!(batches[1].pks.len(), 1); + } + + #[test] + fn cte_outer_write_is_collected() { + let ctx = ctx(); + let plans = vec![SqlPlan::Cte { + definitions: Vec::new(), + outer: Box::new(insert("users", "id", vec![row(Some("c"))])), + }]; + let batches = plan_key_batches(&plans, &ctx); + assert_eq!(batches.len(), 1); + assert_eq!(keys(&batches[0].pks), vec![&b"c"[..]]); + } + + #[test] + fn document_point_read_inside_a_union_is_collected() { + let ctx = ctx(); + let point_get = |key: &str| SqlPlan::PointGet { + collection: "users".to_string(), + alias: None, + engine: EngineType::DocumentStrict, + key_column: "id".to_string(), + key_value: SqlValue::String(key.to_string()), + projection: Vec::new(), + }; + let plans = vec![SqlPlan::Union { + inputs: vec![point_get("p"), point_get("q")], + distinct: false, + }]; + let batches = plan_key_batches(&plans, &ctx); + assert_eq!(batches.len(), 2); + assert!(batches.iter().all(|batch| !batch.binds)); + assert_eq!(keys(&batches[0].pks), vec![&b"p"[..]]); + assert_eq!(keys(&batches[1].pks), vec![&b"q"[..]]); + } + + #[test] + fn a_plan_with_no_keys_yields_no_batch() { + let ctx = ctx(); + let plans = vec![insert("users", "id", Vec::new())]; + assert!(plan_key_batches(&plans, &ctx).is_empty()); + } + + #[test] + fn timeseries_ingest_draws_one_fresh_identity_per_row() { + let ctx = ctx(); + let plans = vec![SqlPlan::TimeseriesIngest { + collection: "metrics".to_string(), + rows: vec![Vec::new(), Vec::new(), Vec::new()], + volatile_defaults: false, + }]; + let batches = plan_key_batches(&plans, &ctx); + assert_eq!(batches.len(), 1); + assert_eq!(batches[0].collection, "metrics"); + assert!(batches[0].pks.is_empty()); + assert_eq!(batches[0].fresh, 3); + } +} diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs index f5f073635..c20ec0ba4 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/delete.rs @@ -115,7 +115,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( // CRDT gate: a `crdt = true` collection routes to `CrdtOp::DocDelete`. // Only a PK-targeted DELETE is representable; a predicate DELETE is - // rejected — no silent fallthrough that would bypass CRDT convergence. + // rejected — no silent fallthrough that will bypass CRDT convergence. let is_crdt = super::super::crdt_gate::document_collection_is_crdt(ctx, collection)?; if is_crdt && target_keys.is_empty() { return Err(crate::Error::BadRequest { @@ -165,10 +165,10 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_delete( for key in target_keys { let pk_string = sql_value_to_string(key); let pk_bytes = pk_string.clone().into_bytes(); - // Read-only resolution: a task always exists (the write hook still - // runs, an unbound row_key affects 0 rows, and the clone CoW - // resolver intercepts the ZERO sentinel), but a key this statement - // never creates must never mint a binding. + // Read-only resolution: a key this statement never creates must + // never mint a binding. A delete of an unbound key carries `None`: + // it affects 0 rows, so the statement still answers `DELETE 0`, + // and the clone CoW resolver still sees a document task. let surrogate = ctx.surrogate_for_existing_pk(collection_key, &pk_bytes)?; let plan = if is_crdt { PhysicalPlan::Crdt(CrdtOp::DocDelete { @@ -244,16 +244,16 @@ mod tests { CredentialStore::open(&dir.path().join("system.redb")).expect("open credential store"); { let catalog = store.catalog(); - let mut edges = StoredCollection::new(0, "edges", "owner"); + let mut edges = StoredCollection::stamped_for_test(0, "edges", "owner"); edges.has_implicit_edges = true; catalog .put_collection(crate::types::DatabaseId::DEFAULT, &edges) .expect("put edges collection"); - let plain = StoredCollection::new(0, "plain", "owner"); + let plain = StoredCollection::stamped_for_test(0, "plain", "owner"); catalog .put_collection(crate::types::DatabaseId::DEFAULT, &plain) .expect("put plain collection"); - let mut crdt_coll = StoredCollection::new(0, "crdt_coll", "owner"); + let mut crdt_coll = StoredCollection::stamped_for_test(0, "crdt_coll", "owner"); crdt_coll.crdt = true; catalog .put_collection(crate::types::DatabaseId::DEFAULT, &crdt_coll) @@ -266,7 +266,8 @@ mod tests { array_catalog: None, credentials: Some(Arc::new(store)), wal: None, - surrogate_assigner: None, + surrogate_assigner: + crate::control::planner::sql_plan_convert::test_support::test_assigner(), cluster_enabled: false, bitemporal_retention_registry: None, max_vector_dim: 0, @@ -276,6 +277,7 @@ mod tests { shuffle_agg_num_parts: 0, broadcast_threshold_bytes: 8 * 1024 * 1024, shuffle_agg_threshold: 10_000, + prefetched: Default::default(), database_id: crate::types::DatabaseId::DEFAULT, tenant_id: crate::types::TenantId::new(0), }; @@ -331,9 +333,16 @@ mod tests { ); } - #[test] - fn delete_by_pk_on_crdt_routes_doc_delete() { + #[tokio::test] + async fn delete_by_pk_on_crdt_routes_doc_delete() { let (ctx, _dir) = ctx_with_catalog(); + // The row exists: its key is bound, as the insert that created it + // bound it. + let bound = ctx + .surrogate_assigner + .assign(ctx.collection_key("crdt_coll"), ctx.tenant_id, b"k1") + .await + .expect("bind k1"); let keys = vec![SqlValue::String("k1".to_string())]; let tasks = convert_delete( "crdt_coll", @@ -346,11 +355,51 @@ mod tests { .expect("convert_delete"); assert_eq!(tasks.len(), 1); match &tasks[0].plan { - PhysicalPlan::Crdt(CrdtOp::DocDelete { document_id, .. }) => { + PhysicalPlan::Crdt(CrdtOp::DocDelete { + document_id, + surrogate, + .. + }) => { assert_eq!(document_id, "k1"); + assert_eq!(*surrogate, Some(bound)); + } + other => panic!("expected CrdtOp::DocDelete, got {other:?}"), + } + } + + /// A key no row holds still plans its delete, so the statement answers + /// `DELETE 0`. The delete carries no surrogate and mints no binding. + #[test] + fn delete_of_an_unbound_crdt_key_plans_a_delete_matching_nothing() { + let (ctx, _dir) = ctx_with_catalog(); + let keys = vec![SqlValue::String("ghost".to_string())]; + let tasks = convert_delete( + "crdt_coll", + &EngineType::DocumentSchemaless, + &[], + &keys, + TenantId::new(0), + &ctx, + ) + .expect("convert_delete"); + assert_eq!(tasks.len(), 1); + match &tasks[0].plan { + PhysicalPlan::Crdt(CrdtOp::DocDelete { + document_id, + surrogate, + .. + }) => { + assert_eq!(document_id, "ghost"); + assert_eq!(*surrogate, None); } other => panic!("expected CrdtOp::DocDelete, got {other:?}"), } + assert_eq!( + ctx.surrogate_for_existing_pk(ctx.collection_key("crdt_coll"), b"ghost") + .expect("lookup"), + None, + "planning the delete bound no key" + ); } #[test] diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/shared.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/shared.rs index f71aa744b..7133770b5 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/shared.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/shared.rs @@ -15,7 +15,7 @@ use crate::control::planner::sql_plan_convert::value::sql_value_to_string; /// edge-bearing gate in `execute.rs`. /// /// A genuine catalog READ error propagates (misrouting a delete on a real I/O -/// fault would silently skip edge cleanup → dangling edges). An ABSENT +/// fault will silently skip edge cleanup → dangling edges). An ABSENT /// credential store or catalog, or an absent collection row (`Ok(None)`), is /// treated as non-edge-bearing (`Ok(false)`). pub(super) fn document_collection_is_edge_bearing( @@ -35,6 +35,31 @@ pub(super) fn document_collection_is_edge_bearing( .unwrap_or(false)) } +/// Returns `true` when `collection` (db-qualified by the caller) is a clone +/// that still reads through to its source: `Shadowed` or `Materializing`. +/// Its writes copy rows up one point op at a time, so a write to it keeps +/// its point form. +/// +/// A catalog read error propagates. An absent credential store, catalog row, +/// or clone source is `Ok(false)`. +pub(super) fn document_collection_is_shadowed_clone( + ctx: &ConvertContext, + collection: &str, +) -> crate::Result { + let Some(credentials) = ctx.credentials.as_ref() else { + return Ok(false); + }; + let bare = + crate::control::target_identity::naming::bare_collection_name(ctx.database_id, collection); + Ok(credentials + .catalog() + .get_collection(ctx.database_id, ctx.tenant_id.as_u64(), &bare)? + .is_some_and(|c| { + c.cloned_from.is_some() + && !matches!(c.clone_status, nodedb_types::CloneStatus::Materialized) + })) +} + /// Effective filter for a PK-pre-resolved write (shared by the columnar UPDATE /// path and the edge-bearing PK-equality DELETE path). /// @@ -73,7 +98,7 @@ pub(super) fn pk_effective_filter( /// `BulkDelete`. Thin wrapper over [`pk_effective_filter`]: serializes the /// user's `WHERE` predicate, then defers to the shared synthesis. The DELETE /// gate only calls this with a non-empty `target_keys`, so the result is NEVER -/// an empty filter (which would match ALL rows). +/// an empty filter (which will match ALL rows). pub(super) fn delete_effective_filter( filters: &[Filter], target_keys: &[SqlValue], diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs index 807b59fcb..fc316bc1d 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/update_delete/update.rs @@ -15,7 +15,9 @@ use crate::control::planner::sql_plan_convert::value::{ }; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; -use super::shared::{document_collection_is_edge_bearing, pk_effective_filter}; +use super::shared::{ + document_collection_is_edge_bearing, document_collection_is_shadowed_clone, pk_effective_filter, +}; /// Parameters for [`convert_update`], bundled to avoid an unwieldy argument /// list. Fields borrow from the caller exactly as the individual arguments @@ -122,7 +124,6 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( .collect(); let key_bytes = sql_value_to_bytes(key)?; // Content-addressed identity: keeps the surrogate the original insert assigned. - // `Surrogate::ZERO` only when no assigner is wired (test / embedded-without-catalog). let surrogate = ctx.surrogate_for_pk(collection_key, &key_bytes)?; tasks.push(PhysicalTask { tenant_id, @@ -196,7 +197,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( }]); } - // CRDT UPDATE never falls through to `DocumentOp`, which would bypass convergence. + // CRDT UPDATE never falls through to `DocumentOp`, which will bypass convergence. let is_crdt = super::super::crdt_gate::document_collection_is_crdt(ctx, collection)?; if is_crdt && target_keys.is_empty() { return Err(crate::Error::BadRequest { @@ -220,13 +221,26 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( && !target_keys.is_empty() && document_collection_is_edge_bearing(ctx, collection)?; - if edge_bearing { + // An UPDATE naming several keys is one unit: UNIQUE and BALANCED hold on + // its post-state, and it writes all its rows or none. One `PointUpdate` + // per key judges each row alone and applies each on its own, so a swap is + // refused and two rows claiming one value both land. `BulkUpdate` judges + // every matched row before the first write. A clone that reads through to + // its source copies rows up one point write at a time, so its UPDATE + // keeps one `PointUpdate` per key. + let multi_key = !is_crdt + && target_keys.len() > 1 + && !document_collection_is_shadowed_clone(ctx, collection)?; + + if edge_bearing || multi_key { // Reject `Expr` RHS to a reserved edge field: reconciliation diffs against // literal SET values only (mirrors the KV/columnar `Expr`-RHS rejection). - if let Some((field, _)) = assignments.iter().find(|(field, expr)| { - matches!(field.as_str(), "_from" | "_to" | "_type") - && !matches!(expr, SqlExpr::Literal(_)) - }) { + if edge_bearing + && let Some((field, _)) = assignments.iter().find(|(field, expr)| { + matches!(field.as_str(), "_from" | "_to" | "_type") + && !matches!(expr, SqlExpr::Literal(_)) + }) + { return Err(crate::Error::BadRequest { detail: format!( "expression updates to reserved edge fields (_from, _to, _type) \ @@ -278,10 +292,10 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_update( rls_filters: Vec::new(), }) } else { - // Read-only resolution: a task always exists (the write hook - // still runs, an unbound row_key affects 0 rows, and the clone - // CoW resolver intercepts the ZERO sentinel), but an UPDATE - // creates no row, so it must never mint a binding. + // Read-only resolution: an UPDATE creates no row, so it must + // never mint a binding. An unbound key carries `None`: it + // affects 0 rows here, and the clone CoW resolver still sees + // the task and copies the source row up first. let surrogate = ctx.surrogate_for_existing_pk(collection_key, &pk_bytes)?; PhysicalPlan::Document(DocumentOp::PointUpdate { collection: qualified_collection.clone(), @@ -364,16 +378,16 @@ mod tests { CredentialStore::open(&dir.path().join("system.redb")).expect("open credential store"); { let catalog = store.catalog(); - let mut edges = StoredCollection::new(0, "edges", "owner"); + let mut edges = StoredCollection::stamped_for_test(0, "edges", "owner"); edges.has_implicit_edges = true; catalog .put_collection(crate::types::DatabaseId::DEFAULT, &edges) .expect("put edges collection"); - let plain = StoredCollection::new(0, "plain", "owner"); + let plain = StoredCollection::stamped_for_test(0, "plain", "owner"); catalog .put_collection(crate::types::DatabaseId::DEFAULT, &plain) .expect("put plain collection"); - let mut crdt_coll = StoredCollection::new(0, "crdt_coll", "owner"); + let mut crdt_coll = StoredCollection::stamped_for_test(0, "crdt_coll", "owner"); crdt_coll.crdt = true; catalog .put_collection(crate::types::DatabaseId::DEFAULT, &crdt_coll) @@ -386,7 +400,8 @@ mod tests { array_catalog: None, credentials: Some(Arc::new(store)), wal: None, - surrogate_assigner: None, + surrogate_assigner: + crate::control::planner::sql_plan_convert::test_support::test_assigner(), cluster_enabled: false, bitemporal_retention_registry: None, max_vector_dim: 0, @@ -396,6 +411,7 @@ mod tests { shuffle_agg_num_parts: 0, broadcast_threshold_bytes: 8 * 1024 * 1024, shuffle_agg_threshold: 10_000, + prefetched: Default::default(), database_id: crate::types::DatabaseId::DEFAULT, tenant_id: crate::types::TenantId::new(0), }; diff --git a/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs b/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs index e57e34c91..f0ca30c23 100644 --- a/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs +++ b/nodedb/src/control/planner/sql_plan_convert/dml/vector_primary.rs @@ -232,10 +232,16 @@ fn write_targets( if target_keys.is_empty() { return Ok(VectorWriteTargets::Predicate(serialize_filters(filters)?)); } + let pk_all: Vec> = target_keys + .iter() + .map(|key| sql_value_to_string(key).into_bytes()) + .collect(); + // A key unbound in this database names no row, so it targets nothing. let mut surrogates: Vec = Vec::with_capacity(target_keys.len()); - for key in target_keys { - let pk_bytes = sql_value_to_string(key).into_bytes(); - surrogates.push(ctx.surrogate_for_existing_pk(collection, &pk_bytes)?); + for pk_bytes in &pk_all { + if let Some(surrogate) = ctx.surrogate_for_existing_pk(collection, pk_bytes)? { + surrogates.push(surrogate); + } } Ok(VectorWriteTargets::Surrogates(surrogates)) } @@ -360,7 +366,8 @@ mod tests { array_catalog: None, credentials: None, wal: None, - surrogate_assigner: None, + surrogate_assigner: + crate::control::planner::sql_plan_convert::test_support::test_assigner(), cluster_enabled: false, bitemporal_retention_registry: None, max_vector_dim, @@ -370,6 +377,7 @@ mod tests { shuffle_agg_num_parts: 0, broadcast_threshold_bytes: 8 * 1024 * 1024, shuffle_agg_threshold: 10_000, + prefetched: Default::default(), database_id: crate::types::DatabaseId::DEFAULT, tenant_id: crate::types::TenantId::new(0), } @@ -379,7 +387,6 @@ mod tests { let mut payload_fields = std::collections::HashMap::new(); payload_fields.insert("id".to_string(), SqlValue::String(id.to_string())); VectorPrimaryRow { - surrogate: nodedb_types::Surrogate::ZERO, vector: vec![0.0f32; dim], payload_fields, } diff --git a/nodedb/src/control/planner/sql_plan_convert/kv_counter_shape.rs b/nodedb/src/control/planner/sql_plan_convert/kv_counter_shape.rs index 4fe129c2b..8325bb1d0 100644 --- a/nodedb/src/control/planner/sql_plan_convert/kv_counter_shape.rs +++ b/nodedb/src/control/planner/sql_plan_convert/kv_counter_shape.rs @@ -32,6 +32,7 @@ pub(crate) fn kv_counter_shape( ) -> crate::Result { let catalog = OriginCatalog::new( Arc::clone(&state.credentials), + state.array_catalog.clone(), tenant_id.as_u64(), database_id, Some(Arc::clone(&state.retention_policy_registry)), diff --git a/nodedb/src/control/planner/sql_plan_convert/mod.rs b/nodedb/src/control/planner/sql_plan_convert/mod.rs index db0b41a2a..98041e6d2 100644 --- a/nodedb/src/control/planner/sql_plan_convert/mod.rs +++ b/nodedb/src/control/planner/sql_plan_convert/mod.rs @@ -19,8 +19,12 @@ pub mod output_schema_types; pub mod scan; pub mod scan_params; pub mod set_ops; +pub mod surrogate_prefetch; +#[cfg(test)] +pub(crate) mod test_support; pub mod value; pub mod visitor; pub use cache_verdict::batch_cache_eligibility; pub use convert::{ConvertContext, PlanningPurpose, convert}; +pub use surrogate_prefetch::{PrefetchedSurrogates, convert_bound}; diff --git a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs index bd1e39c4e..61abfa37d 100644 --- a/nodedb/src/control/planner/sql_plan_convert/scan/core.rs +++ b/nodedb/src/control/planner/sql_plan_convert/scan/core.rs @@ -261,21 +261,11 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_point_get( EngineType::DocumentSchemaless | EngineType::DocumentStrict => { let pk_string = sql_value_to_string(key_value); let pk_bytes = pk_string.clone().into_bytes(); - let surrogate = match ctx.surrogate_assigner.as_ref() { - Some(a) => match a.lookup(collection_key, ctx.tenant_id, &pk_bytes)? { - Some(s) => s, - None => { - // No surrogate bound in the target database yet. - // Emit a sentinel task so the clone CoW resolver can - // intercept and fetch the row from the source database. - // For non-clone databases the Data Plane looks up - // the sentinel key, finds nothing, and returns empty - // — identical behaviour to the zero-tasks path. - nodedb_types::Surrogate::ZERO - } - }, - None => nodedb_types::Surrogate::ZERO, - }; + // An unbound key carries `None`: it matches no row in this + // database, so the Data Plane returns empty. The clone CoW + // resolver still sees the task and fetches the row from the + // source database. + let surrogate = ctx.surrogate_for_existing_pk(collection_key, &pk_bytes)?; PhysicalPlan::Document(DocumentOp::PointGet { collection: qualified_collection.clone(), document_id: pk_string, @@ -318,7 +308,7 @@ pub(in crate::control::planner::sql_plan_convert) fn convert_point_get( computed_columns: Vec::new(), }) } - // Timeseries should never reach here — nodedb-sql rejects point gets. + // Timeseries must never reach here — nodedb-sql rejects point gets. EngineType::Timeseries => { return Err(crate::Error::PlanError { detail: format!( diff --git a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs index 5e9c892d0..d498969d2 100644 --- a/nodedb/src/control/planner/sql_plan_convert/set_ops.rs +++ b/nodedb/src/control/planner/sql_plan_convert/set_ops.rs @@ -23,7 +23,7 @@ pub(super) fn convert_constant_result( ) -> crate::Result> { // A constant row is one object, which cannot hold two cells under one // key. `SELECT nextval('s'), nextval('s')` legally repeats an output name; - // keying both cells by the name would collapse them to the last value. Use + // keying both cells by the name will collapse them to the last value. Use // the same unique per-column keys every response encoder derives, so each // column keeps its own cell. let cell_keys = crate::control::server::response_shape::project::cell_keys(columns); @@ -291,7 +291,7 @@ pub(super) fn convert_cte( } /// Lower `SqlPlan::Subquery` — relational post-processing over a subquery body -/// whose leaf could not absorb the outer constraints — into a coordinator- +/// whose leaf cannot absorb the outer constraints — into a coordinator- /// resolved `QueryOp::PostProcess`. /// /// The body lowers to ONE physical relation through @@ -324,7 +324,7 @@ pub(super) fn convert_subquery( // computed columns, and window specs must address the same shape — an // unqualified key resolves to NULL on every merged row, and a sort where // every key is NULL is a no-op that silently answers an ordered query in - // the body's own order. The body may sit under the `Exchange{Gather}` + // the body's own order. The body can sit under the `Exchange{Gather}` // wrapper, so the detection looks through it. let merged_doc_body = is_merged_doc_body(&child); @@ -480,7 +480,8 @@ mod tests { array_catalog: None, credentials: None, wal: None, - surrogate_assigner: None, + surrogate_assigner: + crate::control::planner::sql_plan_convert::test_support::test_assigner(), cluster_enabled: false, bitemporal_retention_registry: None, max_vector_dim: 0, @@ -490,6 +491,7 @@ mod tests { shuffle_agg_num_parts: 0, broadcast_threshold_bytes: 8 * 1024 * 1024, shuffle_agg_threshold: 10_000, + prefetched: Default::default(), database_id: crate::types::DatabaseId::DEFAULT, tenant_id: crate::types::TenantId::new(0), } @@ -616,7 +618,8 @@ mod tests { array_catalog: None, credentials: None, wal: None, - surrogate_assigner: None, + surrogate_assigner: + crate::control::planner::sql_plan_convert::test_support::test_assigner(), cluster_enabled: false, bitemporal_retention_registry: None, max_vector_dim: 0, @@ -626,6 +629,7 @@ mod tests { shuffle_agg_num_parts: 0, broadcast_threshold_bytes: 8 * 1024 * 1024, shuffle_agg_threshold: 10_000, + prefetched: Default::default(), database_id: crate::types::DatabaseId::DEFAULT, tenant_id: crate::types::TenantId::new(0), }, @@ -675,7 +679,8 @@ mod tests { array_catalog: None, credentials: None, wal: None, - surrogate_assigner: None, + surrogate_assigner: + crate::control::planner::sql_plan_convert::test_support::test_assigner(), cluster_enabled: false, bitemporal_retention_registry: None, max_vector_dim: 0, @@ -685,6 +690,7 @@ mod tests { shuffle_agg_num_parts: 0, broadcast_threshold_bytes: 8 * 1024 * 1024, shuffle_agg_threshold: 10_000, + prefetched: Default::default(), database_id: crate::types::DatabaseId::DEFAULT, tenant_id: crate::types::TenantId::new(0), }, diff --git a/nodedb/src/control/planner/sql_plan_convert/surrogate_prefetch/bind.rs b/nodedb/src/control/planner/sql_plan_convert/surrogate_prefetch/bind.rs new file mode 100644 index 000000000..0087272a9 --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/surrogate_prefetch/bind.rs @@ -0,0 +1,300 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Execution conversion: a pure planning pass per plan, with an async bind +//! step that resolves every surrogate the pass needs before its tasks leave. +//! +//! Conversion stays a synchronous, deterministic function of the plan and +//! the resolved answers, so a plan converted again has the same structure. +//! Every draw and every request to a key's home is awaited here, and none +//! blocks a runtime worker, on any runtime flavor. + +use nodedb_physical::physical_task::PhysicalTask; +use nodedb_sql::types::SqlPlan; + +use super::super::convert::{ConvertContext, convert_plan}; +use super::cache::SurrogateMisses; +use super::resolve::{prefetch_plan_surrogates, resolve_misses}; +use crate::types::TenantId; + +/// Convert `plans` to physical tasks with every surrogate they use resolved. +/// +/// The keys read off the plans resolve first, in one pass. Each plan then +/// converts in order. A pass that records a miss is converted again once the +/// miss resolves, so a key the up-front read does not derive still resolves +/// without a synchronous request. Plans convert one at a time, so a plan +/// that persists catalog state converts once, before the plans after it. +pub async fn convert_bound( + plans: &[SqlPlan], + tenant_id: TenantId, + ctx: &mut ConvertContext, +) -> crate::Result> { + ctx.prefetched = prefetch_plan_surrogates(plans, ctx).await?; + convert_resolving_misses(plans, tenant_id, ctx).await +} + +/// Convert each of `plans` with the answers `ctx` holds, resolving the misses +/// of each pass and converting that plan again until a pass records none. +pub(super) async fn convert_resolving_misses( + plans: &[SqlPlan], + tenant_id: TenantId, + ctx: &mut ConvertContext, +) -> crate::Result> { + let mut tasks = Vec::new(); + for plan in plans { + tasks.extend(convert_plan_bound(plan, tenant_id, ctx).await?); + } + Ok(tasks) +} + +async fn convert_plan_bound( + plan: &SqlPlan, + tenant_id: TenantId, + ctx: &mut ConvertContext, +) -> crate::Result> { + let mark = ctx.prefetched.fresh_mark(); + let mut resolved: Option = None; + loop { + // A pass with misses planned placeholders, so its result, an error + // included, is replaced by the next pass. + let converted = convert_plan(plan, tenant_id, ctx); + let misses = ctx.prefetched.take_misses(); + if misses.is_empty() { + return converted; + } + drop(converted); + // Each pass asks only for keys no earlier pass resolved. The same + // misses twice means the answers do not reach conversion. + if resolved.as_ref() == Some(&misses) { + return Err(crate::Error::Internal { + detail: format!( + "surrogate bind: conversion asked again for the keys of {} after they \ + resolved", + misses.collection_names() + ), + }); + } + ctx.prefetched.rewind_fresh(&mark); + resolve_misses(ctx, &misses).await?; + resolved = Some(misses); + } +} + +#[cfg(test)] +mod tests { + use std::collections::HashMap; + use std::sync::atomic::{AtomicBool, Ordering}; + use std::sync::{Arc, Mutex, RwLock}; + + use futures::FutureExt; + use futures::future::BoxFuture; + use nodedb_physical::physical_plan::{KvOp, TimeseriesOp}; + use nodedb_physical::physical_task::PhysicalTask; + use nodedb_sql::types::{EngineType, SqlExpr, SqlPlan, SqlValue}; + use nodedb_types::{CollectionKey, Surrogate}; + + use super::{convert_bound, convert_resolving_misses}; + use crate::bridge::envelope::PhysicalPlan; + use crate::control::planner::sql_plan_convert::{ConvertContext, PlanningPurpose}; + use crate::control::security::credential::CredentialStore; + use crate::control::surrogate::registry::SurrogateRegistry; + use crate::control::surrogate::wal_appender::{NoopWalAppender, SurrogateWalAppender}; + use crate::control::surrogate::{HomeSurrogateAuthority, SurrogateAssigner}; + use crate::types::{DatabaseId, TenantId}; + + const TENANT: TenantId = TenantId::new(1); + + /// A cluster-mode assigner whose reserved batch is empty. + fn cluster_assigner() -> Arc { + let credentials = Arc::new(CredentialStore::new().expect("in-memory credential store")); + let registry = Arc::new(RwLock::new(SurrogateRegistry::from_persisted_cluster(0, 0))); + let wal: Arc = Arc::new(NoopWalAppender); + Arc::new(SurrogateAssigner::new(registry, credentials, wal)) + } + + fn cluster_ctx(assigner: &Arc) -> ConvertContext { + ConvertContext { + purpose: PlanningPurpose::Execute, + retention_registry: None, + array_catalog: None, + credentials: None, + wal: None, + surrogate_assigner: Arc::clone(assigner), + cluster_enabled: true, + bitemporal_retention_registry: None, + max_vector_dim: 0, + database_id: DatabaseId::DEFAULT, + tenant_id: TENANT, + force_shuffle_join: false, + shuffle_num_parts: 0, + force_shuffle_agg: false, + shuffle_agg_num_parts: 0, + broadcast_threshold_bytes: 0, + shuffle_agg_threshold: 0, + prefetched: Default::default(), + } + } + + fn timeseries_ingest(rows: usize) -> Vec { + vec![SqlPlan::TimeseriesIngest { + collection: "metrics".to_string(), + rows: vec![vec![("value".to_string(), SqlValue::Int(1))]; rows], + volatile_defaults: false, + }] + } + + fn ingest_surrogates(tasks: &[PhysicalTask]) -> Vec { + match &tasks[0].plan { + PhysicalPlan::Timeseries(TimeseriesOp::Ingest { surrogates, .. }) => { + surrogates.iter().map(|s| s.as_u32()).collect() + } + other => panic!("expected a timeseries ingest, got {other:?}"), + } + } + + /// Install the reserved batch `[100, 116)` once the conversion is parked + /// on the empty one. Fails when the conversion finished without it. + async fn install_batch(assigner: &SurrogateAssigner, converted: &AtomicBool) { + for _ in 0..8 { + tokio::task::yield_now().await; + } + assert!( + !converted.load(Ordering::SeqCst), + "conversion finished with no batch installed" + ); + let _waiter = assigner.await_reservation_for_test(1); + assigner.complete_reservation(1, 100, 116); + } + + /// A cluster conversion on a current-thread runtime awaits the refill of + /// an empty batch: its one worker stays free to install the batch, and + /// the rows draw from it. + #[tokio::test(flavor = "current_thread")] + async fn current_thread_cluster_conversion_awaits_an_empty_batch() { + let assigner = cluster_assigner(); + let mut ctx = cluster_ctx(&assigner); + let plans = timeseries_ingest(3); + let converted = AtomicBool::new(false); + + let conversion = async { + let tasks = convert_bound(&plans, TENANT, &mut ctx).await; + converted.store(true, Ordering::SeqCst); + tasks + }; + let (tasks, ()) = tokio::join!(conversion, install_batch(&assigner, &converted)); + + let surrogates = ingest_surrogates(&tasks.expect("conversion")); + assert_eq!(surrogates.len(), 3); + assert!(surrogates.iter().all(|s| (100..116).contains(s))); + let mut distinct = surrogates.clone(); + distinct.dedup(); + assert_eq!(distinct, surrogates, "each row draws its own surrogate"); + } + + /// A draw the up-front key read does not cover is recorded by the pass, + /// awaited, and the plan converts again with the drawn values. + #[tokio::test(flavor = "current_thread")] + async fn a_missed_fresh_draw_is_awaited_and_the_plan_converts_again() { + let assigner = cluster_assigner(); + let mut ctx = cluster_ctx(&assigner); + let plans = timeseries_ingest(2); + let converted = AtomicBool::new(false); + + let conversion = async { + let tasks = convert_resolving_misses(&plans, TENANT, &mut ctx).await; + converted.store(true, Ordering::SeqCst); + tasks + }; + let (tasks, ()) = tokio::join!(conversion, install_batch(&assigner, &converted)); + + let surrogates = ingest_surrogates(&tasks.expect("conversion")); + assert_eq!(surrogates.len(), 2); + assert!(surrogates.iter().all(|s| (100..116).contains(s))); + assert_ne!(surrogates[0], surrogates[1]); + assert!(ctx.prefetched.take_misses().is_empty()); + } + + /// A home that binds each new key to the next value from 500, after a + /// real await point. + #[derive(Default)] + struct CountingHome { + bound: Mutex, Surrogate>>, + } + + impl HomeSurrogateAuthority for CountingHome { + fn assign<'a>( + &'a self, + _key: CollectionKey<'a>, + _tenant_id: nodedb_types::TenantId, + pks: &'a [&'a [u8]], + ) -> BoxFuture<'a, crate::Result>> { + async move { + tokio::task::yield_now().await; + let mut bound = self.bound.lock().unwrap_or_else(|p| p.into_inner()); + let mut answers = Vec::with_capacity(pks.len()); + for pk in pks { + let next = Surrogate::new(500 + bound.len() as u32); + answers.push(*bound.entry(pk.to_vec()).or_insert(next)); + } + Ok(answers) + } + .boxed() + } + + fn lookup_many<'a>( + &'a self, + _key: CollectionKey<'a>, + _tenant_id: nodedb_types::TenantId, + pks: &'a [&'a [u8]], + ) -> BoxFuture<'a, crate::Result>>> { + async move { + tokio::task::yield_now().await; + let bound = self.bound.lock().unwrap_or_else(|p| p.into_inner()); + Ok(pks.iter().map(|pk| bound.get(*pk).copied()).collect()) + } + .boxed() + } + } + + /// A keyed write whose keys the up-front read misses asks the keys' home + /// through an awaited request on a current-thread runtime, and plans with + /// the home's values. + #[tokio::test(flavor = "current_thread")] + async fn missed_keys_are_bound_at_their_home_without_blocking() { + let assigner = cluster_assigner(); + assigner.install_home_authority(Arc::new(CountingHome::default())); + let mut ctx = cluster_ctx(&assigner); + let plans = vec![SqlPlan::Update { + collection: "cache".to_string(), + engine: EngineType::KeyValue, + assignments: vec![("v".to_string(), SqlExpr::Literal(SqlValue::Int(1)))], + filters: Vec::new(), + target_keys: vec![ + SqlValue::String("k1".to_string()), + SqlValue::String("k2".to_string()), + ], + returning: false, + }]; + + let tasks = convert_resolving_misses(&plans, TENANT, &mut ctx) + .await + .expect("conversion"); + + let cache = CollectionKey::from_bare(DatabaseId::DEFAULT, "cache"); + let mut planned = Vec::new(); + for task in &tasks { + match &task.plan { + PhysicalPlan::Kv(KvOp::FieldSet { key, surrogate, .. }) => { + assert_eq!( + assigner.lookup_bound(cache, TENANT, key).expect("lookup"), + Some(*surrogate), + "the home's value is kept in this node's catalog" + ); + planned.push(surrogate.as_u32()); + } + other => panic!("expected a KV field set, got {other:?}"), + } + } + planned.sort_unstable(); + assert_eq!(planned, vec![500, 501]); + } +} diff --git a/nodedb/src/control/planner/sql_plan_convert/surrogate_prefetch/cache.rs b/nodedb/src/control/planner/sql_plan_convert/surrogate_prefetch/cache.rs new file mode 100644 index 000000000..f5c6df2bc --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/surrogate_prefetch/cache.rs @@ -0,0 +1,289 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The surrogate answers one plan batch's bind step resolved, and the keys a +//! conversion pass asked for without an answer. +//! +//! Conversion never draws a surrogate or asks a key's home. It reads the +//! answers held here. A key with no answer is recorded as a miss, and the +//! pass goes on with a placeholder. The bind step then resolves the misses, +//! awaiting each draw and home request, and converts the plan again. + +use std::collections::{HashMap, HashSet}; +use std::sync::Mutex; + +use nodedb_types::{CollectionKey, DatabaseId, Surrogate}; + +/// One collection: its database and bare catalog name. +type CollectionSlot = (DatabaseId, String); + +/// A fresh identity drawn ahead: its surrogate and its identity string. +type FreshIdentity = (Surrogate, String); + +/// The bind step's answer for each `(collection, primary key)`: the bound +/// surrogate, or `None` when the key's home binds none. Also the fresh +/// identities drawn ahead for rows that name no key, and the misses of the +/// running conversion pass. All hold for the one plan batch they belong to. +#[derive(Debug, Default)] +pub struct PrefetchedSurrogates { + answers: HashMap<(CollectionSlot, Vec), Option>, + /// Conversion reads through a shared reference, so the fresh pool and + /// the misses sit behind locks. + fresh: Mutex, + misses: Mutex, +} + +/// Fresh identities per collection, each already bound under its identity, +/// handed out in draw order. +#[derive(Debug, Default)] +struct FreshPool { + drawn: HashMap>, + /// How many of each collection's identities conversion has taken. + taken: HashMap, +} + +/// How far conversion has taken each collection's fresh identities. A pass +/// that is converted again rewinds to the mark taken before it. +#[derive(Debug, Clone, Default)] +pub struct FreshMark(HashMap); + +/// The keys one conversion pass asked for without an answer, per collection. +#[derive(Debug, Default, PartialEq, Eq)] +pub struct SurrogateMisses { + collections: HashMap, +} + +/// The misses of one collection. +#[derive(Debug, Default, PartialEq, Eq)] +pub struct CollectionMisses { + /// Keys a write binds when the key has no surrogate. + pub binds: HashSet>, + /// Keys a read or a key-preserving write only looks up at their home. + pub lookups: HashSet>, + /// Fresh identities asked for beyond those drawn. + pub fresh: usize, +} + +impl SurrogateMisses { + pub fn is_empty(&self) -> bool { + self.collections.is_empty() + } + + /// How many answers the misses ask for: each key, and each fresh + /// identity. + pub fn len(&self) -> usize { + self.collections + .values() + .map(|m| m.binds.len() + m.lookups.len() + m.fresh) + .sum() + } + + /// Each collection's key and misses. + pub fn iter(&self) -> impl Iterator, &CollectionMisses)> { + self.collections + .iter() + .map(|((database_id, name), misses)| { + (CollectionKey::from_bare(*database_id, name), misses) + }) + } + + /// The names of the collections with misses, sorted, for an error + /// message. + pub fn collection_names(&self) -> String { + let mut names: Vec<&str> = self + .collections + .keys() + .map(|(_, name)| name.as_str()) + .collect(); + names.sort_unstable(); + names.join(", ") + } + + fn collection(&mut self, key: CollectionKey<'_>) -> &mut CollectionMisses { + self.collections.entry(slot_of(key)).or_default() + } +} + +impl PrefetchedSurrogates { + /// Keep the answer for `pk` in `key`. + pub fn insert(&mut self, key: CollectionKey<'_>, pk: &[u8], answer: Option) { + self.answers.insert((slot_of(key), pk.to_vec()), answer); + } + + /// The answer for `pk` in `key`. `None` when the key has no answer. + pub fn get(&self, key: CollectionKey<'_>, pk: &[u8]) -> Option> { + if self.answers.is_empty() { + return None; + } + self.answers.get(&(slot_of(key), pk.to_vec())).copied() + } + + /// The bound surrogate of `pk` in `key`, when the answer binds one. + pub fn bound(&self, key: CollectionKey<'_>, pk: &[u8]) -> Option { + self.get(key, pk).flatten() + } + + /// Keep a fresh identity drawn for a row of `key` that names no key. + pub fn push_fresh(&mut self, key: CollectionKey<'_>, fresh: FreshIdentity) { + self.fresh + .get_mut() + .unwrap_or_else(|p| p.into_inner()) + .drawn + .entry(slot_of(key)) + .or_default() + .push(fresh); + } + + /// Take the next fresh identity drawn for `key`. `None` once the drawn + /// identities run out. + pub fn take_fresh(&self, key: CollectionKey<'_>) -> Option { + let mut pool = self.fresh.lock().unwrap_or_else(|p| p.into_inner()); + if pool.drawn.is_empty() { + return None; + } + let slot = slot_of(key); + let taken = pool.taken.get(&slot).copied().unwrap_or(0); + let fresh = pool.drawn.get(&slot)?.get(taken)?.clone(); + pool.taken.insert(slot, taken + 1); + Some(fresh) + } + + /// Where conversion has taken each collection's fresh identities up to. + pub fn fresh_mark(&self) -> FreshMark { + let pool = self.fresh.lock().unwrap_or_else(|p| p.into_inner()); + FreshMark(pool.taken.clone()) + } + + /// Hand the fresh identities taken since `mark` out again, to a pass + /// that converts the same plan again. + pub fn rewind_fresh(&self, mark: &FreshMark) { + let mut pool = self.fresh.lock().unwrap_or_else(|p| p.into_inner()); + pool.taken = mark.0.clone(); + } + + /// Record that a write asked to bind `pk` in `key` and found no answer. + pub fn record_bind_miss(&self, key: CollectionKey<'_>, pk: &[u8]) { + self.lock_misses().collection(key).binds.insert(pk.to_vec()); + } + + /// Record that a read asked for the binding of `pk` in `key` and found + /// no answer. + pub fn record_lookup_miss(&self, key: CollectionKey<'_>, pk: &[u8]) { + self.lock_misses() + .collection(key) + .lookups + .insert(pk.to_vec()); + } + + /// Record that a row of `key` asked for a fresh identity beyond those + /// drawn. + pub fn record_fresh_miss(&self, key: CollectionKey<'_>) { + self.lock_misses().collection(key).fresh += 1; + } + + /// Take the misses recorded since the last take. + pub fn take_misses(&self) -> SurrogateMisses { + std::mem::take(&mut *self.lock_misses()) + } + + fn lock_misses(&self) -> std::sync::MutexGuard<'_, SurrogateMisses> { + self.misses.lock().unwrap_or_else(|p| p.into_inner()) + } +} + +fn slot_of(key: CollectionKey<'_>) -> CollectionSlot { + (key.database_id(), key.name().to_string()) +} + +#[cfg(test)] +mod tests { + use nodedb_types::{CollectionKey, DatabaseId, Surrogate}; + + use super::PrefetchedSurrogates; + + #[test] + fn answers_are_kept_per_collection_and_key() { + let users = CollectionKey::from_bare(DatabaseId::DEFAULT, "users"); + let orders = CollectionKey::from_bare(DatabaseId::DEFAULT, "orders"); + let mut prefetched = PrefetchedSurrogates::default(); + prefetched.insert(users, b"alice", Some(Surrogate::new(7))); + prefetched.insert(users, b"bob", None); + + assert_eq!( + prefetched.get(users, b"alice"), + Some(Some(Surrogate::new(7))) + ); + assert_eq!(prefetched.bound(users, b"alice"), Some(Surrogate::new(7))); + assert_eq!(prefetched.get(users, b"bob"), Some(None)); + assert_eq!(prefetched.bound(users, b"bob"), None); + assert_eq!(prefetched.get(users, b"carol"), None); + assert_eq!(prefetched.get(orders, b"alice"), None); + } + + #[test] + fn fresh_identities_are_taken_in_draw_order_per_collection() { + let users = CollectionKey::from_bare(DatabaseId::DEFAULT, "users"); + let orders = CollectionKey::from_bare(DatabaseId::DEFAULT, "orders"); + let mut prefetched = PrefetchedSurrogates::default(); + prefetched.push_fresh(users, (Surrogate::new(3), "3".to_string())); + prefetched.push_fresh(users, (Surrogate::new(4), "4".to_string())); + + assert_eq!(prefetched.take_fresh(orders), None); + assert_eq!( + prefetched.take_fresh(users), + Some((Surrogate::new(3), "3".to_string())) + ); + assert_eq!( + prefetched.take_fresh(users), + Some((Surrogate::new(4), "4".to_string())) + ); + assert_eq!(prefetched.take_fresh(users), None); + } + + #[test] + fn a_rewound_pass_takes_the_same_fresh_identities_again() { + let users = CollectionKey::from_bare(DatabaseId::DEFAULT, "users"); + let mut prefetched = PrefetchedSurrogates::default(); + prefetched.push_fresh(users, (Surrogate::new(3), "3".to_string())); + prefetched.push_fresh(users, (Surrogate::new(4), "4".to_string())); + assert_eq!( + prefetched.take_fresh(users).map(|f| f.0), + Some(Surrogate::new(3)) + ); + + let mark = prefetched.fresh_mark(); + assert_eq!( + prefetched.take_fresh(users).map(|f| f.0), + Some(Surrogate::new(4)) + ); + prefetched.rewind_fresh(&mark); + assert_eq!( + prefetched.take_fresh(users).map(|f| f.0), + Some(Surrogate::new(4)) + ); + } + + #[test] + fn misses_are_kept_per_collection_until_taken() { + let users = CollectionKey::from_bare(DatabaseId::DEFAULT, "users"); + let orders = CollectionKey::from_bare(DatabaseId::DEFAULT, "orders"); + let prefetched = PrefetchedSurrogates::default(); + prefetched.record_bind_miss(users, b"a"); + prefetched.record_bind_miss(users, b"a"); + prefetched.record_lookup_miss(orders, b"o"); + prefetched.record_fresh_miss(users); + prefetched.record_fresh_miss(users); + + let misses = prefetched.take_misses(); + assert_eq!(misses.len(), 4); + assert_eq!(misses.collection_names(), "orders, users"); + for (key, recorded) in misses.iter() { + if key.name() == "users" { + assert_eq!(recorded.binds.len(), 1); + assert_eq!(recorded.fresh, 2); + } else { + assert!(recorded.lookups.contains(b"o".as_slice())); + } + } + assert!(prefetched.take_misses().is_empty()); + } +} diff --git a/nodedb/src/control/planner/sql_plan_convert/surrogate_prefetch/mod.rs b/nodedb/src/control/planner/sql_plan_convert/surrogate_prefetch/mod.rs new file mode 100644 index 000000000..e848b1e76 --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/surrogate_prefetch/mod.rs @@ -0,0 +1,11 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Resolve a batch of plans' surrogate keys, async, around the synchronous +//! conversion that reads the answers. + +pub mod bind; +pub mod cache; +pub mod resolve; + +pub use bind::convert_bound; +pub use cache::PrefetchedSurrogates; diff --git a/nodedb/src/control/planner/sql_plan_convert/surrogate_prefetch/resolve.rs b/nodedb/src/control/planner/sql_plan_convert/surrogate_prefetch/resolve.rs new file mode 100644 index 000000000..257ddad4a --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/surrogate_prefetch/resolve.rs @@ -0,0 +1,154 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The async resolution of a plan batch's surrogate keys: the keys read off +//! the plans before conversion, and the misses a conversion pass records. + +use std::sync::Arc; + +use nodedb_sql::types::SqlPlan; +use nodedb_types::CollectionKey; + +use super::super::convert::ConvertContext; +use super::super::dml::plan_key_batches; +use super::cache::{PrefetchedSurrogates, SurrogateMisses}; +use crate::control::surrogate::SurrogateAssigner; +use crate::types::TenantId; + +/// Resolve every surrogate key `plans` convert with before conversion runs, +/// one batch per plan. +/// +/// A write that creates its rows binds each absent key: at the key's +/// collection home in a cluster, in this node's catalog on a single node. A +/// metadata plan, and a write that only reads its rows, never binds. Each +/// answer is kept in the returned set, and conversion answers these keys +/// without a request. +/// +/// A row that names no key takes a fresh identity, drawn here too. An empty +/// cluster reservation is awaited from the refill loop, never blocked on. A +/// statement that then fails leaves its drawn identities unused. +pub async fn prefetch_plan_surrogates( + plans: &[SqlPlan], + ctx: &ConvertContext, +) -> crate::Result { + let resolver = Resolver::new(ctx)?; + let mut prefetched = PrefetchedSurrogates::default(); + for batch in plan_key_batches(plans, ctx) { + let pks: Vec<&[u8]> = batch.pks.iter().map(Vec::as_slice).collect(); + let (binds, lookups) = if batch.binds { + (pks, Vec::new()) + } else { + (Vec::new(), pks) + }; + let request = KeyRequest { + key: ctx.collection_key(batch.collection), + binds: &binds, + lookups: &lookups, + fresh: batch.fresh, + }; + resolver.resolve(request, &mut prefetched).await?; + } + Ok(prefetched) +} + +/// Resolve the misses one conversion pass recorded into `ctx`'s answers. +pub(super) async fn resolve_misses( + ctx: &mut ConvertContext, + misses: &SurrogateMisses, +) -> crate::Result<()> { + let resolver = Resolver::new(ctx)?; + for (key, recorded) in misses.iter() { + let binds: Vec<&[u8]> = recorded.binds.iter().map(Vec::as_slice).collect(); + let lookups: Vec<&[u8]> = recorded.lookups.iter().map(Vec::as_slice).collect(); + let request = KeyRequest { + key, + binds: &binds, + lookups: &lookups, + fresh: recorded.fresh, + }; + resolver.resolve(request, &mut ctx.prefetched).await?; + } + Ok(()) +} + +/// The keys of one collection to resolve. +struct KeyRequest<'r> { + key: CollectionKey<'r>, + /// Keys bound when absent. + binds: &'r [&'r [u8]], + /// Keys only looked up. + lookups: &'r [&'r [u8]], + /// Fresh identities to draw. + fresh: usize, +} + +/// Awaits each draw and home request a batch's keys need. +struct Resolver { + assigner: Arc, + tenant_id: TenantId, + /// A metadata plan never binds and draws nothing. + metadata: bool, + /// Whether keys resolve at a collection home: a cluster node. A single + /// node's catalog answers every lookup, so conversion reads it directly. + at_home: bool, +} + +impl Resolver { + fn new(ctx: &ConvertContext) -> crate::Result { + let assigner = Arc::clone(&ctx.surrogate_assigner); + let at_home = assigner.resolves_at_home()?; + Ok(Self { + assigner, + tenant_id: ctx.tenant_id, + metadata: ctx.is_metadata(), + at_home, + }) + } + + async fn resolve( + &self, + request: KeyRequest<'_>, + into: &mut PrefetchedSurrogates, + ) -> crate::Result<()> { + let KeyRequest { + key, + binds, + lookups, + fresh, + } = request; + if self.metadata { + // A metadata plan binds nothing: its binding keys are lookups. + self.look_up(key, binds, into).await?; + return self.look_up(key, lookups, into).await; + } + for _ in 0..fresh { + let drawn = self.assigner.assign_fresh(key, self.tenant_id).await?; + into.push_fresh(key, drawn); + } + if !binds.is_empty() { + let bound = self + .assigner + .assign_many(key, self.tenant_id, binds) + .await?; + for (pk, surrogate) in binds.iter().zip(bound) { + into.insert(key, pk, Some(surrogate)); + } + } + self.look_up(key, lookups, into).await + } + + async fn look_up( + &self, + key: CollectionKey<'_>, + pks: &[&[u8]], + into: &mut PrefetchedSurrogates, + ) -> crate::Result<()> { + if pks.is_empty() || !self.at_home { + return Ok(()); + } + let found = self.assigner.lookup_many(key, self.tenant_id, pks).await?; + for (pk, answer) in pks.iter().zip(found) { + into.insert(key, pk, answer); + } + Ok(()) + } +} diff --git a/nodedb/src/control/planner/sql_plan_convert/test_support.rs b/nodedb/src/control/planner/sql_plan_convert/test_support.rs new file mode 100644 index 000000000..d1876cbed --- /dev/null +++ b/nodedb/src/control/planner/sql_plan_convert/test_support.rs @@ -0,0 +1,19 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A real surrogate assigner for converter unit tests: a single-node registry +//! over an in-memory credential store. + +use std::sync::{Arc, RwLock}; + +use crate::control::security::credential::CredentialStore; +use crate::control::surrogate::SurrogateAssigner; +use crate::control::surrogate::registry::SurrogateRegistry; +use crate::control::surrogate::wal_appender::{NoopWalAppender, SurrogateWalAppender}; + +/// A single-node assigner that binds every key in its own in-memory catalog. +pub(crate) fn test_assigner() -> Arc { + let credentials = Arc::new(CredentialStore::new().expect("in-memory credential store")); + let registry = Arc::new(RwLock::new(SurrogateRegistry::new())); + let wal: Arc = Arc::new(NoopWalAppender); + Arc::new(SurrogateAssigner::new(registry, credentials, wal)) +} diff --git a/nodedb/src/control/planner/sql_plan_convert/value/convert.rs b/nodedb/src/control/planner/sql_plan_convert/value/convert.rs index 186d79286..b803f5726 100644 --- a/nodedb/src/control/planner/sql_plan_convert/value/convert.rs +++ b/nodedb/src/control/planner/sql_plan_convert/value/convert.rs @@ -31,7 +31,7 @@ pub(crate) fn sql_value_to_string(v: &SqlValue) -> String { SqlValue::Decimal(d) => d.to_string(), SqlValue::Bool(b) => b.to_string(), SqlValue::Timestamp(dt) | SqlValue::Timestamptz(dt) => dt.to_iso8601(), - SqlValue::Bytes(b) => format!("\\x{}", hex_encode(b)), + SqlValue::Bytes(b) => format!("\\x{}", hex::encode(b)), SqlValue::Array(arr) => format_pg_array(arr), SqlValue::Null => String::new(), } @@ -60,14 +60,6 @@ fn pg_array_string_needs_quotes(value: &str) -> bool { .any(|c| c.is_whitespace() || matches!(c, ',' | '{' | '}' | '"' | '\\')) } -fn hex_encode(bytes: &[u8]) -> String { - let mut s = String::with_capacity(bytes.len() * 2); - for b in bytes { - s.push_str(&format!("{b:02x}")); - } - s -} - /// Raw bytes for a KV key or a single-`value` column body. /// /// A scalar encodes through `nodedb_types::scalar_to_raw_bytes`, the same @@ -105,7 +97,7 @@ mod tests { /// The read-side stringifier (`sql_value_to_string`, which the index-range /// read-set capture uses) and the write-side index-key stringifier /// (`json_scalar_to_string`) MUST agree on the canonical string for every - /// scalar type — otherwise a captured `IndexEq` value would never match the + /// scalar type — otherwise a captured `IndexEq` value will never match the /// index key a write records. This parity is the load-bearing guarantee the /// per-value comparison (a later change) will rest on. #[test] diff --git a/nodedb/src/control/planner/wasm/runtime.rs b/nodedb/src/control/planner/wasm/runtime.rs index 41cf13035..272922516 100644 --- a/nodedb/src/control/planner/wasm/runtime.rs +++ b/nodedb/src/control/planner/wasm/runtime.rs @@ -102,11 +102,7 @@ pub fn sha256(data: &[u8]) -> [u8; 32] { /// Format a SHA-256 hash as a hex string. pub fn sha256_hex(data: &[u8]) -> String { - hex_encode(&sha256(data)) -} - -fn hex_encode(bytes: &[u8]) -> String { - bytes.iter().map(|b| format!("{b:02x}")).collect() + hex::encode(sha256(data)) } #[cfg(test)] diff --git a/nodedb/src/control/propose_outcome.rs b/nodedb/src/control/propose_outcome.rs index 1afb629b6..db687f002 100644 --- a/nodedb/src/control/propose_outcome.rs +++ b/nodedb/src/control/propose_outcome.rs @@ -2,10 +2,9 @@ //! What happened to a `CatalogEntry` handed to the metadata proposer. //! -//! These three outcomes demand three different follow-ups from the caller and -//! must never be conflated: a bare log index cannot distinguish "nothing was -//! replicated, write the catalog yourself" from "held for COMMIT, touch -//! nothing", and doing the former for the latter durably leaks rolled-back DDL. +//! A caller never writes the catalog itself. A replicated entry already ran +//! both post-apply lanes on this node. A buffered one must touch nothing: +//! applying it durably leaks rolled-back DDL. /// Result of proposing one `CatalogEntry`. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -14,39 +13,33 @@ pub enum ProposeOutcome { /// log index. The applier has already written the catalog. Replicated { log_index: u64 }, /// Captured by the connection's DDL transaction buffer. Nothing is durable - /// yet and nothing may be applied: COMMIT proposes the whole batch, + /// yet and nothing is applied: COMMIT proposes the whole batch, /// ROLLBACK discards it. Buffered, - /// Nothing was replicated — no metadata raft group (single node) or - /// rolling-upgrade compat mode. The caller owns the catalog write and any - /// local side effects. - LocalOnly, } impl ProposeOutcome { - /// True when the caller must write the catalog and run its local side - /// effects itself. False for both `Replicated` and `Buffered`. - pub fn needs_local_apply(self) -> bool { - matches!(self, Self::LocalOnly) - } - /// True when a raft applier has already landed the entry on this node. pub fn is_replicated(self) -> bool { matches!(self, Self::Replicated { .. }) } - /// True when the entry is held for COMMIT and no side effect may run yet. + /// True when the entry is held for COMMIT and no side effect runs yet. pub fn is_buffered(self) -> bool { matches!(self, Self::Buffered) } - /// The replicated log index, or 0 when nothing was replicated. Use only - /// for logging and for wire fields that already carry 0 as "not - /// replicated"; never to decide whether to apply locally. + /// True when the entry landed on this node. + pub fn is_durable(self) -> bool { + !self.is_buffered() + } + + /// The replicated log index, or 0 for a buffered entry. Use only for + /// logging and for wire fields that already carry 0 as "not replicated". pub fn log_index(self) -> u64 { match self { Self::Replicated { log_index } => log_index, - Self::Buffered | Self::LocalOnly => 0, + Self::Buffered => 0, } } } @@ -56,32 +49,14 @@ mod tests { use super::*; #[test] - fn only_local_only_applies_locally() { - assert!(ProposeOutcome::LocalOnly.needs_local_apply()); - assert!(!ProposeOutcome::Buffered.needs_local_apply()); - assert!(!ProposeOutcome::Replicated { log_index: 7 }.needs_local_apply()); + fn only_buffered_is_not_durable() { + assert!(!ProposeOutcome::Buffered.is_durable()); + assert!(ProposeOutcome::Replicated { log_index: 7 }.is_durable()); } #[test] - fn log_index_is_zero_for_non_replicated() { + fn log_index_is_zero_for_buffered() { assert_eq!(ProposeOutcome::Replicated { log_index: 7 }.log_index(), 7); assert_eq!(ProposeOutcome::Buffered.log_index(), 0); - assert_eq!(ProposeOutcome::LocalOnly.log_index(), 0); - } - - #[test] - fn predicates_are_mutually_exclusive() { - for outcome in [ - ProposeOutcome::Replicated { log_index: 1 }, - ProposeOutcome::Buffered, - ProposeOutcome::LocalOnly, - ] { - let set = [ - outcome.is_replicated(), - outcome.is_buffered(), - outcome.needs_local_apply(), - ]; - assert_eq!(set.iter().filter(|flag| **flag).count(), 1); - } } } diff --git a/nodedb/src/control/request_tracker.rs b/nodedb/src/control/request_tracker.rs index 6fac98f4c..7fbafd085 100644 --- a/nodedb/src/control/request_tracker.rs +++ b/nodedb/src/control/request_tracker.rs @@ -1,7 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 use std::collections::HashMap; -use std::sync::{Mutex, MutexGuard}; +use std::sync::{Arc, Mutex, MutexGuard, Weak}; use tokio::sync::{mpsc, oneshot}; @@ -22,13 +22,41 @@ struct PendingRequest { final_tx: oneshot::Sender, } +type PendingMap = Mutex>; + /// The receiving end of one tracked request. /// /// Partial responses arrive through a bounded channel. The final response /// has a slot of its own, so a full partial channel never drops it. +/// +/// Dropped before its final response, the receiver removes its tracker +/// entry: nobody reads the rest of that request's answer. pub struct ResponseReceiver { partials: mpsc::Receiver, final_rx: Option>, + /// The tracker entry this receiver reads. `None` for a receiver no + /// tracker registered. + entry: Option<(Weak, RequestId)>, +} + +impl Drop for ResponseReceiver { + fn drop(&mut self) { + if self.final_rx.is_none() { + return; + } + if let Some((pending, request_id)) = &self.entry + && let Some(pending) = pending.upgrade() + { + lock(&pending).remove(request_id); + } + } +} + +fn lock(pending: &PendingMap) -> MutexGuard<'_, HashMap> { + match pending.lock() { + Ok(guard) => guard, + Err(poisoned) => poisoned.into_inner(), + } } impl ResponseReceiver { @@ -76,6 +104,7 @@ impl ResponseReceiver { Self { partials, final_rx: None, + entry: None, } } } @@ -92,21 +121,16 @@ impl ResponseReceiver { /// request's final slot and request removed from the map. #[derive(Default)] pub struct RequestTracker { - pending: Mutex>, + pending: Arc, } impl RequestTracker { pub fn new() -> Self { - Self { - pending: Mutex::new(HashMap::new()), - } + Self::default() } fn lock_pending(&self) -> MutexGuard<'_, HashMap> { - match self.pending.lock() { - Ok(guard) => guard, - Err(poisoned) => poisoned.into_inner(), - } + lock(&self.pending) } /// Register a pending request. Returns the receiver the session awaits. @@ -127,6 +151,7 @@ impl RequestTracker { ResponseReceiver { partials: partials_rx, final_rx: Some(final_rx), + entry: Some((Arc::downgrade(&self.pending), id)), } } @@ -315,4 +340,37 @@ mod tests { let last = rx.recv().await.expect("final response"); assert_eq!(last.request_id, RequestId::new(12)); } + + /// A receiver dropped before its final response removes its entry, so + /// no entry outlives its reader. + #[test] + fn a_receiver_dropped_before_its_final_removes_its_entry() { + let tracker = RequestTracker::new(); + let rx = tracker.register(RequestId::new(13)); + assert!(tracker.complete(Response { + partial: true, + ..make_response(13) + })); + assert_eq!(tracker.in_flight(), 1); + + drop(rx); + + assert_eq!(tracker.in_flight(), 0); + assert!(!tracker.complete(make_response(13))); + } + + /// A receiver dropped after its final response leaves other entries in + /// place. + #[tokio::test] + async fn a_receiver_dropped_after_its_final_touches_no_other_entry() { + let tracker = RequestTracker::new(); + let mut done = tracker.register(RequestId::new(14)); + let _other = tracker.register(RequestId::new(15)); + assert!(tracker.complete(make_response(14))); + assert!(done.recv().await.is_some()); + + drop(done); + + assert_eq!(tracker.in_flight(), 1); + } } diff --git a/nodedb/src/control/rolling_upgrade/mod.rs b/nodedb/src/control/rolling_upgrade/mod.rs index df3eb1fdc..38bea2e4b 100644 --- a/nodedb/src/control/rolling_upgrade/mod.rs +++ b/nodedb/src/control/rolling_upgrade/mod.rs @@ -1,28 +1,21 @@ // SPDX-License-Identifier: BUSL-1.1 -//! N-1 rolling upgrade compatibility. +//! Cluster wire-version view. //! -//! The on-disk and in-flight wire format is versioned so a node -//! running release N can operate in a cluster that still contains -//! nodes running release N-1. Feature flags gate on the -//! cluster-wide minimum version: a feature introduced at version -//! V only activates once every node reports `wire_version >= V`, -//! and the minimum is derived on demand from the live -//! `ClusterTopology` (see [`view::ClusterVersionView`]). +//! There is no rolling-upgrade window before 1.0: `nodedb_types::wire_version` +//! pins `MIN_WIRE_FORMAT_VERSION == WIRE_FORMAT_VERSION` (floor == ceiling), +//! and every join and wire-version handshake additionally requires exact +//! `WIRE_BUILD_ID` equality — a cluster can only ever contain nodes on one +//! build. No code path gates on a version. //! //! Layout: //! -//! - [`versions`] — wire-version constants and static compatibility -//! helpers (`accept_message`, `should_compat_mode`). +//! - [`versions`] — `should_compat_mode`. //! - [`view`] — `ClusterVersionView` plus `compute_from_topology` -//! and the feature-gate predicates. Pure functions, no shared -//! mutable state. +//! and the version predicates. Pure functions, no shared mutable state. pub mod versions; pub mod view; -pub use versions::{ - DESCRIPTOR_DRAIN_VERSION, DESCRIPTOR_VERSIONING_VERSION, DISTRIBUTED_CATALOG_VERSION, - accept_message, should_compat_mode, -}; +pub use versions::should_compat_mode; pub use view::{ClusterVersionView, compute_from_topology}; diff --git a/nodedb/src/control/rolling_upgrade/versions.rs b/nodedb/src/control/rolling_upgrade/versions.rs index 19127d92c..9a338a8b6 100644 --- a/nodedb/src/control/rolling_upgrade/versions.rs +++ b/nodedb/src/control/rolling_upgrade/versions.rs @@ -1,102 +1,18 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Wire version constants + static compatibility checks. +//! Static compatibility checks. //! //! See `view::ClusterVersionView` for the live-topology-derived -//! feature-gate predicates. +//! predicates. +//! +//! `MIN_WIRE_FORMAT_VERSION == WIRE_FORMAT_VERSION`, so a node rejects any +//! peer whose version differs and a mixed-version cluster never forms. No +//! feature gates on a version: every node of a cluster runs one build. use super::view::ClusterVersionView; -#[cfg(test)] -use crate::version::WIRE_FORMAT_VERSION; - -// PRE-1.0: every gate below is pinned to 1, the value of -// `WIRE_FORMAT_VERSION`, so each feature is unconditionally active. -// -// `MIN_WIRE_FORMAT_VERSION == WIRE_FORMAT_VERSION` (floor == ceiling), so a -// node rejects any peer whose version differs and a mixed-version cluster can -// never form. Inside a cluster that exists, every node is therefore on this -// exact version, which makes `min_version >= V` constant-true for any -// `V <= WIRE_FORMAT_VERSION` and constant-false above it. These gates cannot -// discriminate, so a value above 1 does not protect a rolling upgrade — it just -// switches the feature OFF permanently and silently routes to a legacy -// fallback. -// -// Do NOT raise these while `WIRE_FORMAT_VERSION` is 1 (see -// `nodedb_types::wire_version` for why it stays there until 1.0). The gate -// machinery is kept, not deleted, because it becomes meaningful the moment a -// real support window (`MIN_WIRE_FORMAT_VERSION < WIRE_FORMAT_VERSION`) is -// introduced post-1.0 — at which point these regain their original meanings, -// recorded below. - -/// Wire-format version that introduced the replicated catalog DDL -/// path (`CatalogEntry` proposed via the metadata raft group). -/// -/// Before this version, catalog DDL was applied directly on the -/// originating node and never replicated. Mixing the two paths in -/// a rolling upgrade window would silently diverge state across -/// nodes, so [`crate::control::metadata_proposer::propose_catalog_entry`] -/// gates on this constant via -/// [`ClusterVersionView::can_activate_feature`] and falls back to -/// the legacy direct-write path until every node in the cluster -/// has caught up. -pub const DISTRIBUTED_CATALOG_VERSION: u16 = 1; - -/// Wire-format version that introduced monotonic descriptor -/// versioning (`descriptor_version: u64` + `modification_hlc: Hlc` -/// on every `Stored*` type stamped by the metadata applier at -/// commit time). -/// -/// Before this version, `Stored*` records had no version / HLC -/// fields on the wire. In a mixed-version cluster during rolling -/// upgrade, an older applier would fail to re-stamp on -/// write-through (it has no stamp logic), so we keep the stamping -/// path disabled in compat mode and let resolvers treat -/// `descriptor_version == 0` as "unknown, always re-fetch". Once -/// every node reports `wire_version >= 3`, the applier transitions -/// to stamping. -pub const DESCRIPTOR_VERSIONING_VERSION: u16 = 1; - -/// Wire version that introduced the replicated -/// `DescriptorDrainStart` / `DescriptorDrainEnd` metadata entries. -/// Mixed-version clusters below this version skip drain via the -/// compat-mode fallback in `drain_for_ddl`. -pub const DESCRIPTOR_DRAIN_VERSION: u16 = 1; - -/// Check if a message from a remote node should be accepted. -/// -/// Accepts only messages with the exact current wire format version. -/// Any other version is rejected (floor == ceiling; no rolling-upgrade window). -pub fn accept_message(remote_version: u16) -> crate::Result<()> { - crate::version::check_wire_compatibility(remote_version) -} -/// Determine if this node should operate in compatibility mode. -/// -/// Compat mode is active when the cluster has mixed versions. In -/// compat mode, new features that require the latest version are -/// disabled. +/// Whether the cluster reports mixed versions. Observability only: a +/// cluster that formed never reports them. pub fn should_compat_mode(view: &ClusterVersionView) -> bool { view.is_mixed_version() } - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn accept_same_version() { - assert!(accept_message(WIRE_FORMAT_VERSION).is_ok()); - } - - #[test] - fn reject_newer() { - assert!(accept_message(WIRE_FORMAT_VERSION + 1).is_err()); - } - - #[test] - fn reject_older() { - if WIRE_FORMAT_VERSION > 0 { - assert!(accept_message(WIRE_FORMAT_VERSION - 1).is_err()); - } - } -} diff --git a/nodedb/src/control/router/vshard.rs b/nodedb/src/control/router/vshard.rs index 794d3c2dc..cbcd7d2cd 100644 --- a/nodedb/src/control/router/vshard.rs +++ b/nodedb/src/control/router/vshard.rs @@ -6,9 +6,8 @@ use crate::types::VShardId; /// Maps virtual shards to Data Plane core IDs. /// -/// In single-node mode, vShards are distributed round-robin across cores. -/// In cluster mode, the routing table is maintained by Raft consensus and -/// updated atomically during vShard migrations. +/// vShards are distributed round-robin across this node's cores. Which node +/// owns a vShard is the cluster routing table's concern, not this map's. pub struct VShardRouter { /// vShard -> core ID mapping. Core ID is the index into the Data Plane /// core array (0..data_plane_cores-1). @@ -17,11 +16,12 @@ pub struct VShardRouter { } impl VShardRouter { - /// Create a round-robin router for single-node mode. + /// Create a round-robin router over `num_cores` cores. pub fn round_robin(num_cores: usize) -> Self { let mut routes = HashMap::with_capacity(VShardId::COUNT as usize); for i in 0..VShardId::COUNT { - routes.insert(VShardId::new(i), i as usize % num_cores); + let vshard = VShardId::new(i); + routes.insert(vshard, crate::types::core_for_vshard(vshard, num_cores)); } Self { routes, num_cores } } diff --git a/nodedb/src/control/scatter_gather/envelope.rs b/nodedb/src/control/scatter_gather/envelope.rs deleted file mode 100644 index ac9e19705..000000000 --- a/nodedb/src/control/scatter_gather/envelope.rs +++ /dev/null @@ -1,132 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! The scatter side of a hop: what is being sent where. -//! -//! Owns the accumulation of cross-shard destinations into one vectorized -//! envelope per hop level, and the local/remote partition that decides which -//! node ids belong in it at all. Kept apart from the fan-out policy that reads -//! the envelope and from the dispatch that consumes the batches, because this -//! is pure grouping: it holds no routing decision, performs no I/O, and is the -//! only place that knows the envelope's internal shape. - -use std::collections::HashMap; - -use crate::types::VShardId; - -/// A batch of node IDs targeted at a specific shard. -/// -/// Produced by the scatter phase when graph traversal discovers nodes -/// that live on a different shard than the current core. -#[derive(Debug, Clone)] -pub struct ScatterBatch { - /// Target shard for this batch of node IDs. - pub target_shard: VShardId, - /// Node IDs that need to be explored on the target shard. - pub node_ids: Vec, -} - -/// Vectorized scatter envelope for one hop level. -/// -/// Groups all cross-shard destinations by target shard, preventing -/// scatter amplification. -#[derive(Debug, Clone, Default)] -pub struct ScatterEnvelope { - /// Batches grouped by target shard. - batches: HashMap>, -} - -impl ScatterEnvelope { - pub fn new() -> Self { - Self::default() - } - - /// Add a node ID destined for a specific shard. - pub fn add(&mut self, shard: VShardId, node_id: String) { - self.batches.entry(shard).or_default().push(node_id); - } - - /// Number of distinct shards in this envelope. - pub fn shard_count(&self) -> usize { - self.batches.len() - } - - /// Consume into scatter batches. - pub fn into_batches(self) -> Vec { - self.batches - .into_iter() - .map(|(shard, node_ids)| ScatterBatch { - target_shard: shard, - node_ids, - }) - .collect() - } - - /// Total number of node IDs across all shards. - pub fn total_nodes(&self) -> usize { - self.batches.values().map(|v| v.len()).sum() - } - - /// Check if the envelope is empty. - pub fn is_empty(&self) -> bool { - self.batches.is_empty() - } -} - -/// Partition a set of node IDs into local nodes (served by this node) and -/// a `ScatterEnvelope` grouping remote nodes by their target shard. -/// -/// "Local" means the shard's leader is `local_node_id`. Any node whose -/// `VShardId::from_key` maps to a shard led by a different node is remote. -/// -/// When `cluster_routing` is `None` (single-node mode), all nodes are -/// considered local and the envelope is empty. -pub fn partition_local_remote( - node_ids: &[String], - local_node_id: u64, - routing: &nodedb_cluster::RoutingTable, -) -> (Vec, ScatterEnvelope) { - let mut local = Vec::new(); - let mut envelope = ScatterEnvelope::new(); - - for node_id in node_ids { - let shard = VShardId::from_key(node_id.as_bytes()); - let leader = routing - .leader_for_vshard(shard.as_u32()) - .unwrap_or(local_node_id); - - if leader == local_node_id { - local.push(node_id.clone()); - } else { - envelope.add(shard, node_id.clone()); - } - } - - (local, envelope) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn scatter_envelope_grouping() { - let mut env = ScatterEnvelope::new(); - env.add(VShardId::new(0), "a".into()); - env.add(VShardId::new(0), "b".into()); - env.add(VShardId::new(1), "c".into()); - - assert_eq!(env.shard_count(), 2); - assert_eq!(env.total_nodes(), 3); - - let batches = env.into_batches(); - assert_eq!(batches.len(), 2); - } - - #[test] - fn empty_envelope() { - let env = ScatterEnvelope::new(); - assert!(env.is_empty()); - assert_eq!(env.shard_count(), 0); - assert_eq!(env.total_nodes(), 0); - } -} diff --git a/nodedb/src/control/scatter_gather/fan_out.rs b/nodedb/src/control/scatter_gather/fan_out.rs deleted file mode 100644 index 52dc62de7..000000000 --- a/nodedb/src/control/scatter_gather/fan_out.rs +++ /dev/null @@ -1,211 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Adaptive fan-out policy: how wide a single hop is allowed to scatter. -//! -//! Separate from the envelope that carries the destinations and from the hop -//! that dispatches them, because this is the one place a traversal is allowed -//! to be narrowed or refused. It answers a policy question — soft warn, hard -//! stop, or partial — from `GraphTraversalOptions` alone, touching no routing -//! table, no gateway, and no network, so the decision is testable in isolation -//! and cannot drift into the dispatch loop. - -use crate::engine::graph::traversal_options::{GraphResponseMeta, GraphTraversalOptions}; - -use super::envelope::{ScatterBatch, ScatterEnvelope}; - -/// Result of applying adaptive fan-out limits to a scatter envelope. -#[derive(Debug)] -pub enum FanOutDecision { - /// All batches can proceed. No limits hit. - Proceed { - batches: Vec, - meta: GraphResponseMeta, - }, - /// Soft limit exceeded but continuing. Response annotated with warning. - ProceedWithWarning { - batches: Vec, - meta: GraphResponseMeta, - }, - /// Hard limit exceeded. If fan_out_partial, return partial results. - /// Otherwise, return FAN_OUT_EXCEEDED error. - Exceeded { - /// Batches that were dispatched before limit was hit (for partial mode). - dispatched: Vec, - /// Batches that were skipped. - skipped: Vec, - meta: GraphResponseMeta, - }, -} - -/// Apply adaptive fan-out limits to a scatter envelope. -/// - Soft limit (default 12): query continues, response annotated with warning -/// - Hard limit (default 16): query terminates with FAN_OUT_EXCEEDED unless -/// fan_out_partial is true, in which case partial results are returned -pub fn apply_fan_out_limits( - envelope: ScatterEnvelope, - options: &GraphTraversalOptions, -) -> FanOutDecision { - let shard_count = envelope.shard_count() as u16; - - if shard_count <= options.fan_out_soft { - // Under soft limit — all clear. - FanOutDecision::Proceed { - batches: envelope.into_batches(), - meta: GraphResponseMeta { - shards_reached: shard_count, - ..Default::default() - }, - } - } else if shard_count <= options.fan_out_hard { - // Between soft and hard limit — proceed with warning. - let batches = envelope.into_batches(); - let meta = GraphResponseMeta::with_warning(shard_count, 0, options.fan_out_hard); - FanOutDecision::ProceedWithWarning { batches, meta } - } else { - // Exceeded hard limit. - let mut all_batches = envelope.into_batches(); - let hard = options.fan_out_hard as usize; - let skipped = all_batches.split_off(hard); - let skipped_count = skipped.len() as u16; - let dispatched_count = all_batches.len() as u16; - - let meta = if options.fan_out_partial { - GraphResponseMeta::with_truncation(dispatched_count, skipped_count) - } else { - GraphResponseMeta { - shards_reached: dispatched_count, - shards_skipped: skipped_count, - truncated: true, - fan_out_warning: None, - approximate: true, - } - }; - - FanOutDecision::Exceeded { - dispatched: all_batches, - skipped, - meta, - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::types::VShardId; - - #[test] - fn fan_out_under_soft_limit() { - let mut env = ScatterEnvelope::new(); - for i in 0..5u32 { - env.add(VShardId::new(i), format!("node_{i}")); - } - - let decision = apply_fan_out_limits(env, &GraphTraversalOptions::default()); - match decision { - FanOutDecision::Proceed { batches, meta } => { - assert_eq!(batches.len(), 5); - assert!(meta.is_clean()); - assert_eq!(meta.shards_reached, 5); - } - _ => panic!("expected Proceed"), - } - } - - #[test] - fn fan_out_between_soft_and_hard() { - let mut env = ScatterEnvelope::new(); - for i in 0..14u32 { - env.add(VShardId::new(i), format!("node_{i}")); - } - - let decision = apply_fan_out_limits(env, &GraphTraversalOptions::default()); - match decision { - FanOutDecision::ProceedWithWarning { batches, meta } => { - assert_eq!(batches.len(), 14); - assert!(!meta.is_clean()); - assert!(meta.approximate); - assert_eq!(meta.fan_out_warning, Some("14/16".to_string())); - } - _ => panic!("expected ProceedWithWarning"), - } - } - - #[test] - fn fan_out_exceeded_no_partial() { - let mut env = ScatterEnvelope::new(); - for i in 0..20u32 { - env.add(VShardId::new(i), format!("node_{i}")); - } - - let opts = GraphTraversalOptions { - fan_out_partial: false, - ..Default::default() - }; - let decision = apply_fan_out_limits(env, &opts); - match decision { - FanOutDecision::Exceeded { - dispatched, - skipped, - meta, - } => { - assert_eq!(dispatched.len(), 16); - assert_eq!(skipped.len(), 4); - assert!(meta.truncated); - assert_eq!(meta.shards_reached, 16); - assert_eq!(meta.shards_skipped, 4); - } - _ => panic!("expected Exceeded"), - } - } - - #[test] - fn fan_out_exceeded_with_partial() { - let mut env = ScatterEnvelope::new(); - for i in 0..20u32 { - env.add(VShardId::new(i), format!("node_{i}")); - } - - let opts = GraphTraversalOptions { - fan_out_partial: true, - ..Default::default() - }; - let decision = apply_fan_out_limits(env, &opts); - match decision { - FanOutDecision::Exceeded { - dispatched, meta, .. - } => { - assert_eq!(dispatched.len(), 16); - assert!(meta.truncated); - } - _ => panic!("expected Exceeded"), - } - } - - #[test] - fn custom_limits() { - let mut env = ScatterEnvelope::new(); - for i in 0..10u32 { - env.add(VShardId::new(i), format!("node_{i}")); - } - - let opts = GraphTraversalOptions { - fan_out_soft: 4, - fan_out_hard: 8, - fan_out_partial: true, - max_visited: 100_000, - }; - let decision = apply_fan_out_limits(env, &opts); - match decision { - FanOutDecision::Exceeded { - dispatched, - skipped, - .. - } => { - assert_eq!(dispatched.len(), 8); - assert_eq!(skipped.len(), 2); - } - _ => panic!("expected Exceeded"), - } - } -} diff --git a/nodedb/src/control/scatter_gather/hop.rs b/nodedb/src/control/scatter_gather/hop.rs deleted file mode 100644 index cd436aa37..000000000 --- a/nodedb/src/control/scatter_gather/hop.rs +++ /dev/null @@ -1,308 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Dispatch of one cross-shard hop: plan, admit, fan out, await, gather. -//! -//! This is the only file in the module that touches shared state, the routing -//! table, the gateway, and tokio. Everything it decides has been factored out -//! around it — the envelope holds the destinations, the fan-out policy decides -//! how wide it may go, the SQL builder produces what the remote node re-plans, -//! and the merge folds the replies. What remains here is the orchestration -//! those pieces are composed by, so the async/lease/spawn machinery lives in -//! exactly one place. - -use tracing::{debug, warn}; - -use crate::control::state::SharedState; -use crate::engine::graph::traversal_options::{GraphResponseMeta, GraphTraversalOptions}; -use crate::types::{DatabaseId, TenantId, TraceId}; - -use super::envelope::ScatterEnvelope; -use super::fan_out::{FanOutDecision, apply_fan_out_limits}; -use super::merge_results::merge_traversal_results; -use super::remote_sql::{RemoteTraverseSql, build_graph_traverse_sql}; - -/// Parameters for a cross-shard graph traversal hop. -pub struct CrossShardHopParams<'a> { - pub local_nodes: Vec, - pub envelope: ScatterEnvelope, - pub options: &'a GraphTraversalOptions, - /// Collection whose edges the remote hop walks. The owning node re-plans - /// the walk from the SQL this builds, and a traversal that names no - /// collection cannot be authorized there — so the scope travels with it. - pub collection: &'a str, - pub edge_label: Option<&'a str>, - pub direction: crate::engine::graph::edge_store::Direction, - pub remaining_depth: usize, - /// Session database scope. Threaded into the per-traversal - /// `QueryContext` and SQL plan so remote shard hops route and plan - /// against the caller's database, not the hardcoded default. - pub database_id: DatabaseId, -} - -/// Coordinate a single cross-shard graph hop from the Control Plane. -/// -/// Given a set of locally-discovered node IDs and a pre-built scatter envelope, -/// this function: -/// 1. Applies adaptive fan-out limits to the envelope. -/// 2. For each shard batch that passes the limit check, forwards a -/// `GRAPH TRAVERSE FROM '' DEPTH 1` query to the leader node that -/// owns that shard via the cluster transport. -/// 3. Merges all remote results with `local_nodes` via deduplication. -/// -/// Returns the merged node list and the aggregate `GraphResponseMeta`. -/// -/// # Cluster mode only -/// -/// This function assumes `shared.cluster_routing` and `shared.gateway` -/// are `Some`. Callers must check `shared.cluster_routing.is_some()` before -/// calling this function. -pub async fn coordinate_cross_shard_hop( - shared: &SharedState, - tenant_id: TenantId, - params: CrossShardHopParams<'_>, -) -> crate::Result<(Vec, GraphResponseMeta)> { - let CrossShardHopParams { - local_nodes, - envelope: cross_shard_targets, - options, - collection, - edge_label, - direction, - remaining_depth, - database_id, - } = params; - // Fast path: nothing to scatter. - if cross_shard_targets.is_empty() { - return Ok((local_nodes, GraphResponseMeta::default())); - } - - let decision = apply_fan_out_limits(cross_shard_targets, options); - - let (batches, mut meta) = match decision { - FanOutDecision::Proceed { batches, meta } => (batches, meta), - FanOutDecision::ProceedWithWarning { batches, meta } => { - debug!( - shards = meta.shards_reached, - warning = ?meta.fan_out_warning, - "cross-shard hop: fan-out soft limit exceeded, continuing" - ); - (batches, meta) - } - FanOutDecision::Exceeded { - dispatched, - skipped, - meta, - } => { - if options.fan_out_partial { - debug!( - dispatched = dispatched.len(), - skipped = skipped.len(), - "cross-shard hop: hard fan-out limit, returning partial results" - ); - (dispatched, meta) - } else { - return Err(crate::Error::FanOutExceeded { - shards_touched: meta.shards_reached + meta.shards_skipped, - limit: options.fan_out_hard, - }); - } - } - }; - - // Acquire the routing table and gateway once. - let routing = match &shared.cluster_routing { - Some(r) => r, - None => { - // Should not happen — callers must check. Return local results. - warn!("coordinate_cross_shard_hop called without cluster routing"); - return Ok((local_nodes, meta)); - } - }; - let gateway = match shared.gateway.get() { - Some(g) => g.clone(), - None => { - warn!("coordinate_cross_shard_hop called without gateway"); - return Ok((local_nodes, meta)); - } - }; - - // We always traverse exactly 1 depth per scatter batch because the caller - // drives the outer BFS loop. `remaining_depth` is included for completeness - // but each forwarded request probes depth 1 so the Control Plane maintains - // authoritative hop counting. - let hop_depth = remaining_depth.min(1); - - // Fan out to all batches in parallel. - let mut join_handles = Vec::with_capacity(batches.len()); - - for batch in batches { - let shard_id = batch.target_shard; - let leader_node = { - let rt = routing.read().unwrap_or_else(|p| p.into_inner()); - match rt.leader_for_vshard(shard_id.as_u32()) { - Ok(node) => node, - Err(e) => { - warn!(%shard_id, error = %e, "no leader for shard, skipping batch"); - continue; - } - } - }; - - // Skip batches that target the local node — those nodes are already - // covered by the local BFS that was executed before this call. - if leader_node == shared.node_id { - continue; - } - - let tenant_id_u64 = tenant_id.as_u64(); - let edge_label = edge_label.map(str::to_owned); - let collection = collection.to_owned(); - let mut any_error = false; - let mut work = Vec::with_capacity(batch.node_ids.len()); - - // The fan-out leg of a traversal the originating query already resolved - // policy for — this synthesizes internal SQL per remote shard and has no - // requester of its own, so it plans as the system. - let security = crate::control::planner::context::SystemPlanSecurity::new( - crate::types::TenantId::new(tenant_id_u64), - "_system_scatter_gather", - ); - - // Plan and admit every traversal before spawning. The resulting work - // owns its descriptor lease scope, so the spawned closure does not - // need to retain or reconstruct SharedState. - for node_id in batch.node_ids { - let sql = build_graph_traverse_sql(RemoteTraverseSql { - collection: &collection, - node_id: &node_id, - depth: hop_depth, - edge_label: edge_label.as_deref(), - direction, - }); - let gw_ctx = crate::control::gateway::core::QueryContext { - tenant_id: crate::types::TenantId::new(tenant_id_u64), - trace_id: TraceId::generate(), - database_id, - txn_id: None, - }; - let plan_ctx = crate::control::planner::context::QueryContext::for_state(shared); - let (tasks, _output_schema, versions, _) = match plan_ctx - .plan_sql_with_rls_and_versions( - &sql, - crate::types::TenantId::new(tenant_id_u64), - database_id, - &security.context(shared), - None, - ) - .await - { - Ok(planned) => planned, - Err(e) => { - warn!( - shard = %shard_id, - error = %e, - "remote graph traverse plan failed" - ); - any_error = true; - continue; - } - }; - // Each planned remote query gets an independent scope. Keep it - // through gateway execution and response payload consumption; - // do not retain it across the next node in this batch. - let lease_scope = match shared.acquire_plan_lease_scope(&versions) { - Ok(scope) => scope, - Err(e) => { - warn!( - shard = %shard_id, - error = %e, - "remote graph traverse rejected by descriptor lease admission" - ); - any_error = true; - continue; - } - }; - let physical_plan = match tasks.into_iter().next().map(|task| task.plan) { - Some(plan) => plan, - None => { - any_error = true; - continue; - } - }; - - work.push((gw_ctx, physical_plan, lease_scope)); - } - - let gateway_clone = gateway.clone(); - join_handles.push(tokio::spawn(async move { - let mut shard_results: Vec = Vec::new(); - - for (gw_ctx, physical_plan, lease_scope) in work { - match gateway_clone.execute_internal(&gw_ctx, physical_plan).await { - Ok(payloads) => { - for payload in payloads { - // `execute_graph_hop` encodes its `Vec` of - // node ids with `response_codec::encode`, which is - // MessagePack — so `decode_payload` is the - // counterpart. A JSON parser here would fail on - // every payload, and an `if let Ok` would drop - // each one — not a tolerated shard failure but a - // silent one: the traversal would return only the - // nodes the local shard found, and report that as - // the complete answer. A shard whose reply - // cannot be read is flagged like a shard that failed - // to answer, so the caller sees a partial result - // rather than a wrong complete one. - match crate::data::executor::response_codec::decode_payload::>( - &payload, - ) { - Ok(nodes) => shard_results.extend(nodes), - Err(e) => { - warn!( - shard = %shard_id, - error = %e, - "remote graph traverse reply could not be decoded" - ); - any_error = true; - } - } - } - } - Err(e) => { - warn!( - shard = %shard_id, - error = %e, - "remote graph traverse dispatch failed" - ); - any_error = true; - } - } - drop(lease_scope); - } - - (shard_results, any_error) - })); - } - - // Collect all remote results. - let mut remote_results: Vec> = Vec::with_capacity(join_handles.len()); - for handle in join_handles { - match handle.await { - Ok((nodes, _had_error)) => { - if !nodes.is_empty() { - remote_results.push(nodes); - } - } - Err(e) => { - warn!(error = %e, "cross-shard hop task panicked"); - } - } - } - - // Update meta with the number of shards that actually responded. - meta.shards_reached = remote_results.len() as u16; - - // Deduplicate and merge local + remote results. - let merged = merge_traversal_results(local_nodes, &remote_results); - Ok((merged, meta)) -} diff --git a/nodedb/src/control/scatter_gather/merge_results.rs b/nodedb/src/control/scatter_gather/merge_results.rs deleted file mode 100644 index 713d4ffa0..000000000 --- a/nodedb/src/control/scatter_gather/merge_results.rs +++ /dev/null @@ -1,56 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! The gather side of a hop: folding per-shard partial results into one answer. -//! -//! Split from the dispatch that produces those partials because deduplication -//! is where a traversal's correctness is decided independently of how many -//! shards replied — the same local + remote node lists must collapse to the -//! same set whether they arrived from one shard or twelve. Isolating it keeps -//! that property testable without a cluster. - -use std::collections::HashSet; - -/// Merge partial traversal results from multiple shards. -/// -/// Deduplicates node IDs and accumulates all discovered nodes. -pub fn merge_traversal_results( - local_nodes: Vec, - shard_results: &[Vec], -) -> Vec { - let mut seen: HashSet = HashSet::new(); - let mut merged = Vec::new(); - - for node in local_nodes { - if seen.insert(node.clone()) { - merged.push(node); - } - } - - for result in shard_results { - for node in result { - if seen.insert(node.clone()) { - merged.push(node.clone()); - } - } - } - - merged -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn merge_deduplicates() { - let local = vec!["a".into(), "b".into(), "c".into()]; - let shard1 = vec!["b".into(), "d".into()]; - let shard2 = vec!["c".into(), "e".into()]; - - let merged = merge_traversal_results(local, &[shard1, shard2]); - assert_eq!(merged.len(), 5); - assert!(merged.contains(&"a".to_string())); - assert!(merged.contains(&"d".to_string())); - assert!(merged.contains(&"e".to_string())); - } -} diff --git a/nodedb/src/control/scatter_gather/mod.rs b/nodedb/src/control/scatter_gather/mod.rs deleted file mode 100644 index eb42e1059..000000000 --- a/nodedb/src/control/scatter_gather/mod.rs +++ /dev/null @@ -1,29 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Scatter-gather coordinator for cross-shard graph traversals. -//! -//! Cross-shard hops are NOT handled by forwarding the traversal to -//! the remote shard. Instead, the Data Plane returns partial results (the set -//! of cross-shard edge targets) to the Control Plane, which batches and -//! dispatches them to the appropriate target cores. -//! -//! This keeps the Data Plane stateless per-request and avoids distributed -//! deadlocks from recursive cross-shard calls. -//! -//! ## Vectorized Scatter Envelopes -//! -//! The Data Plane MUST NOT emit one SPSC message per unresolved cross-shard edge. -//! Instead, for each hop level, cross-shard destinations are accumulated into a -//! single vectorized envelope grouped by target shard: -//! `{ shard_id -> [node_id, ...] }`. - -pub mod envelope; -pub mod fan_out; -pub mod hop; -pub mod merge_results; -pub mod remote_sql; - -pub use envelope::{ScatterBatch, ScatterEnvelope, partition_local_remote}; -pub use fan_out::{FanOutDecision, apply_fan_out_limits}; -pub use hop::{CrossShardHopParams, coordinate_cross_shard_hop}; -pub use merge_results::merge_traversal_results; diff --git a/nodedb/src/control/scatter_gather/remote_sql.rs b/nodedb/src/control/scatter_gather/remote_sql.rs deleted file mode 100644 index 08be49100..000000000 --- a/nodedb/src/control/scatter_gather/remote_sql.rs +++ /dev/null @@ -1,83 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Construction of the `GRAPH TRAVERSE` text a remote shard is asked to re-plan. -//! -//! Its own file because this is the module's only injection surface: the hop -//! synthesizes SQL from caller-supplied node ids, edge labels, and a collection -//! name, and the owning node re-plans whatever arrives. Keeping every fragment -//! of that text — quoting, the optional LABEL clause, the DIRECTION keyword — -//! in one place is what makes "is any of this unescaped?" answerable by reading -//! a single short file rather than auditing the dispatch loop. - -use crate::engine::graph::edge_store::Direction; - -/// Fields the remote `GRAPH TRAVERSE` text is built from. -pub(super) struct RemoteTraverseSql<'a> { - pub(super) collection: &'a str, - pub(super) node_id: &'a str, - pub(super) depth: usize, - pub(super) edge_label: Option<&'a str>, - pub(super) direction: Direction, -} - -/// The traversal direction as its SQL keyword. -/// -/// The value comes from a closed enum, never from caller text, so every arm is -/// a fixed keyword and there is nothing to escape. -fn canonical_direction_sql(direction: Direction) -> &'static str { - match direction { - Direction::In => "in", - Direction::Out => "out", - Direction::Both => "both", - } -} - -/// The optional edge label as a ` LABEL ` clause, or empty. -/// -/// The label is caller-supplied text, so it goes through the shared literal -/// quoter — this is the only place the clause is built. -fn canonical_label_sql(edge_label: Option<&str>) -> String { - match edge_label { - Some(label) => format!(" LABEL {}", ::nodedb_types::quote_literal(label)), - None => String::new(), - } -} - -pub(super) fn build_graph_traverse_sql(params: RemoteTraverseSql<'_>) -> String { - let RemoteTraverseSql { - collection, - node_id, - depth, - edge_label, - direction, - } = params; - format!( - "GRAPH TRAVERSE IN {} FROM {} DEPTH {}{} DIRECTION {}", - ::nodedb_types::quote_literal(collection), - ::nodedb_types::quote_literal(node_id), - ::nodedb_types::Value::Integer(if depth == 0 { 0 } else { 1 }).to_sql_literal(), - canonical_label_sql(edge_label), - canonical_direction_sql(direction), - ) -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn graph_traverse_sql_quotes_node_and_label_literals() { - let sql = build_graph_traverse_sql(RemoteTraverseSql { - collection: "audit'; --", - node_id: "node'; DROP GRAPH audit; --", - depth: 1, - edge_label: Some("label'; --"), - direction: Direction::Out, - }); - assert_eq!( - sql, - "GRAPH TRAVERSE IN 'audit''; --' FROM 'node''; DROP GRAPH audit; --' DEPTH 1 \ - LABEL 'label''; --' DIRECTION out" - ); - } -} diff --git a/nodedb/src/control/security/auth_apikey.rs b/nodedb/src/control/security/auth_apikey.rs index 3355ada5a..7a576b9bd 100644 --- a/nodedb/src/control/security/auth_apikey.rs +++ b/nodedb/src/control/security/auth_apikey.rs @@ -144,7 +144,7 @@ impl AuthApiKeyStore { } else { 0 }; - let hash_hex = hex_encode(&secret_hash); + let hash_hex = hex::encode(&secret_hash); let record = AuthApiKey { key_id, secret_hash, @@ -202,7 +202,7 @@ impl AuthApiKeyStore { } let secret = parts[1]; let secret_hash = hash_secret(secret); - let hash_hex = hex_encode(&secret_hash); + let hash_hex = hex::encode(&secret_hash); let state = self.state.read(); let key_id = state.hash_index.get(&hash_hex)?; @@ -332,15 +332,12 @@ fn generate_secret() -> String { use std::sync::atomic::{AtomicU64, Ordering}; static COUNTER: AtomicU64 = AtomicU64::new(0); let seq = COUNTER.fetch_add(1, Ordering::Relaxed); - // 32 bytes of cryptographic randomness + sequential suffix for uniqueness. + // 32 bytes of OS randomness + sequential suffix for uniqueness. The + // secret never falls back to a guessable value. + use argon2::password_hash::rand_core::{OsRng, RngCore}; let mut random_bytes = [0u8; 32]; - getrandom::fill(&mut random_bytes).unwrap_or_else(|_| { - // Fallback: use timestamp + counter if getrandom unavailable (e.g., early boot). - let ts = now_secs(); - random_bytes[..8].copy_from_slice(&ts.to_le_bytes()); - random_bytes[8..16].copy_from_slice(&seq.to_le_bytes()); - }); - format!("{}{seq:08x}", hex_encode(&random_bytes)) + OsRng.fill_bytes(&mut random_bytes); + format!("{}{seq:08x}", hex::encode(random_bytes)) } fn hash_secret(secret: &str) -> Vec { @@ -352,11 +349,6 @@ fn now_secs() -> u64 { crate::control::security::time::now_secs() } -/// Hex-encode a byte slice (lowercase). -fn hex_encode(bytes: &[u8]) -> String { - bytes.iter().map(|b| format!("{b:02x}")).collect() -} - #[cfg(test)] mod tests { use super::*; @@ -374,7 +366,7 @@ mod tests { let key = store .verify(&token) .expect("lookup after interrupted update"); - let hash_hex = hex_encode(&key.secret_hash); + let hash_hex = hex::encode(&key.secret_hash); let state = store.state.read(); assert_eq!(state.hash_index.get(&hash_hex), Some(&key.key_id)); assert!(state.keys.contains_key(&key.key_id)); @@ -476,11 +468,11 @@ mod tests { let new_token = store.rotate(&old_key.key_id, 24).unwrap(); assert!(new_token.starts_with("nda_")); - // Both old and new should be valid during overlap. + // Both old and new must be valid during overlap. assert!(store.verify(&old_token).is_some()); assert!(store.verify(&new_token).is_some()); - // New key should inherit scopes. + // New key must inherit scopes. let new_key = store.verify(&new_token).unwrap(); assert_eq!(new_key.scopes, vec!["scope_a"]); assert!(new_key.replaces_key_id.is_some()); @@ -508,7 +500,7 @@ mod tests { #[test] fn rotate_chain_invalidates_intermediate_key() { - // Spec: rotation chains invalidate every superseded key, not just the + // Spec: rotation chains invalidate every superseded key, not only the // first. After A→B→C with zero overlap, both A and B must be rejected. let store = AuthApiKeyStore::new(); let token_a = store.create_key("u1", 1, vec![], 0, 0, 0); @@ -533,8 +525,8 @@ mod tests { #[test] fn list_for_user_excludes_superseded_keys_past_overlap() { // Spec: listing active keys must reflect supersession, not only the - // is_revoked bit. An operator auditing active credentials should not - // see a rotated-out key as still active after its overlap elapsed. + // is_revoked bit. An operator auditing active credentials never + // sees a rotated-out key as still active after its overlap elapsed. let store = AuthApiKeyStore::new(); let token_old = store.create_key("u1", 1, vec![], 0, 0, 0); let old_id = store.verify(&token_old).unwrap().key_id; @@ -558,12 +550,9 @@ mod tests { #[test] fn rotate_persists_invalidation_marker_on_old_key() { - // Regression guard against the specific silent-failure mode: rotate() - // historically decorated only the NEW key and left the OLD key - // untouched, so verify() had no way to learn the old key was being - // retired. The fix must leave an invalidation marker on the OLD - // record itself so verify() can reject by inspecting the old record - // alone — no cross-record lookup. + // rotate() must leave an invalidation marker on the OLD record + // itself, not only decorate the NEW key. verify() then rejects by + // inspecting the old record alone, with no cross-record lookup. let store = AuthApiKeyStore::new(); let old_token = store.create_key("u1", 1, vec![], 0, 0, 0); let old_id = store.verify(&old_token).unwrap().key_id; diff --git a/nodedb/src/control/security/auth_fence/mod.rs b/nodedb/src/control/security/auth_fence/mod.rs index ea63f5a9f..0c4067005 100644 --- a/nodedb/src/control/security/auth_fence/mod.rs +++ b/nodedb/src/control/security/auth_fence/mod.rs @@ -7,5 +7,5 @@ pub mod tree_defs; pub mod view; pub use state::AuthorizationFence; -pub use tree_defs::{PendingTreeDefs, TreeDefChange}; -pub use view::permission_view; +pub use tree_defs::{PendingTreeDefs, TreeDefChange, cache_from_catalog, load_tree_defs}; +pub use view::{admit_permission_view, permission_view}; diff --git a/nodedb/src/control/security/auth_fence/tree_defs.rs b/nodedb/src/control/security/auth_fence/tree_defs.rs index 728d35745..10519de06 100644 --- a/nodedb/src/control/security/auth_fence/tree_defs.rs +++ b/nodedb/src/control/security/auth_fence/tree_defs.rs @@ -10,86 +10,76 @@ use std::sync::Mutex; -use crate::control::security::catalog::StoredCollection; -use crate::control::security::permission_tree::{PermissionCache, PermissionTreeDef, SourceIndex}; -use crate::types::DatabaseId; +use crate::control::security::catalog::{StoredCollection, SystemCatalog}; +use crate::control::security::permission_tree::{ + PermissionCache, PermissionTreeDef, SourceIndex, TreeKey, +}; /// One committed change to a collection's tree definition. #[derive(Debug, Clone, PartialEq, Eq)] pub enum TreeDefChange { Register { - tenant_id: u64, - collection: String, + key: TreeKey, def: PermissionTreeDef, }, Unregister { - tenant_id: u64, - collection: String, + key: TreeKey, }, } impl TreeDefChange { - /// The change a committed collection descriptor makes. Tree definitions - /// live on default-database collections only, as the DDL writes them. + /// The change a committed collection descriptor of any database makes. /// An inactive collection governs nothing. - pub fn from_collection(stored: &StoredCollection) -> crate::Result> { - if stored.database_id != DatabaseId::DEFAULT { - return Ok(None); - } - let tenant_id = stored.tenant_id; - let collection = stored.name.clone(); + pub fn from_collection(stored: &StoredCollection) -> crate::Result { + let key = TreeKey::new(stored.database_id, stored.tenant_id, stored.name.clone()); let def = match (&stored.permission_tree_def, stored.is_active) { (Some(json), true) => json, - _ => { - return Ok(Some(Self::Unregister { - tenant_id, - collection, - })); - } + _ => return Ok(Self::Unregister { key }), }; let def: PermissionTreeDef = sonic_rs::from_str(def).map_err(|e| crate::Error::Serialization { format: "json".into(), - detail: format!("PERMISSION_TREE of collection '{collection}': {e}"), + detail: format!("PERMISSION_TREE of collection '{}': {e}", key.collection), })?; - Ok(Some(Self::Register { - tenant_id, - collection, - def, - })) + Ok(Self::Register { key, def }) } /// Record this committed change in the source index, ahead of the cache. pub fn note_committed(&self, sources: &SourceIndex) { match self { - Self::Register { - tenant_id, - collection, - def, - } => sources.note_committed(*tenant_id, collection, Some(def)), - Self::Unregister { - tenant_id, - collection, - } => sources.note_committed(*tenant_id, collection, None), + Self::Register { key, def } => sources.note_committed(key, Some(def)), + Self::Unregister { key } => sources.note_committed(key, None), } } /// Apply this change to `cache`. pub fn apply(self, cache: &mut PermissionCache) { match self { - Self::Register { - tenant_id, - collection, - def, - } => cache.register_tree_def(tenant_id, &collection, def), - Self::Unregister { - tenant_id, - collection, - } => cache.unregister_tree_def(tenant_id, &collection), + Self::Register { key, def } => cache.register_tree_def(key, def), + Self::Unregister { key } => cache.unregister_tree_def(&key), } } } +/// A cache holding the tree definition of every collection `catalog` stores, +/// in every database. Boot builds the cache with this; the edges and grants +/// load once the data groups replayed. +pub fn cache_from_catalog(catalog: &SystemCatalog) -> crate::Result { + let mut cache = PermissionCache::new(); + load_tree_defs(&mut cache, catalog)?; + Ok(cache) +} + +/// Apply the tree definition of every collection `catalog` stores, in every +/// database, to `cache` in place. The cache keeps its source index, so an +/// authorization fence built on that index sees every definition loaded. +pub fn load_tree_defs(cache: &mut PermissionCache, catalog: &SystemCatalog) -> crate::Result<()> { + for stored in &catalog.load_all_collections_across_databases()? { + TreeDefChange::from_collection(stored)?.apply(cache); + } + Ok(()) +} + /// The queue of committed changes not yet in the cache. #[derive(Debug, Default)] pub struct PendingTreeDefs { @@ -126,6 +116,7 @@ impl PendingTreeDefs { #[cfg(test)] mod tests { use super::*; + use crate::types::DatabaseId; fn def() -> PermissionTreeDef { sonic_rs::from_str( @@ -134,21 +125,20 @@ mod tests { .expect("tree def") } + fn key(collection: &str) -> TreeKey { + TreeKey::new(DatabaseId::DEFAULT, 1, collection) + } + #[test] fn queued_changes_apply_in_commit_order() { let pending = PendingTreeDefs::default(); pending.push(TreeDefChange::Register { - tenant_id: 1, - collection: "docs".into(), + key: key("docs"), def: def(), }); - pending.push(TreeDefChange::Unregister { - tenant_id: 1, - collection: "docs".into(), - }); + pending.push(TreeDefChange::Unregister { key: key("docs") }); pending.push(TreeDefChange::Register { - tenant_id: 1, - collection: "notes".into(), + key: key("notes"), def: def(), }); assert!(!pending.is_empty()); @@ -157,7 +147,62 @@ mod tests { pending.apply_to(&mut cache); assert!(pending.is_empty()); - assert!(cache.get_tree_def(1, "docs").is_none()); - assert_eq!(cache.get_tree_def(1, "notes"), Some(&def())); + assert!(cache.get_tree_def(&key("docs")).is_none()); + assert_eq!(cache.get_tree_def(&key("notes")), Some(&def())); + } + + /// A tree on a named database's collection registers under that + /// database, never the default one. Boot and the metadata applier both + /// read collections through this path. + #[test] + fn a_named_database_collection_registers_its_tree() { + let db = DatabaseId::new(7); + let mut stored = StoredCollection::new(1, "docs", "alice"); + stored.database_id = db; + stored.is_active = true; + stored.permission_tree_def = Some(sonic_rs::to_string(&def()).expect("serialize tree def")); + let change = TreeDefChange::from_collection(&stored).expect("parse tree def"); + assert_eq!( + change, + TreeDefChange::Register { + key: TreeKey::new(db, 1, "docs"), + def: def(), + } + ); + + let mut cache = PermissionCache::new(); + change.apply(&mut cache); + assert!(cache.get_tree_def(&TreeKey::new(db, 1, "docs")).is_some()); + assert!(cache.get_tree_def(&key("docs")).is_none()); + } + + /// Boot registers the tree of a named database's collection under that + /// database. + #[test] + fn boot_loads_trees_of_every_database() { + let dir = tempfile::tempdir().expect("tempdir"); + let catalog = + SystemCatalog::open(&dir.path().join("system.redb")).expect("open system catalog"); + let db = DatabaseId::new(7); + let json = sonic_rs::to_string(&def()).expect("serialize tree def"); + for (database_id, name) in [(db, "docs"), (DatabaseId::DEFAULT, "notes")] { + let mut stored = StoredCollection::stamped_for_test(1, name, "alice"); + stored.database_id = database_id; + stored.is_active = true; + stored.permission_tree_def = Some(json.clone()); + catalog + .put_collection(database_id, &stored) + .expect("put collection"); + } + + let cache = cache_from_catalog(&catalog).expect("load tree defs"); + + assert_eq!( + cache.get_tree_def(&TreeKey::new(db, 1, "docs")), + Some(&def()) + ); + assert!(cache.get_tree_def(&key("docs")).is_none()); + assert_eq!(cache.get_tree_def(&key("notes")), Some(&def())); + assert!(cache.get_tree_def(&TreeKey::new(db, 1, "notes")).is_none()); } } diff --git a/nodedb/src/control/security/auth_fence/view.rs b/nodedb/src/control/security/auth_fence/view.rs index 595db3fdf..7bfc92f17 100644 --- a/nodedb/src/control/security/auth_fence/view.rs +++ b/nodedb/src/control/security/auth_fence/view.rs @@ -19,18 +19,28 @@ //! leads the metadata group as its only voter holds a pinned lease, which //! never expires: every barrier waits for its coverage instead. //! -//! The lease is checked after the cache guard is taken. The guard fixes the -//! cache for the whole plan, and a change acknowledged after the check was -//! acknowledged after the statement started planning. +//! A lease lapses when one renewal round runs long, though the next round +//! is a renewal interval away. Before it takes the cache guard, a statement +//! waits up to the lease's lapse grace for a round to grant it again. The +//! wait holds no guard, so the leader's own floor load can take the cache's +//! write lock meanwhile. The check under the guard then decides. +//! +//! The lease is checked after the cache guard is taken, so the guarded cache +//! holds every change acknowledged before the check. A change acknowledged +//! after the check was acknowledged after the statement started planning. +//! The cache only moves forward, so a later read of it holds the fenced +//! state or newer. A planning path that awaits a request (a surrogate at its +//! collection home) runs the fence with [`admit_permission_view`], holds no +//! guard across that request, and reads the live cache once it resumes. use std::time::Instant; use tokio::sync::RwLockReadGuard; -use crate::control::security::auth_lease::lease_status; +use crate::control::security::auth_lease::{lease_status, planning_admitted_within}; use crate::control::security::permission_tree::{PermissionCache, reload}; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId}; +use crate::types::TenantId; use super::cluster::{behind, group_of_vshard, hosts_group}; @@ -42,22 +52,24 @@ pub async fn permission_view( apply_committed_tree_defs(state).await; reload::reload_if_stale(state).await?; + if let Some(timing) = state.authorization_fence.timing() { + // The check under the guard below decides. This wait only lets a + // round that is about to renew finish first. + planning_admitted_within(state, timing.lapse_grace()).await; + } let cache = state.permission_cache.read().await; if state.cluster_routing.is_some() && cache.has_tree_defs_for_tenant(tenant_id.as_u64()) { for source in cache .tree_sources() .into_iter() - .filter(|source| source.tenant_id == tenant_id.as_u64()) + .filter(|source| source.key.scope.tenant_id == tenant_id.as_u64()) { - let vshard = - nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &source.collection) - .vshard(); - let group_id = group_of_vshard(state, vshard.as_u32())?; + let group_id = group_of_vshard(state, source.key.vshard().as_u32())?; if !hosts_group(state, group_id) { return Err(behind(format!( "this node does not replicate raft group {group_id}, which homes \ permission source '{}'; run the statement on a node that does", - source.collection + source.key.qualified() ))); } } @@ -71,6 +83,13 @@ pub async fn permission_view( Ok(cache) } +/// Run the fence [`permission_view`] runs for a statement of `tenant_id`, +/// and release the view. The statement's planning reads the live cache after +/// its last await. +pub async fn admit_permission_view(state: &SharedState, tenant_id: TenantId) -> crate::Result<()> { + permission_view(state, tenant_id).await.map(drop) +} + /// Move the tree-definition changes the metadata applier committed into the /// cache. The applier queues each change before it advances the applied /// index, so the queue holds every change this node applied. diff --git a/nodedb/src/control/security/auth_lease/barrier.rs b/nodedb/src/control/security/auth_lease/barrier.rs index b96676cdb..f0cdfae8c 100644 --- a/nodedb/src/control/security/auth_lease/barrier.rs +++ b/nodedb/src/control/security/auth_lease/barrier.rs @@ -3,24 +3,19 @@ //! The writer's side: hold an authorization change's acknowledgement until //! no node can plan against the state before it. //! -//! - **In a cluster** the metadata leader holds the barrier until every node -//! with an unexpired lease covered the targets, or its lease expired. The -//! writing node is a lease holder too, so its own Event Plane lag closes -//! the same way. -//! - **On a single node** there is no lease. The barrier waits until the -//! local permission cache reflects every event the cores emitted, which -//! covers the change just applied. +//! The metadata leader holds the barrier until every node with an unexpired +//! lease covered the targets, or its lease expired. The writing node is a +//! lease holder too, so its own Event Plane lag closes the same way. A +//! single-node cluster runs the same barrier against its one lease. use std::time::{Duration, Instant}; use nodedb_cluster::calvin::SEQUENCER_GROUP_ID; use nodedb_cluster::{AuthBarrierOutcome, AuthBarrierRequest, GroupCoverage, RaftRpc}; -use tokio::runtime::RuntimeFlavor; use crate::control::security::auth_fence::cluster::behind; use crate::control::state::SharedState; -use super::coverage::permission_step_covers_now; use super::leadership::{metadata_leader, send_to_leader}; /// Hold the acknowledgement of a change until it binds every node. @@ -36,17 +31,14 @@ use super::leadership::{metadata_leader, send_to_leader}; /// cannot renew adds up to one lease duration (the election timeout), until /// its lease expires. A pinned holder, the leader that is the only voter of /// the metadata group, has no expiry: the barrier waits for its next renewal -/// however late it runs. On a single node the wait is the permission step's -/// lag only. +/// however late it runs. pub async fn authorization_barrier( state: &SharedState, targets: Vec, ) -> crate::Result<()> { let deadline_secs = state.tuning.network.default_deadline_secs; let deadline = Instant::now() + Duration::from_secs(deadline_secs); - let Some(timing) = state.authorization_fence.timing() else { - return await_local_coverage(state, deadline).await; - }; + let timing = lease_timing(state)?; loop { let remaining = deadline.saturating_duration_since(Instant::now()); if remaining.is_zero() { @@ -70,7 +62,7 @@ pub async fn authorization_barrier( state, leader_id, RaftRpc::AuthBarrierRequest(request), - remaining + timing.lease, + timing.rpc_read_timeout(remaining), ) .await { @@ -106,11 +98,7 @@ pub async fn authorization_barrier( /// group's commit index now. A node covers that index only once its own /// replicas applied every acknowledged transaction below it. pub async fn calvin_write_barrier(state: &SharedState) -> crate::Result<()> { - if state.authorization_fence.timing().is_none() { - let deadline = - Instant::now() + Duration::from_secs(state.tuning.network.default_deadline_secs); - return await_local_coverage(state, deadline).await; - } + lease_timing(state)?; let commit_index = state .raft_status_fn .get() @@ -131,33 +119,16 @@ pub async fn calvin_write_barrier(state: &SharedState) -> crate::Result<()> { .await } -/// Run [`authorization_barrier`] from synchronous code on a Tokio worker. -pub fn block_on_barrier(state: &SharedState, targets: Vec) -> crate::Result<()> { - let handle = tokio::runtime::Handle::try_current().map_err(|_| crate::Error::Internal { - detail: "authorization barrier: called outside a Tokio runtime".into(), - })?; - if handle.runtime_flavor() != RuntimeFlavor::MultiThread { - return Err(crate::Error::Internal { - detail: "authorization barrier: synchronous callers need a multi-thread runtime".into(), - }); - } - tokio::task::block_in_place(|| handle.block_on(authorization_barrier(state, targets))) -} - -/// Wait until the permission step covers every event the cores emitted -/// before this call, which includes the write just applied. -/// -/// This is a writer's acknowledgement wait: it only waits, and never reloads -/// or dispatches. A cache that needs a reload is reloaded by the next -/// statement's planning, before it reads the cache. -pub async fn await_local_coverage(state: &SharedState, deadline: Instant) -> crate::Result<()> { - let remaining = deadline.saturating_duration_since(Instant::now()); - if permission_step_covers_now(state, remaining).await { - return Ok(()); - } - Err(committed_but_pending( - "the permission cache did not catch up with the change", - )) +/// The lease timing `start_raft` installs. +pub(crate) fn lease_timing(state: &SharedState) -> crate::Result { + state + .authorization_fence + .timing() + .ok_or(crate::Error::Internal { + detail: "the authorization lease is not installed: start_raft has not run on this \ + node" + .to_owned(), + }) } fn committed_but_pending(detail: impl std::fmt::Display) -> crate::Error { diff --git a/nodedb/src/control/security/auth_lease/calvin_acks.rs b/nodedb/src/control/security/auth_lease/calvin_acks.rs index af6596b51..3d6ed6362 100644 --- a/nodedb/src/control/security/auth_lease/calvin_acks.rs +++ b/nodedb/src/control/security/auth_lease/calvin_acks.rs @@ -138,7 +138,7 @@ mod tests { }]) .expect("save"); - let recovered = recover_applied(&wal, &catalog, 7).expect("recover"); + let recovered = recover_applied(&wal, &catalog, 7, None).expect("recover"); let mirrors = AppliedMirrors::default(); mirrors.register(7, recovered.fully_applied_epoch, &recovered.applied_tail); diff --git a/nodedb/src/control/security/auth_lease/coverage.rs b/nodedb/src/control/security/auth_lease/coverage.rs index 0bf20c5c4..e0836b587 100644 --- a/nodedb/src/control/security/auth_lease/coverage.rs +++ b/nodedb/src/control/security/auth_lease/coverage.rs @@ -155,42 +155,3 @@ pub(crate) async fn permission_step_reaches_now( let _ = tokio::time::timeout(until - now, notified).await; } } - -/// Whether the permission step covers every event the cores emitted before -/// this call, within `wait`. A writer holding its acknowledgement calls this. -/// -/// It never reloads. A cache that only a reload can bring to the targets is -/// stale, and [`reload::reload_if_stale`] reloads it before the next -/// statement plans. That reload reads each core after the write applied, so -/// the write already binds every later plan. Before the Event Plane starts no -/// permission step counts writes, so the cache is marked for a reload. -pub(crate) async fn permission_step_covers_now(state: &SharedState, wait: Duration) -> bool { - let fence = &state.authorization_fence; - let Some(targets) = fence.emitted_snapshot() else { - state - .permission_cache - .write() - .await - .progress_mut() - .mark_reload_needed(); - return true; - }; - let until = Instant::now() + wait; - loop { - let notified = fence.permission_applied().notified(); - tokio::pin!(notified); - notified.as_mut().enable(); - { - let cache = state.permission_cache.read().await; - let progress = cache.progress(); - if progress.caught_up(&targets) || progress.needs_reload_for(&targets) { - return true; - } - } - let now = Instant::now(); - if now >= until { - return false; - } - let _ = tokio::time::timeout(until - now, notified).await; - } -} diff --git a/nodedb/src/control/security/auth_lease/holder.rs b/nodedb/src/control/security/auth_lease/holder.rs index 0d940b3e4..e7a7975ca 100644 --- a/nodedb/src/control/security/auth_lease/holder.rs +++ b/nodedb/src/control/security/auth_lease/holder.rs @@ -13,13 +13,126 @@ use std::sync::Mutex; use std::time::Instant; -/// The end of this node's lease, if it holds one. -#[derive(Debug, Default)] +use nodedb_cluster::GroupCoverage; +use tokio::sync::watch; + +/// How this node's last renewal round ended. +#[derive(Debug, Clone, PartialEq, Eq)] +pub enum RenewAttempt { + /// No round ran yet. + NotAttempted, + /// This node's coverage was not computed, so nothing was sent. + CoverageFailed { error: String }, + /// This node knows no metadata leader, so nothing was sent. + NoLeader, + /// This node leads the metadata group but runs no lease service. + NoLeaderService, + /// The renewal did not reach `leader_id`. + NotDelivered { leader_id: u64, error: String }, + /// `leader_id` answered with a message that is not a renewal reply. + UnexpectedReply { leader_id: u64 }, + /// `leader_id` withheld the lease: `coverage` misses one of its floors. + Withheld { + leader_id: u64, + coverage: Vec, + }, + /// `leader_id` no longer leads the metadata group. + NotLeader { + leader_id: u64, + leader_hint: Option, + }, + /// `leader_id` granted the lease. + Granted { leader_id: u64 }, +} + +impl std::fmt::Display for RenewAttempt { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match self { + Self::NotAttempted => write!(f, "no renewal round ran"), + Self::CoverageFailed { error } => { + write!(f, "coverage could not be computed: {error}") + } + Self::NoLeader => write!(f, "no metadata leader is known"), + Self::NoLeaderService => { + write!( + f, + "this node leads the metadata group but runs no lease service" + ) + } + Self::NotDelivered { leader_id, error } => { + write!(f, "renewal did not reach leader {leader_id}: {error}") + } + Self::UnexpectedReply { leader_id } => { + write!( + f, + "leader {leader_id} sent a reply that is not a renewal reply" + ) + } + Self::Withheld { + leader_id, + coverage, + } => { + write!(f, "leader {leader_id} withheld the lease; coverage sent:")?; + for group in coverage { + write!(f, " group {} through {}", group.group_id, group.through)?; + } + Ok(()) + } + Self::NotLeader { + leader_id, + leader_hint, + } => write!( + f, + "node {leader_id} no longer leads the metadata group (hint {leader_hint:?})" + ), + Self::Granted { leader_id } => write!(f, "leader {leader_id} granted the lease"), + } + } +} + +/// The end of this node's lease, if it holds one, and how the last renewal +/// round ended. +#[derive(Debug)] pub struct LeaseHolder { valid_until: Mutex>, + last_attempt: Mutex, + /// Renewal rounds ended so far. A planning path that found the lease + /// lapsed waits on it for the next round. + rounds: watch::Sender, +} + +impl Default for LeaseHolder { + fn default() -> Self { + Self { + valid_until: Mutex::new(None), + last_attempt: Mutex::new(RenewAttempt::NotAttempted), + rounds: watch::Sender::new(0), + } + } } impl LeaseHolder { + /// Record how the latest renewal round ended, and wake every wait for + /// the round. A grant is installed before its round is recorded. + pub fn record_attempt(&self, attempt: RenewAttempt) { + *self.last_attempt.lock().unwrap_or_else(|p| p.into_inner()) = attempt; + self.rounds + .send_modify(|rounds| *rounds = rounds.wrapping_add(1)); + } + + /// A receiver that sees each renewal round recorded after this call. + pub fn subscribe_rounds(&self) -> watch::Receiver { + self.rounds.subscribe() + } + + /// How the latest renewal round ended. + pub fn last_attempt(&self) -> RenewAttempt { + self.last_attempt + .lock() + .unwrap_or_else(|p| p.into_inner()) + .clone() + } + /// Extend the lease to `until`. A grant never shortens a lease already /// held: the leader granted each one against the state it covers. pub fn install(&self, until: Instant) { @@ -65,6 +178,18 @@ mod tests { assert!(!holder.is_valid_at(sent_at + timing.lease)); } + #[tokio::test] + async fn a_recorded_round_wakes_a_subscriber() { + let holder = LeaseHolder::default(); + let mut rounds = holder.subscribe_rounds(); + assert!(!rounds.has_changed().expect("sender alive")); + holder.record_attempt(RenewAttempt::Granted { leader_id: 1 }); + tokio::time::timeout(Duration::from_secs(1), rounds.changed()) + .await + .expect("woken") + .expect("sender alive"); + } + #[test] fn an_older_grant_never_shortens_the_lease() { let holder = LeaseHolder::default(); diff --git a/nodedb/src/control/security/auth_lease/leadership.rs b/nodedb/src/control/security/auth_lease/leadership.rs index 086537480..a3d6ec38c 100644 --- a/nodedb/src/control/security/auth_lease/leadership.rs +++ b/nodedb/src/control/security/auth_lease/leadership.rs @@ -44,11 +44,12 @@ pub(crate) fn sole_voter_term(state: &SharedState) -> Option { .map(|group| group.term) } -/// The leader hint to send back with a refusal. -pub(crate) fn leader_hint(state: &SharedState) -> Option { - metadata_leader(state) - .map(|(leader_id, _)| leader_id) - .filter(|leader_id| *leader_id != 0) +/// The leader hint to send back with a refusal, and the term this node +/// knows it at: `(leader_hint, term)`. +pub(crate) fn leader_hint(state: &SharedState) -> (Option, u64) { + metadata_leader(state).map_or((None, 0), |(leader_id, term)| { + ((leader_id != 0).then_some(leader_id), term) + }) } /// Send `rpc` to the metadata leader `leader_id` and return its answer. diff --git a/nodedb/src/control/security/auth_lease/mod.rs b/nodedb/src/control/security/auth_lease/mod.rs index 907385bad..4486aa3f9 100644 --- a/nodedb/src/control/security/auth_lease/mod.rs +++ b/nodedb/src/control/security/auth_lease/mod.rs @@ -12,11 +12,9 @@ pub mod table; pub mod timing; pub mod withheld_warn; -pub use barrier::{ - authorization_barrier, await_local_coverage, block_on_barrier, calvin_write_barrier, -}; +pub use barrier::{authorization_barrier, calvin_write_barrier}; pub use calvin_acks::CalvinAckCoverage; -pub use holder::LeaseHolder; +pub use holder::{LeaseHolder, RenewAttempt}; pub use service::LeaderLeaseService; -pub use status::{LeaseStatus, await_planning_admitted, lease_status}; +pub use status::{LeaseStatus, await_planning_admitted, lease_status, planning_admitted_within}; pub use timing::LeaseTiming; diff --git a/nodedb/src/control/security/auth_lease/renew_loop.rs b/nodedb/src/control/security/auth_lease/renew_loop.rs index 10a9f879e..227c4e075 100644 --- a/nodedb/src/control/security/auth_lease/renew_loop.rs +++ b/nodedb/src/control/security/auth_lease/renew_loop.rs @@ -8,7 +8,7 @@ //! extends nothing, so the lease lapses unless a later renewal succeeds. use std::sync::Arc; -use std::time::Instant; +use std::time::{Duration, Instant}; use nodedb_cluster::{ AuthLeaseRenewOutcome, AuthLeaseRenewRequest, AuthLeaseRenewResponse, GroupCoverage, RaftRpc, @@ -18,6 +18,7 @@ use crate::control::shutdown::ShutdownReceiver; use crate::control::state::SharedState; use super::coverage::confirmed_coverage; +use super::holder::RenewAttempt; use super::leadership::{metadata_leader, send_to_leader}; use super::timing::LeaseTiming; @@ -51,13 +52,25 @@ async fn renew_round(state: &SharedState, timing: LeaseTiming, confirmed: &mut V } Err(error) => { tracing::warn!(%error, "authorization lease: coverage could not be computed"); + record( + state, + RenewAttempt::CoverageFailed { + error: error.to_string(), + }, + ); } } } +/// Record how this round ended, so a refusal to plan can name it. +fn record(state: &SharedState, attempt: RenewAttempt) { + state.authorization_fence.holder().record_attempt(attempt); +} + /// Send one renewal and install a granted lease. async fn renew_once(state: &SharedState, timing: LeaseTiming, coverage: &[GroupCoverage]) { let Some((leader_id, _)) = metadata_leader(state).filter(|(leader, _)| *leader != 0) else { + record(state, RenewAttempt::NoLeader); return; }; let request = AuthLeaseRenewRequest { @@ -68,14 +81,17 @@ async fn renew_once(state: &SharedState, timing: LeaseTiming, coverage: &[GroupC let response = if leader_id == state.node_id { match state.authorization_fence.leader() { Some(service) => service.renew_lease(request).await, - None => return, + None => { + record(state, RenewAttempt::NoLeaderService); + return; + } } } else { match send_to_leader( state, leader_id, RaftRpc::AuthLeaseRenewRequest(request), - timing.lease, + timing.rpc_read_timeout(Duration::ZERO), ) .await { @@ -85,39 +101,65 @@ async fn renew_once(state: &SharedState, timing: LeaseTiming, coverage: &[GroupC leader_id, "authorization lease: unexpected renewal reply {other:?}" ); + record(state, RenewAttempt::UnexpectedReply { leader_id }); return; } Err(error) => { tracing::debug!(%error, "authorization lease: renewal not delivered"); + record( + state, + RenewAttempt::NotDelivered { + leader_id, + error: error.to_string(), + }, + ); return; } } }; - install(state, timing, sent_at, response); + install(state, timing, sent_at, leader_id, coverage, response); } fn install( state: &SharedState, timing: LeaseTiming, sent_at: Instant, + leader_id: u64, + coverage: &[GroupCoverage], response: AuthLeaseRenewResponse, ) { match response.outcome { AuthLeaseRenewOutcome::Granted { lease_ms } => { - let granted = std::time::Duration::from_millis(lease_ms); + let granted = Duration::from_millis(lease_ms); state .authorization_fence .holder() .install(timing.holder_expiry(sent_at, granted)); + record(state, RenewAttempt::Granted { leader_id }); } AuthLeaseRenewOutcome::Withheld => { tracing::debug!("authorization lease: renewal withheld until coverage catches up"); + record( + state, + RenewAttempt::Withheld { + leader_id, + coverage: coverage.to_vec(), + }, + ); } - AuthLeaseRenewOutcome::NotLeader { leader_hint } => { + AuthLeaseRenewOutcome::NotLeader { leader_hint, term } => { tracing::debug!( ?leader_hint, + term, "authorization lease: renewal reached a non-leader" ); + record( + state, + RenewAttempt::NotLeader { + leader_id, + leader_hint, + }, + ); } } } diff --git a/nodedb/src/control/security/auth_lease/service.rs b/nodedb/src/control/security/auth_lease/service.rs index 98f8abad4..8e695fc7f 100644 --- a/nodedb/src/control/security/auth_lease/service.rs +++ b/nodedb/src/control/security/auth_lease/service.rs @@ -18,6 +18,11 @@ //! meanwhile answers `NotLeader`, so no lease it grants and no barrier it //! releases outlives its term unseen. //! +//! Each reply has a deadline: the leader budget of [`LeaseTiming`] after the +//! request arrived, or after a barrier's own wait. It ends a reply margin +//! before the holder's read timeout, so the reply arrives before the holder +//! gives up. +//! //! A leader that is the only voter of the metadata group pins its own lease //! (see [`super::table`]). No other node can lead the group then, so every //! barrier releases here, and each one waits for this node's coverage. @@ -103,8 +108,13 @@ impl LeaderLeaseService { edit(table); } - /// Make the table of `term` current and load its floors. - async fn table_ready(&self, state: &SharedState, term: u64) -> crate::Result<()> { + /// Make the table of `term` current and load its floors by `reply_by`. + async fn table_ready( + &self, + state: &SharedState, + term: u64, + reply_by: Instant, + ) -> crate::Result<()> { { let mut table = self.table(); if table.as_ref().is_none_or(|t| t.term() != term) { @@ -122,7 +132,7 @@ impl LeaderLeaseService { { return Ok(()); } - let floors = self.load_floors(state).await?; + let floors = self.load_floors(state, reply_by).await?; if let Some(table) = self.table().as_mut().filter(|t| t.term() == term) { table.load_floors(&floors); } @@ -130,13 +140,16 @@ impl LeaderLeaseService { Ok(()) } - /// The floors a new term starts from. - async fn load_floors(&self, state: &SharedState) -> crate::Result> { - let timeout = self.timing.lease; - let metadata = confirmed_read_index(state, METADATA_GROUP_ID, timeout).await?; + /// The floors a new term starts from, loaded by `reply_by`. + async fn load_floors( + &self, + state: &SharedState, + reply_by: Instant, + ) -> crate::Result> { + let metadata = confirmed_read_index(state, METADATA_GROUP_ID, time_left(reply_by)).await?; // The source set comes from tree definitions, which live in the // metadata group. Apply it through the read index first. - wait_applied(state, METADATA_GROUP_ID, metadata, timeout).await?; + wait_applied(state, METADATA_GROUP_ID, metadata, time_left(reply_by)).await?; apply_committed_tree_defs(state).await; let mut groups: HashSet = HashSet::new(); @@ -144,6 +157,7 @@ impl LeaderLeaseService { groups.insert(group_of_vshard(state, vshard_id)?); } groups.insert(SEQUENCER_GROUP_ID); + let timeout = time_left(reply_by); let group_floors = try_join_all(groups.into_iter().map(|group_id| async move { confirmed_read_index(state, group_id, timeout) .await @@ -160,27 +174,25 @@ impl LeaderLeaseService { } /// Whether this node still leads the metadata group in `term`, confirmed - /// against a quorum now. - async fn confirm_leadership(&self, state: &SharedState, term: u64) -> bool { - confirmed_read_index(state, METADATA_GROUP_ID, self.timing.lease) + /// against a quorum by `reply_by`. + async fn confirm_leadership(&self, state: &SharedState, term: u64, reply_by: Instant) -> bool { + confirmed_read_index(state, METADATA_GROUP_ID, time_left(reply_by)) .await .is_ok() && leading_term(state) == Some(term) } fn not_leader_renewal(state: Option<&SharedState>) -> AuthLeaseRenewResponse { + let (leader_hint, term) = state.map_or((None, 0), leader_hint); AuthLeaseRenewResponse { - outcome: AuthLeaseRenewOutcome::NotLeader { - leader_hint: state.and_then(leader_hint), - }, + outcome: AuthLeaseRenewOutcome::NotLeader { leader_hint, term }, } } fn not_leader_barrier(state: Option<&SharedState>) -> AuthBarrierResponse { + let (leader_hint, term) = state.map_or((None, 0), leader_hint); AuthBarrierResponse { - outcome: AuthBarrierOutcome::NotLeader { - leader_hint: state.and_then(leader_hint), - }, + outcome: AuthBarrierOutcome::NotLeader { leader_hint, term }, } } @@ -192,7 +204,8 @@ impl LeaderLeaseService { let Some(term) = leading_term(&state) else { return Self::not_leader_renewal(Some(&state)); }; - if let Err(error) = self.table_ready(&state, term).await { + let reply_by = Instant::now() + self.timing.leader_budget(); + if let Err(error) = self.table_ready(&state, term, reply_by).await { if self .withheld_warnings .should_warn(req.node_id, Instant::now()) @@ -251,7 +264,7 @@ impl LeaderLeaseService { self.changed.notify_waiters(); let outcome = match decision { RenewDecision::Withheld => AuthLeaseRenewOutcome::Withheld, - RenewDecision::Granted if self.confirm_leadership(&state, term).await => { + RenewDecision::Granted if self.confirm_leadership(&state, term, reply_by).await => { AuthLeaseRenewOutcome::Granted { lease_ms: u64::try_from(self.timing.lease.as_millis()).unwrap_or(u64::MAX), } @@ -268,6 +281,7 @@ impl LeaderLeaseService { pub async fn hold_barrier(&self, req: AuthBarrierRequest) -> AuthBarrierResponse { let started = Instant::now(); let deadline = started + Duration::from_millis(req.timeout_ms); + let reply_by = deadline + self.timing.leader_budget(); let timed_out = || AuthBarrierResponse { outcome: AuthBarrierOutcome::Timeout { waited_ms: u64::try_from(started.elapsed().as_millis()).unwrap_or(u64::MAX), @@ -280,7 +294,7 @@ impl LeaderLeaseService { let Some(term) = leading_term(&state) else { return Self::not_leader_barrier(Some(&state)); }; - if let Err(error) = self.table_ready(&state, term).await { + if let Err(error) = self.table_ready(&state, term, reply_by).await { tracing::debug!(%error, "authorization barrier: floors not loaded yet"); if Instant::now() >= deadline { return timed_out(); @@ -305,7 +319,7 @@ impl LeaderLeaseService { }; match status { BarrierState::Released => { - return if self.confirm_leadership(&state, term).await { + return if self.confirm_leadership(&state, term, reply_by).await { AuthBarrierResponse { outcome: AuthBarrierOutcome::Released, } @@ -334,6 +348,11 @@ impl LeaderLeaseService { } } +/// Time left until `reply_by`, zero once it passed. +fn time_left(reply_by: Instant) -> Duration { + reply_by.saturating_duration_since(Instant::now()) +} + #[async_trait::async_trait] impl nodedb_cluster::AuthLeaseService for LeaderLeaseService { async fn renew(&self, req: AuthLeaseRenewRequest) -> AuthLeaseRenewResponse { diff --git a/nodedb/src/control/security/auth_lease/status.rs b/nodedb/src/control/security/auth_lease/status.rs index bec4b3a68..00727d625 100644 --- a/nodedb/src/control/security/auth_lease/status.rs +++ b/nodedb/src/control/security/auth_lease/status.rs @@ -80,23 +80,66 @@ pub(crate) fn lease_status_of( } } -/// Wait until this node can plan permission-checked statements, polling -/// every `poll`, or refuse once `timeout` passes. -pub async fn await_planning_admitted( - state: &SharedState, - timeout: Duration, - poll: Duration, -) -> crate::Result<()> { - let deadline = Instant::now() + timeout; - while !lease_status(state, Instant::now()).admits_planning() { - if Instant::now() >= deadline { - return Err(crate::Error::AuthorizationStateBehind { - detail: format!("no authorization lease was granted within {timeout:?}"), - }); +/// Wait until this node can plan permission-checked statements, or refuse +/// once `timeout` passes. +pub async fn await_planning_admitted(state: &SharedState, timeout: Duration) -> crate::Result<()> { + if planning_admitted_within(state, timeout).await { + return Ok(()); + } + Err(crate::Error::AuthorizationStateBehind { + detail: format!( + "no authorization lease was granted within {timeout:?}; last renewal: {}", + state.authorization_fence.holder().last_attempt() + ), + }) +} + +/// Wait at most `within` until this node can plan permission-checked +/// statements, and return whether it can. Each renewal round wakes the +/// wait, so it ends as soon as a round grants a lease. +pub async fn planning_admitted_within(state: &SharedState, within: Duration) -> bool { + admitted_within( + &state.authorization_fence, + || sole_voter_term(state), + state.node_id, + within, + ) + .await +} + +/// [`planning_admitted_within`] for node `node_id` behind `fence`. +pub(crate) async fn admitted_within( + fence: &AuthorizationFence, + sole_voter_term: impl Fn() -> Option, + node_id: u64, + within: Duration, +) -> bool { + let admitted = + || lease_status_of(fence, &sole_voter_term, node_id, Instant::now()).admits_planning(); + if admitted() { + return true; + } + let deadline = tokio::time::Instant::now() + within; + // Subscribed before the next check, so no round between the two is missed. + let mut rounds = fence.holder().subscribe_rounds(); + loop { + if admitted() { + return true; + } + if tokio::time::Instant::now() >= deadline { + return false; + } + tokio::select! { + changed = rounds.changed() => { + // The holder lives as long as `fence`, so its sender does too. + // Without one, only the deadline is left to wait for. + if changed.is_err() { + tokio::time::sleep_until(deadline).await; + } + } + _ = tokio::time::sleep_until(deadline) => {} } - tokio::time::sleep(poll).await; } - Ok(()) } /// Append the lease gauges. A single node holds no lease and emits none. @@ -289,6 +332,45 @@ mod tests { assert!(!status.admits_planning()); } + /// A lease that lapsed while a round ran long admits planning as soon as + /// that round grants it, well before the wait's bound. + #[tokio::test] + async fn a_lapsed_lease_admits_once_the_next_round_grants() { + let fence = Arc::new(fence()); + assert!(fence.install_timing(timing())); + let within = Duration::from_secs(10); + let renewer = Arc::clone(&fence); + let round = tokio::spawn(async move { + tokio::time::sleep(Duration::from_millis(50)).await; + renewer + .holder() + .install(Instant::now() + Duration::from_secs(5)); + renewer.holder().record_attempt( + crate::control::security::auth_lease::RenewAttempt::Granted { leader_id: NODE }, + ); + }); + + let started = Instant::now(); + assert!(admitted_within(&fence, || None, NODE, within).await); + assert!(started.elapsed() < within / 2, "woken by the round"); + round.await.expect("round"); + } + + /// Without a granting round the wait ends at the grace and refuses. + #[tokio::test] + async fn a_lapsed_lease_refuses_after_the_grace() { + let fence = fence(); + assert!(fence.install_timing(timing())); + let grace = Duration::from_millis(50); + fence + .holder() + .record_attempt(crate::control::security::auth_lease::RenewAttempt::NoLeader); + + let started = Instant::now(); + assert!(!admitted_within(&fence, || None, NODE, grace).await); + assert!(started.elapsed() >= grace); + } + /// A pin belongs to the leadership term that granted it. #[test] fn a_pin_from_an_earlier_term_does_not_count() { diff --git a/nodedb/src/control/security/auth_lease/timing.rs b/nodedb/src/control/security/auth_lease/timing.rs index 5cebf13ad..4b3515fc9 100644 --- a/nodedb/src/control/security/auth_lease/timing.rs +++ b/nodedb/src/control/security/auth_lease/timing.rs @@ -11,6 +11,9 @@ //! - **Renewal:** every heartbeat interval. A holder that covers a change //! renews within one heartbeat, so an acknowledgement waits about one //! heartbeat when every node is healthy. +//! - **Reply margin:** the heartbeat interval. The leader answers a renewal +//! or a barrier this much before the holder's read timeout ends. The reply +//! then reaches the holder before the holder gives up on it. use std::time::{Duration, Instant}; @@ -23,6 +26,9 @@ pub struct LeaseTiming { pub skew_margin: Duration, /// How often a holder renews. pub renew_every: Duration, + /// How much earlier the leader's reply deadline ends than the holder's + /// read timeout. + pub reply_margin: Duration, } impl LeaseTiming { @@ -43,9 +49,34 @@ impl LeaseTiming { lease: election_timeout_min, skew_margin: heartbeat_interval, renew_every: heartbeat_interval, + reply_margin: heartbeat_interval, }) } + /// The longest the leader spends confirming its leadership and loading + /// floors for one reply. It ends a reply margin inside the lease. + /// [`Self::from_raft`] keeps it non-zero: the heartbeat is below the lease. + pub fn leader_budget(&self) -> Duration { + self.lease.saturating_sub(self.reply_margin) + } + + /// How long a planning path waits for the next renewal once it found + /// the lease lapsed: one renewal interval for the next round to start, + /// plus the reply margin for it to finish. + pub fn lapse_grace(&self) -> Duration { + self.renew_every + self.reply_margin + } + + /// The holder's read timeout for an RPC whose leader handler first waits + /// up to `wait`, then spends up to [`Self::leader_budget`]. + /// + /// It exceeds the handler's own deadline by the reply margin. A slow + /// reply then still arrives, and a read timeout means the leader is + /// unreachable rather than busy. + pub fn rpc_read_timeout(&self, wait: Duration) -> Duration { + wait + self.leader_budget() + self.reply_margin + } + /// When a lease the leader granted for `granted`, requested at `sent_at` /// on the holder's clock, ends on the holder's clock. /// @@ -69,6 +100,24 @@ mod tests { assert_eq!(timing.lease, Duration::from_millis(150)); assert_eq!(timing.skew_margin, Duration::from_millis(50)); assert_eq!(timing.renew_every, Duration::from_millis(50)); + assert_eq!(timing.reply_margin, Duration::from_millis(50)); + assert_eq!(timing.lapse_grace(), Duration::from_millis(100)); + } + + /// The leader's reply deadline ends strictly inside the holder's read + /// timeout, for a renewal and for a barrier that waits first. + #[test] + fn the_leader_answers_before_the_holder_times_out() { + let timing = LeaseTiming::from_raft(Duration::from_millis(500), Duration::from_millis(50)) + .expect("timing"); + assert_eq!(timing.leader_budget(), Duration::from_millis(450)); + assert_eq!(timing.rpc_read_timeout(Duration::ZERO), timing.lease); + for wait in [Duration::ZERO, Duration::from_secs(3)] { + let leader_done = wait + timing.leader_budget(); + let holder_gives_up = timing.rpc_read_timeout(wait); + assert!(leader_done < holder_gives_up); + assert_eq!(holder_gives_up - leader_done, timing.reply_margin); + } } #[test] diff --git a/nodedb/src/control/security/catalog/arrays.rs b/nodedb/src/control/security/catalog/arrays.rs index 3c83da595..a7e1ee83a 100644 --- a/nodedb/src/control/security/catalog/arrays.rs +++ b/nodedb/src/control/security/catalog/arrays.rs @@ -5,7 +5,7 @@ //! Mirrors the `triggers.rs` shape: typed put/get/delete plus a bulk //! loader for startup. Keyed by `name` (already globally scoped in the //! `ArrayCatalogEntry` via `ArrayId`'s tenant field — a second-level -//! tenant prefix would only duplicate that information). +//! tenant prefix only duplicates that information). use redb::{ReadableDatabase, ReadableTable}; @@ -114,6 +114,29 @@ impl SystemCatalog { tenant_id: nodedb_types::TenantId, database_id: nodedb_types::DatabaseId, name: &str, + ) -> crate::Result { + self.remove_array_row(tenant_id, database_id, name, None) + } + + /// Atomically delete the array row under `from` and move its surrogate + /// bindings to `to`. The cells keep their surrogates across a MOVE + /// TENANT rekey. + pub fn move_array_surrogates_in_database( + &self, + tenant_id: nodedb_types::TenantId, + from: nodedb_types::DatabaseId, + to: nodedb_types::DatabaseId, + name: &str, + ) -> crate::Result { + self.remove_array_row(tenant_id, from, name, Some(to)) + } + + fn remove_array_row( + &self, + tenant_id: nodedb_types::TenantId, + database_id: nodedb_types::DatabaseId, + name: &str, + move_to: Option, ) -> crate::Result { let write_txn = self .db @@ -160,6 +183,15 @@ impl SystemCatalog { reverse .remove((db_id, tid, name, surrogate)) .map_err(|e| catalog_err("remove surrogate_pk_rev", e))?; + if let Some(to) = move_to { + let to = to.as_u64(); + forward + .insert((to, tid, name, pk.as_slice()), surrogate) + .map_err(|e| catalog_err("move surrogate_pk", e))?; + reverse + .insert((to, tid, name, surrogate), pk.as_slice()) + .map_err(|e| catalog_err("move surrogate_pk_rev", e))?; + } } } write_txn @@ -242,6 +274,8 @@ mod tests { prefix_bits: 8, audit_retain_ms: None, minimum_audit_retain_ms: None, + modification_hlc: nodedb_types::Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, } } @@ -321,4 +355,47 @@ mod tests { .unwrap() ); } + + /// A MOVE TENANT rekey carries every surrogate binding to the target + /// database and leaves none under the source. + #[test] + fn move_carries_surrogates_to_the_target_database() { + let catalog = catalog(); + let (from, to) = (DatabaseId::new(3), DatabaseId::new(4)); + catalog.put_array(&entry(1, from, "cells", 1)).unwrap(); + let key = |db| nodedb_types::CollectionKey::from_bare(db, "cells"); + catalog + .put_surrogate( + key(from), + TenantId::new(1), + b"coord:1", + nodedb_types::Surrogate::new(9), + ) + .unwrap(); + + assert!( + catalog + .move_array_surrogates_in_database(TenantId::new(1), from, to, "cells") + .unwrap() + ); + + assert!( + catalog + .get_array_in_database(TenantId::new(1), from, "cells") + .unwrap() + .is_none() + ); + assert!( + catalog + .get_surrogate_for_pk(key(from), TenantId::new(1), b"coord:1") + .unwrap() + .is_none() + ); + assert_eq!( + catalog + .get_surrogate_for_pk(key(to), TenantId::new(1), b"coord:1") + .unwrap(), + Some(nodedb_types::Surrogate::new(9)) + ); + } } diff --git a/nodedb/src/control/security/catalog/backup_schedule_marks.rs b/nodedb/src/control/security/catalog/backup_schedule_marks.rs new file mode 100644 index 000000000..d7ea8d13f --- /dev/null +++ b/nodedb/src/control/security/catalog/backup_schedule_marks.rs @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `_system.backup_schedule_marks`: how far each scheduled backup has run. +//! +//! The metadata group replicates every mark, so every node holds the same +//! rows. The node that runs scheduled backups reads them to pick the due +//! minute, and a node that takes that role later reads the same rows. +//! +//! A row is keyed by the job and its config incarnation. The incarnation is +//! a fingerprint of the schedule's database, target and cron. A changed +//! schedule is another incarnation, so an old mark never suppresses a run +//! of the new schedule. + +use redb::{ReadableDatabase, ReadableTable, TableError}; + +use super::types::{SystemCatalog, catalog_err}; + +/// Redb table: mark key -> MessagePack [`StoredBackupScheduleMark`]. +pub(super) const BACKUP_SCHEDULE_MARKS: redb::TableDefinition<&str, &[u8]> = + redb::TableDefinition::new("_system.backup_schedule_marks"); + +/// Every scheduled minute at or below `through_minute` is settled for one +/// schedule incarnation: its backup completed, or it lies before the +/// schedule was armed. +#[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct StoredBackupScheduleMark { + /// The job name, `backup::`. + pub job: String, + /// The config incarnation of the schedule. + pub incarnation: u64, + /// Unix minute. Scheduled minutes at or below it never run again. + pub through_minute: u64, +} + +impl StoredBackupScheduleMark { + fn key(job: &str, incarnation: u64) -> String { + format!("{incarnation:016x}:{job}") + } +} + +impl SystemCatalog { + /// Write `mark` unless the stored mark of its incarnation already reaches + /// as far. A replayed or late mark never moves a mark back. Returns + /// whether the row changed. + pub fn raise_backup_schedule_mark( + &self, + mark: &StoredBackupScheduleMark, + ) -> crate::Result { + let key = StoredBackupScheduleMark::key(&mark.job, mark.incarnation); + let bytes = zerompk::to_msgpack_vec(mark) + .map_err(|e| catalog_err("encode backup schedule mark", e))?; + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("backup schedule mark write txn", e))?; + let raised = { + let mut table = txn + .open_table(BACKUP_SCHEDULE_MARKS) + .map_err(|e| catalog_err("open backup schedule marks", e))?; + let current = match table + .get(key.as_str()) + .map_err(|e| catalog_err("read backup schedule mark", e))? + { + Some(value) => Some( + zerompk::from_msgpack::(value.value()) + .map_err(|e| catalog_err("decode backup schedule mark", e))?, + ), + None => None, + }; + if current.is_some_and(|current| current.through_minute >= mark.through_minute) { + false + } else { + table + .insert(key.as_str(), bytes.as_slice()) + .map_err(|e| catalog_err("insert backup schedule mark", e))?; + true + } + }; + txn.commit() + .map_err(|e| catalog_err("backup schedule mark commit", e))?; + Ok(raised) + } + + /// The mark of `job` at `incarnation`, if one was written. + pub fn backup_schedule_mark( + &self, + job: &str, + incarnation: u64, + ) -> crate::Result> { + let txn = self + .db + .begin_read() + .map_err(|e| catalog_err("backup schedule mark read txn", e))?; + let table = match txn.open_table(BACKUP_SCHEDULE_MARKS) { + Ok(table) => table, + Err(TableError::TableDoesNotExist(_)) => return Ok(None), + Err(e) => return Err(catalog_err("open backup schedule marks", e)), + }; + let key = StoredBackupScheduleMark::key(job, incarnation); + match table + .get(key.as_str()) + .map_err(|e| catalog_err("read backup schedule mark", e))? + { + Some(value) => zerompk::from_msgpack(value.value()) + .map(Some) + .map_err(|e| catalog_err("decode backup schedule mark", e)), + None => Ok(None), + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn mark(incarnation: u64, through_minute: u64) -> StoredBackupScheduleMark { + StoredBackupScheduleMark { + job: "backup:sales:s3://b/p".into(), + incarnation, + through_minute, + } + } + + #[test] + fn a_mark_only_moves_forward_within_its_incarnation() { + let catalog = SystemCatalog::open_in_memory().unwrap(); + let job = "backup:sales:s3://b/p"; + assert_eq!(catalog.backup_schedule_mark(job, 7).unwrap(), None); + assert!(catalog.raise_backup_schedule_mark(&mark(7, 100)).unwrap()); + assert!(!catalog.raise_backup_schedule_mark(&mark(7, 90)).unwrap()); + assert!(!catalog.raise_backup_schedule_mark(&mark(7, 100)).unwrap()); + assert_eq!( + catalog.backup_schedule_mark(job, 7).unwrap(), + Some(mark(7, 100)) + ); + assert!(catalog.raise_backup_schedule_mark(&mark(7, 101)).unwrap()); + + // Another incarnation has its own mark. + assert_eq!(catalog.backup_schedule_mark(job, 8).unwrap(), None); + assert!(catalog.raise_backup_schedule_mark(&mark(8, 5)).unwrap()); + assert_eq!( + catalog + .backup_schedule_mark(job, 7) + .unwrap() + .map(|m| m.through_minute), + Some(101) + ); + } +} diff --git a/nodedb/src/control/security/catalog/bootstrap_tables.rs b/nodedb/src/control/security/catalog/bootstrap_tables.rs index 86f5ae1fa..35b42b2f8 100644 --- a/nodedb/src/control/security/catalog/bootstrap_tables.rs +++ b/nodedb/src/control/security/catalog/bootstrap_tables.rs @@ -8,7 +8,7 @@ //! sequence of `open_table` calls in `open`, makes it structurally //! impossible to declare a table and read it in production code without //! also creating it at startup: a table that is consulted on a fresh -//! catalog but missing from this list would fail the first reader with +//! catalog but missing from this list fails the first reader with //! `Table '…' does not exist`. The `bootstrap_creates_every_registered_table` //! unit test re-opens every entry read-only against a freshly-bootstrapped //! catalog to keep the registry and the init path in lockstep. @@ -18,7 +18,7 @@ //! `_system.surrogate_pk_rev_v2` — superseded key layouts) are intentionally //! absent: they are read only by the idempotent migration path, which already //! tolerates their absence, and materialising empty copies on every fresh -//! server would misrepresent the catalog state. +//! server misrepresents the catalog state. use redb::{ReadTransaction, TableError, WriteTransaction}; @@ -61,6 +61,7 @@ pub(super) const BOOTSTRAP_TABLES: &[BootstrapTable] = bootstrap_tables![ "permissions" => PERMISSIONS, "owners" => OWNERS, "tenants" => TENANTS, + "tenant_id_hwm" => super::tenant_id_hwm::TENANT_ID_HWM, "audit_log" => AUDIT_LOG, "blacklist" => BLACKLIST, "auth_users" => AUTH_USERS, @@ -76,9 +77,20 @@ pub(super) const BOOTSTRAP_TABLES: &[BootstrapTable] = bootstrap_tables![ "metadata" => METADATA, "wal_tombstones" => WAL_TOMBSTONES, "tenant_group_marks" => super::tenant_group_marks::TENANT_GROUP_MARKS, + "tenant_group_restore_marks" => super::tenant_group_marks::TENANT_GROUP_RESTORE_MARKS, "calvin_applied" => super::calvin_applied::CALVIN_APPLIED, + "calvin_base" => super::calvin_base::CALVIN_BASE, + "calvin_sequencer_install" => super::calvin_base::CALVIN_SEQUENCER_INSTALL, "l2_cleanup_queue" => L2_CLEANUP_QUEUE, "pending_reclaim" => PENDING_RECLAIM, + // ── Metadata-group host state ── + "metadata_leases" => super::metadata_host::leases::METADATA_LEASES, + "metadata_drains" => super::metadata_host::drains::METADATA_DRAINS, + "metadata_host_scalars" => super::metadata_host::scalars::METADATA_HOST_SCALARS, + "pending_ddl" => super::metadata_host::ddl::PENDING_DDL, + "pending_history_compaction" => super::pending_history_compaction::PENDING_HISTORY_COMPACTION, + "crdt_compaction_points" => super::crdt_compaction_points::CRDT_COMPACTION_POINTS, + "pending_leave_cleanup" => super::pending_leave_cleanup::PENDING_LEAVE_CLEANUP, "column_stats" => COLUMN_STATS, "vector_model_metadata" => VECTOR_MODEL_METADATA, "vector_index_params" => VECTOR_INDEX_PARAMS, @@ -121,6 +133,7 @@ pub(super) const BOOTSTRAP_TABLES: &[BootstrapTable] = bootstrap_tables![ "alert_rules" => ALERT_RULES, "topics_ep" => TOPICS_EP, "topic_messages" => TOPIC_MESSAGES, + "topic_publish_marks" => super::topic_publish_marks::TOPIC_PUBLISH_MARKS, "streaming_mvs" => STREAMING_MVS, // ── Database catalog + quotas ── "databases" => DATABASES, @@ -134,6 +147,12 @@ pub(super) const BOOTSTRAP_TABLES: &[BootstrapTable] = bootstrap_tables![ "clone_tombstones" => CLONE_TOMBSTONES, "clone_kv_tombstones" => CLONE_KV_TOMBSTONES, "clone_lineage" => CLONE_LINEAGE, + "clone_source_drains" => super::clone_source_drains::CLONE_SOURCE_DRAINS, + // ── Cluster restore points ── + "restore_points" => super::restore_points::RESTORE_POINTS, + "cut_floors" => super::cut_floors::CUT_FLOORS, + // ── Scheduled backups ── + "backup_schedule_marks" => super::backup_schedule_marks::BACKUP_SCHEDULE_MARKS, "mirror_collection_map" => MIRROR_COLLECTION_MAP, "mirror_lag" => MIRROR_LAG, // ── Tenant relocation ── diff --git a/nodedb/src/control/security/catalog/calvin_applied.rs b/nodedb/src/control/security/catalog/calvin_applied.rs index db18de836..18ca14c20 100644 --- a/nodedb/src/control/security/catalog/calvin_applied.rs +++ b/nodedb/src/control/security/catalog/calvin_applied.rs @@ -101,6 +101,39 @@ impl SystemCatalog { })) } + /// Replace the saved state of each vShard in `states` with the given one. + /// + /// A data-group snapshot install replaces the vShard's storage with the + /// leader's, so the applied state it brought replaces this node's: a + /// position this node applied and the snapshot does not hold is no + /// longer applied here. + pub fn replace_calvin_applied(&self, states: &[StoredCalvinApplied]) -> crate::Result<()> { + if states.is_empty() { + return Ok(()); + } + let write_txn = self + .db + .begin_write() + .map_err(|e| catalog_err("replace_calvin_applied txn", e))?; + { + let mut table = write_txn + .open_table(CALVIN_APPLIED) + .map_err(|e| catalog_err("open calvin_applied", e))?; + for state in states { + let tail = encode_tail(&state.tail)?; + table + .insert( + state.vshard_id, + (state.fully_applied_epoch, tail.as_slice()), + ) + .map_err(|e| catalog_err("insert calvin_applied", e))?; + } + } + write_txn + .commit() + .map_err(|e| catalog_err("commit calvin_applied", e)) + } + /// Save `states` in one transaction. Each is merged with the state /// already saved for its vShard, so the saved state only grows. pub fn save_calvin_applied(&self, states: Vec) -> crate::Result<()> { diff --git a/nodedb/src/control/security/catalog/calvin_base.rs b/nodedb/src/control/security/catalog/calvin_base.rs new file mode 100644 index 000000000..3ca1ed9f0 --- /dev/null +++ b/nodedb/src/control/security/catalog/calvin_base.rs @@ -0,0 +1,258 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Persistent Calvin base of each vShard, backing `_system.calvin_base`. +//! +//! A replica's state of a vShard is its data group's log through some index +//! plus every Calvin input the sequencer log holds for the vShard. The +//! sequencer compacts its log on its own schedule, so a replica whose Calvin +//! state is not whole up to the first index the log still holds can never +//! catch it up from the log. The base records where this node's Calvin state +//! of each vShard is whole from: +//! +//! - [`CalvinBase::Snapshot`]: a data-group snapshot brought every input +//! sequenced at or below `through`, and no scheduler has started from it. +//! - [`CalvinBase::Kept`]: a scheduler started from a whole base and caught +//! up from sequencer index `from`. While it runs, the sequencer log keeps +//! every input it has not made durable, so the state stays whole across a +//! restart. A sequencer snapshot installed here at or above `from` skips +//! entries the scheduler never received, and the base is whole no more. +//! +//! A vShard with no row has no base: this node never held its Calvin state. +//! +//! `_system.calvin_sequencer_install` holds the index of the last sequencer +//! snapshot this node installed. It is durable before the sequencer log +//! adopts the snapshot. + +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::types::{SystemCatalog, catalog_err}; + +/// Table: `vshard_id` -> `(kind, index)`. `kind` is [`KIND_SNAPSHOT`] or +/// [`KIND_KEPT`]; `index` is the snapshot's `through` or the kept `from`. +pub(super) const CALVIN_BASE: TableDefinition = + TableDefinition::new("_system.calvin_base"); + +/// Table: [`SEQUENCER_INSTALL_KEY`] -> the index of the last sequencer +/// snapshot this node installed. +pub(super) const CALVIN_SEQUENCER_INSTALL: TableDefinition = + TableDefinition::new("_system.calvin_sequencer_install"); + +const SEQUENCER_INSTALL_KEY: u8 = 0; + +const KIND_SNAPSHOT: u8 = 1; +const KIND_KEPT: u8 = 2; + +/// Where this node's Calvin state of one vShard is whole from. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum CalvinBase { + /// A snapshot brought every input sequenced at or below `through`. + Snapshot { through: u64 }, + /// A scheduler that caught up from sequencer index `from` keeps the + /// state whole, while no sequencer snapshot installs at or above it. + Kept { from: u64 }, +} + +impl CalvinBase { + /// Whether a scheduler that catches up from `sequencer_first`, the first + /// index the sequencer log holds, misses no input: the base holds every + /// input below it. + pub fn reaches(base: Option, sequencer_first: u64) -> bool { + match base { + Some(Self::Kept { .. }) => true, + Some(Self::Snapshot { through }) => sequencer_first <= through.saturating_add(1), + None => sequencer_first <= 1, + } + } + + /// The first sequencer index a scheduler started from this base must + /// replay: the sequencer log must keep it until the scheduler starts. + /// `None` for a kept base, whose scheduler keeps its own range. + pub fn replay_start(base: Option) -> Option { + match base { + Some(Self::Kept { .. }) => None, + Some(Self::Snapshot { through }) => Some(through.saturating_add(1)), + None => Some(1), + } + } + + /// Whether this is a kept base. + pub fn is_kept(base: Option) -> bool { + matches!(base, Some(Self::Kept { .. })) + } + + /// The base still whole after the last sequencer snapshot this node + /// installed, at `installed` (`0` for none): a kept base whose scheduler + /// caught up from at or below it lost the entries the install skipped. + pub fn after_install(base: Option, installed: u64) -> Option { + match base { + Some(Self::Kept { from }) if installed >= from => None, + base => base, + } + } + + fn encode(self) -> (u8, u64) { + match self { + Self::Snapshot { through } => (KIND_SNAPSHOT, through), + Self::Kept { from } => (KIND_KEPT, from), + } + } + + fn decode(kind: u8, index: u64) -> crate::Result { + match kind { + KIND_SNAPSHOT => Ok(Self::Snapshot { through: index }), + KIND_KEPT => Ok(Self::Kept { from: index }), + other => Err(crate::Error::Internal { + detail: format!("calvin base: unknown kind {other} in _system.calvin_base"), + }), + } + } +} + +impl SystemCatalog { + /// Every saved Calvin base, by vShard. + pub fn load_calvin_bases(&self) -> crate::Result> { + let read_txn = self + .db + .begin_read() + .map_err(|e| catalog_err("load_calvin_bases read txn", e))?; + let table = read_txn + .open_table(CALVIN_BASE) + .map_err(|e| catalog_err("open calvin_base", e))?; + let mut out = Vec::new(); + for row in table + .iter() + .map_err(|e| catalog_err("iter calvin_base", e))? + { + let (vshard, value) = row.map_err(|e| catalog_err("read calvin_base", e))?; + let (kind, through) = value.value(); + out.push((vshard.value(), CalvinBase::decode(kind, through)?)); + } + Ok(out) + } + + /// Save each `(vshard_id, base)` in one transaction: `Some` as the + /// vShard's Calvin base, `None` removes its row. + pub fn save_calvin_bases(&self, bases: &[(u32, Option)]) -> crate::Result<()> { + if bases.is_empty() { + return Ok(()); + } + let write_txn = self + .db + .begin_write() + .map_err(|e| catalog_err("save_calvin_bases txn", e))?; + { + let mut table = write_txn + .open_table(CALVIN_BASE) + .map_err(|e| catalog_err("open calvin_base", e))?; + for &(vshard_id, base) in bases { + match base { + Some(base) => { + table + .insert(vshard_id, base.encode()) + .map_err(|e| catalog_err("insert calvin_base", e))?; + } + None => { + table + .remove(vshard_id) + .map_err(|e| catalog_err("remove calvin_base", e))?; + } + } + } + } + write_txn + .commit() + .map_err(|e| catalog_err("commit calvin_base", e)) + } + + /// The index of the last sequencer snapshot this node installed, `0` + /// for none. + pub fn load_calvin_sequencer_install(&self) -> crate::Result { + let read_txn = self + .db + .begin_read() + .map_err(|e| catalog_err("load_calvin_sequencer_install read txn", e))?; + let table = read_txn + .open_table(CALVIN_SEQUENCER_INSTALL) + .map_err(|e| catalog_err("open calvin_sequencer_install", e))?; + let index = table + .get(SEQUENCER_INSTALL_KEY) + .map_err(|e| catalog_err("read calvin_sequencer_install", e))? + .map_or(0, |value| value.value()); + Ok(index) + } + + /// Save `index` as the last sequencer snapshot this node installed. + pub fn save_calvin_sequencer_install(&self, index: u64) -> crate::Result<()> { + let write_txn = self + .db + .begin_write() + .map_err(|e| catalog_err("save_calvin_sequencer_install txn", e))?; + { + let mut table = write_txn + .open_table(CALVIN_SEQUENCER_INSTALL) + .map_err(|e| catalog_err("open calvin_sequencer_install", e))?; + table + .insert(SEQUENCER_INSTALL_KEY, index) + .map_err(|e| catalog_err("insert calvin_sequencer_install", e))?; + } + write_txn + .commit() + .map_err(|e| catalog_err("commit calvin_sequencer_install", e)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn bases_save_reload_and_remove() { + let dir = tempfile::tempdir().expect("tempdir"); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).expect("catalog"); + assert!(catalog.load_calvin_bases().expect("load").is_empty()); + + catalog + .save_calvin_bases(&[ + (3, Some(CalvinBase::Snapshot { through: 40 })), + (5, Some(CalvinBase::Kept { from: 7 })), + ]) + .expect("save"); + catalog + .save_calvin_bases(&[(3, Some(CalvinBase::Kept { from: 41 })), (5, None)]) + .expect("save and remove"); + assert_eq!( + catalog.load_calvin_bases().expect("load"), + vec![(3, CalvinBase::Kept { from: 41 })] + ); + + assert_eq!(catalog.load_calvin_sequencer_install().expect("load"), 0); + catalog.save_calvin_sequencer_install(88).expect("save"); + assert_eq!(catalog.load_calvin_sequencer_install().expect("load"), 88); + } + + /// A sequencer snapshot installed at or above a kept base's `from` + /// skipped entries its scheduler never received. + #[test] + fn a_sequencer_install_ends_the_kept_bases_it_skips() { + let kept = Some(CalvinBase::Kept { from: 10 }); + assert_eq!(CalvinBase::after_install(kept, 9), kept); + assert_eq!(CalvinBase::after_install(kept, 10), None); + let snapshot = Some(CalvinBase::Snapshot { through: 5 }); + assert_eq!(CalvinBase::after_install(snapshot, 80), snapshot); + } + + #[test] + fn a_base_reaches_the_log_only_without_a_gap() { + assert!(CalvinBase::reaches(None, 1), "the whole log is still held"); + assert!(!CalvinBase::reaches(None, 2)); + let snapshot = Some(CalvinBase::Snapshot { through: 40 }); + assert!(CalvinBase::reaches(snapshot, 41)); + assert!(CalvinBase::reaches(snapshot, 12)); + assert!(!CalvinBase::reaches(snapshot, 42), "index 41 is gone"); + let kept = Some(CalvinBase::Kept { from: 3 }); + assert!(CalvinBase::reaches(kept, 9_000)); + assert_eq!(CalvinBase::replay_start(snapshot), Some(41)); + assert_eq!(CalvinBase::replay_start(None), Some(1)); + assert_eq!(CalvinBase::replay_start(kept), None); + } +} diff --git a/nodedb/src/control/security/catalog/change_streams.rs b/nodedb/src/control/security/catalog/change_streams.rs index 0d446b810..bd69b4a9c 100644 --- a/nodedb/src/control/security/catalog/change_streams.rs +++ b/nodedb/src/control/security/catalog/change_streams.rs @@ -102,16 +102,12 @@ impl SystemCatalog { } fn stream_key(database_id: DatabaseId, tenant_id: u64, name: &str) -> String { - let mut encoded = String::with_capacity(name.len() * 2); - for byte in name.as_bytes() { - use std::fmt::Write; - let _ = write!(&mut encoded, "{byte:02x}"); - } format!( - "v2/{:016x}/{:016x}/{:08x}/{encoded}", + "v2/{:016x}/{:016x}/{:08x}/{}", database_id.as_u64(), tenant_id, - name.len() + name.len(), + hex::encode(name) ) } @@ -142,6 +138,7 @@ mod tests { owner: "admin".into(), created_at: 1000, subscriber_roles: Vec::new(), + modification_hlc: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/security/catalog/checkpoints.rs b/nodedb/src/control/security/catalog/checkpoints.rs index f1e026bd4..3deeaa777 100644 --- a/nodedb/src/control/security/catalog/checkpoints.rs +++ b/nodedb/src/control/security/catalog/checkpoints.rs @@ -200,7 +200,7 @@ impl SystemCatalog { Ok(keys) } - /// Count the checkpoints a `delete_checkpoints_before` call would remove. + /// Count the checkpoints a `delete_checkpoints_before` call removes. /// /// The leader reports this to the client before proposing the range delete. pub fn count_checkpoints_before( @@ -303,6 +303,49 @@ impl SystemCatalog { write_txn.commit().map_err(|e| catalog_err("commit", e))?; Ok(copied.len()) } + + /// Remove every checkpoint of one collection, across its documents. + /// Returns how many rows went. + pub fn delete_checkpoints_for_collection( + &self, + database_id: u64, + tenant_id: u64, + collection: &str, + ) -> crate::Result { + let lower = format!("{database_id}:{tenant_id}:{collection}:"); + let upper = format!("{database_id}:{tenant_id}:{collection};"); + let write_txn = self + .db + .begin_write() + .map_err(|e| catalog_err("write txn", e))?; + let removed = { + let mut table = write_txn + .open_table(CHECKPOINTS) + .map_err(|e| catalog_err("open checkpoints", e))?; + // A collection name can hold ':', so the range only yields + // candidates: the stored collection must match exactly. + let mut keys = Vec::new(); + for entry in table + .range(lower.as_str()..upper.as_str()) + .map_err(|e| catalog_err("range scan checkpoints", e))? + { + let (key, value) = entry.map_err(|e| catalog_err("iterate checkpoints", e))?; + let record: CheckpointRecord = zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("deserialize checkpoint", e))?; + if record.collection == collection { + keys.push(key.value().to_string()); + } + } + for key in &keys { + table + .remove(key.as_str()) + .map_err(|e| catalog_err("remove checkpoint", e))?; + } + keys.len() + }; + write_txn.commit().map_err(|e| catalog_err("commit", e))?; + Ok(removed) + } } #[cfg(test)] @@ -417,4 +460,21 @@ mod tests { 3 ); } + + #[test] + fn delete_for_collection_removes_only_that_collection() { + let (_dir, catalog) = make_catalog(); + catalog.put_checkpoint(&record(2, "a", 1)).unwrap(); + catalog.put_checkpoint(&record(2, "b", 2)).unwrap(); + catalog.put_checkpoint(&record(3, "a", 1)).unwrap(); + + assert_eq!( + catalog + .delete_checkpoints_for_collection(2, TENANT, COLLECTION) + .unwrap(), + 2 + ); + assert!(catalog.list_checkpoints(doc(2), 0).unwrap().is_empty()); + assert_eq!(catalog.list_checkpoints(doc(3), 0).unwrap().len(), 1); + } } diff --git a/nodedb/src/control/security/catalog/clone_catalog.rs b/nodedb/src/control/security/catalog/clone_catalog.rs index d22959945..6e73bd413 100644 --- a/nodedb/src/control/security/catalog/clone_catalog.rs +++ b/nodedb/src/control/security/catalog/clone_catalog.rs @@ -312,6 +312,31 @@ impl SystemCatalog { Ok(set) } + /// Delete every KV tombstone of `target_collection_key`. + pub fn delete_all_kv_clone_tombstones_for_collection( + &self, + target_collection_key: &str, + ) -> crate::Result { + let keys = self.list_kv_clone_tombstones(target_collection_key)?; + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("clone_kv_tombstones reap begin_write", e))?; + { + let mut table = txn + .open_table(CLONE_KV_TOMBSTONES) + .map_err(|e| catalog_err("open clone_kv_tombstones reap", e))?; + for kv_key in &keys { + table + .remove((target_collection_key, kv_key.as_str())) + .map_err(|e| catalog_err("remove clone_kv_tombstones reap", e))?; + } + } + txn.commit() + .map_err(|e| catalog_err("clone_kv_tombstones reap commit", e))?; + Ok(keys.len() as u64) + } + // ── clone_lineage ───────────────────────────────────────────────────────── /// Return the list of child database ids that are clones of `source_db_id`. @@ -337,6 +362,24 @@ impl SystemCatalog { } } + /// The clone children of `source_db_id` whose database still exists. + /// + /// A dropped child keeps its lineage edge, because the edge marks its + /// clone as applied for metadata-log replay. Every user-visible listing + /// and every dependency check reads this instead of the raw edge. + pub fn get_live_clone_children( + &self, + source_db_id: DatabaseId, + ) -> crate::Result> { + let mut live = Vec::new(); + for child in self.get_clone_children(source_db_id)? { + if self.get_database(child)?.is_some() { + live.push(child); + } + } + Ok(live) + } + /// Add `child_db_id` as a clone child of `source_db_id`. /// Idempotent — safe to call multiple times with the same child. pub fn add_clone_child( @@ -528,4 +571,33 @@ mod tests { // Idempotent — removing absent child is fine. cat.remove_clone_child(db(1), db(2)).unwrap(); } + + /// A dropped child keeps its edge but leaves the live listing. + #[test] + fn live_children_skip_a_dropped_child() { + use crate::control::security::catalog::database_types::{ + DatabaseDescriptor, DatabaseStatus, + }; + + let (_dir, cat) = open_catalog(); + for child in [db(2), db(3)] { + cat.put_database(&DatabaseDescriptor { + id: child, + name: format!("child_{}", child.as_u64()), + status: DatabaseStatus::Active, + created_at_lsn: 0, + quota_ref: 0, + parent_clone: None, + mirror_origin: None, + audit_dml: nodedb_types::AuditDmlMode::None, + idle_session_timeout_secs: 0, + }) + .unwrap(); + cat.add_clone_child(db(1), child).unwrap(); + } + cat.delete_database(db(2)).unwrap(); + + assert_eq!(cat.get_clone_children(db(1)).unwrap(), vec![db(2), db(3)]); + assert_eq!(cat.get_live_clone_children(db(1)).unwrap(), vec![db(3)]); + } } diff --git a/nodedb/src/control/security/catalog/clone_source_drains.rs b/nodedb/src/control/security/catalog/clone_source_drains.rs new file mode 100644 index 000000000..788f0be22 --- /dev/null +++ b/nodedb/src/control/security/catalog/clone_source_drains.rs @@ -0,0 +1,138 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `_system.clone_source_drains`: one row per clone collection whose +//! materializer drains its KV source. +//! +//! The row is written through the metadata log before the drain starts and +//! removed after it ends, so every node, and every later singleton worker, +//! knows which source drains a materialization owns. Recovery ends a drain +//! whose rows all name clone collections that no longer need a copy. + +use nodedb_types::DatabaseId; +use redb::{ReadableDatabase, ReadableTable}; + +use super::types::{SystemCatalog, catalog_err}; + +/// Key: `(clone database, tenant, clone collection)`. Value: zerompk +/// [`CloneSourceDrain`]. +pub const CLONE_SOURCE_DRAINS: redb::TableDefinition<(u64, u64, &str), &[u8]> = + redb::TableDefinition::new("_system.clone_source_drains"); + +/// One materialization's claim on its source collection's drain. +#[derive(Debug, Clone, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct CloneSourceDrain { + pub clone_database: u64, + pub tenant_id: u64, + pub clone_collection: String, + pub source_database: u64, + pub source_collection: String, +} + +impl CloneSourceDrain { + pub fn clone_database_id(&self) -> DatabaseId { + DatabaseId::new(self.clone_database) + } + + pub fn source_database_id(&self) -> DatabaseId { + DatabaseId::new(self.source_database) + } +} + +impl SystemCatalog { + /// Record a materialization's drain claim. Overwrites the same claim. + pub fn put_clone_source_drain(&self, row: &CloneSourceDrain) -> crate::Result<()> { + let bytes = zerompk::to_msgpack_vec(row) + .map_err(|e| catalog_err("serialize clone_source_drain", e))?; + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("clone_source_drains write txn", e))?; + { + let mut table = txn + .open_table(CLONE_SOURCE_DRAINS) + .map_err(|e| catalog_err("open clone_source_drains", e))?; + table + .insert( + ( + row.clone_database, + row.tenant_id, + row.clone_collection.as_str(), + ), + bytes.as_slice(), + ) + .map_err(|e| catalog_err("insert clone_source_drains", e))?; + } + txn.commit() + .map_err(|e| catalog_err("clone_source_drains commit", e)) + } + + /// Remove a drain claim. Idempotent. + pub fn delete_clone_source_drain( + &self, + clone_database: u64, + tenant_id: u64, + clone_collection: &str, + ) -> crate::Result<()> { + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("clone_source_drains delete txn", e))?; + { + let mut table = txn + .open_table(CLONE_SOURCE_DRAINS) + .map_err(|e| catalog_err("open clone_source_drains", e))?; + table + .remove((clone_database, tenant_id, clone_collection)) + .map_err(|e| catalog_err("remove clone_source_drains", e))?; + } + txn.commit() + .map_err(|e| catalog_err("clone_source_drains delete commit", e)) + } + + /// Every drain claim. + pub fn list_clone_source_drains(&self) -> crate::Result> { + let txn = self + .db + .begin_read() + .map_err(|e| catalog_err("clone_source_drains read txn", e))?; + let table = txn + .open_table(CLONE_SOURCE_DRAINS) + .map_err(|e| catalog_err("open clone_source_drains", e))?; + let mut rows = Vec::new(); + for entry in table + .iter() + .map_err(|e| catalog_err("iter clone_source_drains", e))? + { + let (_, value) = entry.map_err(|e| catalog_err("read clone_source_drains", e))?; + rows.push( + zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("deser clone_source_drain", e))?, + ); + } + Ok(rows) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn claims_roundtrip_and_delete_idempotently() { + let dir = tempfile::tempdir().unwrap(); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).unwrap(); + let row = CloneSourceDrain { + clone_database: 1025, + tenant_id: 1, + clone_collection: "kv".into(), + source_database: 1024, + source_collection: "kv".into(), + }; + catalog.put_clone_source_drain(&row).unwrap(); + catalog.put_clone_source_drain(&row).unwrap(); + assert_eq!(catalog.list_clone_source_drains().unwrap(), vec![row]); + catalog.delete_clone_source_drain(1025, 1, "kv").unwrap(); + catalog.delete_clone_source_drain(1025, 1, "kv").unwrap(); + assert!(catalog.list_clone_source_drains().unwrap().is_empty()); + } +} diff --git a/nodedb/src/control/security/catalog/collection.rs b/nodedb/src/control/security/catalog/collection.rs index d33e64fe2..30c5edbc1 100644 --- a/nodedb/src/control/security/catalog/collection.rs +++ b/nodedb/src/control/security/catalog/collection.rs @@ -88,6 +88,13 @@ pub struct StoredCollection { /// applier at commit time. Strictly monotonic per descriptor. #[msgpack(default)] pub modification_hlc: Hlc, + /// The collection's identity across ALTER, UNDROP and MOVE TENANT: the + /// `modification_hlc` of the put that created it. A drop and a same-name + /// create start a new one. Every replicated write carries the incarnation + /// it was planned against, and a replica applies it only while the + /// collection still holds it. + #[msgpack(default)] + pub incarnation: Hlc, /// Optional field type declarations. Empty = schemaless. #[msgpack(default)] pub fields: Vec<(String, String)>, @@ -134,7 +141,7 @@ pub struct StoredCollection { /// Type guard field constraints for schemaless collections. #[msgpack(default)] pub type_guards: Vec, - /// General CHECK constraints (Control Plane enforcement, may contain subqueries). + /// General CHECK constraints (Control Plane enforcement, can contain subqueries). #[msgpack(default)] pub check_constraints: Vec, /// Materialized sum definitions. @@ -189,8 +196,8 @@ pub struct StoredCollection { /// Defaults to `CollectionHomed` on deserialization so catalog entries /// written before this field was added continue to behave correctly — /// every pre-existing collection is collection-homed, making `default` - /// the safe zero-migration value (unlike a surrogate where a default would - /// be wrong). No wire-version bump is required. + /// the safe zero-migration value (unlike a surrogate where a default is + /// wrong). No wire-version bump is required. #[msgpack(default)] pub partition_strategy: nodedb_types::PartitionStrategy, /// Best-effort estimate of this collection's on-core data size in @@ -248,7 +255,7 @@ pub struct StoredCollection { pub has_implicit_edges: bool, /// Declared `PRIMARY KEY` column name from the `CREATE COLLECTION` / - /// `CREATE TABLE` column list, when one was present. May differ from + /// `CREATE TABLE` column list, when one was present. Can differ from /// the built-in `id` field on schemaless document collections (which /// otherwise always uses `id` as its document key). `None` means no /// PRIMARY KEY was declared and the engine falls back to its default @@ -294,6 +301,7 @@ impl StoredCollection { constraint_version: 0, crdt_signing_required: false, modification_hlc: Hlc::ZERO, + incarnation: Hlc::ZERO, fields: Vec::new(), field_defs: Vec::new(), event_defs: Vec::new(), @@ -333,6 +341,21 @@ impl StoredCollection { } } + /// A collection a test fixture writes straight to the catalog, stamped as + /// the proposer stamps a create: a fresh incarnation and descriptor + /// version 1. Each call names a new incarnation. + #[cfg(test)] + pub fn stamped_for_test(tenant_id: u64, name: &str, owner: &str) -> Self { + static CLOCK: std::sync::OnceLock = std::sync::OnceLock::new(); + let hlc = CLOCK.get_or_init(nodedb_types::HlcClock::new).now(); + Self { + descriptor_version: 1, + modification_hlc: hlc, + incarnation: hlc, + ..Self::new(tenant_id, name, owner) + } + } + /// Parse the timeseries config JSON, if present. pub fn get_timeseries_config(&self) -> Option { self.timeseries_config diff --git a/nodedb/src/control/security/catalog/collection_incarnation.rs b/nodedb/src/control/security/catalog/collection_incarnation.rs new file mode 100644 index 000000000..76008b8bb --- /dev/null +++ b/nodedb/src/control/security/catalog/collection_incarnation.rs @@ -0,0 +1,110 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Every committed collection row names an incarnation. +//! +//! Every writer stamps a collection row before it reaches the catalog: the +//! proposer stamps each entry, with or without a metadata group, and every +//! apply that rewrites a row carries the row's own incarnation. The catalog +//! refuses a row with a zero incarnation, so none is ever committed. + +use nodedb_types::{DatabaseId, Hlc}; + +use super::types::{StoredCollection, SystemCatalog}; + +fn key_of(database_id: DatabaseId, collection: &str) -> nodedb_types::CollectionKey<'_> { + nodedb_types::CollectionKey::from_qualified_str(database_id, collection) + .unwrap_or_else(|_| nodedb_types::CollectionKey::from_bare(database_id, collection)) +} + +impl SystemCatalog { + /// The incarnation this node's catalog holds for `collection`, as a plan + /// names it in `database_id`, with a transaction's buffered DDL merged in. + /// `Hlc::ZERO` when no row exists. + pub fn incarnation_of( + &self, + database_id: DatabaseId, + tenant_id: u64, + collection: &str, + ) -> crate::Result { + let key = key_of(database_id, collection); + Ok(self + .get_collection(key.database_id(), tenant_id, key.name())? + .map_or(Hlc::ZERO, |row| row.incarnation)) + } + + /// Whether the committed row of `collection`, as a plan names it in + /// `database_id`, still holds `incarnation`. + pub fn holds_incarnation( + &self, + database_id: DatabaseId, + tenant_id: u64, + collection: &str, + incarnation: Hlc, + ) -> crate::Result { + let key = key_of(database_id, collection); + Ok(self + .get_committed_collection(key.database_id(), tenant_id, key.name())? + .is_some_and(|row| row.incarnation == incarnation)) + } +} + +/// Refuse `coll` when it carries no incarnation. +pub(super) fn require_incarnation( + database_id: DatabaseId, + coll: &StoredCollection, +) -> crate::Result<()> { + if coll.incarnation == Hlc::ZERO { + return Err(crate::Error::CollectionUnstamped { + database_id: database_id.as_u64(), + tenant_id: coll.tenant_id, + name: coll.name.clone(), + }); + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// The catalog refuses an unstamped row by name and writes nothing. A + /// stamped row commits with its own incarnation. + #[test] + fn an_unstamped_row_is_refused() { + let tmp = tempfile::tempdir().expect("tmpdir"); + let catalog = SystemCatalog::open(&tmp.path().join("system.redb")).expect("open"); + + let unstamped = StoredCollection::new(1, "orders", "admin"); + for refused in [ + catalog.put_collection(DatabaseId::DEFAULT, &unstamped), + catalog + .put_collection_if_absent(DatabaseId::DEFAULT, &unstamped) + .map(|_| ()), + ] { + assert!( + matches!( + refused, + Err(crate::Error::CollectionUnstamped { ref name, .. }) if name == "orders" + ), + "{refused:?}" + ); + } + assert!( + catalog + .get_committed_collection(DatabaseId::DEFAULT, 1, "orders") + .expect("read") + .is_none() + ); + + let stamped = StoredCollection::stamped_for_test(1, "orders", "admin"); + catalog + .put_collection(DatabaseId::DEFAULT, &stamped) + .expect("a stamped row commits"); + let row = catalog + .get_committed_collection(DatabaseId::DEFAULT, 1, "orders") + .expect("read") + .expect("row"); + assert_eq!(row.incarnation, stamped.incarnation); + assert_eq!(row.descriptor_version, 1); + } +} diff --git a/nodedb/src/control/security/catalog/collections.rs b/nodedb/src/control/security/catalog/collections.rs index 0cfe2bbd9..76b86f11d 100644 --- a/nodedb/src/control/security/catalog/collections.rs +++ b/nodedb/src/control/security/catalog/collections.rs @@ -3,24 +3,12 @@ //! Collection metadata operations for the system catalog. //! //! The storage key is `(database_id: u64, "{tenant_id}:{name}")`. -//! The inner key preserves the legacy `"{tenant_id}:{name}"` encoding so -//! existing catalog-resolver call sites only need to add `database_id` -//! (always `DatabaseId::DEFAULT` in Tier 1). -//! -//! ## Migration -//! -//! On first boot against pre-migration storage, `migrate_collections()` -//! reads all rows from `_system.collections` (the legacy bare-String-keyed -//! table) and rewrites them under `_system.collections_v2` with -//! `DatabaseId::DEFAULT` prepended. The migration is idempotent: if the -//! v2 table already has rows the migration skips; if the legacy table is -//! absent or empty it is also a no-op. use nodedb_types::DatabaseId; use nodedb_types::columnar::schema::{TS_SYSTEM, TS_VALID_FROM, TS_VALID_UNTIL}; -use redb::{ReadableDatabase, ReadableTable, ReadableTableMetadata}; +use redb::{ReadableDatabase, ReadableTable}; -use super::types::{COLLECTIONS, COLLECTIONS_LEGACY, StoredCollection, SystemCatalog, catalog_err}; +use super::types::{COLLECTIONS, StoredCollection, SystemCatalog, catalog_err}; /// Union inferred ingest fields into a collection's schema projection. /// @@ -28,18 +16,25 @@ use super::types::{COLLECTIONS, COLLECTIONS_LEGACY, StoredCollection, SystemCata /// appended. Bitemporal collections always expose their reserved BIGINT /// fields exactly once. Returns `true` when the projection changed. /// +/// `time_column` is the column an ingest inferred for the row time. A +/// timeseries collection whose declared time key names one of its fields +/// stores the row time in that field. Its projection therefore never gains +/// the inferred time column. Every other collection takes it as a field. +/// /// Deliberately pure: a collection descriptor is replicated catalog state, so /// the merged record has to reach storage through the replicated metadata path /// (see `catalog_entry::persist_collection`) rather than a local write. Mutating -/// the persisted record in place would leave this node's copy at descriptor +/// the persisted record in place leaves this node's copy at descriptor /// version N no longer byte-equal to the replicated entry at version N, and -/// replaying that entry after a restart would wedge the metadata applier. +/// replaying that entry after a restart wedges the metadata applier. pub fn merge_inferred_fields( collection: &mut StoredCollection, + time_column: Option<&(String, String)>, inferred_fields: &[(String, String)], ) -> bool { + let time_column = time_column.filter(|_| !declares_time_key_field(collection)); let mut changed = false; - for (field, field_type) in inferred_fields { + for (field, field_type) in time_column.into_iter().chain(inferred_fields) { // Reserved bitemporal columns are schema-owned. Never let an ingest // projection supply their type or add a duplicate; normalization below // owns them entirely. @@ -87,8 +82,23 @@ pub fn merge_inferred_fields( changed } +/// Whether a timeseries collection's declared time key names one of its +/// fields. The Data Plane then builds its memtable from that declaration and +/// writes each row's time into that field. +fn declares_time_key_field(collection: &StoredCollection) -> bool { + let nodedb_types::CollectionType::Columnar(nodedb_types::ColumnarProfile::Timeseries { + time_key, + .. + }) = &collection.collection_type + else { + return false; + }; + collection.fields.iter().any(|(field, _)| field == time_key) +} + impl SystemCatalog { - /// Store a collection record. + /// Store a collection record. A record with no incarnation is refused + /// with [`crate::Error::CollectionUnstamped`]. pub fn put_collection( &self, database_id: DatabaseId, @@ -104,6 +114,7 @@ impl SystemCatalog { "injected collection write failure", )); } + super::collection_incarnation::require_incarnation(database_id, coll)?; let inner_key = format!("{}:{}", coll.tenant_id, coll.name); let bytes = zerompk::to_msgpack_vec(coll).map_err(|e| catalog_err("serialize collection", e))?; @@ -134,6 +145,7 @@ impl SystemCatalog { database_id: DatabaseId, coll: &StoredCollection, ) -> crate::Result { + super::collection_incarnation::require_incarnation(database_id, coll)?; let inner_key = format!("{}:{}", coll.tenant_id, coll.name); let bytes = zerompk::to_msgpack_vec(coll).map_err(|e| catalog_err("serialize collection", e))?; @@ -306,7 +318,7 @@ impl SystemCatalog { /// Committed-only read, bypassing the transaction DDL overlay. The /// descriptor stamper reads through this: a version derived from an - /// uncommitted overlay row would stamp two entries at the same version. + /// uncommitted overlay row stamps two entries at the same version. pub fn get_committed_collection( &self, database_id: DatabaseId, @@ -331,85 +343,6 @@ impl SystemCatalog { Err(e) => Err(catalog_err("get collection", e)), } } - - /// Idempotent migration: reads all rows from the legacy - /// `_system.collections` table (bare `"{tenant_id}:{name}"` key) and - /// rewrites them under `_system.collections_v2` with - /// `DatabaseId::DEFAULT` prepended. - /// - /// Safe to call on: - /// - Fresh boot: legacy table absent or empty → no-op. - /// - Pre-migration boot: legacy rows present → migrated to v2. - /// - Already-migrated boot: v2 rows already exist → no-op (skips if - /// v2 table is non-empty; any duplicate put is an idempotent - /// overwrite because the key+value are identical). - pub fn migrate_collections(&self) -> crate::Result<()> { - // Check legacy table existence and emptiness. - let legacy_rows: Vec<(String, Vec)> = { - let txn = self - .db - .begin_read() - .map_err(|e| catalog_err("migrate_collections read txn", e))?; - match txn.open_table(COLLECTIONS_LEGACY) { - Ok(table) => { - let iter = table - .iter() - .map_err(|e| catalog_err("migrate_collections iter", e))?; - let mut rows = Vec::new(); - for row in iter { - let (k, v) = row.map_err(|e| catalog_err("migrate_collections row", e))?; - rows.push((k.value().to_string(), v.value().to_vec())); - } - rows - } - Err(_) => Vec::new(), // legacy table does not exist yet - } - }; - - if legacy_rows.is_empty() { - return Ok(()); - } - - // Check if v2 is already populated (already-migrated boot). - let v2_empty = { - let txn = self - .db - .begin_read() - .map_err(|e| catalog_err("migrate_collections v2 check txn", e))?; - match txn.open_table(COLLECTIONS) { - Ok(table) => table - .is_empty() - .map_err(|e| catalog_err("migrate_collections v2 is_empty", e))?, - Err(_) => true, - } - }; - if !v2_empty { - // Already migrated — idempotent no-op. - return Ok(()); - } - - // Write all legacy rows into v2 under DatabaseId::DEFAULT. - let db_id = DatabaseId::DEFAULT.as_u64(); - let write_txn = self - .db - .begin_write() - .map_err(|e| catalog_err("migrate_collections write txn", e))?; - { - let mut table = write_txn - .open_table(COLLECTIONS) - .map_err(|e| catalog_err("migrate_collections open v2", e))?; - for (inner_key, bytes) in &legacy_rows { - table - .insert((db_id, inner_key.as_str()), bytes.as_slice()) - .map_err(|e| catalog_err("migrate_collections insert v2", e))?; - } - } - write_txn - .commit() - .map_err(|e| catalog_err("migrate_collections commit", e))?; - // The migration wrote rows outside `put_collection`. - self.reload_event_definitions() - } } /// Body of [`SystemCatalog::load_collections_for_tenant`], over an already-open @@ -479,7 +412,7 @@ mod tests { use nodedb_types::CollectionType; use super::*; - use crate::control::security::catalog::types::{COLLECTIONS_LEGACY, StoredCollection}; + use crate::control::security::catalog::types::StoredCollection; fn open_catalog() -> (tempfile::TempDir, SystemCatalog) { let dir = tempfile::tempdir().unwrap(); @@ -488,7 +421,7 @@ mod tests { } fn make_coll(tenant_id: u64, name: &str) -> StoredCollection { - let mut c = StoredCollection::new(tenant_id, name, "admin"); + let mut c = StoredCollection::stamped_for_test(tenant_id, name, "admin"); c.collection_type = CollectionType::document(); c } @@ -602,15 +535,18 @@ mod tests { assert!(merge_inferred_fields( &mut coll, + None, &[("first".to_owned(), "BIGINT".to_owned())] )); assert!(merge_inferred_fields( &mut coll, + None, &[("second".to_owned(), "FLOAT".to_owned())] )); // A known name never re-types an existing column, and reports no change. assert!(!merge_inferred_fields( &mut coll, + None, &[("first".to_owned(), "BOOLEAN".to_owned())] )); @@ -624,6 +560,103 @@ mod tests { ); } + fn ilp_time_column() -> (String, String) { + ("timestamp".to_owned(), "TIMESTAMP".to_owned()) + } + + /// A collection declared with `ts BIGINT TIME_KEY` stores the ILP line + /// time in `ts`. The inferred `timestamp` column must not reach its + /// projection, or every flush proposes a new descriptor version. + #[test] + fn merge_skips_the_inferred_time_column_for_a_declared_time_key() { + let mut coll = make_coll(1, "crash_ilp_ts_bulk"); + coll.collection_type = CollectionType::timeseries("ts", "1h"); + coll.fields = vec![ + ("ts".to_owned(), "BIGINT TIME_KEY".to_owned()), + ("value".to_owned(), "BIGINT".to_owned()), + ]; + + assert!( + !merge_inferred_fields( + &mut coll, + Some(&ilp_time_column()), + &[("value".to_owned(), "BIGINT".to_owned())] + ), + "an ILP batch carrying only declared fields changes nothing" + ); + assert!( + merge_inferred_fields( + &mut coll, + Some(&ilp_time_column()), + &[ + ("host".to_owned(), "VARCHAR".to_owned()), + ("value".to_owned(), "BIGINT".to_owned()), + ("load".to_owned(), "FLOAT".to_owned()), + ] + ), + "a new tag and a new field still reach the projection" + ); + assert_eq!( + coll.fields, + vec![ + ("ts".to_owned(), "BIGINT TIME_KEY".to_owned()), + ("value".to_owned(), "BIGINT".to_owned()), + ("host".to_owned(), "VARCHAR".to_owned()), + ("load".to_owned(), "FLOAT".to_owned()), + ] + ); + } + + /// A field literally called `timestamp` is a field, not the line time. + /// It reaches the projection of a collection with a declared time key. + #[test] + fn merge_keeps_a_field_named_timestamp_for_a_declared_time_key() { + let mut coll = make_coll(1, "metrics"); + coll.collection_type = CollectionType::timeseries("ts", "1h"); + coll.fields = vec![("ts".to_owned(), "TIMESTAMP".to_owned())]; + + assert!(merge_inferred_fields( + &mut coll, + Some(&ilp_time_column()), + &[("timestamp".to_owned(), "BIGINT".to_owned())] + )); + assert_eq!( + coll.fields, + vec![ + ("ts".to_owned(), "TIMESTAMP".to_owned()), + ("timestamp".to_owned(), "BIGINT".to_owned()), + ] + ); + } + + /// With no declared time key among its fields, the Data Plane infers the + /// schema and stores the line time under the inferred name. The + /// projection follows it. + #[test] + fn merge_adds_the_inferred_time_column_without_a_declared_time_key() { + let mut undeclared = make_coll(1, "events"); + assert!(merge_inferred_fields( + &mut undeclared, + Some(&ilp_time_column()), + &[("value".to_owned(), "FLOAT".to_owned())] + )); + assert_eq!( + undeclared.fields, + vec![ilp_time_column(), ("value".to_owned(), "FLOAT".to_owned())] + ); + + // A time key absent from the field list resolves no declaration, so + // the Data Plane infers here too. + let mut unresolved = make_coll(1, "cpu"); + unresolved.collection_type = CollectionType::timeseries("ts", "1h"); + assert!(merge_inferred_fields( + &mut unresolved, + Some(&ilp_time_column()), + &[] + )); + assert_eq!(unresolved.fields, vec![ilp_time_column()]); + } + #[test] fn merge_inferred_fields_adds_bitemporal_reserved_fields_once() { let mut coll = make_coll(1, "audit"); @@ -632,9 +665,10 @@ mod tests { assert!(merge_inferred_fields( &mut coll, + None, &[("value".to_owned(), "FLOAT".to_owned())] )); - assert!(!merge_inferred_fields(&mut coll, &[])); + assert!(!merge_inferred_fields(&mut coll, None, &[])); for reserved in [TS_SYSTEM, TS_VALID_FROM, TS_VALID_UNTIL] { assert_eq!( coll.fields @@ -658,6 +692,7 @@ mod tests { assert!(merge_inferred_fields( &mut coll, + None, &[ (TS_VALID_FROM.to_owned(), "VARCHAR".to_owned()), (TS_VALID_UNTIL.to_owned(), "BOOLEAN".to_owned()), @@ -697,69 +732,4 @@ mod tests { .unwrap(); assert_eq!(t2.len(), 1); } - - // ── Migration tests ────────────────────────────────────────────────── - - /// Helper: write a legacy (bare string key) row directly so we can - /// test the migration without going through put_collection. - fn insert_legacy_row(cat: &SystemCatalog, coll: &StoredCollection) { - let key = format!("{}:{}", coll.tenant_id, coll.name); - let bytes = zerompk::to_msgpack_vec(coll).unwrap(); - let txn = cat.db.begin_write().unwrap(); - { - let mut table = txn.open_table(COLLECTIONS_LEGACY).unwrap(); - table.insert(key.as_str(), bytes.as_slice()).unwrap(); - } - txn.commit().unwrap(); - } - - #[test] - fn fresh_boot_migration_is_noop() { - let (_dir, cat) = open_catalog(); - // No legacy rows → migration is a no-op. - cat.migrate_collections().unwrap(); - assert!( - cat.load_all_collections(DatabaseId::DEFAULT) - .unwrap() - .is_empty() - ); - } - - #[test] - fn pre_migration_boot_migrates_all_rows() { - let (_dir, cat) = open_catalog(); - let coll1 = make_coll(1, "widgets"); - let coll2 = make_coll(2, "orders"); - insert_legacy_row(&cat, &coll1); - insert_legacy_row(&cat, &coll2); - - cat.migrate_collections().unwrap(); - - let w = cat - .get_collection(DatabaseId::DEFAULT, 1, "widgets") - .unwrap(); - assert!(w.is_some(), "widgets must be accessible after migration"); - let o = cat - .get_collection(DatabaseId::DEFAULT, 2, "orders") - .unwrap(); - assert!(o.is_some(), "orders must be accessible after migration"); - } - - #[test] - fn already_migrated_boot_is_idempotent() { - let (_dir, cat) = open_catalog(); - // Write a v2 row directly (simulating already-migrated). - cat.put_collection(DatabaseId::DEFAULT, &make_coll(1, "existing")) - .unwrap(); - - // Also insert a legacy row that would conflict if re-migrated. - let coll_legacy = make_coll(1, "existing"); - insert_legacy_row(&cat, &coll_legacy); - - // Migration should be a no-op (v2 non-empty). - cat.migrate_collections().unwrap(); - - let all = cat.load_all_collections(DatabaseId::DEFAULT).unwrap(); - assert_eq!(all.len(), 1, "should still be 1 row, not duplicated"); - } } diff --git a/nodedb/src/control/security/catalog/column_stats.rs b/nodedb/src/control/security/catalog/column_stats.rs index 29f22fcad..6f224335f 100644 --- a/nodedb/src/control/security/catalog/column_stats.rs +++ b/nodedb/src/control/security/catalog/column_stats.rs @@ -133,6 +133,50 @@ impl SystemCatalog { } } +impl SystemCatalog { + /// Remove every statistics row of one collection. Returns how many went. + pub fn delete_column_stats_for_collection( + &self, + database_id: u64, + tenant_id: u64, + collection: &str, + ) -> crate::Result { + let prefix = format!("{database_id}:{tenant_id}:{collection}:"); + let upper = prefix_upper_bound(database_id, tenant_id, collection); + let write_txn = self + .db + .begin_write() + .map_err(|e| catalog_err("write txn", e))?; + let removed = { + let mut table = write_txn + .open_table(COLUMN_STATS) + .map_err(|e| catalog_err("open column_stats", e))?; + // A collection name can hold ':', so the range only yields + // candidates: the stored collection must match exactly. + let mut keys = Vec::new(); + for row in table + .range(prefix.as_str()..upper.as_str()) + .map_err(|e| catalog_err("range column_stats", e))? + { + let (key, value) = row.map_err(|e| catalog_err("scan column_stats", e))?; + let decoded: StoredColumnStats = zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("deser column_stats", e))?; + if decoded.collection == collection { + keys.push(key.value().to_string()); + } + } + for key in &keys { + table + .remove(key.as_str()) + .map_err(|e| catalog_err("remove column_stats", e))?; + } + keys.len() + }; + write_txn.commit().map_err(|e| catalog_err("commit", e))?; + Ok(removed) + } +} + fn stats_key(database_id: u64, tenant_id: u64, collection: &str, column: &str) -> String { format!("{database_id}:{tenant_id}:{collection}:{column}") } @@ -208,6 +252,28 @@ mod tests { assert_eq!(second[0].row_count, 42); } + #[test] + fn delete_for_collection_removes_only_that_collection() { + let (_dir, cat) = make_catalog(); + let mut sibling = sample(2, "email"); + sibling.collection = "users_archive".into(); + cat.put_column_stats_batch(&[sample(2, "email"), sample(2, "name"), sibling]) + .unwrap(); + cat.put_column_stats(&sample(3, "email")).unwrap(); + + assert_eq!( + cat.delete_column_stats_for_collection(2, 1, "users") + .unwrap(), + 2 + ); + assert!(cat.load_column_stats(2, 1, "users").unwrap().is_empty()); + assert_eq!( + cat.load_column_stats(2, 1, "users_archive").unwrap().len(), + 1 + ); + assert_eq!(cat.load_column_stats(3, 1, "users").unwrap().len(), 1); + } + #[test] fn a_batch_write_lands_every_column() { let (_dir, cat) = make_catalog(); diff --git a/nodedb/src/control/security/catalog/consumer_groups.rs b/nodedb/src/control/security/catalog/consumer_groups.rs index 25312616a..e526bd619 100644 --- a/nodedb/src/control/security/catalog/consumer_groups.rs +++ b/nodedb/src/control/security/catalog/consumer_groups.rs @@ -65,6 +65,33 @@ impl SystemCatalog { Ok(inserted) } + /// Read one committed consumer group by its durable identity. + pub fn get_consumer_group( + &self, + database_id: DatabaseId, + tenant_id: u64, + stream: &str, + group: &str, + ) -> crate::Result> { + let key = group_key(database_id, tenant_id, stream, group); + let read_txn = self + .db + .begin_read() + .map_err(|e| catalog_err("read txn", e))?; + let table = read_txn + .open_table(CONSUMER_GROUPS) + .map_err(|e| catalog_err("open consumer_groups", e))?; + let Some(value) = table + .get(key.as_str()) + .map_err(|e| catalog_err("get consumer_group", e))? + else { + return Ok(None); + }; + decode_consumer_group(value.value()) + .map(Some) + .ok_or_else(|| catalog_err("deser consumer_group", key)) + } + /// Delete a consumer group. pub fn delete_consumer_group( &self, @@ -187,6 +214,7 @@ impl From for ConsumerGroupDef { owner: legacy.owner, created_at: legacy.created_at, database_id: DatabaseId::DEFAULT, + modification_hlc: nodedb_types::Hlc::ZERO, } } } @@ -216,6 +244,7 @@ mod tests { owner: "admin".into(), created_at: 0, database_id, + modification_hlc: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/security/catalog/crdt_compaction_points.rs b/nodedb/src/control/security/catalog/crdt_compaction_points.rs new file mode 100644 index 000000000..303c14c4f --- /dev/null +++ b/nodedb/src/control/security/catalog/crdt_compaction_points.rs @@ -0,0 +1,144 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `_system.crdt_compaction_points`: the last CRDT history compaction each +//! collection committed. +//! +//! Applying `CompactHistory` writes the collection's row. The metadata group +//! replicates it, so a metadata image carries every collection's point. A +//! node that installs an image skips the `CompactHistory` entries it covers. +//! The install compares each point with the one it held before and owes a +//! compaction for every point that moved. Purging the collection removes its +//! row, so a later collection of the same name never inherits a point. +//! +//! Table: `{database_id}:{tenant_id}:{collection}` -> MessagePack row. + +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::types::{SystemCatalog, catalog_err}; + +pub(super) const CRDT_COMPACTION_POINTS: TableDefinition<&str, &[u8]> = + TableDefinition::new("_system.crdt_compaction_points"); + +/// The version one collection's CRDT history was last compacted to. +#[derive(zerompk::ToMessagePack, zerompk::FromMessagePack, Debug, Clone, PartialEq, Eq)] +#[msgpack(map, allow_unknown_fields)] +pub struct StoredCompactionPoint { + pub database_id: u64, + pub tenant_id: u64, + pub collection: String, + /// Loro version vector of the last committed `CompactHistory`. + pub target_version_json: String, +} + +fn point_key(database_id: u64, tenant_id: u64, collection: &str) -> String { + format!("{database_id}:{tenant_id}:{collection}") +} + +impl SystemCatalog { + /// Record `point` as its collection's compaction point, replacing any + /// earlier one. + pub fn put_compaction_point(&self, point: &StoredCompactionPoint) -> crate::Result<()> { + let bytes = zerompk::to_msgpack_vec(point) + .map_err(|e| catalog_err("encode crdt_compaction_points row", e))?; + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("crdt_compaction_points write txn", e))?; + { + let mut table = txn + .open_table(CRDT_COMPACTION_POINTS) + .map_err(|e| catalog_err("open crdt_compaction_points", e))?; + let key = point_key(point.database_id, point.tenant_id, &point.collection); + table + .insert(key.as_str(), bytes.as_slice()) + .map_err(|e| catalog_err("insert crdt_compaction_points row", e))?; + } + txn.commit() + .map_err(|e| catalog_err("commit crdt_compaction_points put", e)) + } + + /// Remove one collection's compaction point. Removing an absent row + /// succeeds. + pub fn delete_compaction_point( + &self, + database_id: u64, + tenant_id: u64, + collection: &str, + ) -> crate::Result<()> { + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("crdt_compaction_points write txn", e))?; + { + let mut table = txn + .open_table(CRDT_COMPACTION_POINTS) + .map_err(|e| catalog_err("open crdt_compaction_points", e))?; + let key = point_key(database_id, tenant_id, collection); + table + .remove(key.as_str()) + .map_err(|e| catalog_err("remove crdt_compaction_points row", e))?; + } + txn.commit() + .map_err(|e| catalog_err("commit crdt_compaction_points delete", e)) + } + + /// Every collection's compaction point. + pub fn load_compaction_points(&self) -> crate::Result> { + let txn = self + .db + .begin_read() + .map_err(|e| catalog_err("crdt_compaction_points read txn", e))?; + let table = txn + .open_table(CRDT_COMPACTION_POINTS) + .map_err(|e| catalog_err("open crdt_compaction_points", e))?; + let mut out = Vec::new(); + for item in table + .range(..) + .map_err(|e| catalog_err("range crdt_compaction_points", e))? + { + let (_, value) = item.map_err(|e| catalog_err("read crdt_compaction_points", e))?; + out.push( + zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("decode crdt_compaction_points row", e))?, + ); + } + Ok(out) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use tempfile::TempDir; + + fn point(collection: &str, target: &str) -> StoredCompactionPoint { + StoredCompactionPoint { + database_id: 3, + tenant_id: 1, + collection: collection.to_string(), + target_version_json: target.to_string(), + } + } + + #[test] + fn a_later_point_replaces_the_earlier_and_delete_removes_it() { + let tmp = TempDir::new().unwrap(); + let cat = SystemCatalog::open(&tmp.path().join("system.redb")).unwrap(); + cat.put_compaction_point(&point("docs", "{\"1\":4}")) + .unwrap(); + cat.put_compaction_point(&point("docs", "{\"1\":9}")) + .unwrap(); + cat.put_compaction_point(&point("notes", "{\"1\":2}")) + .unwrap(); + assert_eq!( + cat.load_compaction_points().unwrap(), + vec![point("docs", "{\"1\":9}"), point("notes", "{\"1\":2}")] + ); + cat.delete_compaction_point(3, 1, "docs").unwrap(); + cat.delete_compaction_point(3, 1, "docs").unwrap(); + assert_eq!( + cat.load_compaction_points().unwrap(), + vec![point("notes", "{\"1\":2}")] + ); + } +} diff --git a/nodedb/src/control/security/catalog/cut_floors.rs b/nodedb/src/control/security/catalog/cut_floors.rs new file mode 100644 index 000000000..d9808c9c5 --- /dev/null +++ b/nodedb/src/control/security/catalog/cut_floors.rs @@ -0,0 +1,173 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `_system.cut_floors`: the cut barriers each data group applied on this +//! node, by log index, with each barrier's watermark. +//! +//! Every entry a group's log places after a barrier records a commit HLC +//! above the barrier's watermark. After a restart the apply loop applies the +//! entries above the group's durable applied index again, and some follow a +//! barrier it applied before the restart. These rows give those barriers +//! back, whatever a checkpoint truncated from the WAL. The table is local: +//! every replica writes the same rows from the same log. + +use redb::{ReadableDatabase, ReadableTable, TableError}; + +use super::types::{SystemCatalog, catalog_err}; + +/// Redb table: group id -> MessagePack [`StoredFloors`]. +pub(super) const CUT_FLOORS: redb::TableDefinition = + redb::TableDefinition::new("_system.cut_floors"); + +/// One barrier a group applied. +#[derive(Debug, Clone, Copy, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct StoredBarrier { + /// The barrier's log index. + pub index: u64, + /// The barrier's watermark: HLC wall time in nanoseconds. + pub watermark: u64, +} + +#[derive(Debug, Clone, Default, zerompk::ToMessagePack, zerompk::FromMessagePack)] +struct StoredFloors { + /// In index order. + barriers: Vec, +} + +fn decode(bytes: &[u8]) -> crate::Result { + zerompk::from_msgpack(bytes).map_err(|e| catalog_err("decode cut floors", e)) +} + +impl SystemCatalog { + /// Record the barrier `barrier` group `group_id` applied. + /// + /// Entries apply in log order and a restart resumes above the durable + /// applied index `durable_applied`. So of the barriers at or below it, + /// only the one with the highest watermark still binds an entry: the + /// rest go. + pub fn put_cut_floor( + &self, + group_id: u64, + barrier: StoredBarrier, + durable_applied: u64, + ) -> crate::Result<()> { + crate::fail_point_err!("cut_floor::before_persist", |detail: String| { + crate::Error::Internal { + detail: format!("fail point: {detail}"), + } + }); + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("cut floor write txn", e))?; + { + let mut table = txn + .open_table(CUT_FLOORS) + .map_err(|e| catalog_err("open cut floors", e))?; + let mut floors = match table + .get(group_id) + .map_err(|e| catalog_err("get cut floors", e))? + { + Some(bytes) => decode(bytes.value())?, + None => StoredFloors::default(), + }; + floors.barriers.retain(|held| held.index != barrier.index); + floors.barriers.push(barrier); + floors.barriers.sort_by_key(|held| held.index); + let settled = floors + .barriers + .iter() + .filter(|held| held.index <= durable_applied) + .map(|held| held.watermark) + .max(); + if let Some(watermark) = settled { + let highest = floors + .barriers + .iter() + .filter(|held| held.index <= durable_applied) + .map(|held| held.index) + .max() + .unwrap_or(0); + floors.barriers.retain(|held| held.index > durable_applied); + floors.barriers.insert( + 0, + StoredBarrier { + index: highest, + watermark, + }, + ); + } + let bytes = zerompk::to_msgpack_vec(&floors) + .map_err(|e| catalog_err("encode cut floors", e))?; + table + .insert(group_id, bytes.as_slice()) + .map_err(|e| catalog_err("insert cut floors", e))?; + } + txn.commit().map_err(|e| catalog_err("cut floor commit", e)) + } + + /// Every recorded barrier, as `(group_id, barrier)`. + pub fn load_cut_floors(&self) -> crate::Result> { + let txn = self + .db + .begin_read() + .map_err(|e| catalog_err("cut floor read txn", e))?; + let table = match txn.open_table(CUT_FLOORS) { + Ok(table) => table, + Err(TableError::TableDoesNotExist(_)) => return Ok(Vec::new()), + Err(e) => return Err(catalog_err("open cut floors", e)), + }; + let mut all = Vec::new(); + for row in table + .range(..) + .map_err(|e| catalog_err("range cut floors", e))? + { + let (group_id, value) = row.map_err(|e| catalog_err("read cut floors", e))?; + let group_id = group_id.value(); + all.extend( + decode(value.value())? + .barriers + .into_iter() + .map(|barrier| (group_id, barrier)), + ); + } + Ok(all) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn barrier(index: u64, watermark: u64) -> StoredBarrier { + StoredBarrier { index, watermark } + } + + #[test] + fn settled_barriers_fold_into_the_one_that_still_binds() { + let dir = tempfile::tempdir().unwrap(); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).unwrap(); + catalog.put_cut_floor(3, barrier(10, 100), 0).unwrap(); + catalog.put_cut_floor(3, barrier(20, 200), 5).unwrap(); + catalog.put_cut_floor(4, barrier(7, 70), 0).unwrap(); + assert_eq!( + catalog.load_cut_floors().unwrap(), + [ + (3, barrier(10, 100)), + (3, barrier(20, 200)), + (4, barrier(7, 70)) + ] + ); + + // Durable through 25: both barriers of group 3 lie below every entry + // a restart delivers again, and the higher watermark binds them all. + catalog.put_cut_floor(3, barrier(30, 300), 25).unwrap(); + assert_eq!( + catalog.load_cut_floors().unwrap(), + [ + (3, barrier(20, 200)), + (3, barrier(30, 300)), + (4, barrier(7, 70)) + ] + ); + } +} diff --git a/nodedb/src/control/security/catalog/database.rs b/nodedb/src/control/security/catalog/database.rs index e4f15996c..96935c59c 100644 --- a/nodedb/src/control/security/catalog/database.rs +++ b/nodedb/src/control/security/catalog/database.rs @@ -13,9 +13,13 @@ use redb::{ReadableDatabase, ReadableTable, ReadableTableMetadata}; use super::database_types::DatabaseDescriptor; use super::types::{DATABASE_HWM, DATABASES, DATABASES_BY_NAME, SystemCatalog, catalog_err}; -/// Singleton row key for the hwm table. +/// Row key of the highest issued database id. const HWM_KEY: &str = "global"; +/// Row key of the highest metadata log index whose `DatabaseIdReserve` is +/// folded into the hwm. Replay skips every reservation at or below it. +const RESERVE_INDEX_KEY: &str = "reserve_index"; + impl SystemCatalog { // ── database_hwm ────────────────────────────────────────────────────── @@ -37,8 +41,40 @@ impl SystemCatalog { .map_err(|e| catalog_err("database_hwm commit", e)) } + /// Persist the hwm and the applied-reservation cursor in one write txn. + /// A crash between two separate writes makes the next replay count + /// a reservation twice or skip it. + pub fn put_database_reserve_state(&self, hwm: u64, reserve_index: u64) -> crate::Result<()> { + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("database_reserve_state write txn", e))?; + { + let mut table = txn + .open_table(DATABASE_HWM) + .map_err(|e| catalog_err("open database_hwm", e))?; + table + .insert(HWM_KEY, hwm) + .map_err(|e| catalog_err("insert database_hwm", e))?; + table + .insert(RESERVE_INDEX_KEY, reserve_index) + .map_err(|e| catalog_err("insert database reserve_index", e))?; + } + txn.commit() + .map_err(|e| catalog_err("database_reserve_state commit", e)) + } + /// Load the persisted database hwm, or `0` if none recorded yet. pub fn get_database_hwm(&self) -> crate::Result { + self.get_database_hwm_row(HWM_KEY) + } + + /// Load the applied-reservation cursor, or `0` if none recorded yet. + pub fn get_database_reserve_index(&self) -> crate::Result { + self.get_database_hwm_row(RESERVE_INDEX_KEY) + } + + fn get_database_hwm_row(&self, key: &str) -> crate::Result { let txn = self .db .begin_read() @@ -47,7 +83,7 @@ impl SystemCatalog { .open_table(DATABASE_HWM) .map_err(|e| catalog_err("open database_hwm", e))?; match table - .get(HWM_KEY) + .get(key) .map_err(|e| catalog_err("get database_hwm", e))? { Some(v) => Ok(v.value()), @@ -269,6 +305,20 @@ mod tests { assert_eq!(cat.get_database_hwm().unwrap(), 9999); } + #[test] + fn reserve_state_roundtrip_survives_reopen() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("system.redb"); + { + let cat = SystemCatalog::open(&path).unwrap(); + assert_eq!(cat.get_database_reserve_index().unwrap(), 0); + cat.put_database_reserve_state(1030, 77).unwrap(); + } + let cat = SystemCatalog::open(&path).unwrap(); + assert_eq!(cat.get_database_hwm().unwrap(), 1030); + assert_eq!(cat.get_database_reserve_index().unwrap(), 77); + } + /// Opening a catalog must leave the default database resolvable. /// /// Every fail-closed database lookup — the native handshake among them — diff --git a/nodedb/src/control/security/catalog/database_quotas.rs b/nodedb/src/control/security/catalog/database_quotas.rs index 39372c1c4..beeb8322a 100644 --- a/nodedb/src/control/security/catalog/database_quotas.rs +++ b/nodedb/src/control/security/catalog/database_quotas.rs @@ -37,7 +37,7 @@ impl SystemCatalog { // ── database_quotas ─────────────────────────────────────────────────────── /// Retrieve the quota record for a database. Returns `None` if no explicit - /// quota has been configured (callers should fall back to `QuotaRecord::DEFAULT`). + /// quota has been configured (callers fall back to `QuotaRecord::DEFAULT`). pub fn get_database_quota(&self, db_id: DatabaseId) -> crate::Result> { let txn = self .db @@ -90,13 +90,38 @@ impl SystemCatalog { || ceiling.max_qps > 0 || ceiling.max_connections > 0 { - self.check_database_quota_ceiling(db_id, record, ceiling)?; + self.check_database_quota_ceiling(Some(db_id), &[record], ceiling)?; + } + Ok(()) + } + + /// Check the quotas of several databases not yet in the catalog, together. + /// + /// Each record is validated, and the sum of every existing quota plus all + /// of `records` must fit `ceiling`. A restore runs this before it creates + /// any of the databases, so a refusal leaves the catalog unchanged. + pub fn check_new_database_quotas( + &self, + records: &[&QuotaRecord], + ceiling: &GlobalQuotaCeiling, + ) -> crate::Result<()> { + for record in records { + record.validate().map_err(|e| crate::Error::BadRequest { + detail: e.to_string(), + })?; + } + if ceiling.max_memory_bytes > 0 + || ceiling.max_storage_bytes > 0 + || ceiling.max_qps > 0 + || ceiling.max_connections > 0 + { + self.check_database_quota_ceiling(None, records, ceiling)?; } Ok(()) } /// Write a database quota record consensus already accepted. No validation - /// and no ceiling check — a rejection here would diverge nodes. + /// and no ceiling check — a rejection here diverges nodes. pub fn write_database_quota( &self, db_id: DatabaseId, @@ -216,8 +241,8 @@ impl SystemCatalog { /// sum past any non-zero ceiling dimension. fn check_database_quota_ceiling( &self, - db_id: DatabaseId, - proposed: &QuotaRecord, + replaced: Option, + proposed: &[&QuotaRecord], ceiling: &GlobalQuotaCeiling, ) -> crate::Result<()> { let all = self.list_database_quotas()?; @@ -229,7 +254,7 @@ impl SystemCatalog { let mut sum_connections: u64 = 0; for (id, rec) in &all { - if *id == db_id { + if Some(*id) == replaced { continue; // Will be replaced by `proposed`. } sum_memory = sum_memory.saturating_add(rec.max_memory_bytes); @@ -239,10 +264,12 @@ impl SystemCatalog { } // Add the proposed values. - sum_memory = sum_memory.saturating_add(proposed.max_memory_bytes); - sum_storage = sum_storage.saturating_add(proposed.max_storage_bytes); - sum_qps = sum_qps.saturating_add(proposed.max_qps as u64); - sum_connections = sum_connections.saturating_add(proposed.max_connections as u64); + for rec in proposed { + sum_memory = sum_memory.saturating_add(rec.max_memory_bytes); + sum_storage = sum_storage.saturating_add(rec.max_storage_bytes); + sum_qps = sum_qps.saturating_add(rec.max_qps as u64); + sum_connections = sum_connections.saturating_add(rec.max_connections as u64); + } if ceiling.max_memory_bytes > 0 && sum_memory > ceiling.max_memory_bytes { return Err(crate::Error::QuotaOvercommit { @@ -394,7 +421,7 @@ mod tests { cat.put_database_quota(DatabaseId::new(1), &r1, &ceiling) .unwrap(); - // Second database would push total to 3 GB, exceeding 2 GB ceiling. + // Second database pushes total to 3 GB, exceeding 2 GB ceiling. let r2 = QuotaRecord { max_memory_bytes: 1_500_000_000, ..QuotaRecord::DEFAULT @@ -418,7 +445,7 @@ mod tests { }; cat.put_database_quota(DatabaseId::new(1), &r, &ceiling) .unwrap(); - // Updating the same database to 1.8 GB should succeed (replaces, not adds). + // Updating the same database to 1.8 GB succeeds (replaces, not adds). let r2 = QuotaRecord { max_memory_bytes: 1_800_000_000, ..QuotaRecord::DEFAULT diff --git a/nodedb/src/control/security/catalog/event_defs_index.rs b/nodedb/src/control/security/catalog/event_defs_index.rs index 35116e546..8f8b1a83a 100644 --- a/nodedb/src/control/security/catalog/event_defs_index.rs +++ b/nodedb/src/control/security/catalog/event_defs_index.rs @@ -26,6 +26,8 @@ use nodedb_types::DatabaseId; use redb::{ReadableDatabase, ReadableTable}; +use crate::event::interest::{Interest, InterestSlice}; + use super::collection::StoredCollection; use super::collection_constraints::EventDefinition; use super::system_catalog::SystemCatalog; @@ -40,6 +42,8 @@ type IndexKey = (DatabaseId, u64, String); #[derive(Debug, Default)] pub struct EventDefsIndex { by_collection: RwLock>>, + /// The collections with definitions, republished after every change. + interest: Arc, } impl EventDefsIndex { @@ -47,6 +51,20 @@ impl EventDefsIndex { Self::default() } + /// The collections whose write events a definition of this index reads. + pub fn interest(&self) -> Arc { + Arc::clone(&self.interest) + } + + /// Republish the collections that carry definitions. + fn publish_interest(&self, map: &HashMap>) { + let mut interest = Interest::default(); + for (database_id, _, collection) in map.keys() { + interest.insert(*database_id, collection); + } + self.interest.publish(interest); + } + /// Replace the whole index with the definitions of `rows`, each keyed /// under the database its table row is stored in. pub fn load_all(&self, rows: &[(DatabaseId, StoredCollection)]) { @@ -56,7 +74,9 @@ impl EventDefsIndex { map.insert(key_of(*database_id, row), defs); } } - *self.write() = map; + let mut current = self.write(); + *current = map; + self.publish_interest(¤t); } /// Record `row`, committed under `database_id`. An inactive row, or a @@ -72,12 +92,14 @@ impl EventDefsIndex { map.remove(&key); } } + self.publish_interest(&map); } /// Forget a collection whose row was deleted. pub fn remove(&self, database_id: DatabaseId, tenant_id: u64, collection: &str) { - self.write() - .remove(&(database_id, tenant_id, collection.to_owned())); + let mut map = self.write(); + map.remove(&(database_id, tenant_id, collection.to_owned())); + self.publish_interest(&map); } /// The committed event definitions of a collection. `None` when it has @@ -104,6 +126,11 @@ impl EventDefsIndex { } impl SystemCatalog { + /// The collections with committed DEFINE EVENT definitions. + pub fn event_definition_interest(&self) -> Arc { + self.event_defs.interest() + } + /// Rebuild the event-definition index from every committed collection /// row, keyed by the database each row is stored under. pub fn reload_event_definitions(&self) -> crate::Result<()> { @@ -179,6 +206,19 @@ mod tests { .unwrap_or_default() } + #[test] + fn a_collection_with_definitions_consumes_write_events() { + let index = EventDefsIndex::new(); + let interest = index.interest(); + index.install(DB, &row(vec![def("a")])); + assert!(interest.contains(DB, "orders")); + index.install(DB, &row(Vec::new())); + assert!(!interest.contains(DB, "orders")); + index.install(DB, &row(vec![def("a")])); + index.remove(DB, 7, "orders"); + assert!(!interest.contains(DB, "orders")); + } + #[test] fn install_records_the_definitions() { let index = EventDefsIndex::new(); diff --git a/nodedb/src/control/security/catalog/metadata_host/ddl.rs b/nodedb/src/control/security/catalog/metadata_host/ddl.rs new file mode 100644 index 000000000..36c98dd5b --- /dev/null +++ b/nodedb/src/control/security/catalog/metadata_host/ddl.rs @@ -0,0 +1,109 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! In-flight `DdlPendingPropose` records, keyed by fencing token. + +use nodedb_cluster::PendingDdlObject; +use nodedb_types::Hlc; +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::super::types::{SystemCatalog, catalog_err}; + +/// Table: fencing token -> MessagePack `StoredPendingDdl`. +pub(in crate::control::security::catalog) const PENDING_DDL: TableDefinition = + TableDefinition::new("_system.pending_ddl"); + +/// One persisted pending DDL record. +#[derive(zerompk::ToMessagePack, zerompk::FromMessagePack, Debug, Clone, PartialEq)] +#[msgpack(map)] +pub struct StoredPendingDdl { + pub token: u64, + pub objects: Vec, + pub proposed_at: Hlc, +} + +impl SystemCatalog { + /// Write or replace the pending record of `record.token`. + pub fn put_pending_ddl(&self, record: &StoredPendingDdl) -> crate::Result<()> { + let value = + zerompk::to_msgpack_vec(record).map_err(|e| catalog_err("encode pending_ddl", e))?; + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("pending_ddl write txn", e))?; + { + let mut table = txn + .open_table(PENDING_DDL) + .map_err(|e| catalog_err("open pending_ddl", e))?; + table + .insert(record.token, value.as_slice()) + .map_err(|e| catalog_err("insert pending_ddl", e))?; + } + txn.commit() + .map_err(|e| catalog_err("commit pending_ddl", e)) + } + + /// Remove the pending record of `token`. Idempotent. + pub fn remove_pending_ddl(&self, token: u64) -> crate::Result<()> { + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("pending_ddl remove txn", e))?; + { + let mut table = txn + .open_table(PENDING_DDL) + .map_err(|e| catalog_err("open pending_ddl", e))?; + table + .remove(token) + .map_err(|e| catalog_err("remove pending_ddl", e))?; + } + txn.commit() + .map_err(|e| catalog_err("commit pending_ddl remove", e)) + } + + /// Every persisted pending record. + pub fn load_pending_ddl(&self) -> crate::Result> { + let txn = self + .db + .begin_read() + .map_err(|e| catalog_err("pending_ddl read txn", e))?; + let table = txn + .open_table(PENDING_DDL) + .map_err(|e| catalog_err("open pending_ddl", e))?; + let mut records = Vec::new(); + for item in table + .range(..) + .map_err(|e| catalog_err("range pending_ddl", e))? + { + let (_, value) = item.map_err(|e| catalog_err("read pending_ddl", e))?; + records.push( + zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("decode pending_ddl", e))?, + ); + } + Ok(records) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn pending_ddl_survives_reopen_and_remove() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("system.redb"); + let record = |token| StoredPendingDdl { + token, + objects: Vec::new(), + proposed_at: Hlc::new(5, 1), + }; + { + let catalog = SystemCatalog::open(&path).unwrap(); + catalog.put_pending_ddl(&record(7)).unwrap(); + catalog.put_pending_ddl(&record(9)).unwrap(); + catalog.remove_pending_ddl(9).unwrap(); + } + let catalog = SystemCatalog::open(&path).unwrap(); + assert_eq!(catalog.load_pending_ddl().unwrap(), vec![record(7)]); + } +} diff --git a/nodedb/src/control/security/catalog/metadata_host/drains.rs b/nodedb/src/control/security/catalog/metadata_host/drains.rs new file mode 100644 index 000000000..f2549a2b7 --- /dev/null +++ b/nodedb/src/control/security/catalog/metadata_host/drains.rs @@ -0,0 +1,160 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Descriptor drains, one row per `(descriptor, owner)`. + +use nodedb_cluster::{DescriptorId, DrainOwner}; +use nodedb_types::Hlc; +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::super::types::{SystemCatalog, catalog_err}; + +/// Table: MessagePack `DrainKey` -> MessagePack `StoredDrain`. +pub(in crate::control::security::catalog) const METADATA_DRAINS: TableDefinition<&[u8], &[u8]> = + TableDefinition::new("_system.metadata_drains"); + +/// Row key. A map encoding keeps it deterministic. +#[derive(zerompk::ToMessagePack)] +#[msgpack(map)] +struct DrainKey { + descriptor_id: DescriptorId, + owner: DrainOwner, +} + +/// One owner's drain of one descriptor. +#[derive(zerompk::ToMessagePack, zerompk::FromMessagePack, Debug, Clone, PartialEq, Eq)] +#[msgpack(map)] +pub struct StoredDrain { + pub descriptor_id: DescriptorId, + pub owner: DrainOwner, + pub up_to_version: u64, + pub expires_at: Hlc, + pub proposer_node_id: u64, +} + +fn drain_key(descriptor_id: &DescriptorId, owner: &DrainOwner) -> crate::Result> { + zerompk::to_msgpack_vec(&DrainKey { + descriptor_id: descriptor_id.clone(), + owner: owner.clone(), + }) + .map_err(|e| catalog_err("encode metadata drain key", e)) +} + +impl SystemCatalog { + /// Write or replace `drain.owner`'s drain of `drain.descriptor_id`. + pub fn put_descriptor_drain(&self, drain: &StoredDrain) -> crate::Result<()> { + let key = drain_key(&drain.descriptor_id, &drain.owner)?; + let value = + zerompk::to_msgpack_vec(drain).map_err(|e| catalog_err("encode metadata drain", e))?; + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("metadata drain write txn", e))?; + { + let mut table = txn + .open_table(METADATA_DRAINS) + .map_err(|e| catalog_err("open metadata_drains", e))?; + table + .insert(key.as_slice(), value.as_slice()) + .map_err(|e| catalog_err("insert metadata drain", e))?; + } + txn.commit() + .map_err(|e| catalog_err("commit metadata drain", e)) + } + + /// Remove each `(descriptor, owner)` drain row. Other owners' rows stay. + /// Idempotent. + pub fn remove_descriptor_drains( + &self, + drains: &[(DescriptorId, DrainOwner)], + ) -> crate::Result<()> { + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("metadata drain remove txn", e))?; + { + let mut table = txn + .open_table(METADATA_DRAINS) + .map_err(|e| catalog_err("open metadata_drains", e))?; + for (descriptor_id, owner) in drains { + let key = drain_key(descriptor_id, owner)?; + table + .remove(key.as_slice()) + .map_err(|e| catalog_err("remove metadata drain", e))?; + } + } + txn.commit() + .map_err(|e| catalog_err("commit metadata drain remove", e)) + } + + /// Every persisted drain. + pub fn load_descriptor_drains(&self) -> crate::Result> { + let txn = self + .db + .begin_read() + .map_err(|e| catalog_err("metadata drain read txn", e))?; + let table = txn + .open_table(METADATA_DRAINS) + .map_err(|e| catalog_err("open metadata_drains", e))?; + let mut drains = Vec::new(); + for item in table + .range(..) + .map_err(|e| catalog_err("range metadata_drains", e))? + { + let (_, value) = item.map_err(|e| catalog_err("read metadata drain", e))?; + drains.push( + zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("decode metadata drain", e))?, + ); + } + Ok(drains) + } +} + +#[cfg(test)] +mod tests { + use nodedb_cluster::DescriptorKind; + + use super::*; + + fn drain(owner: DrainOwner, up_to: u64) -> StoredDrain { + StoredDrain { + descriptor_id: DescriptorId::new(0, 1, DescriptorKind::Collection, "orders"), + owner, + up_to_version: up_to, + expires_at: Hlc::new(50, 0), + proposer_node_id: 3, + } + } + + /// Drains survive a reopen, and removing one owner's row leaves the + /// other owner's drain of the same descriptor. + #[test] + fn drains_survive_reopen_and_end_per_owner() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("system.redb"); + let moving = DrainOwner::MoveTenant { + tenant_id: 1, + source_db_id: 0, + }; + { + let catalog = SystemCatalog::open(&path).unwrap(); + catalog + .put_descriptor_drain(&drain(DrainOwner::Ddl, 4)) + .unwrap(); + catalog + .put_descriptor_drain(&drain(moving.clone(), 6)) + .unwrap(); + catalog + .remove_descriptor_drains(&[( + drain(DrainOwner::Ddl, 0).descriptor_id, + DrainOwner::Ddl, + )]) + .unwrap(); + } + let catalog = SystemCatalog::open(&path).unwrap(); + assert_eq!( + catalog.load_descriptor_drains().unwrap(), + vec![drain(moving, 6)] + ); + } +} diff --git a/nodedb/src/control/security/catalog/metadata_host/leases.rs b/nodedb/src/control/security/catalog/metadata_host/leases.rs new file mode 100644 index 000000000..6ccf70670 --- /dev/null +++ b/nodedb/src/control/security/catalog/metadata_host/leases.rs @@ -0,0 +1,144 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Descriptor leases of `MetadataCache.leases`, one row per +//! `(descriptor, holder node)`. + +use nodedb_cluster::{DescriptorId, DescriptorLease}; +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::super::types::{SystemCatalog, catalog_err}; + +/// Table: MessagePack `LeaseKey` -> MessagePack `DescriptorLease`. +pub(in crate::control::security::catalog) const METADATA_LEASES: TableDefinition<&[u8], &[u8]> = + TableDefinition::new("_system.metadata_leases"); + +/// Row key. A map encoding keeps it deterministic, so the same lease always +/// lands on the same row. +#[derive(zerompk::ToMessagePack)] +#[msgpack(map)] +struct LeaseKey { + descriptor_id: DescriptorId, + node_id: u64, +} + +fn lease_key(descriptor_id: &DescriptorId, node_id: u64) -> crate::Result> { + zerompk::to_msgpack_vec(&LeaseKey { + descriptor_id: descriptor_id.clone(), + node_id, + }) + .map_err(|e| catalog_err("encode metadata lease key", e)) +} + +impl SystemCatalog { + /// Write or replace the lease `lease.node_id` holds on its descriptor. + pub fn put_descriptor_lease(&self, lease: &DescriptorLease) -> crate::Result<()> { + let key = lease_key(&lease.descriptor_id, lease.node_id)?; + let value = + zerompk::to_msgpack_vec(lease).map_err(|e| catalog_err("encode metadata lease", e))?; + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("metadata lease write txn", e))?; + { + let mut table = txn + .open_table(METADATA_LEASES) + .map_err(|e| catalog_err("open metadata_leases", e))?; + table + .insert(key.as_slice(), value.as_slice()) + .map_err(|e| catalog_err("insert metadata lease", e))?; + } + txn.commit() + .map_err(|e| catalog_err("commit metadata lease", e)) + } + + /// Remove the leases `node_id` holds on `descriptor_ids`. Idempotent. + pub fn remove_descriptor_leases( + &self, + node_id: u64, + descriptor_ids: &[DescriptorId], + ) -> crate::Result<()> { + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("metadata lease remove txn", e))?; + { + let mut table = txn + .open_table(METADATA_LEASES) + .map_err(|e| catalog_err("open metadata_leases", e))?; + for descriptor_id in descriptor_ids { + let key = lease_key(descriptor_id, node_id)?; + table + .remove(key.as_slice()) + .map_err(|e| catalog_err("remove metadata lease", e))?; + } + } + txn.commit() + .map_err(|e| catalog_err("commit metadata lease remove", e)) + } + + /// Every persisted lease. + pub fn load_descriptor_leases(&self) -> crate::Result> { + let txn = self + .db + .begin_read() + .map_err(|e| catalog_err("metadata lease read txn", e))?; + let table = txn + .open_table(METADATA_LEASES) + .map_err(|e| catalog_err("open metadata_leases", e))?; + let mut leases = Vec::new(); + for item in table + .range(..) + .map_err(|e| catalog_err("range metadata_leases", e))? + { + let (_, value) = item.map_err(|e| catalog_err("read metadata lease", e))?; + leases.push( + zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("decode metadata lease", e))?, + ); + } + Ok(leases) + } +} + +#[cfg(test)] +mod tests { + use nodedb_cluster::DescriptorKind; + use nodedb_types::Hlc; + + use super::*; + + fn lease(name: &str, node_id: u64, expires: u64) -> DescriptorLease { + DescriptorLease { + descriptor_id: DescriptorId::new(0, 1, DescriptorKind::Collection, name), + version: 3, + node_id, + expires_at: Hlc::new(expires, 0), + } + } + + /// Leases survive a reopen of the catalog file, and a release removes + /// only the holder's row. + #[test] + fn leases_survive_reopen_and_release_removes_one_holder() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("system.redb"); + { + let catalog = SystemCatalog::open(&path).unwrap(); + catalog + .put_descriptor_lease(&lease("orders", 1, 10)) + .unwrap(); + catalog + .put_descriptor_lease(&lease("orders", 2, 20)) + .unwrap(); + catalog + .put_descriptor_lease(&lease("orders", 1, 30)) + .unwrap(); + catalog + .remove_descriptor_leases(2, &[lease("orders", 2, 0).descriptor_id]) + .unwrap(); + } + let catalog = SystemCatalog::open(&path).unwrap(); + let leases = catalog.load_descriptor_leases().unwrap(); + assert_eq!(leases, vec![lease("orders", 1, 30)]); + } +} diff --git a/nodedb/src/control/security/catalog/metadata_host/mod.rs b/nodedb/src/control/security/catalog/metadata_host/mod.rs new file mode 100644 index 000000000..de78fe4ad --- /dev/null +++ b/nodedb/src/control/security/catalog/metadata_host/mod.rs @@ -0,0 +1,12 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Durable images of the metadata group's host-side state. +//! +//! Each metadata apply that changes one of these writes its row before it +//! returns, and boot seeds the in-memory state from the rows. No state here +//! depends on replaying the metadata log. + +pub mod ddl; +pub mod drains; +pub mod leases; +pub mod scalars; diff --git a/nodedb/src/control/security/catalog/metadata_host/scalars.rs b/nodedb/src/control/security/catalog/metadata_host/scalars.rs new file mode 100644 index 000000000..ce3ce5557 --- /dev/null +++ b/nodedb/src/control/security/catalog/metadata_host/scalars.rs @@ -0,0 +1,172 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Single-value host state of the metadata group: the cluster version, the +//! owner token of the DDL preparation lease, the highest applied stamp, and +//! the metadata timeline. + +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::super::types::{SystemCatalog, catalog_err}; + +/// Table: scalar name -> value. +pub(in crate::control::security::catalog) const METADATA_HOST_SCALARS: TableDefinition<&str, u64> = + TableDefinition::new("_system.metadata_host_scalars"); + +const CLUSTER_VERSION: &str = "cluster_version"; +const DDL_OWNER_TOKEN: &str = "ddl_owner_token"; +/// The node that owns the DDL preparation lease. Written with the token. +const DDL_OWNER_NODE: &str = "ddl_owner_node"; +/// The highest metadata entry stamp applied. The node's HLC starts above it +/// at boot and after a snapshot install, so the stamps a leader takes rise +/// above every entry the log ever held. +const METADATA_STAMP_HWM: &str = "metadata_stamp_hwm"; +/// The metadata timeline of the catalog's history. +const METADATA_TIMELINE: &str = "metadata_timeline"; + +impl SystemCatalog { + fn put_scalar(&self, name: &str, value: Option) -> crate::Result<()> { + self.put_scalars(&[(name, value)]) + } + + /// Write every `(name, value)` in one transaction. `None` removes it. + fn put_scalars(&self, values: &[(&str, Option)]) -> crate::Result<()> { + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("metadata scalar write txn", e))?; + { + let mut table = txn + .open_table(METADATA_HOST_SCALARS) + .map_err(|e| catalog_err("open metadata_host_scalars", e))?; + for (name, value) in values { + match value { + Some(value) => { + table + .insert(*name, *value) + .map_err(|e| catalog_err("insert metadata scalar", e))?; + } + None => { + table + .remove(*name) + .map_err(|e| catalog_err("remove metadata scalar", e))?; + } + } + } + } + txn.commit() + .map_err(|e| catalog_err("commit metadata scalar", e)) + } + + fn get_scalar(&self, name: &str) -> crate::Result> { + let txn = self + .db + .begin_read() + .map_err(|e| catalog_err("metadata scalar read txn", e))?; + let table = txn + .open_table(METADATA_HOST_SCALARS) + .map_err(|e| catalog_err("open metadata_host_scalars", e))?; + Ok(table + .get(name) + .map_err(|e| catalog_err("get metadata scalar", e))? + .map(|value| value.value())) + } + + /// Record the applied cluster version. + pub fn put_cluster_version(&self, version: u16) -> crate::Result<()> { + self.put_scalar(CLUSTER_VERSION, Some(u64::from(version))) + } + + /// The applied cluster version, `None` before any bump applied. + pub fn load_cluster_version(&self) -> crate::Result> { + self.get_scalar(CLUSTER_VERSION)? + .map(|value| { + u16::try_from(value).map_err(|_| { + catalog_err( + "decode cluster_version", + format!("{value} does not fit a cluster version"), + ) + }) + }) + .transpose() + } + + /// Record the owner of the DDL preparation lease as `(token, node_id)`, + /// or clear it. Both values change in one transaction. + pub fn put_ddl_owner(&self, owner: Option<(u64, u64)>) -> crate::Result<()> { + self.put_scalars(&[ + (DDL_OWNER_TOKEN, owner.map(|(token, _)| token)), + (DDL_OWNER_NODE, owner.map(|(_, node_id)| node_id)), + ]) + } + + /// The owner of the DDL preparation lease as `(token, node_id)`, if a + /// lease is held. A token stored without its node is a corrupt row. + pub fn load_ddl_owner(&self) -> crate::Result> { + match ( + self.get_scalar(DDL_OWNER_TOKEN)?, + self.get_scalar(DDL_OWNER_NODE)?, + ) { + (Some(token), Some(node_id)) => Ok(Some((token, node_id))), + (None, None) => Ok(None), + (token, node_id) => Err(catalog_err( + "decode ddl_owner", + format!( + "the DDL preparation owner row is partial: token {token:?}, node {node_id:?}" + ), + )), + } + } + + /// Raise the highest metadata entry stamp this node applied to `stamp`. + /// A lower `stamp` writes nothing. + pub fn raise_metadata_stamp_hwm(&self, stamp: u64) -> crate::Result<()> { + if self + .load_metadata_stamp_hwm()? + .is_some_and(|hwm| hwm >= stamp) + { + return Ok(()); + } + self.put_scalar(METADATA_STAMP_HWM, Some(stamp)) + } + + /// The highest metadata entry stamp this node applied, `None` before any. + pub fn load_metadata_stamp_hwm(&self) -> crate::Result> { + self.get_scalar(METADATA_STAMP_HWM) + } + + /// Record the metadata timeline the catalog belongs to. A restore sets + /// it, and a snapshot install carries it to every node. + pub fn put_metadata_timeline(&self, timeline: u64) -> crate::Result<()> { + self.put_scalar(METADATA_TIMELINE, Some(timeline)) + } + + /// The metadata timeline the catalog belongs to. A catalog no restore + /// touched belongs to the root timeline. + pub fn load_metadata_timeline(&self) -> crate::Result { + Ok(self + .get_scalar(METADATA_TIMELINE)? + .unwrap_or(crate::storage::metadata_timeline::ROOT_TIMELINE)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn scalars_survive_reopen() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("system.redb"); + { + let catalog = SystemCatalog::open(&path).unwrap(); + assert_eq!(catalog.load_cluster_version().unwrap(), None); + catalog.put_cluster_version(4).unwrap(); + catalog.put_ddl_owner(Some((77, 3))).unwrap(); + } + let catalog = SystemCatalog::open(&path).unwrap(); + assert_eq!(catalog.load_cluster_version().unwrap(), Some(4)); + assert_eq!(catalog.load_ddl_owner().unwrap(), Some((77, 3))); + catalog.put_ddl_owner(None).unwrap(); + assert_eq!(catalog.load_ddl_owner().unwrap(), None); + } +} diff --git a/nodedb/src/control/security/catalog/mod.rs b/nodedb/src/control/security/catalog/mod.rs index e316a06a8..0f37cebd1 100644 --- a/nodedb/src/control/security/catalog/mod.rs +++ b/nodedb/src/control/security/catalog/mod.rs @@ -5,24 +5,30 @@ pub mod arrays; pub mod audit; pub mod auth_types; pub mod auth_users; +pub mod backup_schedule_marks; pub mod blacklist; pub mod bootstrap_tables; pub mod calvin_applied; +pub mod calvin_base; pub mod change_streams; pub mod checkpoint; pub mod checkpoints; pub mod clone_catalog; +pub mod clone_source_drains; pub mod collection; pub mod collection_constraints; pub mod collection_descriptor_convert; +pub mod collection_incarnation; pub mod collections; pub mod column_stats; pub mod constraint_translate; pub mod consumer_groups; pub mod continuous_aggregate; pub mod continuous_aggregates; +pub mod crdt_compaction_points; pub mod custom_type_oid_hwm; pub mod custom_types; +pub mod cut_floors; pub mod database; pub mod database_grants; pub mod database_quotas; @@ -38,6 +44,7 @@ pub mod lockout; pub mod materialized_view; pub mod materialized_views; pub mod metadata; +pub mod metadata_host; pub mod mirror; pub mod move_tenant_journal; pub mod move_tenant_journal_types; @@ -45,11 +52,16 @@ pub mod oidc_providers; pub mod orgs; pub mod owner_rewrite; pub mod ownership_fallback; +pub mod pending_history_compaction; +pub mod pending_leave_cleanup; pub mod pending_reclaim; pub mod procedure_types; pub mod procedures; pub mod read_only; pub mod redaction; +pub mod replicated_image; +pub mod replicated_image_merge; +pub mod restore_points; pub mod retention_policy; pub mod rls; pub mod schedules; @@ -68,6 +80,9 @@ pub mod tables; pub mod tenant_group_marks; pub mod tenant_id_hwm; pub mod tenant_quotas; +pub mod topic_lookup; +pub mod topic_messages; +pub mod topic_publish_marks; pub mod topics; pub mod trigger_types; pub mod triggers; @@ -87,6 +102,7 @@ pub use collection_constraints::{ }; pub use collections::merge_inferred_fields; pub use constraint_translate::collection_constraints; +pub use crdt_compaction_points::StoredCompactionPoint; pub use custom_type_oid_hwm::USER_TYPE_OID_BASE; pub use custom_types::{CompositeField, CustomTypeDef, StoredCustomType, UNASSIGNED_OID}; pub use database_grants::DatabaseGrant; @@ -98,9 +114,12 @@ pub use function_types::{ pub use index_record::{IndexKind, StoredIndexRecord}; pub use l2_cleanup_queue::StoredL2CleanupEntry; pub use lockout::StoredLockoutRecord; +pub use metadata_host::ddl::StoredPendingDdl; +pub use metadata_host::drains::StoredDrain; pub use move_tenant_journal_types::{MovePhase, MoveTenantJournalEntry}; pub use oidc_providers::{StoredClaimMappingRule, StoredOidcProvider}; pub use orgs::{StoredOrg, StoredOrgMember}; +pub use pending_history_compaction::StoredPendingHistoryCompaction; pub use pending_reclaim::StoredPendingReclaim; pub use procedure_types::StoredProcedure; pub use read_only::{ReadOnlyOpenError, ReadOnlySystemCatalog}; diff --git a/nodedb/src/control/security/catalog/pending_history_compaction.rs b/nodedb/src/control/security/catalog/pending_history_compaction.rs new file mode 100644 index 000000000..5139a6995 --- /dev/null +++ b/nodedb/src/control/security/catalog/pending_history_compaction.rs @@ -0,0 +1,260 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Owed CRDT history compactions of this node. +//! +//! Applying a `CompactHistory` entry writes one row here before it deletes the +//! checkpoint rows. Post-apply removes the row once every local core compacted +//! and checkpointed the collection's oplog. A row that survives a crash or a +//! failed fan-out is re-driven by the boot drain and the retry worker. +//! +//! One row per collection. A later compaction of the same collection replaces +//! the row: compacting to its target also discards what the earlier one owed. +//! +//! Table: `{database_id}:{tenant_id}:{collection}` -> MessagePack row. + +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::types::{SystemCatalog, catalog_err}; + +pub(super) const PENDING_HISTORY_COMPACTION: TableDefinition<&str, &[u8]> = + TableDefinition::new("_system.pending_history_compaction"); + +/// One owed compaction: "compact this collection's oplog to this version". +#[derive(zerompk::ToMessagePack, zerompk::FromMessagePack, Debug, Clone, PartialEq, Eq)] +#[msgpack(map, allow_unknown_fields)] +pub struct StoredPendingHistoryCompaction { + pub database_id: u64, + pub tenant_id: u64, + pub collection: String, + /// Loro version vector the committed entry carries. + pub target_version_json: String, + /// Last error a retry observed. Empty until a retry fails. + #[msgpack(default)] + pub last_error: String, + /// Failed retries of this row. + #[msgpack(default)] + pub attempts: u32, +} + +fn compaction_key(database_id: u64, tenant_id: u64, collection: &str) -> String { + format!("{database_id}:{tenant_id}:{collection}") +} + +impl SystemCatalog { + /// Record an owed compaction, replacing any row for the same collection. + pub fn enqueue_pending_history_compaction( + &self, + entry: &StoredPendingHistoryCompaction, + ) -> crate::Result<()> { + let bytes = zerompk::to_msgpack_vec(entry) + .map_err(|e| catalog_err("encode pending_history_compaction row", e))?; + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("enqueue_pending_history_compaction txn", e))?; + { + let mut table = txn + .open_table(PENDING_HISTORY_COMPACTION) + .map_err(|e| catalog_err("open pending_history_compaction", e))?; + let key = compaction_key(entry.database_id, entry.tenant_id, &entry.collection); + table + .insert(key.as_str(), bytes.as_slice()) + .map_err(|e| catalog_err("insert pending_history_compaction row", e))?; + } + txn.commit() + .map_err(|e| catalog_err("commit pending_history_compaction enqueue", e)) + } + + /// Every owed compaction on this node. + pub fn load_pending_history_compactions( + &self, + ) -> crate::Result> { + let txn = self + .db + .begin_read() + .map_err(|e| catalog_err("load_pending_history_compactions read txn", e))?; + let table = txn + .open_table(PENDING_HISTORY_COMPACTION) + .map_err(|e| catalog_err("open pending_history_compaction", e))?; + let mut out = Vec::new(); + for item in table + .range(..) + .map_err(|e| catalog_err("range pending_history_compaction", e))? + { + let (_, v) = item.map_err(|e| catalog_err("read pending_history_compaction", e))?; + out.push( + zerompk::from_msgpack(v.value()) + .map_err(|e| catalog_err("decode pending_history_compaction row", e))?, + ); + } + Ok(out) + } + + /// Bump `attempts` and store `last_error` on the row, if it still owes + /// `target_version_json`. + pub fn record_pending_history_compaction_attempt( + &self, + database_id: u64, + tenant_id: u64, + collection: &str, + target_version_json: &str, + last_error: &str, + ) -> crate::Result<()> { + self.update_owed_compaction( + database_id, + tenant_id, + collection, + target_version_json, + |mut row| { + row.attempts = row.attempts.saturating_add(1); + row.last_error = last_error.to_string(); + Some(row) + }, + ) + } + + /// Remove the row once `target_version_json` is durably compacted. + /// + /// A row that a later entry replaced with another target stays: that + /// compaction is still owed. + pub fn remove_pending_history_compaction( + &self, + database_id: u64, + tenant_id: u64, + collection: &str, + target_version_json: &str, + ) -> crate::Result<()> { + self.update_owed_compaction( + database_id, + tenant_id, + collection, + target_version_json, + |_| None, + ) + } + + /// Rewrite (`Some`) or remove (`None`) the collection's row, when it owes + /// `target_version_json`. Any other row, or none, is left as it is. + fn update_owed_compaction( + &self, + database_id: u64, + tenant_id: u64, + collection: &str, + target_version_json: &str, + update: impl FnOnce(StoredPendingHistoryCompaction) -> Option, + ) -> crate::Result<()> { + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("pending_history_compaction update txn", e))?; + { + let mut table = txn + .open_table(PENDING_HISTORY_COMPACTION) + .map_err(|e| catalog_err("open pending_history_compaction", e))?; + let key = compaction_key(database_id, tenant_id, collection); + let existing = table + .get(key.as_str()) + .map_err(|e| catalog_err("get pending_history_compaction row", e))? + .map(|g| g.value().to_vec()); + let Some(raw) = existing else { + return Ok(()); + }; + let row: StoredPendingHistoryCompaction = zerompk::from_msgpack(&raw) + .map_err(|e| catalog_err("decode pending_history_compaction row", e))?; + if row.target_version_json != target_version_json { + return Ok(()); + } + match update(row) { + Some(row) => { + let bytes = zerompk::to_msgpack_vec(&row) + .map_err(|e| catalog_err("encode pending_history_compaction row", e))?; + table + .insert(key.as_str(), bytes.as_slice()) + .map_err(|e| catalog_err("update pending_history_compaction row", e))?; + } + None => { + table + .remove(key.as_str()) + .map_err(|e| catalog_err("remove pending_history_compaction row", e))?; + } + } + } + txn.commit() + .map_err(|e| catalog_err("commit pending_history_compaction update", e)) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use tempfile::TempDir; + + fn cat() -> (SystemCatalog, TempDir) { + let tmp = TempDir::new().unwrap(); + let cat = SystemCatalog::open(&tmp.path().join("system.redb")).unwrap(); + (cat, tmp) + } + + fn row(collection: &str, target: &str) -> StoredPendingHistoryCompaction { + StoredPendingHistoryCompaction { + database_id: 3, + tenant_id: 7, + collection: collection.to_string(), + target_version_json: target.to_string(), + last_error: String::new(), + attempts: 0, + } + } + + #[test] + fn enqueue_then_load_roundtrip() { + let (c, _t) = cat(); + c.enqueue_pending_history_compaction(&row("docs", "v1")) + .unwrap(); + assert_eq!( + c.load_pending_history_compactions().unwrap(), + vec![row("docs", "v1")] + ); + } + + #[test] + fn a_later_compaction_replaces_the_row() { + let (c, _t) = cat(); + c.enqueue_pending_history_compaction(&row("docs", "v1")) + .unwrap(); + c.enqueue_pending_history_compaction(&row("docs", "v2")) + .unwrap(); + assert_eq!( + c.load_pending_history_compactions().unwrap(), + vec![row("docs", "v2")] + ); + } + + #[test] + fn remove_takes_only_the_row_that_owes_the_target() { + let (c, _t) = cat(); + c.enqueue_pending_history_compaction(&row("docs", "v2")) + .unwrap(); + c.remove_pending_history_compaction(3, 7, "docs", "v1") + .unwrap(); + assert_eq!(c.load_pending_history_compactions().unwrap().len(), 1); + c.remove_pending_history_compaction(3, 7, "docs", "v2") + .unwrap(); + assert!(c.load_pending_history_compactions().unwrap().is_empty()); + // Idempotent. + c.remove_pending_history_compaction(3, 7, "docs", "v2") + .unwrap(); + } + + #[test] + fn record_attempt_updates_in_place() { + let (c, _t) = cat(); + c.enqueue_pending_history_compaction(&row("docs", "v1")) + .unwrap(); + c.record_pending_history_compaction_attempt(3, 7, "docs", "v1", "core 0 refused") + .unwrap(); + let rows = c.load_pending_history_compactions().unwrap(); + assert_eq!(rows[0].attempts, 1); + assert_eq!(rows[0].last_error, "core 0 refused"); + } +} diff --git a/nodedb/src/control/security/catalog/pending_leave_cleanup.rs b/nodedb/src/control/security/catalog/pending_leave_cleanup.rs new file mode 100644 index 000000000..157b5be9f --- /dev/null +++ b/nodedb/src/control/security/catalog/pending_leave_cleanup.rs @@ -0,0 +1,100 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Cleanup owed for nodes that left the cluster. +//! +//! Applying `TopologyChange::Leave` writes one row here before it returns. +//! The row stays until the metadata cache holds no lease of that node and +//! no drain it proposed. The Leave post-apply, the boot drain, and the retry +//! worker drive the release and drain-end proposals; the singleton worker +//! makes them. +//! +//! Table: `node_id` -> the log index of the `Leave` entry. + +use redb::{ReadableDatabase, ReadableTable, TableDefinition}; + +use super::types::{SystemCatalog, catalog_err}; + +pub(super) const PENDING_LEAVE_CLEANUP: TableDefinition = + TableDefinition::new("_system.pending_leave_cleanup"); + +impl SystemCatalog { + /// Record that `node_id` left at log index `raft_index` and owes cleanup. + pub fn enqueue_pending_leave_cleanup( + &self, + node_id: u64, + raft_index: u64, + ) -> crate::Result<()> { + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("pending_leave_cleanup write txn", e))?; + { + let mut table = txn + .open_table(PENDING_LEAVE_CLEANUP) + .map_err(|e| catalog_err("open pending_leave_cleanup", e))?; + table + .insert(node_id, raft_index) + .map_err(|e| catalog_err("insert pending_leave_cleanup", e))?; + } + txn.commit() + .map_err(|e| catalog_err("pending_leave_cleanup commit", e)) + } + + /// Drop the row of `node_id`. Removing an absent row succeeds. + pub fn remove_pending_leave_cleanup(&self, node_id: u64) -> crate::Result<()> { + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("pending_leave_cleanup write txn", e))?; + { + let mut table = txn + .open_table(PENDING_LEAVE_CLEANUP) + .map_err(|e| catalog_err("open pending_leave_cleanup", e))?; + table + .remove(node_id) + .map_err(|e| catalog_err("remove pending_leave_cleanup", e))?; + } + txn.commit() + .map_err(|e| catalog_err("pending_leave_cleanup commit", e)) + } + + /// Every node that owes cleanup. + pub fn load_pending_leave_cleanups(&self) -> crate::Result> { + let txn = self + .db + .begin_read() + .map_err(|e| catalog_err("pending_leave_cleanup read txn", e))?; + let table = txn + .open_table(PENDING_LEAVE_CLEANUP) + .map_err(|e| catalog_err("open pending_leave_cleanup", e))?; + let mut out = Vec::new(); + for item in table + .iter() + .map_err(|e| catalog_err("iterate pending_leave_cleanup", e))? + { + let (node_id, _) = item.map_err(|e| catalog_err("read pending_leave_cleanup", e))?; + out.push(node_id.value()); + } + Ok(out) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn rows_survive_reopen_and_remove() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("system.redb"); + { + let catalog = SystemCatalog::open(&path).unwrap(); + catalog.enqueue_pending_leave_cleanup(4, 10).unwrap(); + catalog.enqueue_pending_leave_cleanup(5, 11).unwrap(); + catalog.remove_pending_leave_cleanup(4).unwrap(); + catalog.remove_pending_leave_cleanup(9).unwrap(); + } + let catalog = SystemCatalog::open(&path).unwrap(); + assert_eq!(catalog.load_pending_leave_cleanups().unwrap(), vec![5]); + } +} diff --git a/nodedb/src/control/security/catalog/pending_reclaim.rs b/nodedb/src/control/security/catalog/pending_reclaim.rs index edccd3d7a..fb298062a 100644 --- a/nodedb/src/control/security/catalog/pending_reclaim.rs +++ b/nodedb/src/control/security/catalog/pending_reclaim.rs @@ -23,9 +23,10 @@ //! Surface: `SystemCatalog::{enqueue,load,record_attempt,remove}_pending_reclaim`. //! Structure mirrors `l2_cleanup_queue.rs`. +use nodedb_types::Hlc; use redb::{ReadableDatabase, ReadableTable}; -use super::types::{PENDING_RECLAIM, SystemCatalog, catalog_err}; +use super::types::{PENDING_RECLAIM, StoredCollection, SystemCatalog, catalog_err}; /// One queue entry: "engine storage purge for this collection is still owed". #[derive(zerompk::ToMessagePack, zerompk::FromMessagePack, Debug, Clone)] @@ -48,6 +49,36 @@ pub struct StoredPendingReclaim { /// Number of purge attempts this entry has survived (post-first-failure). #[msgpack(default)] pub attempts: u32, + /// Incarnation the reclaim targets. For a purge: the collection row's + /// `modification_hlc` when the reclaim was queued, `None` when no row + /// existed then. For a cancelled create: the create's own clock. + #[msgpack(default)] + pub target_hlc: Option, + /// The entry tears down a cancelled create, which never committed a row. + #[msgpack(default)] + pub cancelled_create: bool, +} + +impl StoredPendingReclaim { + /// The key the row's drain hold is owned under: the row's own key. + pub fn owner(&self) -> crate::bridge::quiesce::ReclaimOwner { + crate::bridge::quiesce::ReclaimOwner::new(self.database_id, self.tenant_id, &self.name) + } + + /// Whether the name still belongs to the incarnation this entry reclaims, + /// given the committed row now under the name. + /// + /// No row: nothing newer claimed the name. A cancelled create never + /// committed a row, so any row is a later incarnation. A purge owns only + /// the row it was queued against, which `prepare_purge` keeps with its + /// clock unchanged. + pub fn owns(&self, live: Option<&StoredCollection>) -> bool { + match live { + None => true, + Some(_) if self.cancelled_create => false, + Some(row) => self.target_hlc == Some(row.modification_hlc), + } + } } fn pending_key(database_id: u64, tenant_id: u64, name: &str) -> String { @@ -183,9 +214,45 @@ mod tests { enqueued_at_ns: 100, last_error: String::new(), attempts: 0, + target_hlc: None, + cancelled_create: false, } } + fn row_at(hlc: Hlc) -> StoredCollection { + let mut row = StoredCollection::new(1, "events", "tester"); + row.modification_hlc = hlc; + row + } + + /// A purge owns only the row it was queued against. + #[test] + fn a_purge_owns_only_its_own_incarnation() { + let mut purge = entry(1, "events", 500); + purge.target_hlc = Some(Hlc::new(10, 0)); + assert!(purge.owns(None)); + assert!(purge.owns(Some(&row_at(Hlc::new(10, 0))))); + assert!(!purge.owns(Some(&row_at(Hlc::new(20, 0))))); + } + + /// A purge queued when no row existed owns no row that appears later. + #[test] + fn a_purge_of_an_absent_row_owns_no_later_row() { + let purge = entry(1, "events", 500); + assert!(purge.owns(None)); + assert!(!purge.owns(Some(&row_at(Hlc::new(20, 0))))); + } + + /// A cancelled create committed no row, so every row is a later one. + #[test] + fn a_cancelled_create_owns_no_committed_row() { + let mut cancel = entry(1, "events", 500); + cancel.target_hlc = Some(Hlc::new(10, 0)); + cancel.cancelled_create = true; + assert!(cancel.owns(None)); + assert!(!cancel.owns(Some(&row_at(Hlc::new(10, 0))))); + } + #[test] fn enqueue_then_load_roundtrip() { let (c, _t) = cat(); diff --git a/nodedb/src/control/security/catalog/replicated_image.rs b/nodedb/src/control/security/catalog/replicated_image.rs new file mode 100644 index 000000000..c59552a6f --- /dev/null +++ b/nodedb/src/control/security/catalog/replicated_image.rs @@ -0,0 +1,406 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The `_system.*` tables metadata Raft group 0 replicates, as raw rows. +//! +//! A group 0 snapshot carries exactly these tables. [`REPLICATED_TABLES`] +//! lists every table a committed metadata entry writes: the tables +//! `CatalogEntry::apply_to` writes (except `wal_tombstones` and the +//! surrogate bindings, below), the metadata-group host tables, join +//! tokens, enrollment pre-authorizations, the surrogate and database-id +//! watermarks with their reserve cursors, the sync producer registry, the +//! cluster restore points, and the scheduled backup marks. +//! +//! Every other bootstrap table stays out of the image. Each one is written +//! only by this node, or travels with a data-group snapshot: +//! - `audit_log`: this node's audit trail, numbered by a node-local sequence. +//! - `blacklist`, `orgs`, `org_members`, `scopes`: written only by the node +//! that ran the statement. Group 0 never applies them. +//! - `lockout_state`: this node's failed-login counters. +//! - `wal_tombstones`: purge boundaries in this node's WAL LSN space. Every +//! node records its own when it reclaims a collection. A replicated +//! `RecordWalTombstone` (backup restore) also writes one, but another +//! node's rows move this node's replay boundaries. +//! - `tenant_group_marks`, `tenant_group_restore_marks`: data-group write +//! marks, carried by data-group snapshots. +//! - `calvin_applied`: this node's Calvin apply ledger. +//! - `l2_cleanup_queue`, `pending_reclaim`, `pending_history_compaction`: +//! work this node still owes its own storage. +//! - `surrogate_pk_v3`, `surrogate_pk_rev_v3`: PK bindings of stored rows, +//! carried by data-group snapshots. A purge apply deletes a collection's +//! bindings; the install's reclaim of that collection does the same here. +//! - `crdt_signing_keys`, `crdt_signing_root_metadata`: key material derived +//! from this node's WAL. +//! - `topic_messages`: this node's published-message buffer. +//! - `topic_publish_marks`: the committed-message marks of this node's +//! topic log, moved with each append. +//! - `mirror_collection_map`, `mirror_lag`: state of a mirror stream, not of +//! group 0. +//! - `move_tenant_journal`: the journal of the node coordinating a move. + +use redb::{ + Key, ReadTransaction, ReadableDatabase, ReadableTable, TableDefinition, TableError, Value, + WriteTransaction, +}; + +use super::tables::*; +use super::types::{SystemCatalog, catalog_err}; + +/// Raw rows of one table: `(key bytes, value bytes)` in key order. +pub type RawRows = Vec<(Vec, Vec)>; + +/// One replicated table: a stable label and thunks that read or replace its +/// rows without naming its key and value types. +pub(super) struct ReplicatedTable { + pub(super) label: &'static str, + dump: fn(&ReadTransaction) -> crate::Result, + /// The rows as the write transaction sees them, for a merge. + current: fn(&WriteTransaction) -> crate::Result, + replace: fn(&WriteTransaction, &RawRows) -> crate::Result<()>, +} + +fn dump_rows( + txn: &ReadTransaction, + def: TableDefinition, +) -> crate::Result { + let table = match txn.open_table(def) { + Ok(table) => table, + Err(TableError::TableDoesNotExist(_)) => return Ok(Vec::new()), + Err(e) => return Err(catalog_err("replicated image: open table", e)), + }; + let mut rows = Vec::new(); + for item in table + .iter() + .map_err(|e| catalog_err("replicated image: iterate table", e))? + { + let (key, value) = item.map_err(|e| catalog_err("replicated image: read row", e))?; + let key = key.value(); + let value = value.value(); + rows.push(( + K::as_bytes(&key).as_ref().to_vec(), + V::as_bytes(&value).as_ref().to_vec(), + )); + } + Ok(rows) +} + +fn current_rows( + txn: &WriteTransaction, + def: TableDefinition, +) -> crate::Result { + let table = txn + .open_table(def) + .map_err(|e| catalog_err("replicated image: open table", e))?; + let mut rows = Vec::new(); + for item in table + .iter() + .map_err(|e| catalog_err("replicated image: iterate table", e))? + { + let (key, value) = item.map_err(|e| catalog_err("replicated image: read row", e))?; + let key = key.value(); + let value = value.value(); + rows.push(( + K::as_bytes(&key).as_ref().to_vec(), + V::as_bytes(&value).as_ref().to_vec(), + )); + } + Ok(rows) +} + +fn replace_rows( + txn: &WriteTransaction, + def: TableDefinition, + rows: &RawRows, +) -> crate::Result<()> { + let mut table = txn + .open_table(def) + .map_err(|e| catalog_err("replicated image: open table", e))?; + table + .retain(|_, _| false) + .map_err(|e| catalog_err("replicated image: clear table", e))?; + for (key, value) in rows { + table + .insert(K::from_bytes(key), V::from_bytes(value)) + .map_err(|e| catalog_err("replicated image: insert row", e))?; + } + Ok(()) +} + +macro_rules! replicated_tables { + ($($label:literal => $def:expr),+ $(,)?) => { + &[$( + ReplicatedTable { + label: $label, + dump: |txn| dump_rows(txn, $def), + current: |txn| current_rows(txn, $def), + replace: |txn, rows| replace_rows(txn, $def, rows), + } + ),+] + }; +} + +/// Every table group 0 replicates. Labels match the bootstrap registry. +pub(super) const REPLICATED_TABLES: &[ReplicatedTable] = replicated_tables![ + // ── Auth / tenancy ── + "users" => USERS, + "api_keys" => API_KEYS, + "roles" => ROLES, + "permissions" => PERMISSIONS, + "owners" => OWNERS, + "tenants" => TENANTS, + "tenant_id_hwm" => super::tenant_id_hwm::TENANT_ID_HWM, + "auth_users" => AUTH_USERS, + "scope_grants" => SCOPE_GRANTS, + "scope_quotas" => SCOPE_QUOTAS, + "oidc_providers" => OIDC_PROVIDERS, + // ── Collections ── + "collections" => COLLECTIONS, + "metadata" => METADATA, + "column_stats" => COLUMN_STATS, + "vector_model_metadata" => VECTOR_MODEL_METADATA, + "vector_index_params" => VECTOR_INDEX_PARAMS, + "index_registry" => INDEX_REGISTRY, + "checkpoints" => CHECKPOINTS, + "crdt_compaction_points" => super::crdt_compaction_points::CRDT_COMPACTION_POINTS, + // ── Metadata-group host state ── + "metadata_leases" => super::metadata_host::leases::METADATA_LEASES, + "metadata_drains" => super::metadata_host::drains::METADATA_DRAINS, + "metadata_host_scalars" => super::metadata_host::scalars::METADATA_HOST_SCALARS, + "pending_ddl" => super::metadata_host::ddl::PENDING_DDL, + "pending_leave_cleanup" => super::pending_leave_cleanup::PENDING_LEAVE_CLEANUP, + // ── Replicated watermarks ── + "surrogate_hwm" => super::surrogate_hwm::SURROGATE_HWM, + "surrogate_reserve_index" => super::surrogate_hwm::SURROGATE_RESERVE_INDEX, + "database_hwm" => DATABASE_HWM, + // ── Sync producers, join tokens, enrollment ── + "sync_producer_hwm" => super::sync_producer::SYNC_PRODUCER_HWM, + "sync_producers" => super::sync_producer::SYNC_PRODUCERS, + "sync_peer_bindings" => super::sync_producer::SYNC_PEER_BINDINGS, + "join_token_states" => super::sync_producer::JOIN_TOKEN_STATES, + "enrollment_preauthorizations" => super::sync_producer::ENROLLMENT_PREAUTHORIZATIONS, + // ── DDL objects ── + "materialized_views" => MATERIALIZED_VIEWS, + "continuous_aggregates" => CONTINUOUS_AGGREGATES, + "functions" => FUNCTIONS, + "procedures" => PROCEDURES, + "triggers" => TRIGGERS, + "arrays" => ARRAYS, + "dependencies" => DEPENDENCIES, + "sequences" => SEQUENCES, + "sequence_state" => SEQUENCE_STATE, + "synonym_groups" => SYNONYM_GROUPS, + "custom_types" => CUSTOM_TYPES, + "custom_type_oid_hwm" => super::custom_type_oid_hwm::CUSTOM_TYPE_OID_HWM, + "wasm_modules" => WASM_MODULES, + "rls_policies" => super::rls::RLS_POLICIES, + "redaction_policies" => super::redaction::REDACTION_POLICIES, + // ── Event Plane definitions ── + "change_streams" => CHANGE_STREAMS, + "consumer_groups" => CONSUMER_GROUPS, + "schedules" => SCHEDULES, + "retention_policies" => RETENTION_POLICIES, + "alert_rules" => ALERT_RULES, + "topics_ep" => TOPICS_EP, + "streaming_mvs" => STREAMING_MVS, + // ── Databases and quotas ── + "databases" => DATABASES, + "databases_by_name" => DATABASES_BY_NAME, + "database_grants" => DATABASE_GRANTS, + "database_quotas" => DATABASE_QUOTAS, + "tenant_quotas" => TENANT_QUOTAS, + // ── Clone copy-on-write ── + "clone_copyups" => CLONE_COPYUPS, + "clone_tombstones" => CLONE_TOMBSTONES, + "clone_kv_tombstones" => CLONE_KV_TOMBSTONES, + "clone_lineage" => CLONE_LINEAGE, + "clone_source_drains" => super::clone_source_drains::CLONE_SOURCE_DRAINS, + // ── Cluster restore points ── + "restore_points" => super::restore_points::RESTORE_POINTS, + // ── Scheduled backups ── + "backup_schedule_marks" => super::backup_schedule_marks::BACKUP_SCHEDULE_MARKS, +]; + +/// The labels of every replicated table, in image order. +pub fn replicated_table_labels() -> Vec<&'static str> { + REPLICATED_TABLES.iter().map(|table| table.label).collect() +} + +/// A read transaction on the system catalog, held to dump the replicated +/// tables at one commit point. +pub struct ReplicatedCatalogRead { + txn: ReadTransaction, +} + +impl ReplicatedCatalogRead { + /// Read every replicated table: `(label, rows)` in image order. + pub fn dump(&self) -> crate::Result> { + REPLICATED_TABLES + .iter() + .map(|table| Ok((table.label.to_string(), (table.dump)(&self.txn)?))) + .collect() + } +} + +impl SystemCatalog { + /// Open a read transaction for [`ReplicatedCatalogRead::dump`]. The dump + /// sees the catalog as of this call, whatever commits after it. + pub fn begin_replicated_read(&self) -> crate::Result { + let txn = self + .db + .begin_read() + .map_err(|e| catalog_err("replicated image: begin read", e))?; + Ok(ReplicatedCatalogRead { txn }) + } + + /// Replace every replicated table with `tables`, in one write + /// transaction. + /// + /// A table that also takes this node's own writes merges instead (see + /// [`super::replicated_image_merge`]): a counter or a fencing epoch never + /// moves down. + /// + /// `tables` must name each replicated table exactly once. A missing, + /// repeated, or unknown label fails before anything is written: the image + /// came from a build with a different table set. + pub fn replace_replicated_tables(&self, tables: &[(String, RawRows)]) -> crate::Result<()> { + if tables.len() != REPLICATED_TABLES.len() { + return Err(crate::Error::Internal { + detail: format!( + "metadata image carries {} tables; this build replicates {}", + tables.len(), + REPLICATED_TABLES.len() + ), + }); + } + let mut ordered: Vec<(&ReplicatedTable, &RawRows)> = Vec::with_capacity(tables.len()); + for table in REPLICATED_TABLES { + let matches: Vec<&RawRows> = tables + .iter() + .filter(|(label, _)| label == table.label) + .map(|(_, rows)| rows) + .collect(); + match matches.as_slice() { + [rows] => ordered.push((table, rows)), + _ => { + return Err(crate::Error::Internal { + detail: format!( + "metadata image must carry table '{}' exactly once, found {}", + table.label, + matches.len() + ), + }); + } + } + } + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("replicated image: begin write", e))?; + for (table, rows) in ordered { + match super::replicated_image_merge::merge_for(table.label) { + Some(merge) => { + let local = (table.current)(&txn)?; + (table.replace)(&txn, &merge(&local, rows)?)?; + } + None => (table.replace)(&txn, rows)?, + } + } + txn.commit() + .map_err(|e| catalog_err("replicated image: commit", e))?; + self.reload_event_definitions()?; + Ok(()) + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::security::catalog::StoredCollection; + use nodedb_types::DatabaseId; + + /// Every replicated table is a bootstrap table, and no label repeats. + #[test] + fn replicated_tables_are_bootstrap_tables() { + let bootstrap: Vec<&str> = super::super::bootstrap_tables::BOOTSTRAP_TABLES + .iter() + .map(|table| table.label) + .collect(); + let labels = replicated_table_labels(); + for label in &labels { + assert!( + bootstrap.contains(label), + "{label} is not a bootstrap table" + ); + } + let mut sorted = labels.clone(); + sorted.sort_unstable(); + sorted.dedup(); + assert_eq!(sorted.len(), labels.len(), "a replicated label repeats"); + } + + /// A dump replaces another catalog's tables exactly: rows it lacks are + /// removed, rows it holds are written. + #[test] + fn dump_then_replace_copies_the_replicated_tables() { + let source = SystemCatalog::open_in_memory().unwrap(); + let target = SystemCatalog::open_in_memory().unwrap(); + source + .put_collection( + DatabaseId::DEFAULT, + &StoredCollection::stamped_for_test(1, "kept", "admin"), + ) + .unwrap(); + target + .put_collection( + DatabaseId::DEFAULT, + &StoredCollection::stamped_for_test(1, "stale", "admin"), + ) + .unwrap(); + + let image = source.begin_replicated_read().unwrap().dump().unwrap(); + target.replace_replicated_tables(&image).unwrap(); + + assert!( + target + .get_collection(DatabaseId::DEFAULT, 1, "kept") + .unwrap() + .is_some() + ); + assert!( + target + .get_collection(DatabaseId::DEFAULT, 1, "stale") + .unwrap() + .is_none() + ); + assert_eq!( + target.begin_replicated_read().unwrap().dump().unwrap(), + image + ); + } + + /// A counter this node advanced past the image keeps its local value, + /// and one the image advanced further takes the image's. + #[test] + fn replace_never_moves_a_local_counter_down() { + let source = SystemCatalog::open_in_memory().unwrap(); + let target = SystemCatalog::open_in_memory().unwrap(); + source.put_surrogate_hwm(10).unwrap(); + source.save_next_user_id(40).unwrap(); + target.put_surrogate_hwm(25).unwrap(); + target.save_next_user_id(7).unwrap(); + + let image = source.begin_replicated_read().unwrap().dump().unwrap(); + target.replace_replicated_tables(&image).unwrap(); + + assert_eq!(target.get_surrogate_hwm().unwrap(), 25); + assert_eq!(target.load_next_user_id().unwrap(), 40); + } + + /// An image with a missing table is refused before any write. + #[test] + fn an_incomplete_image_is_refused() { + let catalog = SystemCatalog::open_in_memory().unwrap(); + let mut image = catalog.begin_replicated_read().unwrap().dump().unwrap(); + image.pop(); + assert!(catalog.replace_replicated_tables(&image).is_err()); + } +} diff --git a/nodedb/src/control/security/catalog/replicated_image_merge.rs b/nodedb/src/control/security/catalog/replicated_image_merge.rs new file mode 100644 index 000000000..d4f2fe500 --- /dev/null +++ b/nodedb/src/control/security/catalog/replicated_image_merge.rs @@ -0,0 +1,289 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Merges for replicated tables that also take this node's own writes. +//! +//! An install replaces most tables with the image's rows. The tables below +//! also hold values this node writes outside group 0 apply, and each such +//! value is monotonic. Taking the image's row can move it down and reissue +//! an id, a sequence number, or a fencing epoch. So the install keeps, per +//! row, the higher of the local and the image value: +//! - `surrogate_hwm`: the surrogate assigner flushes this node's watermark. +//! - `sync_producer_hwm`: the producer registry flushes this node's +//! allocator on every registration. +//! - `sync_producers`: registration and fencing write this node's row before +//! they propose it; the fencing epoch never moves down. +//! - `metadata`: the credential store saves `next_user_id`. +//! - `tenant_id_hwm`: tenant creation allocates the next id on this node. +//! - `topics_ep`: publishing advances a topic's last sequence and position. +//! +//! Two more tables take local writes that are not counters: +//! - `sync_peer_bindings`: a sync session claims a peer id on this node +//! before the claim replicates. Apply keeps the lower producer id, and the +//! merge keeps it too. A claim only this node holds stays until it commits. +//! - `sequence_state`: a GAP_FREE reservation logs its outcome as a `log:` +//! row on the node that ran it. Log rows are node-local, so the merge keeps +//! this node's log rows and drops the image's. Counter rows follow the +//! image: on a cluster node only apply writes them, and `RESTART` lowers +//! them on purpose. +//! +//! `database_hwm`, `surrogate_reserve_index`, and `custom_type_oid_hwm` are +//! written only by group 0 apply on a cluster node. They merge the same way, +//! so no install moves a watermark down even if a local writer appears. + +use std::collections::BTreeMap; + +use super::replicated_image::RawRows; +use super::sync_producer::{StoredPeerBinding, StoredProducerRegistration}; +use super::types::catalog_err; +use crate::event::topic::TopicDef; + +/// Merge `local` rows with `image` rows into the rows the table holds after +/// the install. +pub(super) type MergeFn = fn(&RawRows, &RawRows) -> crate::Result; + +/// The merge for `label`, or `None` when the image replaces the table. +pub(super) fn merge_for(label: &str) -> Option { + let merge: MergeFn = match label { + "surrogate_hwm" | "custom_type_oid_hwm" => max_u32_rows, + "sync_producer_hwm" | "database_hwm" | "surrogate_reserve_index" | "tenant_id_hwm" => { + max_u64_rows + } + "metadata" => merge_metadata, + "sync_producers" => merge_producers, + "topics_ep" => merge_topics, + "sync_peer_bindings" => merge_peer_bindings, + "sequence_state" => merge_sequence_state, + _ => return None, + }; + Some(merge) +} + +fn fixed(bytes: &[u8], what: &str) -> crate::Result<[u8; N]> { + bytes + .try_into() + .map_err(|_| catalog_err("replicated image merge", format!("{what}: bad width"))) +} + +/// Every key of `local` or `image`, each value the higher of the two. +fn max_rows( + local: &RawRows, + image: &RawRows, + higher: fn(&[u8], &[u8]) -> crate::Result>, +) -> crate::Result { + max_rows_with(local, image, higher) +} + +fn higher_u32(a: &[u8], b: &[u8]) -> crate::Result> { + let a = u32::from_le_bytes(fixed(a, "u32 counter")?); + let b = u32::from_le_bytes(fixed(b, "u32 counter")?); + Ok(a.max(b).to_le_bytes().to_vec()) +} + +fn higher_u64(a: &[u8], b: &[u8]) -> crate::Result> { + let a = u64::from_le_bytes(fixed(a, "u64 counter")?); + let b = u64::from_le_bytes(fixed(b, "u64 counter")?); + Ok(a.max(b).to_le_bytes().to_vec()) +} + +fn max_u32_rows(local: &RawRows, image: &RawRows) -> crate::Result { + max_rows(local, image, higher_u32) +} + +fn max_u64_rows(local: &RawRows, image: &RawRows) -> crate::Result { + max_rows(local, image, higher_u64) +} + +/// The image's rows, with `next_user_id` the higher of the two. +fn merge_metadata(local: &RawRows, image: &RawRows) -> crate::Result { + const NEXT_USER_ID: &[u8] = b"next_user_id"; + let mut merged: BTreeMap, Vec> = image.iter().cloned().collect(); + if let Some((_, local_next)) = local.iter().find(|(key, _)| key == NEXT_USER_ID) { + let kept = match merged.get(NEXT_USER_ID) { + Some(from_image) => higher_u64(local_next, from_image)?, + None => local_next.clone(), + }; + merged.insert(NEXT_USER_ID.to_vec(), kept); + } + Ok(merged.into_iter().collect()) +} + +/// The image's registrations, each fencing epoch the higher of the two. +/// A registration only this node holds stays: its proposal can still +/// commit, and its producer id is already spent. +fn merge_producers(local: &RawRows, image: &RawRows) -> crate::Result { + let decode = |bytes: &[u8]| -> crate::Result { + zerompk::from_msgpack(bytes).map_err(|e| catalog_err("decode producer registration", e)) + }; + max_rows_with(local, image, |local_value, image_value| { + let local_row = decode(local_value)?; + let mut row = decode(image_value)?; + row.current_epoch = row.current_epoch.max(local_row.current_epoch); + zerompk::to_msgpack_vec(&row).map_err(|e| catalog_err("encode producer registration", e)) + }) +} + +/// The image's bindings, each key owned by the lower producer id of the two. +/// A claim only this node holds stays: its proposal can still commit. +fn merge_peer_bindings(local: &RawRows, image: &RawRows) -> crate::Result { + let decode = |bytes: &[u8]| -> crate::Result { + zerompk::from_msgpack(bytes).map_err(|e| catalog_err("decode peer binding", e)) + }; + max_rows_with(local, image, |local_value, image_value| { + let local_owner = decode(local_value)?; + let image_owner = decode(image_value)?; + Ok(if local_owner.producer_id < image_owner.producer_id { + local_value.to_vec() + } else { + image_value.to_vec() + }) + }) +} + +/// Whether a `sequence_state` key names a GAP_FREE log row. The key is +/// `"{database_id}:{tenant_id}:{name}"`, and a log row's name starts with +/// `log:`, which no SQL identifier produces. +fn is_sequence_log_key(key: &[u8]) -> bool { + let mut parts = key.splitn(3, |byte| *byte == b':'); + let _database = parts.next(); + let _tenant = parts.next(); + parts.next().is_some_and(|name| name.starts_with(b"log:")) +} + +/// The image's counter rows plus this node's own log rows. +fn merge_sequence_state(local: &RawRows, image: &RawRows) -> crate::Result { + let mut merged: BTreeMap, Vec> = image + .iter() + .filter(|(key, _)| !is_sequence_log_key(key)) + .cloned() + .collect(); + merged.extend( + local + .iter() + .filter(|(key, _)| is_sequence_log_key(key)) + .cloned(), + ); + Ok(merged.into_iter().collect()) +} + +/// The image's topics, each keeping the higher last sequence and the later +/// position of the two. A topic the image lacks was dropped and goes. +fn merge_topics(local: &RawRows, image: &RawRows) -> crate::Result { + let decode = |bytes: &[u8]| -> crate::Result { + zerompk::from_msgpack(bytes).map_err(|e| catalog_err("decode topic", e)) + }; + let local: BTreeMap<&[u8], &[u8]> = local + .iter() + .map(|(key, value)| (key.as_slice(), value.as_slice())) + .collect(); + image + .iter() + .map(|(key, image_value)| { + let Some(&local_value) = local.get(key.as_slice()) else { + return Ok((key.clone(), image_value.clone())); + }; + let local_topic = decode(local_value)?; + let mut topic = decode(image_value.as_slice())?; + topic.last_sequence = topic.last_sequence.max(local_topic.last_sequence); + if (local_topic.last_epoch, local_topic.last_lsn) > (topic.last_epoch, topic.last_lsn) { + topic.last_epoch = local_topic.last_epoch; + topic.last_lsn = local_topic.last_lsn; + } + let bytes = + zerompk::to_msgpack_vec(&topic).map_err(|e| catalog_err("encode topic", e))?; + Ok((key.clone(), bytes)) + }) + .collect() +} + +/// Every key of `local` or `image`. A key in both takes +/// `combine(local, image)`. +fn max_rows_with( + local: &RawRows, + image: &RawRows, + combine: impl Fn(&[u8], &[u8]) -> crate::Result>, +) -> crate::Result { + let mut merged: BTreeMap, Vec> = image.iter().cloned().collect(); + for (key, value) in local { + let kept = match merged.get(key) { + Some(from_image) => combine(value.as_slice(), from_image.as_slice())?, + None => value.clone(), + }; + merged.insert(key.clone(), kept); + } + Ok(merged.into_iter().collect()) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn row(key: &str, value: u64) -> (Vec, Vec) { + (key.as_bytes().to_vec(), value.to_le_bytes().to_vec()) + } + + #[test] + fn a_counter_never_moves_down() { + let merged = max_u64_rows(&vec![row("global", 9)], &vec![row("global", 4)]).unwrap(); + assert_eq!(merged, vec![row("global", 9)]); + let merged = max_u64_rows(&vec![row("global", 2)], &vec![row("global", 4)]).unwrap(); + assert_eq!(merged, vec![row("global", 4)]); + } + + #[test] + fn next_user_id_keeps_the_higher_and_other_rows_follow_the_image() { + let local = vec![row("next_user_id", 12), row("stale", 1)]; + let image = vec![row("next_user_id", 7), row("fresh", 2)]; + let merged = merge_metadata(&local, &image).unwrap(); + assert_eq!(merged, vec![row("fresh", 2), row("next_user_id", 12)]); + } + + #[test] + fn a_fencing_epoch_never_moves_down() { + let reg = |epoch: u64| StoredProducerRegistration { + producer_id: 3, + current_epoch: epoch, + tenant_id: 1, + user_id: 1, + created_ms: 0, + }; + let enc = |r: &StoredProducerRegistration| zerompk::to_msgpack_vec(r).unwrap(); + let key = b"lite-a".to_vec(); + let merged = merge_producers( + &vec![(key.clone(), enc(®(5)))], + &vec![(key.clone(), enc(®(2)))], + ) + .unwrap(); + let kept: StoredProducerRegistration = zerompk::from_msgpack(&merged[0].1).unwrap(); + assert_eq!(kept.current_epoch, 5); + } + + #[test] + fn a_peer_binding_keeps_the_lower_producer() { + let enc = |producer_id: u64| { + zerompk::to_msgpack_vec(&StoredPeerBinding { + producer_id, + bound_ms: 0, + }) + .unwrap() + }; + let key = b"peer".to_vec(); + let pending = b"pending".to_vec(); + let merged = merge_peer_bindings( + &vec![(key.clone(), enc(4)), (pending.clone(), enc(9))], + &vec![(key.clone(), enc(7))], + ) + .unwrap(); + assert_eq!(merged, vec![(key, enc(4)), (pending, enc(9))]); + } + + #[test] + fn sequence_log_rows_stay_node_local() { + let local = vec![row("2:1:log:s:10:committed", 1), row("2:1:s", 50)]; + let image = vec![row("2:1:log:s:11:committed", 2), row("2:1:s", 40)]; + let merged = merge_sequence_state(&local, &image).unwrap(); + assert_eq!( + merged, + vec![row("2:1:log:s:10:committed", 1), row("2:1:s", 40)] + ); + } +} diff --git a/nodedb/src/control/security/catalog/restore_points.rs b/nodedb/src/control/security/catalog/restore_points.rs new file mode 100644 index 000000000..8bbf9d159 --- /dev/null +++ b/nodedb/src/control/security/catalog/restore_points.rs @@ -0,0 +1,99 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `_system.restore_points`: every cluster restore point, keyed by id. +//! +//! The metadata group replicates each point, so every node holds the same +//! rows and any node can list them. + +use redb::{ReadableDatabase, ReadableTable, TableError}; + +use super::types::{SystemCatalog, catalog_err}; + +/// Redb table: restore point id -> MessagePack [`StoredRestorePoint`]. +pub(super) const RESTORE_POINTS: redb::TableDefinition = + redb::TableDefinition::new("_system.restore_points"); + +/// One cluster restore point. +#[derive(Debug, Clone, Copy, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] +pub struct StoredRestorePoint { + /// The metadata log index of the entry that created the point. + pub id: u64, + /// The point's watermark: HLC wall time in nanoseconds. + pub hlc: u64, + /// Wall-clock milliseconds when the point was requested. + pub created_at_ms: u64, +} + +impl SystemCatalog { + /// Record `point`. A second write of the same id overwrites it with the + /// same values. + pub fn put_restore_point(&self, point: &StoredRestorePoint) -> crate::Result<()> { + let bytes = + zerompk::to_msgpack_vec(point).map_err(|e| catalog_err("encode restore point", e))?; + let txn = self + .db + .begin_write() + .map_err(|e| catalog_err("restore point write txn", e))?; + { + let mut table = txn + .open_table(RESTORE_POINTS) + .map_err(|e| catalog_err("open restore points", e))?; + table + .insert(point.id, bytes.as_slice()) + .map_err(|e| catalog_err("insert restore point", e))?; + } + txn.commit() + .map_err(|e| catalog_err("restore point commit", e)) + } + + /// Every restore point, oldest first. + pub fn list_restore_points(&self) -> crate::Result> { + let txn = self + .db + .begin_read() + .map_err(|e| catalog_err("restore point read txn", e))?; + let table = match txn.open_table(RESTORE_POINTS) { + Ok(table) => table, + Err(TableError::TableDoesNotExist(_)) => return Ok(Vec::new()), + Err(e) => return Err(catalog_err("open restore points", e)), + }; + let mut points = Vec::new(); + for row in table + .range(..) + .map_err(|e| catalog_err("range restore points", e))? + { + let (_, value) = row.map_err(|e| catalog_err("read restore point", e))?; + points.push( + zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("decode restore point", e))?, + ); + } + Ok(points) + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn points_list_in_id_order_and_rewrites_are_idempotent() { + let dir = tempfile::tempdir().unwrap(); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).unwrap(); + assert!(catalog.list_restore_points().unwrap().is_empty()); + let later = StoredRestorePoint { + id: 20, + hlc: 2_000, + created_at_ms: 2, + }; + let earlier = StoredRestorePoint { + id: 10, + hlc: 1_000, + created_at_ms: 1, + }; + catalog.put_restore_point(&later).unwrap(); + catalog.put_restore_point(&earlier).unwrap(); + catalog.put_restore_point(&earlier).unwrap(); + assert_eq!(catalog.list_restore_points().unwrap(), vec![earlier, later]); + } +} diff --git a/nodedb/src/control/security/catalog/security.rs b/nodedb/src/control/security/catalog/security.rs index 44a9ee1ed..65dba76b8 100644 --- a/nodedb/src/control/security/catalog/security.rs +++ b/nodedb/src/control/security/catalog/security.rs @@ -198,6 +198,46 @@ impl SystemCatalog { write_txn.commit().map_err(|e| catalog_err("commit", e)) } + /// Remove every permission row granted on exactly `target`. Returns how + /// many rows went. + pub fn delete_permissions_for_target(&self, target: &str) -> crate::Result { + let prefix = format!("{target}:"); + let write_txn = self + .db + .begin_write() + .map_err(|e| catalog_err("write txn", e))?; + let removed = { + let mut table = write_txn + .open_table(PERMISSIONS) + .map_err(|e| catalog_err("open perms", e))?; + // A name can itself hold ':', so a key prefix match is only a + // candidate: the stored target must match exactly. + let mut keys = Vec::new(); + for entry in table + .range(prefix.as_str()..) + .map_err(|e| catalog_err("range perms", e))? + { + let (key, value) = entry.map_err(|e| catalog_err("read perm", e))?; + if !key.value().starts_with(&prefix) { + break; + } + let perm: StoredPermission = zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("deser perm", e))?; + if perm.target == target { + keys.push(key.value().to_string()); + } + } + for key in &keys { + table + .remove(key.as_str()) + .map_err(|e| catalog_err("remove perm", e))?; + } + keys.len() + }; + write_txn.commit().map_err(|e| catalog_err("commit", e))?; + Ok(removed) + } + pub fn load_all_permissions(&self) -> crate::Result> { let read_txn = self .db diff --git a/nodedb/src/control/security/catalog/sequence_types.rs b/nodedb/src/control/security/catalog/sequence_types.rs index 4de8b23e4..74cbcd4bf 100644 --- a/nodedb/src/control/security/catalog/sequence_types.rs +++ b/nodedb/src/control/security/catalog/sequence_types.rs @@ -127,7 +127,8 @@ pub struct SequenceState { pub database_id: u64, pub tenant_id: u64, pub name: String, - /// Current value (last returned by nextval on this node). + /// Last value nextval returned once `is_called`. Before that, the value + /// the next nextval returns. pub current_value: i64, /// Whether nextval has been called at least once on this node. pub is_called: bool, @@ -151,7 +152,7 @@ impl SequenceState { database_id, tenant_id, name, - // Start one step before start_value so the first nextval returns start_value. + // Uncalled: the first nextval returns start_value. current_value: start_value, is_called: false, epoch, diff --git a/nodedb/src/control/security/catalog/surrogate_hwm.rs b/nodedb/src/control/security/catalog/surrogate_hwm.rs index 001f14cf9..5e23b5c0c 100644 --- a/nodedb/src/control/security/catalog/surrogate_hwm.rs +++ b/nodedb/src/control/security/catalog/surrogate_hwm.rs @@ -18,7 +18,7 @@ pub const SURROGATE_HWM: redb::TableDefinition<&str, u32> = /// `SurrogateReserve` has been folded into the global watermark `G` (`u64`). /// Persisted ATOMICALLY with `SURROGATE_HWM` in cluster mode so a crash can /// never leave the seeded `G` and the applied-reserve cursor inconsistent -/// (which would diverge `G` across nodes on the next restart replay). +/// (which diverges `G` across nodes on the next restart replay). pub const SURROGATE_RESERVE_INDEX: redb::TableDefinition<&str, u64> = redb::TableDefinition::new("_system.surrogate_reserve_index"); @@ -29,6 +29,16 @@ impl SystemCatalog { /// Persist the surrogate allocator high-watermark. Overwrites the /// singleton row. pub fn put_surrogate_hwm(&self, hwm: u32) -> crate::Result<()> { + #[cfg(test)] + if self + .fail_next_surrogate_write + .swap(false, std::sync::atomic::Ordering::SeqCst) + { + return Err(catalog_err( + "surrogate watermark write", + "injected surrogate write failure", + )); + } let txn = self .db .begin_write() @@ -47,11 +57,21 @@ impl SystemCatalog { /// Cluster-mode: persist the global watermark `hwm` AND the /// applied-reserve cursor `reserve_index` together in a SINGLE redb write - /// transaction. Atomicity is mandatory: if these two values could be - /// written separately, a crash between them would seed the next restart + /// transaction. Atomicity is mandatory: if these two values were + /// written separately, a crash between them seeds the next restart /// with a mismatched `(G, cursor)` pair, causing metadata-log replay to /// re-apply (or wrongly skip) reservations and diverge `G` across nodes. pub fn put_surrogate_reserve_state(&self, hwm: u32, reserve_index: u64) -> crate::Result<()> { + #[cfg(test)] + if self + .fail_next_surrogate_write + .swap(false, std::sync::atomic::Ordering::SeqCst) + { + return Err(catalog_err( + "surrogate watermark write", + "injected surrogate write failure", + )); + } let txn = self .db .begin_write() diff --git a/nodedb/src/control/security/catalog/surrogate_pk.rs b/nodedb/src/control/security/catalog/surrogate_pk.rs index 5c85bc5d6..8ea246248 100644 --- a/nodedb/src/control/security/catalog/surrogate_pk.rs +++ b/nodedb/src/control/security/catalog/surrogate_pk.rs @@ -24,10 +24,9 @@ use nodedb_types::{CollectionKey, DatabaseId, Surrogate, TenantId}; use redb::{ReadableDatabase, ReadableTable, ReadableTableMetadata}; -#[allow(unused_imports)] // SURROGATE_PK_REV_LEGACY is used only in #[cfg(test)] helpers use super::types::{ - SURROGATE_PK_LEGACY, SURROGATE_PK_REV_LEGACY, SURROGATE_PK_REV_V2, SURROGATE_PK_REV_V3, - SURROGATE_PK_V2, SURROGATE_PK_V3, SystemCatalog, catalog_err, + SURROGATE_PK_LEGACY, SURROGATE_PK_REV_V2, SURROGATE_PK_REV_V3, SURROGATE_PK_V2, + SURROGATE_PK_V3, SystemCatalog, catalog_err, }; impl SystemCatalog { diff --git a/nodedb/src/control/security/catalog/synonym_groups.rs b/nodedb/src/control/security/catalog/synonym_groups.rs index 2f68ed8de..0468c2148 100644 --- a/nodedb/src/control/security/catalog/synonym_groups.rs +++ b/nodedb/src/control/security/catalog/synonym_groups.rs @@ -23,6 +23,10 @@ pub struct StoredSynonymGroup { pub name: String, pub terms: Vec, pub created_at: u64, + /// Stamped at propose time on every put; fences a replayed delete to + /// the incarnation it targeted. + #[serde(default)] + pub modification_hlc: nodedb_types::Hlc, } impl SystemCatalog { @@ -211,6 +215,7 @@ mod tests { name: name.into(), terms: terms.iter().map(|t| (*t).to_string()).collect(), created_at: 1000, + modification_hlc: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/security/catalog/system_catalog.rs b/nodedb/src/control/security/catalog/system_catalog.rs index efa13ba3a..80efd3e78 100644 --- a/nodedb/src/control/security/catalog/system_catalog.rs +++ b/nodedb/src/control/security/catalog/system_catalog.rs @@ -30,6 +30,14 @@ pub struct SystemCatalog { pub(super) fail_next_function_wasm_write: Arc, #[cfg(test)] pub(super) fail_next_collection_write: Arc, + #[cfg(test)] + pub(super) fail_next_surrogate_write: Arc, +} + +impl crate::storage::RedbBacked for SystemCatalog { + fn redb_database(&self) -> &Database { + &self.db + } } impl SystemCatalog { @@ -57,6 +65,8 @@ impl SystemCatalog { fail_next_function_wasm_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), #[cfg(test)] fail_next_collection_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), + #[cfg(test)] + fail_next_surrogate_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), }; catalog.bootstrap_default_database()?; catalog.reload_event_definitions()?; @@ -81,6 +91,8 @@ impl SystemCatalog { fail_next_function_wasm_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), #[cfg(test)] fail_next_collection_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), + #[cfg(test)] + fail_next_surrogate_write: Arc::new(std::sync::atomic::AtomicBool::new(false)), }; catalog.bootstrap_default_database()?; catalog.reload_event_definitions()?; @@ -106,6 +118,14 @@ impl SystemCatalog { .store(true, std::sync::atomic::Ordering::SeqCst); } + /// Make the next surrogate watermark write raise, so an applier's error + /// path runs. + #[cfg(test)] + pub(crate) fn fail_next_surrogate_write_for_test(&self) { + self.fail_next_surrogate_write + .store(true, std::sync::atomic::Ordering::SeqCst); + } + /// Bootstrap every `_system.*` table from the canonical registry — /// but only if at least one is actually missing. Probing read-only /// first keeps `open` byte-idempotent on an already-bootstrapped @@ -113,7 +133,7 @@ impl SystemCatalog { /// page on redb every time, so an unconditional bootstrap rewrites /// `system.redb` on every boot (changing its size/md5) even when /// nothing changed — and a boot that then fails its integrity check - /// would have mutated persistent catalog state on its way out. + /// mutates persistent catalog state on its way out. /// Opening a table in a write transaction creates it if absent; the /// registry is the single source of truth, so a table cannot be /// read in production code without being bootstrapped here. Returns @@ -279,7 +299,7 @@ mod tests { // bootstrap registry, so boot-time readers (integrity walk, // continuous-aggregate replay, …) open existing empty tables // instead of hitting "table does not exist". Re-opening each - // entry read-only would fail with `TableDoesNotExist` if the + // entry read-only fails with `TableDoesNotExist` if the // init path ever stopped iterating the registry. let dir = tempfile::tempdir().unwrap(); let catalog = SystemCatalog::open(&dir.path().join("system.redb")).unwrap(); diff --git a/nodedb/src/control/security/catalog/tables.rs b/nodedb/src/control/security/catalog/tables.rs index 269b83e53..24d55a025 100644 --- a/nodedb/src/control/security/catalog/tables.rs +++ b/nodedb/src/control/security/catalog/tables.rs @@ -35,12 +35,6 @@ pub(super) const OWNERS: TableDefinition<&str, &[u8]> = TableDefinition::new("_s // ── Collections ─────────────────────────────────────────────────────── -/// Table (legacy, pre-database-boundary): `"{tenant_id}:{name}"` -> msgpack -/// collection metadata. Used only by the idempotent migration path that reads -/// legacy rows and rewrites them under `COLLECTIONS` with the database_id key. -pub(super) const COLLECTIONS_LEGACY: TableDefinition<&str, &[u8]> = - TableDefinition::new("_system.collections"); - /// Table: `(database_id: u64, "{tenant_id}:{name}")` -> MessagePack collection metadata. /// /// The compound key prepends `database_id` (as raw `u64`) so every collection @@ -78,8 +72,8 @@ pub(super) const L2_CLEANUP_QUEUE: TableDefinition<(u64, u64, &str), &[u8]> = /// Populated when the synchronous, result-checked engine purge for a /// dropped collection (`clear_collection_all_engines` via /// `MetaOp::UnregisterCollection`) FAILS on this node after the catalog -/// row has already been removed. Left unrecorded, that failure would -/// leave engine storage rows behind a gone catalog row — permanent +/// row has already been removed. Left unrecorded, that failure +/// leaves engine storage rows behind a gone catalog row — permanent /// divergence that resurrects the dropped collection's history on /// re-CREATE. A Tokio worker (and a boot-time drain) retries the engine /// purge for each entry until it succeeds, then removes the row. This @@ -138,7 +132,7 @@ pub(super) const SURROGATE_PK_V3: TableDefinition<(u64, u64, &str, &[u8]), u32> /// Table (legacy): `(collection, surrogate)` -> encoded pk bytes. /// Used only by the idempotent migration that prefixes rows with database_id. -#[allow(dead_code)] +#[cfg(test)] pub(super) const SURROGATE_PK_REV_LEGACY: TableDefinition<(&str, u32), &[u8]> = TableDefinition::new("_system.surrogate_pk_rev"); @@ -238,8 +232,9 @@ pub(super) const DATABASES: TableDefinition = TableDefinition::new(" pub(super) const DATABASES_BY_NAME: TableDefinition<&str, u64> = TableDefinition::new("_system.databases_by_name"); -/// Table: singleton `"global"` -> highest allocated database id (`u64`). -/// Persisted by `DatabaseRegistry::flush`; seeded at startup. +/// Table: `"global"` -> highest issued database id, `"reserve_index"` -> +/// highest metadata log index folded into it (`u64`). Written on every +/// allocation; seeds `DatabaseRegistry` at startup. pub(super) const DATABASE_HWM: TableDefinition<&str, u64> = TableDefinition::new("_system.database_hwm"); @@ -273,7 +268,7 @@ pub(super) const BLACKLIST: TableDefinition<&str, &[u8]> = /// Table: scope name -> MessagePack-serialized `StoredScopeQuota`. /// /// Quota definitions are admin-authored catalog objects, like scope grants: -/// a definition that lived only in process memory would be forgotten by every +/// a definition that lived only in process memory is forgotten by every /// restart, silently lifting every cap it expressed. pub(super) const SCOPE_QUOTAS: TableDefinition<&str, &[u8]> = TableDefinition::new("_system.scope_quotas"); diff --git a/nodedb/src/control/security/catalog/tenant_group_marks.rs b/nodedb/src/control/security/catalog/tenant_group_marks.rs index e47ad2c8d..b37812bff 100644 --- a/nodedb/src/control/security/catalog/tenant_group_marks.rs +++ b/nodedb/src/control/security/catalog/tenant_group_marks.rs @@ -1,7 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 //! Persistent per-group tenant write marks backing -//! `_system.tenant_group_marks`. +//! `_system.tenant_group_marks` and `_system.tenant_group_restore_marks`. //! //! For each data group this node replicates, the newest commit HLC of any //! write of each tenant the group applied. The apply loop writes a group's @@ -9,6 +9,10 @@ //! committed entry is either covered by a persisted mark or above the floor, //! where Raft delivers it again after a restart and the loop derives its mark //! again. A Calvin commit writes its mark before its install is acknowledged. +//! +//! A write a RESTORE re-issued keeps its mark apart, with the id of the +//! restore that wrote it, so a retry of that restore can tell its own writes +//! from every other write. use redb::{ReadableDatabase, ReadableTable, TableDefinition}; @@ -18,6 +22,11 @@ use super::types::{SystemCatalog, catalog_err}; pub(super) const TENANT_GROUP_MARKS: TableDefinition<(u64, u64), (u64, u8, &str)> = TableDefinition::new("_system.tenant_group_marks"); +/// Table: `(group_id, tenant_id)` -> `(commit_hlc, restore_id, collection)` +/// of the newest write a RESTORE re-issued. +pub(super) const TENANT_GROUP_RESTORE_MARKS: TableDefinition<(u64, u64), (u64, u64, &str)> = + TableDefinition::new("_system.tenant_group_restore_marks"); + /// One persisted mark. #[derive(Debug, Clone, PartialEq, Eq)] pub struct StoredGroupMark { @@ -29,19 +38,21 @@ pub struct StoredGroupMark { pub site: u8, /// The collection the write named, empty when it named none. pub collection: String, + /// The restore that re-issued the write, `0` for any other write. + pub restore_id: u64, } impl SystemCatalog { - /// Every persisted mark. + /// Every persisted mark, user and restore alike. pub fn load_tenant_group_marks(&self) -> crate::Result> { let read_txn = self .db .begin_read() .map_err(|e| catalog_err("load_tenant_group_marks read txn", e))?; + let mut marks = Vec::new(); let table = read_txn .open_table(TENANT_GROUP_MARKS) .map_err(|e| catalog_err("open tenant_group_marks", e))?; - let mut marks = Vec::new(); for entry in table .iter() .map_err(|e| catalog_err("iterate tenant_group_marks", e))? @@ -55,13 +66,35 @@ impl SystemCatalog { hlc, site, collection: collection.to_owned(), + restore_id: 0, + }); + } + let restore = read_txn + .open_table(TENANT_GROUP_RESTORE_MARKS) + .map_err(|e| catalog_err("open tenant_group_restore_marks", e))?; + for entry in restore + .iter() + .map_err(|e| catalog_err("iterate tenant_group_restore_marks", e))? + { + let (key, value) = + entry.map_err(|e| catalog_err("read tenant_group_restore_mark", e))?; + let (group_id, tenant_id) = key.value(); + let (hlc, restore_id, collection) = value.value(); + marks.push(StoredGroupMark { + group_id, + tenant_id, + hlc, + site: RESTORE_SITE_CODE, + collection: collection.to_owned(), + restore_id, }); } Ok(marks) } /// Raise every mark in `marks` in one transaction. A persisted mark at or - /// above the new one stays. + /// above the new one stays. A mark with a `restore_id` goes to the + /// restore table. pub fn raise_tenant_group_marks(&self, marks: &[StoredGroupMark]) -> crate::Result<()> { if marks.is_empty() { return Ok(()); @@ -74,22 +107,94 @@ impl SystemCatalog { let mut table = write_txn .open_table(TENANT_GROUP_MARKS) .map_err(|e| catalog_err("open tenant_group_marks", e))?; + let mut restore = write_txn + .open_table(TENANT_GROUP_RESTORE_MARKS) + .map_err(|e| catalog_err("open tenant_group_restore_marks", e))?; for mark in marks { let key = (mark.group_id, mark.tenant_id); - let current = table - .get(key) - .map_err(|e| catalog_err("get tenant_group_mark", e))? - .map(|guard| guard.value().0); - if current.is_some_and(|hlc| hlc >= mark.hlc) { - continue; + if mark.restore_id == 0 { + let current = table + .get(key) + .map_err(|e| catalog_err("get tenant_group_mark", e))? + .map(|guard| guard.value().0); + if current.is_some_and(|hlc| hlc >= mark.hlc) { + continue; + } + table + .insert(key, (mark.hlc, mark.site, mark.collection.as_str())) + .map_err(|e| catalog_err("insert tenant_group_mark", e))?; + } else { + let current = restore + .get(key) + .map_err(|e| catalog_err("get tenant_group_restore_mark", e))? + .map(|guard| guard.value().0); + if current.is_some_and(|hlc| hlc >= mark.hlc) { + continue; + } + restore + .insert(key, (mark.hlc, mark.restore_id, mark.collection.as_str())) + .map_err(|e| catalog_err("insert tenant_group_restore_mark", e))?; } - table - .insert(key, (mark.hlc, mark.site, mark.collection.as_str())) - .map_err(|e| catalog_err("insert tenant_group_mark", e))?; } } write_txn .commit() .map_err(|e| catalog_err("commit tenant_group_marks", e)) } + + /// Replace every persisted mark, user and restore alike, with `marks`. + /// A cluster restore rebuilds the marks at its point, offline: the clear + /// and the writes commit apart, and a failed restore empties the data + /// directory. + pub fn replace_tenant_group_marks(&self, marks: &[StoredGroupMark]) -> crate::Result<()> { + let write_txn = self + .db + .begin_write() + .map_err(|e| catalog_err("replace_tenant_group_marks txn", e))?; + write_txn + .delete_table(TENANT_GROUP_MARKS) + .map_err(|e| catalog_err("clear tenant_group_marks", e))?; + write_txn + .delete_table(TENANT_GROUP_RESTORE_MARKS) + .map_err(|e| catalog_err("clear tenant_group_restore_marks", e))?; + write_txn + .commit() + .map_err(|e| catalog_err("commit tenant_group_marks clear", e))?; + self.raise_tenant_group_marks(marks) + } +} + +/// The site code a restore mark carries. It matches +/// `MarkSite::Restore::code()`. +pub const RESTORE_SITE_CODE: u8 = 3; + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn user_and_restore_marks_persist_apart() { + let dir = tempfile::tempdir().expect("tempdir"); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).expect("catalog"); + let user = StoredGroupMark { + group_id: 2, + tenant_id: 1, + hlc: 100, + site: 0, + collection: "docs".into(), + restore_id: 0, + }; + let restore = StoredGroupMark { + hlc: 200, + site: RESTORE_SITE_CODE, + restore_id: 77, + ..user.clone() + }; + catalog + .raise_tenant_group_marks(&[user.clone(), restore.clone()]) + .expect("raise"); + let mut loaded = catalog.load_tenant_group_marks().expect("load"); + loaded.sort_by_key(|m| m.restore_id); + assert_eq!(loaded, vec![user, restore]); + } } diff --git a/nodedb/src/control/security/catalog/tenant_quotas.rs b/nodedb/src/control/security/catalog/tenant_quotas.rs index 1c22c3d54..e03d254f5 100644 --- a/nodedb/src/control/security/catalog/tenant_quotas.rs +++ b/nodedb/src/control/security/catalog/tenant_quotas.rs @@ -77,7 +77,7 @@ impl SystemCatalog { } /// Write a tenant quota record consensus already accepted. No validation - /// and no ceiling check — a rejection here would diverge nodes. + /// and no ceiling check — a rejection here diverges nodes. pub fn write_tenant_quota( &self, db_id: DatabaseId, @@ -231,6 +231,25 @@ impl SystemCatalog { // ── sum-of-tenant-quotas validation ────────────────────────────────────── + /// Check a tenant quota for a database the caller creates with + /// `db_quota`, before either exists in the catalog. + /// + /// `others` holds every other tenant quota the caller installs in the + /// same new database. Their sum plus `record` must fit `db_quota`. + pub fn check_tenant_quota_in_new_database( + db_quota: Option<&QuotaRecord>, + others: &[&QuotaRecord], + record: &QuotaRecord, + ) -> crate::Result<()> { + record.validate().map_err(|e| crate::Error::BadRequest { + detail: e.to_string(), + })?; + match db_quota { + Some(db_quota) => tenant_sum_fits(db_quota, others.iter().copied(), record), + None => Ok(()), + } + } + fn check_tenant_quota_ceiling( &self, db_id: DatabaseId, @@ -242,85 +261,96 @@ impl SystemCatalog { Some(q) => q, None => return Ok(()), }; - - // Build a GlobalQuotaCeiling from the database quota's non-zero limits. - let ceiling = GlobalQuotaCeiling { - max_memory_bytes: db_quota.max_memory_bytes, - max_storage_bytes: db_quota.max_storage_bytes, - max_qps: db_quota.max_qps as u64, - max_connections: db_quota.max_connections as u64, - }; - - // If all dimensions are zero, the database quota imposes no limits. - if ceiling.max_memory_bytes == 0 - && ceiling.max_storage_bytes == 0 - && ceiling.max_qps == 0 - && ceiling.max_connections == 0 - { - return Ok(()); - } - let tenants = self.list_tenant_quotas_for_database(db_id)?; + tenant_sum_fits( + &db_quota, + tenants + .iter() + .filter(|(tid, _)| *tid != tenant_id) + .map(|(_, rec)| rec), + proposed, + ) + } +} - let mut sum_memory: u64 = 0; - let mut sum_storage: u64 = 0; - let mut sum_qps: u64 = 0; - let mut sum_connections: u64 = 0; +/// Whether the tenant quotas `others` plus `proposed` fit the database quota. +fn tenant_sum_fits<'a>( + db_quota: &QuotaRecord, + others: impl Iterator, + proposed: &QuotaRecord, +) -> crate::Result<()> { + // Build a GlobalQuotaCeiling from the database quota's non-zero limits. + let ceiling = GlobalQuotaCeiling { + max_memory_bytes: db_quota.max_memory_bytes, + max_storage_bytes: db_quota.max_storage_bytes, + max_qps: db_quota.max_qps as u64, + max_connections: db_quota.max_connections as u64, + }; + + // If all dimensions are zero, the database quota imposes no limits. + if ceiling.max_memory_bytes == 0 + && ceiling.max_storage_bytes == 0 + && ceiling.max_qps == 0 + && ceiling.max_connections == 0 + { + return Ok(()); + } - for (tid, rec) in &tenants { - if *tid == tenant_id { - continue; // Will be replaced by `proposed`. - } - sum_memory = sum_memory.saturating_add(rec.max_memory_bytes); - sum_storage = sum_storage.saturating_add(rec.max_storage_bytes); - sum_qps = sum_qps.saturating_add(rec.max_qps as u64); - sum_connections = sum_connections.saturating_add(rec.max_connections as u64); - } + let mut sum_memory: u64 = 0; + let mut sum_storage: u64 = 0; + let mut sum_qps: u64 = 0; + let mut sum_connections: u64 = 0; - sum_memory = sum_memory.saturating_add(proposed.max_memory_bytes); - sum_storage = sum_storage.saturating_add(proposed.max_storage_bytes); - sum_qps = sum_qps.saturating_add(proposed.max_qps as u64); - sum_connections = sum_connections.saturating_add(proposed.max_connections as u64); - - if ceiling.max_memory_bytes > 0 && sum_memory > ceiling.max_memory_bytes { - return Err(crate::Error::QuotaOvercommit { - field: "max_memory_bytes".into(), - detail: format!( - "tenant sum {sum_memory} exceeds database quota {}", - ceiling.max_memory_bytes - ), - }); - } - if ceiling.max_storage_bytes > 0 && sum_storage > ceiling.max_storage_bytes { - return Err(crate::Error::QuotaOvercommit { - field: "max_storage_bytes".into(), - detail: format!( - "tenant sum {sum_storage} exceeds database quota {}", - ceiling.max_storage_bytes - ), - }); - } - if ceiling.max_qps > 0 && sum_qps > ceiling.max_qps { - return Err(crate::Error::QuotaOvercommit { - field: "max_qps".into(), - detail: format!( - "tenant sum {sum_qps} exceeds database quota {}", - ceiling.max_qps - ), - }); - } - if ceiling.max_connections > 0 && sum_connections > ceiling.max_connections { - return Err(crate::Error::QuotaOvercommit { - field: "max_connections".into(), - detail: format!( - "tenant sum {sum_connections} exceeds database quota {}", - ceiling.max_connections - ), - }); - } + for rec in others { + sum_memory = sum_memory.saturating_add(rec.max_memory_bytes); + sum_storage = sum_storage.saturating_add(rec.max_storage_bytes); + sum_qps = sum_qps.saturating_add(rec.max_qps as u64); + sum_connections = sum_connections.saturating_add(rec.max_connections as u64); + } - Ok(()) + sum_memory = sum_memory.saturating_add(proposed.max_memory_bytes); + sum_storage = sum_storage.saturating_add(proposed.max_storage_bytes); + sum_qps = sum_qps.saturating_add(proposed.max_qps as u64); + sum_connections = sum_connections.saturating_add(proposed.max_connections as u64); + + if ceiling.max_memory_bytes > 0 && sum_memory > ceiling.max_memory_bytes { + return Err(crate::Error::QuotaOvercommit { + field: "max_memory_bytes".into(), + detail: format!( + "tenant sum {sum_memory} exceeds database quota {}", + ceiling.max_memory_bytes + ), + }); + } + if ceiling.max_storage_bytes > 0 && sum_storage > ceiling.max_storage_bytes { + return Err(crate::Error::QuotaOvercommit { + field: "max_storage_bytes".into(), + detail: format!( + "tenant sum {sum_storage} exceeds database quota {}", + ceiling.max_storage_bytes + ), + }); + } + if ceiling.max_qps > 0 && sum_qps > ceiling.max_qps { + return Err(crate::Error::QuotaOvercommit { + field: "max_qps".into(), + detail: format!( + "tenant sum {sum_qps} exceeds database quota {}", + ceiling.max_qps + ), + }); + } + if ceiling.max_connections > 0 && sum_connections > ceiling.max_connections { + return Err(crate::Error::QuotaOvercommit { + field: "max_connections".into(), + detail: format!( + "tenant sum {sum_connections} exceeds database quota {}", + ceiling.max_connections + ), + }); } + + Ok(()) } #[cfg(test)] @@ -352,6 +382,26 @@ mod tests { } } + /// Two tenants that each fit a new database's quota alone are refused + /// together when their sum exceeds it. + #[test] + fn new_database_sums_every_tenant_quota() { + let mut db = sample_record(); + db.max_connections = 50; + let mut first = sample_record(); + first.max_connections = 30; + let mut second = sample_record(); + second.max_connections = 30; + + SystemCatalog::check_tenant_quota_in_new_database(Some(&db), &[], &first) + .expect("one tenant fits"); + SystemCatalog::check_tenant_quota_in_new_database(Some(&db), &[], &second) + .expect("the other fits alone"); + let err = SystemCatalog::check_tenant_quota_in_new_database(Some(&db), &[&first], &second) + .expect_err("the two tenants together exceed the database quota"); + assert!(matches!(err, crate::Error::QuotaOvercommit { .. }), "{err}"); + } + /// Bytes redb accepts as a value but zerompk cannot decode. const CORRUPT: &[u8] = &[0xc1, 0xc1, 0xc1]; diff --git a/nodedb/src/control/security/catalog/topic_lookup.rs b/nodedb/src/control/security/catalog/topic_lookup.rs new file mode 100644 index 000000000..1ead4cad8 --- /dev/null +++ b/nodedb/src/control/security/catalog/topic_lookup.rs @@ -0,0 +1,38 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Point read of one durable topic definition. + +use redb::{ReadableDatabase, ReadableTable}; + +use super::topics::{decode_topic, topic_key, validate_topic_identity}; +use super::types::{SystemCatalog, TOPICS_EP, catalog_err}; +use crate::event::topic::TopicDef; +use crate::types::DatabaseId; + +impl SystemCatalog { + /// Committed-only read of one topic definition. + pub fn get_committed_ep_topic( + &self, + database_id: DatabaseId, + tenant_id: u64, + name: &str, + ) -> crate::Result> { + let key = topic_key(database_id, tenant_id, name); + let read_txn = self + .db + .begin_read() + .map_err(|e| catalog_err("read txn", e))?; + let table = read_txn + .open_table(TOPICS_EP) + .map_err(|e| catalog_err("open topics_ep", e))?; + let Some(value) = table + .get(key.as_str()) + .map_err(|e| catalog_err("get topic", e))? + else { + return Ok(None); + }; + let def = decode_topic(value.value())?; + validate_topic_identity(&def, database_id, tenant_id, name)?; + Ok(Some(def)) + } +} diff --git a/nodedb/src/control/security/catalog/topic_messages.rs b/nodedb/src/control/security/catalog/topic_messages.rs new file mode 100644 index 000000000..0a3d6f577 --- /dev/null +++ b/nodedb/src/control/security/catalog/topic_messages.rs @@ -0,0 +1,446 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Durable topic messages: append, load, and retention pruning. +//! +//! Every replica of the topic's home vShard appends each published message at +//! apply, at the Raft entry's position, so every replica holds the same +//! messages with the same sequences. +//! +//! A committed transaction's message carries its origin. The append claims +//! the origin in the same transaction (see `topic_publish_marks`), so a topic +//! holds each committed message once. + +use redb::{ReadableDatabase, ReadableTable}; +use std::time::{SystemTime, UNIX_EPOCH}; + +use super::topic_publish_marks::claim_origin; +use super::topics::find_topic_definition; +use super::topics::topic_key; +use super::types::{SystemCatalog, TOPIC_MESSAGES, TOPICS_EP, catalog_err}; +use crate::event::topic::types::PublishOrigin; +use crate::event::topic::{TopicDef, TopicMessage, validate_topic_name}; +use crate::types::DatabaseId; + +impl SystemCatalog { + /// Append one message at the Raft entry `(epoch, index)` that carries it, + /// and durably advance the topic's high-water marks in the same + /// transaction as retention pruning. + /// + /// Every replica applies the same entries in log order, so each assigns + /// the same sequence. An entry at or below the topic's applied position + /// returns `None`: a re-delivered entry appends nothing twice. So does a + /// message of an `origin` the topic already holds. + pub fn append_replicated_topic_message( + &self, + scope: (DatabaseId, u64, &str), + payload: impl Into, + event_time: u64, + entry: (u64, u64), + origin: Option<&PublishOrigin>, + ) -> crate::Result> { + self.append_topic_message(scope, payload.into(), event_time, entry, origin) + } + + /// Load messages for one exact `(database, tenant, topic)` identity. + pub fn load_ep_topic_messages( + &self, + database_id: DatabaseId, + tenant_id: u64, + topic: &str, + ) -> crate::Result> { + validate_topic_name(topic).map_err(|error| catalog_err("load topic messages", error))?; + self.load_topic_messages(Some((database_id, tenant_id, topic))) + } + + /// Load messages for every topic, sorted by scope and sequence. + pub fn load_all_ep_topic_messages(&self) -> crate::Result> { + self.load_topic_messages(None) + } + + fn load_topic_messages( + &self, + scope: Option<(DatabaseId, u64, &str)>, + ) -> crate::Result> { + let read_txn = self + .db + .begin_read() + .map_err(|e| catalog_err("read topic messages txn", e))?; + let table = read_txn + .open_table(TOPIC_MESSAGES) + .map_err(|e| catalog_err("open topic_messages", e))?; + let mut messages = Vec::new(); + for entry in table + .range(..) + .map_err(|e| catalog_err("range topic_messages", e))? + { + let (key, value) = entry.map_err(|e| catalog_err("read topic message", e))?; + let (database_id, tenant_id, topic, sequence) = parse_topic_message_key(key.value())?; + let message: TopicMessage = zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("decode topic message", e))?; + if ( + message.database_id, + message.tenant_id, + message.topic.as_str(), + message.sequence, + ) != (database_id, tenant_id, topic.as_str(), sequence) + { + return Err(catalog_err( + "decode topic message", + "message identity does not match key", + )); + } + if scope.is_none_or(|(db, tenant, name)| { + (db, tenant, name) == (database_id, tenant_id, topic.as_str()) + }) { + messages.push(message); + } + } + messages.sort_by(|left, right| { + ( + left.database_id.as_u64(), + left.tenant_id, + &left.topic, + left.sequence, + ) + .cmp(&( + right.database_id.as_u64(), + right.tenant_id, + &right.topic, + right.sequence, + )) + }); + Ok(messages) + } + + fn append_topic_message( + &self, + (database_id, tenant_id, topic): (DatabaseId, u64, &str), + payload: String, + event_time: u64, + (epoch, index): (u64, u64), + origin: Option<&PublishOrigin>, + ) -> crate::Result> { + validate_topic_name(topic).map_err(|error| catalog_err("append topic", error))?; + let write_txn = self + .db + .begin_write() + .map_err(|e| catalog_err("append topic txn", e))?; + let message; + { + let mut definitions = write_txn + .open_table(TOPICS_EP) + .map_err(|e| catalog_err("open topics_ep", e))?; + let Some(mut def) = find_topic_definition(&definitions, database_id, tenant_id, topic)? + else { + return Err(catalog_err("append topic", "topic not found")); + }; + if (epoch, index) <= (def.last_epoch, def.last_lsn) { + return Ok(None); + } + let message_lsn = index; + // A committed message the topic already holds, delivered again. + if let Some(origin) = origin + && !claim_origin(&write_txn, (database_id, tenant_id, topic), origin)? + { + return Ok(None); + } + let sequence = def + .last_sequence + .checked_add(1) + .ok_or_else(|| catalog_err("append topic", "topic sequence overflow"))?; + message = TopicMessage { + database_id, + tenant_id, + topic: topic.to_owned(), + sequence, + event_time, + lsn: message_lsn, + epoch, + payload, + }; + let bytes = zerompk::to_msgpack_vec(&message) + .map_err(|e| catalog_err("serialize topic message", e))?; + { + let mut messages = write_txn + .open_table(TOPIC_MESSAGES) + .map_err(|e| catalog_err("open topic_messages", e))?; + let key = topic_message_key(database_id, tenant_id, topic, sequence)?; + messages + .insert(key.as_slice(), bytes.as_slice()) + .map_err(|e| catalog_err("insert topic message", e))?; + prune_topic_messages(&mut messages, &def, database_id, tenant_id, topic)?; + } + def.last_sequence = sequence; + def.last_lsn = message_lsn; + def.last_epoch = epoch; + let bytes = + zerompk::to_msgpack_vec(&def).map_err(|e| catalog_err("serialize topic", e))?; + definitions + .insert( + topic_key(database_id, tenant_id, topic).as_str(), + bytes.as_slice(), + ) + .map_err(|e| catalog_err("update topic high-water marks", e))?; + } + write_txn + .commit() + .map_err(|e| catalog_err("commit topic append", e))?; + Ok(Some(message)) + } +} + +fn prune_topic_messages( + table: &mut redb::Table<&[u8], &[u8]>, + def: &TopicDef, + database_id: DatabaseId, + tenant_id: u64, + topic: &str, +) -> crate::Result<()> { + let cutoff = current_time_ms().saturating_sub(def.retention.max_age_secs.saturating_mul(1_000)); + let mut messages = Vec::new(); + for entry in table + .range(..) + .map_err(|e| catalog_err("range topic_messages", e))? + { + let (key, value) = entry.map_err(|e| catalog_err("read topic message", e))?; + let (db, tenant, stored_topic, sequence) = parse_topic_message_key(key.value())?; + if (db, tenant, stored_topic.as_str()) == (database_id, tenant_id, topic) { + let message: TopicMessage = zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("decode topic message", e))?; + if ( + message.database_id, + message.tenant_id, + message.topic.as_str(), + message.sequence, + ) != (db, tenant, stored_topic.as_str(), sequence) + { + return Err(catalog_err( + "decode topic message", + "message identity does not match key", + )); + } + messages.push((key.value().to_vec(), message)); + } + } + messages.sort_by_key(|(_, message)| message.sequence); + let mut remove: Vec> = messages + .iter() + .filter(|(_, message)| message.event_time < cutoff) + .map(|(key, _)| key.clone()) + .collect(); + let retained: Vec<_> = messages + .into_iter() + .filter(|(key, _)| !remove.iter().any(|removed| removed == key)) + .collect(); + let overflow = retained + .len() + .saturating_sub(def.retention.max_events as usize); + remove.extend(retained.into_iter().take(overflow).map(|(key, _)| key)); + for key in remove { + table + .remove(key.as_slice()) + .map_err(|e| catalog_err("prune topic message", e))?; + } + Ok(()) +} + +pub(super) fn scoped_message_keys( + table: &redb::Table<&[u8], &[u8]>, + database_id: DatabaseId, + tenant_id: u64, + topic: &str, +) -> crate::Result>> { + let mut keys = Vec::new(); + for entry in table + .range(..) + .map_err(|e| catalog_err("range topic_messages", e))? + { + let (key, value) = entry.map_err(|e| catalog_err("read topic message", e))?; + let (db, tenant, stored_topic, sequence) = parse_topic_message_key(key.value())?; + if (db, tenant, stored_topic.as_str()) == (database_id, tenant_id, topic) { + let message: TopicMessage = zerompk::from_msgpack(value.value()) + .map_err(|e| catalog_err("decode topic message", e))?; + if ( + message.database_id, + message.tenant_id, + message.topic.as_str(), + message.sequence, + ) != (db, tenant, stored_topic.as_str(), sequence) + { + return Err(catalog_err( + "decode topic message", + "message identity does not match key", + )); + } + keys.push(key.value().to_vec()); + } + } + Ok(keys) +} + +fn topic_message_key( + database_id: DatabaseId, + tenant_id: u64, + topic: &str, + sequence: u64, +) -> crate::Result> { + let name_len: u16 = topic + .len() + .try_into() + .map_err(|_| catalog_err("topic message key", "topic name exceeds u16 length"))?; + let mut key = Vec::with_capacity(26 + topic.len()); + key.extend_from_slice(&database_id.as_u64().to_be_bytes()); + key.extend_from_slice(&tenant_id.to_be_bytes()); + key.extend_from_slice(&name_len.to_be_bytes()); + key.extend_from_slice(topic.as_bytes()); + key.extend_from_slice(&sequence.to_be_bytes()); + Ok(key) +} + +pub(super) fn parse_topic_message_key(key: &[u8]) -> crate::Result<(DatabaseId, u64, String, u64)> { + if key.len() < 26 { + return Err(catalog_err( + "topic message key", + "key is shorter than fixed fields", + )); + } + let database_id = DatabaseId::new(u64::from_be_bytes( + key[..8] + .try_into() + .map_err(|_| catalog_err("topic message key", "invalid database id"))?, + )); + let tenant_id = u64::from_be_bytes( + key[8..16] + .try_into() + .map_err(|_| catalog_err("topic message key", "invalid tenant id"))?, + ); + let name_len = u16::from_be_bytes( + key[16..18] + .try_into() + .map_err(|_| catalog_err("topic message key", "invalid name length"))?, + ) as usize; + if key.len() != 26 + name_len { + return Err(catalog_err( + "topic message key", + "key length does not match topic name", + )); + } + let topic = std::str::from_utf8(&key[18..18 + name_len]) + .map_err(|e| catalog_err("topic message key", e))? + .to_owned(); + let sequence = u64::from_be_bytes( + key[18 + name_len..] + .try_into() + .map_err(|_| catalog_err("topic message key", "invalid sequence"))?, + ); + Ok((database_id, tenant_id, topic, sequence)) +} + +pub(super) fn current_time_ms() -> u64 { + SystemTime::now() + .duration_since(UNIX_EPOCH) + .unwrap_or_default() + .as_millis() as u64 +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::event::cdc::stream_def::RetentionConfig; + + fn catalog_with_topic() -> (tempfile::TempDir, SystemCatalog) { + let dir = tempfile::tempdir().expect("tempdir"); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).expect("catalog"); + catalog + .put_ep_topic(&TopicDef { + database_id: DatabaseId::DEFAULT, + tenant_id: 1, + name: "events".into(), + retention: RetentionConfig { + max_events: 100, + max_age_secs: 3_600, + }, + owner: "admin".into(), + created_at: 0, + last_sequence: 0, + last_lsn: 0, + last_epoch: 0, + modification_hlc: nodedb_types::Hlc::ZERO, + }) + .expect("put topic"); + (dir, catalog) + } + + #[test] + fn a_redelivered_entry_appends_nothing_twice() { + let (_dir, catalog) = catalog_with_topic(); + let append = |index| { + catalog + .append_replicated_topic_message( + (DatabaseId::DEFAULT, 1, "events"), + "{}", + current_time_ms(), + (0, index), + None, + ) + .expect("append") + }; + let first = append(10).expect("first entry appends"); + let second = append(11).expect("second entry appends"); + assert_eq!((first.sequence, first.lsn), (1, 10)); + assert_eq!((second.sequence, second.lsn), (2, 11)); + assert!(append(11).is_none()); + assert!(append(10).is_none()); + let messages = catalog + .load_ep_topic_messages(DatabaseId::DEFAULT, 1, "events") + .expect("load"); + assert_eq!(messages.len(), 2); + } + + #[test] + fn a_higher_epoch_appends_at_a_lower_index() { + let (_dir, catalog) = catalog_with_topic(); + let append = |epoch, index| { + catalog + .append_replicated_topic_message( + (DatabaseId::DEFAULT, 1, "events"), + "{}", + current_time_ms(), + (epoch, index), + None, + ) + .expect("append") + }; + assert!(append(0, 500).is_some()); + let moved = append(7, 2).expect("new group's entry appends"); + assert_eq!((moved.epoch, moved.lsn, moved.sequence), (7, 2, 2)); + } + + /// A committed message delivered again, by a later lease holder at a + /// later entry, appends nothing. + #[test] + fn a_committed_message_is_appended_once_per_origin() { + let (_dir, catalog) = catalog_with_topic(); + let origin = PublishOrigin { + partition: 3, + position: crate::event::cdc::CdcOffset::data_event(0, 40, 1), + }; + let replicated = |index| { + catalog + .append_replicated_topic_message( + (DatabaseId::DEFAULT, 1, "events"), + "{}", + current_time_ms(), + (0, index), + Some(&origin), + ) + .expect("append") + }; + assert!(replicated(10).is_some()); + assert!(replicated(11).is_none(), "a second entry of one origin"); + let messages = catalog + .load_ep_topic_messages(DatabaseId::DEFAULT, 1, "events") + .expect("load"); + assert_eq!(messages.len(), 1); + } +} diff --git a/nodedb/src/control/security/catalog/topic_publish_marks.rs b/nodedb/src/control/security/catalog/topic_publish_marks.rs new file mode 100644 index 000000000..03a4d4fc6 --- /dev/null +++ b/nodedb/src/control/security/catalog/topic_publish_marks.rs @@ -0,0 +1,158 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-topic, per-partition marks of the committed messages a topic holds. +//! +//! A committed transaction's message is delivered to its topic from the +//! change-feed partition of the record that carries it, in position order. +//! The topic keeps the highest position it appended from each partition. A +//! message at or below the mark is one the topic holds already: a delivery +//! repeated after a lease move appends nothing. The mark moves in the same +//! transaction as the append, and goes with the topic when it is dropped. + +use redb::{ReadableTable, TableDefinition, WriteTransaction}; + +use super::types::catalog_err; +use crate::event::cdc::CdcOffset; +use crate::event::topic::types::PublishOrigin; +use crate::types::DatabaseId; + +/// Table: `[database_id: be u64][tenant_id: be u64][name_len: be u16] +/// [topic_name bytes][partition: be u32]` -> `[epoch][index][sequence]`, each +/// a big-endian u64. +pub(super) const TOPIC_PUBLISH_MARKS: TableDefinition<&[u8], &[u8]> = + TableDefinition::new("_system.topic_publish_marks"); + +/// The key prefix of one topic's marks. +fn topic_prefix(database_id: DatabaseId, tenant_id: u64, topic: &str) -> crate::Result> { + let name_len: u16 = topic + .len() + .try_into() + .map_err(|_| catalog_err("topic publish mark", "topic name exceeds u16 length"))?; + let mut key = Vec::with_capacity(22 + topic.len()); + key.extend_from_slice(&database_id.as_u64().to_be_bytes()); + key.extend_from_slice(&tenant_id.to_be_bytes()); + key.extend_from_slice(&name_len.to_be_bytes()); + key.extend_from_slice(topic.as_bytes()); + Ok(key) +} + +fn encode_mark(position: CdcOffset) -> [u8; 24] { + let mut bytes = [0u8; 24]; + bytes[..8].copy_from_slice(&position.epoch.to_be_bytes()); + bytes[8..16].copy_from_slice(&position.index.to_be_bytes()); + bytes[16..].copy_from_slice(&position.sequence.to_be_bytes()); + bytes +} + +fn decode_mark(bytes: &[u8]) -> crate::Result { + let word = |range: std::ops::Range| -> crate::Result { + bytes + .get(range) + .and_then(|slice| slice.try_into().ok()) + .map(u64::from_be_bytes) + .ok_or_else(|| catalog_err("topic publish mark", "malformed mark")) + }; + if bytes.len() != 24 { + return Err(catalog_err("topic publish mark", "malformed mark")); + } + Ok(CdcOffset::at(word(0..8)?, word(8..16)?, word(16..24)?)) +} + +/// Raise the topic's mark for `origin`'s partition to `origin`'s position. +/// Returns `false`, and moves nothing, when the topic holds a message at or +/// above that position already. +pub(super) fn claim_origin( + txn: &WriteTransaction, + (database_id, tenant_id, topic): (DatabaseId, u64, &str), + origin: &PublishOrigin, +) -> crate::Result { + let mut key = topic_prefix(database_id, tenant_id, topic)?; + key.extend_from_slice(&origin.partition.to_be_bytes()); + let mut marks = txn + .open_table(TOPIC_PUBLISH_MARKS) + .map_err(|e| catalog_err("open topic_publish_marks", e))?; + let held = marks + .get(key.as_slice()) + .map_err(|e| catalog_err("read topic publish mark", e))? + .map(|mark| decode_mark(mark.value())) + .transpose()?; + if held.is_some_and(|held| origin.position <= held) { + return Ok(false); + } + marks + .insert(key.as_slice(), encode_mark(origin.position).as_slice()) + .map_err(|e| catalog_err("write topic publish mark", e))?; + Ok(true) +} + +/// Remove every mark of one topic. +pub(super) fn forget_topic_marks( + txn: &WriteTransaction, + database_id: DatabaseId, + tenant_id: u64, + topic: &str, +) -> crate::Result<()> { + let prefix = topic_prefix(database_id, tenant_id, topic)?; + let mut marks = txn + .open_table(TOPIC_PUBLISH_MARKS) + .map_err(|e| catalog_err("open topic_publish_marks", e))?; + // A topic's marks follow its prefix. A longer topic name that shares the + // prefix differs in its length bytes, so it never falls in this range. + let mut end = prefix.clone(); + end.extend_from_slice(&u32::MAX.to_be_bytes()); + marks + .retain_in(prefix.as_slice()..=end.as_slice(), |_, _| false) + .map_err(|e| catalog_err("delete topic publish marks", e)) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::security::catalog::types::SystemCatalog; + + fn origin(partition: u32, index: u64) -> PublishOrigin { + PublishOrigin { + partition, + position: CdcOffset::data_event(0, index, 1), + } + } + + fn claim(catalog: &SystemCatalog, topic: &str, origin: PublishOrigin) -> bool { + let txn = catalog.db.begin_write().expect("write txn"); + let claimed = claim_origin(&txn, (DatabaseId::DEFAULT, 1, topic), &origin).expect("claim"); + txn.commit().expect("commit"); + claimed + } + + #[test] + fn a_position_is_claimed_once_per_topic_and_partition() { + let catalog = SystemCatalog::open_in_memory().expect("catalog"); + assert!(claim(&catalog, "feed", origin(3, 10))); + assert!( + !claim(&catalog, "feed", origin(3, 10)), + "a repeat is refused" + ); + assert!( + !claim(&catalog, "feed", origin(3, 9)), + "a lower one is refused" + ); + assert!(claim(&catalog, "feed", origin(3, 11))); + assert!( + claim(&catalog, "feed", origin(4, 1)), + "partitions are apart" + ); + assert!(claim(&catalog, "feeds", origin(3, 1)), "topics are apart"); + + let txn = catalog.db.begin_write().expect("write txn"); + forget_topic_marks(&txn, DatabaseId::DEFAULT, 1, "feed").expect("forget"); + txn.commit().expect("commit"); + assert!( + claim(&catalog, "feed", origin(3, 1)), + "a dropped topic has no marks" + ); + assert!( + !claim(&catalog, "feeds", origin(3, 1)), + "another topic keeps its marks" + ); + } +} diff --git a/nodedb/src/control/security/catalog/topics.rs b/nodedb/src/control/security/catalog/topics.rs index 6bc2a811e..cfc00dd83 100644 --- a/nodedb/src/control/security/catalog/topics.rs +++ b/nodedb/src/control/security/catalog/topics.rs @@ -1,15 +1,16 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Durable topic metadata and message operations for the system catalog. +//! Durable topic definitions for the system catalog. Messages live in +//! `topic_messages`. use std::collections::HashMap; use redb::{ReadableDatabase, ReadableTable}; -use std::time::{SystemTime, UNIX_EPOCH}; use super::consumer_groups::decode_consumer_group; +use super::topic_messages::scoped_message_keys; use super::types::{CONSUMER_GROUPS, SystemCatalog, TOPIC_MESSAGES, TOPICS_EP, catalog_err}; -use crate::event::topic::{TopicDef, TopicMessage, validate_topic_name}; +use crate::event::topic::{TopicDef, validate_topic_name}; use crate::types::DatabaseId; impl SystemCatalog { @@ -31,7 +32,10 @@ impl SystemCatalog { let mut stored = def.clone(); if let Some(existing) = existing { stored.last_sequence = stored.last_sequence.max(existing.last_sequence); - stored.last_lsn = stored.last_lsn.max(existing.last_lsn); + if (existing.last_epoch, existing.last_lsn) > (stored.last_epoch, stored.last_lsn) { + stored.last_epoch = existing.last_epoch; + stored.last_lsn = existing.last_lsn; + } } let bytes = zerompk::to_msgpack_vec(&stored).map_err(|e| catalog_err("serialize topic", e))?; @@ -54,7 +58,7 @@ impl SystemCatalog { /// Insert a topic definition without re-checking its name. /// /// Replicated apply uses this: the leader validated before proposing, so a - /// rejection here would diverge this node from the accepted entry. + /// rejection here diverges this node from the accepted entry. pub fn create_ep_topic_unchecked(&self, def: &TopicDef) -> crate::Result { let key = topic_key(def.database_id, def.tenant_id, &def.name); let write_txn = self @@ -80,90 +84,6 @@ impl SystemCatalog { Ok(true) } - /// Append one exact payload to a topic and durably advance its high-water - /// marks in the same transaction as retention pruning. - pub fn append_ep_topic_message( - &self, - database_id: DatabaseId, - tenant_id: u64, - topic: &str, - payload: impl Into, - event_time: u64, - lsn: u64, - ) -> crate::Result { - validate_topic_name(topic).map_err(|error| catalog_err("append topic", error))?; - let write_txn = self - .db - .begin_write() - .map_err(|e| catalog_err("append topic txn", e))?; - let message; - { - let mut definitions = write_txn - .open_table(TOPICS_EP) - .map_err(|e| catalog_err("open topics_ep", e))?; - let Some(mut def) = find_topic_definition(&definitions, database_id, tenant_id, topic)? - else { - return Err(catalog_err("append topic", "topic not found")); - }; - let sequence = def - .last_sequence - .checked_add(1) - .ok_or_else(|| catalog_err("append topic", "topic sequence overflow"))?; - let message_lsn = lsn.max(def.last_lsn); - message = TopicMessage { - database_id, - tenant_id, - topic: topic.to_owned(), - sequence, - event_time, - lsn: message_lsn, - payload: payload.into(), - }; - let bytes = zerompk::to_msgpack_vec(&message) - .map_err(|e| catalog_err("serialize topic message", e))?; - { - let mut messages = write_txn - .open_table(TOPIC_MESSAGES) - .map_err(|e| catalog_err("open topic_messages", e))?; - let key = topic_message_key(database_id, tenant_id, topic, sequence)?; - messages - .insert(key.as_slice(), bytes.as_slice()) - .map_err(|e| catalog_err("insert topic message", e))?; - prune_topic_messages(&mut messages, &def, database_id, tenant_id, topic)?; - } - def.last_sequence = sequence; - def.last_lsn = message_lsn; - let bytes = - zerompk::to_msgpack_vec(&def).map_err(|e| catalog_err("serialize topic", e))?; - definitions - .insert( - topic_key(database_id, tenant_id, topic).as_str(), - bytes.as_slice(), - ) - .map_err(|e| catalog_err("update topic high-water marks", e))?; - } - write_txn - .commit() - .map_err(|e| catalog_err("commit topic append", e))?; - Ok(message) - } - - /// Load messages for one exact `(database, tenant, topic)` identity. - pub fn load_ep_topic_messages( - &self, - database_id: DatabaseId, - tenant_id: u64, - topic: &str, - ) -> crate::Result> { - validate_topic_name(topic).map_err(|error| catalog_err("load topic messages", error))?; - self.load_topic_messages(Some((database_id, tenant_id, topic))) - } - - /// Load messages for every topic, sorted by scope and sequence. - pub fn load_all_ep_topic_messages(&self) -> crate::Result> { - self.load_topic_messages(None) - } - /// Delete a topic and every one of its durable messages atomically. pub fn delete_ep_topic( &self, @@ -195,6 +115,12 @@ impl SystemCatalog { .remove(key.as_slice()) .map_err(|e| catalog_err("delete topic message", e))?; } + super::topic_publish_marks::forget_topic_marks( + &write_txn, + database_id, + tenant_id, + name, + )?; } write_txn.commit().map_err(|e| catalog_err("commit", e))?; Ok(existed) @@ -255,7 +181,7 @@ impl SystemCatalog { /// Delete a topic and its groups without re-checking the topic name. /// /// Replicated apply uses this: the name was validated before the entry was - /// proposed, and a rejection here would leave the row on this node alone. + /// proposed, and a rejection here leaves the row on this node alone. pub fn delete_ep_topic_with_consumer_groups_unchecked( &self, database_id: DatabaseId, @@ -285,6 +211,12 @@ impl SystemCatalog { .remove(key.as_slice()) .map_err(|e| catalog_err("delete topic message", e))?; } + super::topic_publish_marks::forget_topic_marks( + &write_txn, + database_id, + tenant_id, + name, + )?; let mut groups = write_txn .open_table(CONSUMER_GROUPS) .map_err(|e| catalog_err("open consumer_groups", e))?; @@ -346,64 +278,9 @@ impl SystemCatalog { }); Ok(topics) } - - fn load_topic_messages( - &self, - scope: Option<(DatabaseId, u64, &str)>, - ) -> crate::Result> { - let read_txn = self - .db - .begin_read() - .map_err(|e| catalog_err("read topic messages txn", e))?; - let table = read_txn - .open_table(TOPIC_MESSAGES) - .map_err(|e| catalog_err("open topic_messages", e))?; - let mut messages = Vec::new(); - for entry in table - .range(..) - .map_err(|e| catalog_err("range topic_messages", e))? - { - let (key, value) = entry.map_err(|e| catalog_err("read topic message", e))?; - let (database_id, tenant_id, topic, sequence) = parse_topic_message_key(key.value())?; - let message: TopicMessage = zerompk::from_msgpack(value.value()) - .map_err(|e| catalog_err("decode topic message", e))?; - if ( - message.database_id, - message.tenant_id, - message.topic.as_str(), - message.sequence, - ) != (database_id, tenant_id, topic.as_str(), sequence) - { - return Err(catalog_err( - "decode topic message", - "message identity does not match key", - )); - } - if scope.is_none_or(|(db, tenant, name)| { - (db, tenant, name) == (database_id, tenant_id, topic.as_str()) - }) { - messages.push(message); - } - } - messages.sort_by(|left, right| { - ( - left.database_id.as_u64(), - left.tenant_id, - &left.topic, - left.sequence, - ) - .cmp(&( - right.database_id.as_u64(), - right.tenant_id, - &right.topic, - right.sequence, - )) - }); - Ok(messages) - } } -fn find_topic_definition( +pub(super) fn find_topic_definition( table: &redb::Table<&str, &[u8]>, database_id: DatabaseId, tenant_id: u64, @@ -421,7 +298,7 @@ fn find_topic_definition( Ok(Some(def)) } -fn validate_topic_identity( +pub(super) fn validate_topic_identity( def: &TopicDef, database_id: DatabaseId, tenant_id: u64, @@ -436,174 +313,16 @@ fn validate_topic_identity( Ok(()) } -fn prune_topic_messages( - table: &mut redb::Table<&[u8], &[u8]>, - def: &TopicDef, - database_id: DatabaseId, - tenant_id: u64, - topic: &str, -) -> crate::Result<()> { - let cutoff = current_time_ms().saturating_sub(def.retention.max_age_secs.saturating_mul(1_000)); - let mut messages = Vec::new(); - for entry in table - .range(..) - .map_err(|e| catalog_err("range topic_messages", e))? - { - let (key, value) = entry.map_err(|e| catalog_err("read topic message", e))?; - let (db, tenant, stored_topic, sequence) = parse_topic_message_key(key.value())?; - if (db, tenant, stored_topic.as_str()) == (database_id, tenant_id, topic) { - let message: TopicMessage = zerompk::from_msgpack(value.value()) - .map_err(|e| catalog_err("decode topic message", e))?; - if ( - message.database_id, - message.tenant_id, - message.topic.as_str(), - message.sequence, - ) != (db, tenant, stored_topic.as_str(), sequence) - { - return Err(catalog_err( - "decode topic message", - "message identity does not match key", - )); - } - messages.push((key.value().to_vec(), message)); - } - } - messages.sort_by_key(|(_, message)| message.sequence); - let mut remove: Vec> = messages - .iter() - .filter(|(_, message)| message.event_time < cutoff) - .map(|(key, _)| key.clone()) - .collect(); - let retained: Vec<_> = messages - .into_iter() - .filter(|(key, _)| !remove.iter().any(|removed| removed == key)) - .collect(); - let overflow = retained - .len() - .saturating_sub(def.retention.max_events as usize); - remove.extend(retained.into_iter().take(overflow).map(|(key, _)| key)); - for key in remove { - table - .remove(key.as_slice()) - .map_err(|e| catalog_err("prune topic message", e))?; - } - Ok(()) -} - -fn scoped_message_keys( - table: &redb::Table<&[u8], &[u8]>, - database_id: DatabaseId, - tenant_id: u64, - topic: &str, -) -> crate::Result>> { - let mut keys = Vec::new(); - for entry in table - .range(..) - .map_err(|e| catalog_err("range topic_messages", e))? - { - let (key, value) = entry.map_err(|e| catalog_err("read topic message", e))?; - let (db, tenant, stored_topic, sequence) = parse_topic_message_key(key.value())?; - if (db, tenant, stored_topic.as_str()) == (database_id, tenant_id, topic) { - let message: TopicMessage = zerompk::from_msgpack(value.value()) - .map_err(|e| catalog_err("decode topic message", e))?; - if ( - message.database_id, - message.tenant_id, - message.topic.as_str(), - message.sequence, - ) != (db, tenant, stored_topic.as_str(), sequence) - { - return Err(catalog_err( - "decode topic message", - "message identity does not match key", - )); - } - keys.push(key.value().to_vec()); - } - } - Ok(keys) -} - -fn topic_key(database_id: DatabaseId, tenant_id: u64, name: &str) -> String { - let mut encoded = String::with_capacity(name.len() * 2); - for byte in name.as_bytes() { - use std::fmt::Write; - let _ = write!(&mut encoded, "{byte:02x}"); - } +pub(super) fn topic_key(database_id: DatabaseId, tenant_id: u64, name: &str) -> String { format!( - "v2/{:016x}/{:016x}/{:08x}/{encoded}", + "v2/{:016x}/{:016x}/{:08x}/{}", database_id.as_u64(), tenant_id, - name.len() + name.len(), + hex::encode(name) ) } -fn topic_message_key( - database_id: DatabaseId, - tenant_id: u64, - topic: &str, - sequence: u64, -) -> crate::Result> { - let name_len: u16 = topic - .len() - .try_into() - .map_err(|_| catalog_err("topic message key", "topic name exceeds u16 length"))?; - let mut key = Vec::with_capacity(26 + topic.len()); - key.extend_from_slice(&database_id.as_u64().to_be_bytes()); - key.extend_from_slice(&tenant_id.to_be_bytes()); - key.extend_from_slice(&name_len.to_be_bytes()); - key.extend_from_slice(topic.as_bytes()); - key.extend_from_slice(&sequence.to_be_bytes()); - Ok(key) -} - -fn parse_topic_message_key(key: &[u8]) -> crate::Result<(DatabaseId, u64, String, u64)> { - if key.len() < 26 { - return Err(catalog_err( - "topic message key", - "key is shorter than fixed fields", - )); - } - let database_id = DatabaseId::new(u64::from_be_bytes( - key[..8] - .try_into() - .map_err(|_| catalog_err("topic message key", "invalid database id"))?, - )); - let tenant_id = u64::from_be_bytes( - key[8..16] - .try_into() - .map_err(|_| catalog_err("topic message key", "invalid tenant id"))?, - ); - let name_len = u16::from_be_bytes( - key[16..18] - .try_into() - .map_err(|_| catalog_err("topic message key", "invalid name length"))?, - ) as usize; - if key.len() != 26 + name_len { - return Err(catalog_err( - "topic message key", - "key length does not match topic name", - )); - } - let topic = std::str::from_utf8(&key[18..18 + name_len]) - .map_err(|e| catalog_err("topic message key", e))? - .to_owned(); - let sequence = u64::from_be_bytes( - key[18 + name_len..] - .try_into() - .map_err(|_| catalog_err("topic message key", "invalid sequence"))?, - ); - Ok((database_id, tenant_id, topic, sequence)) -} - -fn current_time_ms() -> u64 { - SystemTime::now() - .duration_since(UNIX_EPOCH) - .unwrap_or_default() - .as_millis() as u64 -} - /// Positional wire shape written before topics adopted map encoding. #[derive(zerompk::FromMessagePack, zerompk::ToMessagePack)] #[msgpack(array)] @@ -626,11 +345,13 @@ impl From for TopicDef { database_id: DatabaseId::DEFAULT, last_sequence: 0, last_lsn: 0, + last_epoch: 0, + modification_hlc: nodedb_types::Hlc::ZERO, } } } -fn decode_topic(bytes: &[u8]) -> crate::Result { +pub(super) fn decode_topic(bytes: &[u8]) -> crate::Result { zerompk::from_msgpack(bytes) .or_else(|_| zerompk::from_msgpack::(bytes).map(TopicDef::from)) .map_err(|e| catalog_err("decode topic", e)) @@ -640,6 +361,7 @@ fn decode_topic(bytes: &[u8]) -> crate::Result { mod tests { use std::sync::Arc; + use super::super::topic_messages::current_time_ms; use super::*; use crate::event::cdc::consumer_group::ConsumerGroupDef; use crate::event::cdc::stream_def::RetentionConfig; @@ -663,32 +385,41 @@ mod tests { created_at: 0, last_sequence: 0, last_lsn: 0, + last_epoch: 0, + modification_hlc: nodedb_types::Hlc::ZERO, } } + /// Append one message at the Raft entry `(0, index)`. + fn append_at( + catalog: &SystemCatalog, + scope: (DatabaseId, u64, &str), + payload: &str, + event_time: u64, + index: u64, + ) -> crate::event::topic::TopicMessage { + catalog + .append_replicated_topic_message(scope, payload, event_time, (0, index), None) + .expect("append") + .expect("the entry appends") + } + + /// The apply appends a group's entries in log order: their messages take + /// contiguous sequences, and survive a reopen. #[test] - fn concurrent_appends_are_contiguous_and_survive_reopen() { + fn entry_appends_are_contiguous_and_survive_reopen() { let (dir, catalog) = catalog(); catalog .put_ep_topic(&topic(DatabaseId::new(7), 1, "events", 100)) .expect("topic"); - let catalog = Arc::new(catalog); - let mut workers = Vec::new(); - for number in 0..16 { - let catalog = Arc::clone(&catalog); - workers.push(std::thread::spawn(move || { - catalog.append_ep_topic_message( - DatabaseId::new(7), - 1, - "events", - number.to_string(), - current_time_ms(), - number, - ) - })); - } - for worker in workers { - worker.join().expect("worker").expect("append"); + for index in 1..=16 { + append_at( + &catalog, + (DatabaseId::new(7), 1, "events"), + &index.to_string(), + current_time_ms(), + index, + ); } let messages = catalog .load_ep_topic_messages(DatabaseId::new(7), 1, "events") @@ -763,10 +494,14 @@ mod tests { .put_ep_topic(&topic(DatabaseId::DEFAULT, 1, "events", 2)) .expect("topic"); let now = current_time_ms(); - for sequence in 1..=3 { - catalog - .append_ep_topic_message(DatabaseId::DEFAULT, 1, "events", "{}", now, sequence) - .expect("append"); + for index in 1..=3 { + append_at( + &catalog, + (DatabaseId::DEFAULT, 1, "events"), + "{}", + now, + index, + ); } let messages = catalog .load_ep_topic_messages(DatabaseId::DEFAULT, 1, "events") @@ -800,9 +535,7 @@ mod tests { let mut definition = topic(DatabaseId::DEFAULT, 1, "events", 10); definition.retention.max_age_secs = 1; catalog.put_ep_topic(&definition).expect("topic"); - catalog - .append_ep_topic_message(DatabaseId::DEFAULT, 1, "events", "old", 0, 1) - .expect("append"); + append_at(&catalog, (DatabaseId::DEFAULT, 1, "events"), "old", 0, 1); assert!( catalog .load_ep_topic_messages(DatabaseId::DEFAULT, 1, "events") @@ -833,12 +566,17 @@ mod tests { stream_name: stream_name.into(), owner: "admin".into(), created_at: 0, + modification_hlc: nodedb_types::Hlc::ZERO, }) .expect("group"); } - catalog - .append_ep_topic_message(database_id, 1, "events", "before", current_time_ms(), 1) - .expect("message"); + append_at( + &catalog, + (database_id, 1, "events"), + "before", + current_time_ms(), + 1, + ); assert_eq!( catalog .topic_consumer_group_names(database_id, 1, "events") @@ -872,9 +610,13 @@ mod tests { catalog .create_ep_topic(&topic(database_id, 1, "events", 10)) .expect("create"); - catalog - .append_ep_topic_message(database_id, 1, "events", "before", current_time_ms(), 1) - .expect("append"); + append_at( + &catalog, + (database_id, 1, "events"), + "before", + current_time_ms(), + 1, + ); assert!( catalog .delete_ep_topic(database_id, 1, "events") @@ -885,9 +627,13 @@ mod tests { .create_ep_topic(&topic(database_id, 1, "events", 10)) .expect("recreate") ); - let message = catalog - .append_ep_topic_message(database_id, 1, "events", "after", current_time_ms(), 1) - .expect("append recreated"); + let message = append_at( + &catalog, + (database_id, 1, "events"), + "after", + current_time_ms(), + 1, + ); assert_eq!(message.sequence, 1); assert_eq!( catalog @@ -910,12 +656,8 @@ mod tests { .put_ep_topic(&topic(second.0, second.1, second.2, 10)) .expect("second topic"); let now = current_time_ms(); - catalog - .append_ep_topic_message(first.0, first.1, first.2, "one", now, 1) - .expect("first append"); - catalog - .append_ep_topic_message(second.0, second.1, second.2, "two", now, 1) - .expect("second append"); + append_at(&catalog, first, "one", now, 1); + append_at(&catalog, second, "two", now, 1); assert!( catalog .delete_ep_topic(first.0, first.1, first.2) diff --git a/nodedb/src/control/security/catalog/trigger_types.rs b/nodedb/src/control/security/catalog/trigger_types.rs index 4047d3e1d..a9ba8f589 100644 --- a/nodedb/src/control/security/catalog/trigger_types.rs +++ b/nodedb/src/control/security/catalog/trigger_types.rs @@ -5,6 +5,12 @@ use nodedb_types::id::DatabaseId; /// When the trigger fires relative to the DML operation. +/// +/// A BEFORE or INSTEAD OF body runs in the Control Plane before the +/// triggering write stages. It joins the triggering statement's transaction: +/// the client's transaction block, or an implicit one around the statement. +/// The write and the body's writes commit together or not at all. A body +/// error rolls back the body's own writes and fails the statement. #[derive(Debug, Clone, Copy, PartialEq, Eq, zerompk::ToMessagePack, zerompk::FromMessagePack)] #[repr(u8)] #[msgpack(c_enum)] @@ -50,10 +56,14 @@ impl TriggerEvents { /// Execution mode for AFTER triggers. /// -/// Controls where and when the trigger body executes: -/// - `Async` (default): Event Plane, eventually consistent, zero write latency impact. -/// - `Sync`: Control Plane write path, same logical transaction, adds to write latency. -/// - `Deferred`: Data Plane at COMMIT time, same transaction, batched. +/// Controls where and when the trigger body executes. A body's writes fire +/// no triggers: +/// - `Async` (default): Event Plane, after the triggering write commits. The +/// body commits as its own transaction. +/// - `Sync`: Control Plane write path, after the triggering write stages. The +/// body joins the triggering statement's transaction. +/// - `Deferred`: Event Plane, once the triggering transaction commits. The +/// body commits as its own transaction. #[derive( Debug, Clone, Copy, PartialEq, Eq, Default, zerompk::ToMessagePack, zerompk::FromMessagePack, )] @@ -64,12 +74,15 @@ pub enum TriggerExecutionMode { /// Default. Eventually consistent side effects. Zero write latency impact. #[default] Async = 0, - /// Trigger fires synchronously in the Control Plane write path. - /// ACID (same logical transaction). Adds trigger execution time to write latency. + /// Trigger fires synchronously in the Control Plane write path, after + /// the triggering write stages. Adds trigger execution time to write + /// latency. The body joins the triggering statement's transaction: a + /// failed body fails the statement, and the statement's transaction + /// commits neither the write nor the body's writes. /// Cross-shard SYNC triggers are rejected at CREATE TRIGGER time. Sync = 1, - /// Trigger fires at COMMIT time in the Data Plane, batched. - /// ACID (same transaction). Only adds latency at COMMIT, not per-statement. + /// Trigger fires from the Event Plane once the triggering transaction + /// commits, batched per COMMIT. The body commits as its own transaction. Deferred = 2, } diff --git a/nodedb/src/control/security/catalog/types.rs b/nodedb/src/control/security/catalog/types.rs index 7e6695973..e4d4dbac3 100644 --- a/nodedb/src/control/security/catalog/types.rs +++ b/nodedb/src/control/security/catalog/types.rs @@ -35,18 +35,20 @@ pub use super::system_catalog::SystemCatalog; // Re-exported so existing `super::types::{TABLE_FOO, ...}` imports keep // working unchanged. +#[cfg(test)] +pub(super) use super::tables::SURROGATE_PK_REV_LEGACY; pub(super) use super::tables::{ ALERT_RULES, API_KEYS, ARRAYS, AUDIT_LOG, AUTH_USERS, BLACKLIST, CHANGE_STREAMS, CHECKPOINTS, - CLONE_COPYUPS, CLONE_KV_TOMBSTONES, CLONE_LINEAGE, CLONE_TOMBSTONES, COLLECTIONS, - COLLECTIONS_LEGACY, COLUMN_STATS, CONSUMER_GROUPS, CONTINUOUS_AGGREGATES, CUSTOM_TYPES, - DATABASE_GRANTS, DATABASE_HWM, DATABASE_QUOTAS, DATABASES, DATABASES_BY_NAME, DEPENDENCIES, - FUNCTIONS, INDEX_REGISTRY, L2_CLEANUP_QUEUE, LOCKOUT_STATE, MATERIALIZED_VIEWS, METADATA, - MIRROR_COLLECTION_MAP, MIRROR_LAG, ORG_MEMBERS, ORGS, OWNERS, PENDING_RECLAIM, PERMISSIONS, - PROCEDURES, RETENTION_POLICIES, ROLES, SCHEDULES, SCOPE_GRANTS, SCOPE_QUOTAS, SCOPES, - SEQUENCE_STATE, SEQUENCES, STREAMING_MVS, SURROGATE_PK_LEGACY, SURROGATE_PK_REV_LEGACY, - SURROGATE_PK_REV_V2, SURROGATE_PK_REV_V3, SURROGATE_PK_V2, SURROGATE_PK_V3, SYNONYM_GROUPS, - TENANT_QUOTAS, TENANTS, TOPIC_MESSAGES, TOPICS_EP, TRIGGERS, USERS, VECTOR_INDEX_PARAMS, - VECTOR_MODEL_METADATA, WAL_TOMBSTONES, WASM_MODULES, + CLONE_COPYUPS, CLONE_KV_TOMBSTONES, CLONE_LINEAGE, CLONE_TOMBSTONES, COLLECTIONS, COLUMN_STATS, + CONSUMER_GROUPS, CONTINUOUS_AGGREGATES, CUSTOM_TYPES, DATABASE_GRANTS, DATABASE_HWM, + DATABASE_QUOTAS, DATABASES, DATABASES_BY_NAME, DEPENDENCIES, FUNCTIONS, INDEX_REGISTRY, + L2_CLEANUP_QUEUE, LOCKOUT_STATE, MATERIALIZED_VIEWS, METADATA, MIRROR_COLLECTION_MAP, + MIRROR_LAG, ORG_MEMBERS, ORGS, OWNERS, PENDING_RECLAIM, PERMISSIONS, PROCEDURES, + RETENTION_POLICIES, ROLES, SCHEDULES, SCOPE_GRANTS, SCOPE_QUOTAS, SCOPES, SEQUENCE_STATE, + SEQUENCES, STREAMING_MVS, SURROGATE_PK_LEGACY, SURROGATE_PK_REV_V2, SURROGATE_PK_REV_V3, + SURROGATE_PK_V2, SURROGATE_PK_V3, SYNONYM_GROUPS, TENANT_QUOTAS, TENANTS, TOPIC_MESSAGES, + TOPICS_EP, TRIGGERS, USERS, VECTOR_INDEX_PARAMS, VECTOR_MODEL_METADATA, WAL_TOMBSTONES, + WASM_MODULES, }; // ── Helpers ─────────────────────────────────────────────────────────── diff --git a/nodedb/src/control/security/credential/store/list.rs b/nodedb/src/control/security/credential/store/list.rs index f437ba2fc..b72554445 100644 --- a/nodedb/src/control/security/credential/store/list.rs +++ b/nodedb/src/control/security/credential/store/list.rs @@ -22,18 +22,27 @@ impl CredentialStore { } /// Reload all users from the given catalog into the in-memory cache. - /// Used by the recovery verifier repair path. + /// Used by the recovery verifier repair path and by a metadata image + /// install. + /// + /// The next user id never moves down. It rises to the catalog's counter + /// and past every loaded user id, so a user the catalog gained never + /// shares an id with the next local `CREATE USER`. pub fn reload_from_catalog(&self, catalog: &SystemCatalog) -> crate::Result<()> { let stored_users = catalog.load_all_users()?; + let mut floor = catalog.load_next_user_id()?; let mut replacement = std::collections::HashMap::with_capacity(stored_users.len()); for stored in stored_users { validate_stored_user_credentials(&stored, &self.argon2_config)?; + floor = floor.max(stored.user_id.saturating_add(1)); let record = UserRecord::from_stored(stored); replacement.insert(record.username.clone(), record); } let mut users = write_lock(&self.users); *users = replacement; + let mut next = write_lock(&self.next_user_id); + *next = (*next).max(floor); Ok(()) } diff --git a/nodedb/src/control/security/escalation/violation.rs b/nodedb/src/control/security/escalation/violation.rs index 73f1bebe4..9b4d4d7ba 100644 --- a/nodedb/src/control/security/escalation/violation.rs +++ b/nodedb/src/control/security/escalation/violation.rs @@ -13,7 +13,7 @@ //! `_system.auth_users` record, which [`check_blacklist`](crate::control::server::session_auth::guards::check_blacklist) //! consults on every subsequent request — and replicated as a //! [`CatalogEntry::PutAuthUser`]. A suspension that lived only in a -//! process-local map would evaporate on restart, which is a security control +//! process-local map evaporates on restart, which is a security control //! that silently stops working. use tracing::warn; @@ -43,7 +43,7 @@ pub enum ViolationSubject<'a> { /// Record the audit entry and nothing else. Used where no principal is /// attributable, and where the rejection *is* the previous verdict being /// enforced (an already-suspended account, a blacklisted org) — counting - /// those would let a client in a retry loop drive the ladder from its own + /// those lets a client in a retry loop drive the ladder from its own /// rejections. AuditOnly, } @@ -120,14 +120,50 @@ pub fn record_auth_violation(state: &SharedState, violation: AuthViolation<'_>) } }; - let entry = CatalogEntry::PutAuthUser(Box::new(stored)); - if let Err(e) = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) { + replicate_verdict( + state, + CatalogEntry::PutAuthUser(Box::new(stored)), + subject.log_id, + ); +} + +/// Propose the locally installed verdict `entry` on its own task. +/// +/// The authentication paths that record a violation are synchronous and hold +/// no proposer budget: the verdict already binds this node, so its +/// replication runs off the rejected request. The task has no connection +/// scope, so an open transactional-DDL block never buffers it. +fn replicate_verdict(state: &SharedState, entry: CatalogEntry, log_id: String) { + let owner = match state.self_arc() { + Ok(owner) => owner, + Err(e) => { + warn!( + user_id = %log_id, + error = %e, + "escalation verdict persisted locally but could not be replicated" + ); + return; + } + }; + let Ok(runtime) = tokio::runtime::Handle::try_current() else { warn!( - user_id = %subject.log_id, - error = %e, - "escalation verdict persisted locally but could not be replicated" + user_id = %log_id, + "escalation verdict persisted locally but could not be replicated: \ + no tokio runtime runs on this thread" ); - } + return; + }; + runtime.spawn(async move { + if let Err(e) = + crate::control::metadata_proposer::propose_catalog_entry_async(&owner, &entry).await + { + warn!( + user_id = %log_id, + error = %e, + "escalation verdict persisted locally but could not be replicated" + ); + } + }); } /// The account an escalation applies to. diff --git a/nodedb/src/control/security/identity/plan_permission.rs b/nodedb/src/control/security/identity/plan_permission.rs index 9f43425b5..9e826ab7f 100644 --- a/nodedb/src/control/security/identity/plan_permission.rs +++ b/nodedb/src/control/security/identity/plan_permission.rs @@ -27,7 +27,7 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm | DocumentOp::IndexedFetch { .. } | DocumentOp::EstimateCount { .. } | DocumentOp::MaterializeScan { .. } - // Read-only: reports what the wrapped write would apply; that write is authorized separately. + // Read-only: reports what the wrapped write applies; that write is authorized separately. | DocumentOp::ResolveWrite(_), ) => Permission::Read, @@ -37,7 +37,7 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm | VectorOp::QueryStats { .. } | VectorOp::SparseSearch { .. } | VectorOp::MultiVectorScoreSearch { .. } - // Read-only: reports what the wrapped write would apply; that write is authorized separately. + // Read-only: reports what the wrapped write applies; that write is authorized separately. | VectorOp::ResolveDirectWrite(_), ) => Permission::Read, @@ -51,7 +51,7 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm ) => Permission::Read, PhysicalPlan::Graph( - // Read-only: decides what a governed delete would do; that delete is authorized separately. + // Read-only: decides what a governed delete does; that delete is authorized separately. GraphOp::ResolveEdgeDelete(_) | GraphOp::Hop { .. } | GraphOp::Neighbors { .. } @@ -67,7 +67,10 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm | GraphOp::TemporalAlgorithm { .. } | GraphOp::BspSuperstep(_) | GraphOp::WccSuperstep(_) - | GraphOp::Stats { .. }, + | GraphOp::Stats { .. } + // Never client-issued: a CRDT delete's planner reads which ids + // are stored before it builds the delete's presence guard. + | GraphOp::NodePresenceRead { .. }, ) => Permission::Read, PhysicalPlan::Query( @@ -125,7 +128,7 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm Permission::Read } - // Read-only: reports what a governed ingest would store; that ingest is authorized separately. + // Read-only: reports what a governed ingest stores; that ingest is authorized separately. PhysicalPlan::Timeseries(TimeseriesOp::Scan { .. } | TimeseriesOp::ResolveIngest(_)) => { Permission::Read } @@ -189,7 +192,13 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm | GraphOp::EdgeDelete { .. } | GraphOp::EdgeDeleteBatch { .. } | GraphOp::SetNodeLabels { .. } - | GraphOp::RemoveNodeLabels { .. }, + | GraphOp::RemoveNodeLabels { .. } + // Never client-issued: the planner derives these from a + // document delete or TRUNCATE after that statement is + // authorized. + | GraphOp::NodeEdgeGuard { .. } + | GraphOp::NodePresenceGuard { .. } + | GraphOp::TruncateEdges { .. }, ) => Permission::Write, PhysicalPlan::Meta(MetaOp::WalAppend { .. }) => Permission::Write, @@ -257,7 +266,6 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm | MetaOp::QueryCollectionSize { .. } | MetaOp::AlterArray { .. } | MetaOp::RebuildIndex { .. } - | MetaOp::RenameCollection { .. } | MetaOp::DropTxnOverlay { .. }, ) => Permission::Admin, @@ -275,6 +283,9 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm // Installs a committed transaction's post-images into base state. PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { .. }) => Permission::Write, + // Installs a RESTORE's rows and edge versions into base state. + PhysicalPlan::Meta(MetaOp::RestoreRedo(_)) => Permission::Write, + // KV engine: read operations. PhysicalPlan::Kv( KvOp::Get { .. } @@ -289,7 +300,7 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm | KvOp::SortedIndexCount { .. } | KvOp::SortedIndexScore { .. } | KvOp::SortedIndexTxnRead { .. } - // Read-only: reports what a governed write would apply; that write is authorized separately. + // Read-only: reports what a governed write applies; that write is authorized separately. | KvOp::ResolveWrite(_), ) => Permission::Read, @@ -342,6 +353,13 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm Permission::Read } + // Hash-chain verification reads every stored row of one collection. + PhysicalPlan::Meta(MetaOp::VerifyHashChain { .. }) => Permission::Read, + + // A home version check reads the versions a transaction's own reads + // observed, and no row. + PhysicalPlan::Meta(MetaOp::HomeVersions { .. }) => Permission::Read, + // Array engine: query ops are reads, put/delete are writes, OpenArray is DDL, flush/compact are admin. PhysicalPlan::Array( ArrayOp::Slice { .. } @@ -356,7 +374,7 @@ pub fn required_permission(plan: &crate::bridge::envelope::PhysicalPlan) -> Perm ArrayOp::Flush { .. } | ArrayOp::Compact { .. } | ArrayOp::DropArray { .. } - | ArrayOp::RestoreArrayDrop { .. } + | ArrayOp::RekeyArray { .. } | ArrayOp::PurgeArrayDrop { .. }, ) => Permission::Admin, diff --git a/nodedb/src/control/security/jit/auth_user.rs b/nodedb/src/control/security/jit/auth_user.rs index 60eac520a..143e69550 100644 --- a/nodedb/src/control/security/jit/auth_user.rs +++ b/nodedb/src/control/security/jit/auth_user.rs @@ -133,6 +133,20 @@ impl AuthUserStore { }) } + /// Replace the in-memory records with the catalog's auth users. + pub fn reload_from_catalog(&self, catalog: &SystemCatalog) -> crate::Result<()> { + let users: HashMap = catalog + .load_all_auth_users()? + .iter() + .map(|s| { + let record = AuthUserRecord::from_stored(s); + (record.id.clone(), record) + }) + .collect(); + *self.users.write() = users; + Ok(()) + } + /// Get an auth user by ID. pub fn get(&self, id: &str) -> Option { let users = self.users.read(); @@ -161,7 +175,7 @@ impl AuthUserStore { /// Install a record replicated from another node: update the in-memory /// cache only. The redb row was already written by the catalog applier, - /// so re-writing it here would be a redundant second write. + /// so re-writing it here is a redundant second write. pub fn install_replicated(&self, stored: &StoredAuthUser) { let record = AuthUserRecord::from_stored(stored); let mut users = self.users.write(); diff --git a/nodedb/src/control/security/permission/check.rs b/nodedb/src/control/security/permission/check.rs index 1bc512df8..194910ab2 100644 --- a/nodedb/src/control/security/permission/check.rs +++ b/nodedb/src/control/security/permission/check.rs @@ -96,7 +96,7 @@ impl PermissionStore { return true; } - let target = collection_target(identity.tenant_id, collection); + let target = collection_target(database_id, identity.tenant_id, collection); for role in &identity.roles { if identity::role_grants_permission(role, permission) { @@ -160,7 +160,7 @@ impl PermissionStore { return true; } - let target = function_target(identity.tenant_id, function_name); + let target = function_target(database_id, identity.tenant_id, function_name); for role in &identity.roles { if identity::role_grants_permission(role, Permission::Execute) { @@ -410,11 +410,56 @@ mod tests { )); } + /// A grant on `orders` in one database opens no same-name collection in + /// another database of the same tenant. + #[test] + fn collection_grant_binds_one_database() { + let store = PermissionStore::new(); + let roles = RoleStore::new(); + let granted = DatabaseId::new(1024); + store + .grant( + &collection_target(granted, TenantId::new(1), "orders"), + "user:bob", + Permission::Read, + "admin", + None, + ) + .unwrap(); + + let id = identity("bob", vec![], false); + assert!(store.check(&id, Permission::Read, granted, "orders", &roles, NOOP)); + for other in [DatabaseId::new(1025), DatabaseId::DEFAULT] { + assert!(!store.check(&id, Permission::Read, other, "orders", &roles, NOOP)); + } + } + + /// The same holds for EXECUTE on a function. + #[test] + fn function_grant_binds_one_database() { + let store = PermissionStore::new(); + let roles = RoleStore::new(); + let granted = DatabaseId::new(1024); + store + .grant( + &function_target(granted, TenantId::new(1), "score"), + "user:bob", + Permission::Execute, + "admin", + None, + ) + .unwrap(); + + let id = identity("bob", vec![], false); + assert!(store.check_function(&id, granted, "score", &roles, NOOP)); + assert!(!store.check_function(&id, DatabaseId::new(1025), "score", &roles, NOOP)); + } + #[test] fn explicit_user_grant() { let store = PermissionStore::new(); let roles = RoleStore::new(); - let target = collection_target(TenantId::new(1), "orders"); + let target = collection_target(DatabaseId::DEFAULT, TenantId::new(1), "orders"); store .grant(&target, "user:bob", Permission::Read, "admin", None) .unwrap(); @@ -442,7 +487,7 @@ mod tests { fn grant_on_role() { let store = PermissionStore::new(); let roles = RoleStore::new(); - let target = collection_target(TenantId::new(1), "reports"); + let target = collection_target(DatabaseId::DEFAULT, TenantId::new(1), "reports"); store .grant(&target, "readonly", Permission::Read, "admin", None) .unwrap(); @@ -466,7 +511,7 @@ mod tests { .unwrap(); let perm_store = PermissionStore::new(); - let target = collection_target(TenantId::new(1), "data"); + let target = collection_target(DatabaseId::DEFAULT, TenantId::new(1), "data"); perm_store .grant(&target, "readonly", Permission::Read, "admin", None) .unwrap(); @@ -485,7 +530,7 @@ mod tests { #[test] fn revoke_removes_grant() { let store = PermissionStore::new(); - let target = collection_target(TenantId::new(1), "users"); + let target = collection_target(DatabaseId::DEFAULT, TenantId::new(1), "users"); store .grant(&target, "user:bob", Permission::Read, "admin", None) .unwrap(); @@ -667,7 +712,7 @@ mod tests { &roles, NOOP )); - let target = collection_target(TenantId::new(1), "orders"); + let target = collection_target(DatabaseId::DEFAULT, TenantId::new(1), "orders"); store .grant(&target, "user:bob", Permission::Read, "admin", None) .expect("post-panic grant must succeed"); diff --git a/nodedb/src/control/security/permission/mod.rs b/nodedb/src/control/security/permission/mod.rs index 441363474..f74d9f212 100644 --- a/nodedb/src/control/security/permission/mod.rs +++ b/nodedb/src/control/security/permission/mod.rs @@ -31,6 +31,6 @@ pub mod types; pub use replication::prepare_owner; pub use store::PermissionStore; pub use types::{ - Grant, OwnerRecord, collection_target, format_permission, function_target, owner_key, - parse_permission, procedure_target, tenant_target, + Grant, OwnerRecord, ScopedTarget, collection_target, format_permission, function_target, + owner_key, parse_permission, parse_scoped_target, procedure_target, tenant_target, }; diff --git a/nodedb/src/control/security/permission/store.rs b/nodedb/src/control/security/permission/store.rs index a2cd81bff..d6c796fbb 100644 --- a/nodedb/src/control/security/permission/store.rs +++ b/nodedb/src/control/security/permission/store.rs @@ -17,7 +17,7 @@ use super::types::{Grant, format_permission, owner_key, parse_permission}; /// redb persistence. pub struct PermissionStore { pub(super) grants: RwLock>, - /// "collection:{tenant_id}:{name}" → owner username + /// owner key → owner username pub(super) owners: RwLock>, } @@ -172,15 +172,18 @@ impl PermissionStore { out } - /// List all grants scoped to the given tenant ID prefix. + /// List every collection and function grant of `tenant_id`, across + /// databases. pub fn all_grants(&self, tenant_id: TenantId) -> Vec { let tid = tenant_id.as_u64(); - let col_prefix = format!("collection:{tid}:"); - let func_prefix = format!("function:{tid}:"); self.grants .read() .iter() - .filter(|g| g.target.starts_with(&col_prefix) || g.target.starts_with(&func_prefix)) + .filter(|g| { + super::types::parse_scoped_target(&g.target).is_some_and(|t| { + t.tenant_id == tid && matches!(t.kind, "collection" | "function") + }) + }) .cloned() .collect() } diff --git a/nodedb/src/control/security/permission/types.rs b/nodedb/src/control/security/permission/types.rs index 2fc573c51..ac91d7c36 100644 --- a/nodedb/src/control/security/permission/types.rs +++ b/nodedb/src/control/security/permission/types.rs @@ -4,12 +4,12 @@ //! permission module. use crate::control::security::identity::Permission; -use crate::types::TenantId; +use crate::types::{DatabaseId, TenantId}; /// A permission grant record (in-memory). #[derive(Debug, Clone, PartialEq, Eq, Hash)] pub struct Grant { - /// Target: "cluster", "tenant:1", "collection:1:users" + /// Target: "cluster", "tenant:1", "collection:0:1:users" pub target: String, /// Grantee: role name or "user:username" pub grantee: String, @@ -26,20 +26,66 @@ pub struct OwnerRecord { pub owner_username: String, } -/// Build a `collection:{tenant}:{name}` target string for grants -/// and ownership lookups. -pub fn collection_target(tenant_id: TenantId, collection: &str) -> String { - format!("collection:{}:{}", tenant_id.as_u64(), collection) +/// Build a `collection:{database}:{tenant}:{name}` grant target. A grant +/// binds one database: a same-name collection in another database is a +/// different target. +pub fn collection_target(database_id: DatabaseId, tenant_id: TenantId, collection: &str) -> String { + scoped_target("collection", database_id, tenant_id, collection) } -/// Build a `function:{tenant}:{name}` target string. -pub fn function_target(tenant_id: TenantId, function_name: &str) -> String { - format!("function:{}:{}", tenant_id.as_u64(), function_name) +/// Build a `function:{database}:{tenant}:{name}` grant target. +pub fn function_target( + database_id: DatabaseId, + tenant_id: TenantId, + function_name: &str, +) -> String { + scoped_target("function", database_id, tenant_id, function_name) } -/// Build a `procedure:{tenant}:{name}` target string. -pub fn procedure_target(tenant_id: TenantId, procedure_name: &str) -> String { - format!("procedure:{}:{}", tenant_id.as_u64(), procedure_name) +/// Build a `procedure:{database}:{tenant}:{name}` grant target. +pub fn procedure_target( + database_id: DatabaseId, + tenant_id: TenantId, + procedure_name: &str, +) -> String { + scoped_target("procedure", database_id, tenant_id, procedure_name) +} + +fn scoped_target(kind: &str, database_id: DatabaseId, tenant_id: TenantId, name: &str) -> String { + format!( + "{kind}:{}:{}:{name}", + database_id.as_u64(), + tenant_id.as_u64() + ) +} + +/// A database-scoped grant target split into its parts. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct ScopedTarget<'a> { + /// `collection`, `function`, or `procedure`. + pub kind: &'a str, + pub database_id: u64, + pub tenant_id: u64, + pub name: &'a str, +} + +/// Split a target built by [`collection_target`], [`function_target`], or +/// [`procedure_target`]. `None` for every other target shape. +pub fn parse_scoped_target(target: &str) -> Option> { + let mut parts = target.splitn(4, ':'); + let kind = parts.next()?; + if !matches!(kind, "collection" | "function" | "procedure") { + return None; + } + let database_id = parts.next()?.parse().ok()?; + let tenant_id = parts.next()?.parse().ok()?; + let name = parts.next()?; + Some(ScopedTarget { + kind, + database_id, + tenant_id, + name, + }) } /// Build a `tenant:{id}` target string for tenant-scoped grants — a diff --git a/nodedb/src/control/security/permission_tree/cache.rs b/nodedb/src/control/security/permission_tree/cache.rs index cb798f133..29adc211d 100644 --- a/nodedb/src/control/security/permission_tree/cache.rs +++ b/nodedb/src/control/security/permission_tree/cache.rs @@ -6,19 +6,24 @@ //! and kept current by the Event Plane's permission step. `progress` records //! how far the cache reflects each core's writes. Lives entirely in the //! Control Plane (Send + Sync). +//! +//! Hierarchy and grants are held per [`TreeScope`]: a tree in one database +//! never resolves against another database's rows. The plan-cache version is +//! per tenant, so a change in any database of a tenant stales its plans. use std::collections::{HashMap, HashSet}; use std::sync::Arc; use tracing::{debug, info}; +use super::scope::{TreeKey, TreeScope}; use super::sources::SourceIndex; use super::sync_state::ApplyProgress; use super::types::{PermissionGrant, PermissionTreeDef}; -/// Per-tenant permission state: resource hierarchy + permission grants. +/// Permission state of one scope: resource hierarchy + permission grants. #[derive(Debug, Default)] -struct TenantPermissions { +struct ScopePermissions { /// Resource hierarchy: `child_id → parent_id`. /// Walk this map upward to find ancestors. parent_map: HashMap, @@ -31,11 +36,6 @@ struct TenantPermissions { /// Reverse children map: `parent_id → set of child_ids`. /// Used for cache invalidation: when a grant changes, evict all descendants. children_map: HashMap>, - - /// Monotonic version, bumped on every grant or edge mutation for this - /// tenant. A cached physical plan stamps this at build time so a plan - /// cache can detect a revoked grant instead of replaying a frozen filter. - version: u64, } /// Central permission cache shared across all sessions. @@ -43,12 +43,17 @@ struct TenantPermissions { /// Thread-safe: wrapped in `Arc>` by SharedState. #[derive(Debug)] pub struct PermissionCache { - /// Per-tenant permission state. - tenants: HashMap, + /// Per-scope permission state. + scopes: HashMap, + + /// Monotonic version per tenant, bumped on every grant, edge, or tree + /// mutation in any of the tenant's databases. A cached physical plan + /// stamps this at build time so a plan cache can detect a revoked grant + /// instead of replaying a frozen filter. + versions: HashMap, /// Per-collection permission tree definitions. - /// Key: `(tenant_id, collection_name)`. - tree_defs: HashMap<(u64, String), PermissionTreeDef>, + tree_defs: HashMap, /// How far the cache reflects each core's writes. progress: ApplyProgress, @@ -67,8 +72,7 @@ impl Default for PermissionCache { /// One collection a reload scans, and what its rows carry. #[derive(Debug, Clone, PartialEq, Eq, Hash)] pub struct TreeSource { - pub tenant_id: u64, - pub collection: String, + pub key: TreeKey, pub kind: TreeSourceKind, } @@ -82,10 +86,11 @@ pub enum TreeSourceKind { } impl PermissionCache { - /// An empty cache. It needs a reload before planning may use it. + /// An empty cache. It needs a reload before planning can use it. pub fn new() -> Self { Self { - tenants: HashMap::new(), + scopes: HashMap::new(), + versions: HashMap::new(), tree_defs: HashMap::new(), progress: ApplyProgress::new(), sources: Arc::new(SourceIndex::default()), @@ -111,91 +116,100 @@ impl PermissionCache { /// Every collection a reload scans, deduplicated. pub fn tree_sources(&self) -> Vec { let mut sources: HashSet = HashSet::new(); - for ((tenant_id, collection), def) in &self.tree_defs { + for (key, def) in &self.tree_defs { sources.insert(TreeSource { - tenant_id: *tenant_id, - collection: collection.clone(), + key: key.clone(), kind: TreeSourceKind::Hierarchy, }); sources.insert(TreeSource { - tenant_id: *tenant_id, - collection: def.permission_table.clone(), + key: key.scope.collection(def.permission_table.clone()), kind: TreeSourceKind::Grants, }); } sources.into_iter().collect() } - /// Replace a tenant's hierarchy and grants with a reload's result, and - /// bump its version so a cached plan built from the old state goes stale. - pub fn replace_tenant_state( + /// Replace a scope's hierarchy and grants with a reload's result, and + /// bump its tenant's version so a cached plan built from the old state + /// goes stale. + pub fn replace_scope_state( &mut self, - tenant_id: u64, + scope: TreeScope, edges: &[(String, String)], grants: &[PermissionGrant], ) { - let version = self.tenant_version(tenant_id); - self.tenants.insert( - tenant_id, - TenantPermissions { - version, - ..TenantPermissions::default() - }, - ); - self.load_edges(tenant_id, edges); - self.load_grants(tenant_id, grants); - self.bump_tenant_version(tenant_id); + self.scopes.insert(scope, ScopePermissions::default()); + self.load_edges(scope, edges); + self.load_grants(scope, grants); + self.bump_tenant_version(scope.tenant_id); + } + + /// Drop the state of every scope `keep` does not name. A reload calls + /// this so a scope whose trees were all removed holds no stale grants. + pub fn retain_scopes(&mut self, keep: &HashSet) { + let dropped: Vec = self + .scopes + .keys() + .filter(|scope| !keep.contains(scope)) + .copied() + .collect(); + for scope in dropped { + self.scopes.remove(&scope); + self.bump_tenant_version(scope.tenant_id); + } } /// Register a permission tree definition for a collection. Registering /// the definition already held changes nothing, so the DDL node and the /// metadata applier can both apply one change. - pub fn register_tree_def(&mut self, tenant_id: u64, collection: &str, def: PermissionTreeDef) { - if self.get_tree_def(tenant_id, collection) == Some(&def) { + pub fn register_tree_def(&mut self, key: TreeKey, def: PermissionTreeDef) { + if self.get_tree_def(&key) == Some(&def) { return; } info!( - tenant_id, - collection, + database_id = key.scope.database_id.as_u64(), + tenant_id = key.scope.tenant_id, + collection = %key.collection, levels = ?def.levels, "permission_tree: registered" ); - self.tree_defs - .insert((tenant_id, collection.to_owned()), def); + let tenant_id = key.scope.tenant_id; + self.tree_defs.insert(key, def); self.sources.rebuild(&self.tree_defs); - // The new sources may already hold rows no reload has read. + // The new sources can already hold rows no reload has read. self.progress.mark_reload_needed(); self.bump_tenant_version(tenant_id); } /// Remove a permission tree definition for a collection. Removing an /// absent definition changes nothing. - pub fn unregister_tree_def(&mut self, tenant_id: u64, collection: &str) { - if self - .tree_defs - .remove(&(tenant_id, collection.to_owned())) - .is_none() - { + pub fn unregister_tree_def(&mut self, key: &TreeKey) { + if self.tree_defs.remove(key).is_none() { return; } self.sources.rebuild(&self.tree_defs); self.progress.mark_reload_needed(); - self.bump_tenant_version(tenant_id); - info!(tenant_id, collection, "permission_tree: unregistered"); + self.bump_tenant_version(key.scope.tenant_id); + info!( + database_id = key.scope.database_id.as_u64(), + tenant_id = key.scope.tenant_id, + collection = %key.collection, + "permission_tree: unregistered" + ); } /// Get the permission tree definition for a collection (if any). - pub fn get_tree_def(&self, tenant_id: u64, collection: &str) -> Option<&PermissionTreeDef> { - self.tree_defs.get(&(tenant_id, collection.to_owned())) + pub fn get_tree_def(&self, key: &TreeKey) -> Option<&PermissionTreeDef> { + self.tree_defs.get(key) } /// Load a parent→child edge into the hierarchy. - pub fn put_edge(&mut self, tenant_id: u64, child_id: &str, parent_id: &str) { - let tenant = self.tenants.entry(tenant_id).or_default(); - tenant + pub fn put_edge(&mut self, scope: TreeScope, child_id: &str, parent_id: &str) { + let state = self.scopes.entry(scope).or_default(); + state .parent_map .insert(child_id.to_owned(), parent_id.to_owned()); - tenant + state .children_map .entry(parent_id.to_owned()) .or_default() @@ -203,33 +217,33 @@ impl PermissionCache { } /// Remove a parent→child edge from the hierarchy. - pub fn remove_edge(&mut self, tenant_id: u64, child_id: &str) { - let Some(tenant) = self.tenants.get_mut(&tenant_id) else { + pub fn remove_edge(&mut self, scope: TreeScope, child_id: &str) { + let Some(state) = self.scopes.get_mut(&scope) else { return; }; - if let Some(old_parent) = tenant.parent_map.remove(child_id) - && let Some(children) = tenant.children_map.get_mut(&old_parent) + if let Some(old_parent) = state.parent_map.remove(child_id) + && let Some(children) = state.children_map.get_mut(&old_parent) { children.remove(child_id); if children.is_empty() { - tenant.children_map.remove(&old_parent); + state.children_map.remove(&old_parent); } } } /// Load a permission grant into the cache. - pub fn put_grant(&mut self, tenant_id: u64, grant: &PermissionGrant) { - let tenant = self.tenants.entry(tenant_id).or_default(); - tenant.grants.insert( + pub fn put_grant(&mut self, scope: TreeScope, grant: &PermissionGrant) { + let state = self.scopes.entry(scope).or_default(); + state.grants.insert( (grant.resource_id.clone(), grant.grantee.clone()), (grant.level.clone(), grant.inherited), ); } /// Remove a permission grant from the cache. - pub fn remove_grant(&mut self, tenant_id: u64, resource_id: &str, grantee: &str) { - if let Some(tenant) = self.tenants.get_mut(&tenant_id) { - tenant + pub fn remove_grant(&mut self, scope: TreeScope, resource_id: &str, grantee: &str) { + if let Some(state) = self.scopes.get_mut(&scope) { + state .grants .remove(&(resource_id.to_owned(), grantee.to_owned())); } @@ -239,54 +253,54 @@ impl PermissionCache { /// Returns `None` if no grant exists at this exact resource. pub fn get_grant( &self, - tenant_id: u64, + scope: TreeScope, resource_id: &str, grantee: &str, ) -> Option<(&str, bool)> { - self.tenants - .get(&tenant_id)? + self.scopes + .get(&scope)? .grants .get(&(resource_id.to_owned(), grantee.to_owned())) .map(|(level, inherited)| (level.as_str(), *inherited)) } /// Get the parent of a resource. Returns `None` if root. - pub fn get_parent(&self, tenant_id: u64, resource_id: &str) -> Option<&str> { - self.tenants - .get(&tenant_id)? + pub fn get_parent(&self, scope: TreeScope, resource_id: &str) -> Option<&str> { + self.scopes + .get(&scope)? .parent_map .get(resource_id) .map(|s| s.as_str()) } /// Get all children of a resource (direct, not recursive). - pub fn get_children(&self, tenant_id: u64, resource_id: &str) -> Vec<&str> { - self.tenants - .get(&tenant_id) - .and_then(|t| t.children_map.get(resource_id)) + pub fn get_children(&self, scope: TreeScope, resource_id: &str) -> Vec<&str> { + self.scopes + .get(&scope) + .and_then(|s| s.children_map.get(resource_id)) .map(|children| children.iter().map(|s| s.as_str()).collect()) .unwrap_or_default() } - /// Get all resource IDs for a tenant (for iteration during accessible_resources). - pub fn all_resource_ids(&self, tenant_id: u64) -> Vec<&str> { - self.tenants - .get(&tenant_id) - .map(|t| { + /// Get all resource IDs of a scope (for iteration during accessible_resources). + pub fn all_resource_ids(&self, scope: TreeScope) -> Vec<&str> { + self.scopes + .get(&scope) + .map(|s| { // Collect from all sources: edges (parent_map), reverse edges // (children_map), AND grant keys (for root resources with // direct grants but no parent/child edges). let mut ids: HashSet<&str> = HashSet::new(); - for k in t.parent_map.keys() { + for k in s.parent_map.keys() { ids.insert(k.as_str()); } - for k in t.parent_map.values() { + for k in s.parent_map.values() { ids.insert(k.as_str()); } - for k in t.children_map.keys() { + for k in s.children_map.keys() { ids.insert(k.as_str()); } - for (resource_id, _) in t.grants.keys() { + for (resource_id, _) in s.grants.keys() { ids.insert(resource_id.as_str()); } ids.into_iter().collect() @@ -295,11 +309,11 @@ impl PermissionCache { } /// Get all grantees that have explicit grants for a given resource. - pub fn grantees_for_resource(&self, tenant_id: u64, resource_id: &str) -> Vec<&str> { - self.tenants - .get(&tenant_id) - .map(|t| { - t.grants + pub fn grantees_for_resource(&self, scope: TreeScope, resource_id: &str) -> Vec<&str> { + self.scopes + .get(&scope) + .map(|s| { + s.grants .keys() .filter(|(rid, _)| rid == resource_id) .map(|(_, grantee)| grantee.as_str()) @@ -309,43 +323,42 @@ impl PermissionCache { } /// Bulk load edges from a list of (child_id, parent_id) pairs. - pub fn load_edges(&mut self, tenant_id: u64, edges: &[(String, String)]) { + pub fn load_edges(&mut self, scope: TreeScope, edges: &[(String, String)]) { for (child, parent) in edges { - self.put_edge(tenant_id, child, parent); + self.put_edge(scope, child, parent); } debug!( - tenant_id, + database_id = scope.database_id.as_u64(), + tenant_id = scope.tenant_id, edges = edges.len(), "permission_tree: loaded edges" ); } /// Bulk load grants. - pub fn load_grants(&mut self, tenant_id: u64, grants: &[PermissionGrant]) { + pub fn load_grants(&mut self, scope: TreeScope, grants: &[PermissionGrant]) { for grant in grants { - self.put_grant(tenant_id, grant); + self.put_grant(scope, grant); } debug!( - tenant_id, + database_id = scope.database_id.as_u64(), + tenant_id = scope.tenant_id, grants = grants.len(), "permission_tree: loaded grants" ); } - /// Number of resources tracked for a tenant. - pub fn resource_count(&self, tenant_id: u64) -> usize { - self.tenants - .get(&tenant_id) - .map(|t| t.parent_map.len()) + /// Number of resources tracked for a scope. + pub fn resource_count(&self, scope: TreeScope) -> usize { + self.scopes + .get(&scope) + .map(|s| s.parent_map.len()) .unwrap_or(0) } - /// Number of grants tracked for a tenant. - pub fn grant_count(&self, tenant_id: u64) -> usize { - self.tenants - .get(&tenant_id) - .map(|t| t.grants.len()) - .unwrap_or(0) + /// Number of grants tracked for a scope. + pub fn grant_count(&self, scope: TreeScope) -> usize { + self.scopes.get(&scope).map(|s| s.grants.len()).unwrap_or(0) } /// Check if any permission tree definitions are registered. @@ -353,122 +366,155 @@ impl PermissionCache { !self.tree_defs.is_empty() } - /// Check if any permission tree definition is registered for this tenant. + /// Check if any permission tree definition is registered for this + /// tenant, in any database. /// /// Used by plan-time enforcement for operations that name no collection: /// they cannot be shown to avoid a governed collection, so the question /// widens to the whole tenant. pub fn has_tree_defs_for_tenant(&self, tenant_id: u64) -> bool { - self.tree_defs.keys().any(|(tid, _)| *tid == tenant_id) + self.tree_defs + .keys() + .any(|key| key.scope.tenant_id == tenant_id) } - /// Check if any tree def for this tenant uses `collection` as its permission table. - pub fn tree_defs_using_permission_table(&self, tenant_id: u64, collection: &str) -> bool { - self.tree_defs - .iter() - .any(|((tid, _), def)| *tid == tenant_id && def.permission_table == collection) + /// Check if any tree def in `key`'s scope uses its collection as the + /// permission table. + pub fn tree_defs_using_permission_table(&self, key: &TreeKey) -> bool { + self.tree_defs.iter().any(|(governed, def)| { + governed.scope == key.scope && def.permission_table == key.collection + }) } - /// Check if any tree def for this tenant uses `collection` as its resource graph source. - /// The graph index name references a graph on a collection — we check by matching - /// the collection name itself (the resource hierarchy collection). - pub fn tree_defs_using_graph(&self, tenant_id: u64, collection: &str) -> bool { - self.tree_defs - .iter() - .any(|((tid, coll), _)| *tid == tenant_id && coll == collection) + /// Check if `key` is a governed collection: its rows form the resource + /// hierarchy. + pub fn tree_defs_using_graph(&self, key: &TreeKey) -> bool { + self.tree_defs.contains_key(key) } /// Bump and return the per-tenant permission version. Called from every /// mutator in `invalidation.rs`, in the same critical section as the /// state change, so a stamped plan cache entry goes stale on the spot. pub fn bump_tenant_version(&mut self, tenant_id: u64) -> u64 { - let tenant = self.tenants.entry(tenant_id).or_default(); - tenant.version += 1; - tenant.version + let version = self.versions.entry(tenant_id).or_default(); + *version += 1; + *version } /// Current permission version for a tenant. `0` if the tenant has never /// been mutated. pub fn tenant_version(&self, tenant_id: u64) -> u64 { - self.tenants.get(&tenant_id).map(|t| t.version).unwrap_or(0) + self.versions.get(&tenant_id).copied().unwrap_or(0) } } #[cfg(test)] mod tests { use super::*; + use crate::types::DatabaseId; + + const S1: TreeScope = TreeScope { + database_id: DatabaseId::DEFAULT, + tenant_id: 1, + }; + const S2: TreeScope = TreeScope { + database_id: DatabaseId::DEFAULT, + tenant_id: 2, + }; + + fn def() -> PermissionTreeDef { + sonic_rs::from_str( + r#"{"resource_column":"id","graph_index":"tree","permission_table":"grants"}"#, + ) + .expect("tree def") + } + + fn grant(resource: &str, grantee: &str, level: &str) -> PermissionGrant { + PermissionGrant { + resource_id: resource.into(), + grantee: grantee.into(), + level: level.into(), + inherited: false, + } + } #[test] fn edge_hierarchy() { let mut cache = PermissionCache::new(); // workspace → folder → doc - cache.put_edge(1, "folder-1", "workspace-1"); - cache.put_edge(1, "doc-1", "folder-1"); + cache.put_edge(S1, "folder-1", "workspace-1"); + cache.put_edge(S1, "doc-1", "folder-1"); - assert_eq!(cache.get_parent(1, "doc-1"), Some("folder-1")); - assert_eq!(cache.get_parent(1, "folder-1"), Some("workspace-1")); - assert_eq!(cache.get_parent(1, "workspace-1"), None); // Root. + assert_eq!(cache.get_parent(S1, "doc-1"), Some("folder-1")); + assert_eq!(cache.get_parent(S1, "folder-1"), Some("workspace-1")); + assert_eq!(cache.get_parent(S1, "workspace-1"), None); // Root. - let children = cache.get_children(1, "workspace-1"); + let children = cache.get_children(S1, "workspace-1"); assert_eq!(children, vec!["folder-1"]); } #[test] fn grant_lookup() { let mut cache = PermissionCache::new(); - cache.put_grant( - 1, - &PermissionGrant { - resource_id: "folder-1".into(), - grantee: "user-42".into(), - level: "editor".into(), - inherited: false, - }, - ); + cache.put_grant(S1, &grant("folder-1", "user-42", "editor")); - let (level, inherited) = cache.get_grant(1, "folder-1", "user-42").unwrap(); + let (level, inherited) = cache.get_grant(S1, "folder-1", "user-42").unwrap(); assert_eq!(level, "editor"); assert!(!inherited); - assert!(cache.get_grant(1, "folder-1", "user-99").is_none()); + assert!(cache.get_grant(S1, "folder-1", "user-99").is_none()); } #[test] fn remove_edge() { let mut cache = PermissionCache::new(); - cache.put_edge(1, "doc-1", "folder-1"); - assert_eq!(cache.get_parent(1, "doc-1"), Some("folder-1")); + cache.put_edge(S1, "doc-1", "folder-1"); + assert_eq!(cache.get_parent(S1, "doc-1"), Some("folder-1")); - cache.remove_edge(1, "doc-1"); - assert_eq!(cache.get_parent(1, "doc-1"), None); - assert!(cache.get_children(1, "folder-1").is_empty()); + cache.remove_edge(S1, "doc-1"); + assert_eq!(cache.get_parent(S1, "doc-1"), None); + assert!(cache.get_children(S1, "folder-1").is_empty()); } #[test] fn remove_grant() { let mut cache = PermissionCache::new(); - cache.put_grant( - 1, - &PermissionGrant { - resource_id: "doc-1".into(), - grantee: "user-1".into(), - level: "viewer".into(), - inherited: false, - }, - ); - assert!(cache.get_grant(1, "doc-1", "user-1").is_some()); - cache.remove_grant(1, "doc-1", "user-1"); - assert!(cache.get_grant(1, "doc-1", "user-1").is_none()); + cache.put_grant(S1, &grant("doc-1", "user-1", "viewer")); + assert!(cache.get_grant(S1, "doc-1", "user-1").is_some()); + cache.remove_grant(S1, "doc-1", "user-1"); + assert!(cache.get_grant(S1, "doc-1", "user-1").is_none()); } #[test] fn tenant_isolation() { let mut cache = PermissionCache::new(); - cache.put_edge(1, "doc-1", "folder-1"); - cache.put_edge(2, "doc-1", "folder-2"); + cache.put_edge(S1, "doc-1", "folder-1"); + cache.put_edge(S2, "doc-1", "folder-2"); + + assert_eq!(cache.get_parent(S1, "doc-1"), Some("folder-1")); + assert_eq!(cache.get_parent(S2, "doc-1"), Some("folder-2")); + } + + /// One tenant's hierarchy, grants, and trees in one database are + /// invisible from another database. + #[test] + fn database_isolation() { + let db1 = TreeScope::new(DatabaseId::new(7), 1); + let db2 = TreeScope::new(DatabaseId::new(8), 1); + let mut cache = PermissionCache::new(); + cache.register_tree_def(db1.collection("docs"), def()); + cache.put_edge(db1, "doc-1", "folder-1"); + cache.put_grant(db1, &grant("folder-1", "user-1", "viewer")); - assert_eq!(cache.get_parent(1, "doc-1"), Some("folder-1")); - assert_eq!(cache.get_parent(2, "doc-1"), Some("folder-2")); + assert!(cache.get_tree_def(&db1.collection("docs")).is_some()); + assert!(cache.get_tree_def(&db2.collection("docs")).is_none()); + assert!(cache.tree_defs_using_graph(&db1.collection("docs"))); + assert!(!cache.tree_defs_using_graph(&db2.collection("docs"))); + assert!(cache.tree_defs_using_permission_table(&db1.collection("grants"))); + assert!(!cache.tree_defs_using_permission_table(&db2.collection("grants"))); + assert_eq!(cache.get_parent(db2, "doc-1"), None); + assert!(cache.get_grant(db2, "folder-1", "user-1").is_none()); + assert!(cache.all_resource_ids(db2).is_empty()); } #[test] @@ -495,67 +541,68 @@ mod tests { write_level: "editor".into(), delete_level: "editor".into(), }; - cache.register_tree_def(1, "documents", def.clone()); - assert!(cache.get_tree_def(1, "documents").is_some()); - assert!(cache.get_tree_def(1, "other").is_none()); - assert!(cache.get_tree_def(2, "documents").is_none()); + cache.register_tree_def(S1.collection("documents"), def.clone()); + assert!(cache.get_tree_def(&S1.collection("documents")).is_some()); + assert!(cache.get_tree_def(&S1.collection("other")).is_none()); + assert!(cache.get_tree_def(&S2.collection("documents")).is_none()); // The DDL node and the metadata applier both apply one change. let version = cache.tenant_version(1); - cache.register_tree_def(1, "documents", def); + cache.register_tree_def(S1.collection("documents"), def); assert_eq!(cache.tenant_version(1), version); - cache.unregister_tree_def(1, "documents"); - assert!(cache.get_tree_def(1, "documents").is_none()); + cache.unregister_tree_def(&S1.collection("documents")); + assert!(cache.get_tree_def(&S1.collection("documents")).is_none()); let version = cache.tenant_version(1); - cache.unregister_tree_def(1, "documents"); + cache.unregister_tree_def(&S1.collection("documents")); assert_eq!(cache.tenant_version(1), version); } #[test] - fn a_reload_replaces_the_tenant_state_and_bumps_its_version() { + fn a_reload_replaces_the_scope_state_and_bumps_its_version() { let mut cache = PermissionCache::new(); - cache.put_edge(1, "doc-1", "folder-1"); - cache.put_grant( - 1, - &PermissionGrant { - resource_id: "doc-1".into(), - grantee: "user-1".into(), - level: "viewer".into(), - inherited: false, - }, - ); + cache.put_edge(S1, "doc-1", "folder-1"); + cache.put_grant(S1, &grant("doc-1", "user-1", "viewer")); let before = cache.bump_tenant_version(1); - cache.replace_tenant_state(1, &[("doc-2".into(), "folder-2".into())], &[]); + cache.replace_scope_state(S1, &[("doc-2".into(), "folder-2".into())], &[]); - assert_eq!(cache.get_parent(1, "doc-1"), None); - assert_eq!(cache.get_parent(1, "doc-2"), Some("folder-2")); - assert!(cache.get_grant(1, "doc-1", "user-1").is_none()); + assert_eq!(cache.get_parent(S1, "doc-1"), None); + assert_eq!(cache.get_parent(S1, "doc-2"), Some("folder-2")); + assert!(cache.get_grant(S1, "doc-1", "user-1").is_none()); assert!(cache.tenant_version(1) > before); } + #[test] + fn retaining_scopes_drops_the_rest_and_bumps_their_tenant() { + let mut cache = PermissionCache::new(); + cache.put_grant(S1, &grant("doc-1", "user-1", "viewer")); + cache.put_grant(S2, &grant("doc-1", "user-1", "viewer")); + let before = cache.tenant_version(2); + + cache.retain_scopes(&HashSet::from([S1])); + + assert!(cache.get_grant(S1, "doc-1", "user-1").is_some()); + assert!(cache.get_grant(S2, "doc-1", "user-1").is_none()); + assert!(cache.tenant_version(2) > before); + } + #[test] fn tree_sources_name_the_governed_collection_and_the_permission_table() { + let scope = TreeScope::new(DatabaseId::new(7), 1); let mut cache = PermissionCache::new(); - let def: PermissionTreeDef = sonic_rs::from_str( - r#"{"resource_column":"id","graph_index":"tree","permission_table":"grants"}"#, - ) - .expect("tree def"); - cache.register_tree_def(1, "docs", def); + cache.register_tree_def(scope.collection("docs"), def()); let mut sources = cache.tree_sources(); - sources.sort_by(|a, b| a.collection.cmp(&b.collection)); + sources.sort_by(|a, b| a.key.collection.cmp(&b.key.collection)); assert_eq!( sources, vec![ TreeSource { - tenant_id: 1, - collection: "docs".into(), + key: scope.collection("docs"), kind: TreeSourceKind::Hierarchy, }, TreeSource { - tenant_id: 1, - collection: "grants".into(), + key: scope.collection("grants"), kind: TreeSourceKind::Grants, }, ] diff --git a/nodedb/src/control/security/permission_tree/event_handler.rs b/nodedb/src/control/security/permission_tree/event_handler.rs index 828a27b4f..507027420 100644 --- a/nodedb/src/control/security/permission_tree/event_handler.rs +++ b/nodedb/src/control/security/permission_tree/event_handler.rs @@ -20,6 +20,7 @@ use crate::event::types::WriteEvent; use super::cache::PermissionCache; use super::invalidation; +use super::scope::TreeKey; use super::types::PermissionGrant; /// Apply the permission effect of `events`, taken off core `core_id`'s ring @@ -58,11 +59,19 @@ pub async fn apply_ring_events( /// Apply one data event to the cache when its collection is a permission /// table or a governed collection of a registered tree. fn apply_event(cache: &mut PermissionCache, event: &WriteEvent) { - let collection = event.collection.as_ref(); - let tenant_id = event.tenant_id.as_u64(); + // An event carries the qualified name of its database's collection. A + // name that is not qualified for that database names no tree source. + let Ok(key) = TreeKey::from_qualified( + event.database_id, + event.tenant_id.as_u64(), + event.collection.as_ref(), + ) else { + return; + }; + let scope = key.scope; - let is_permission_table = cache.tree_defs_using_permission_table(tenant_id, collection); - let is_resource_graph = cache.tree_defs_using_graph(tenant_id, collection); + let is_permission_table = cache.tree_defs_using_permission_table(&key); + let is_resource_graph = cache.tree_defs_using_graph(&key); if !is_permission_table && !is_resource_graph { return; @@ -79,13 +88,13 @@ fn apply_event(cache: &mut PermissionCache, event: &WriteEvent) { && let Some(ref val) = new_val && let Some(grant) = extract_grant(val) { - invalidation::on_grant_upsert(cache, tenant_id, &grant); + invalidation::on_grant_upsert(cache, scope, &grant); } if is_resource_graph && let Some(ref val) = new_val && let Some((child_id, parent_id)) = extract_edge(val) { - invalidation::on_edge_upsert(cache, tenant_id, child_id, parent_id); + invalidation::on_edge_upsert(cache, scope, child_id, parent_id); } } crate::event::types::WriteOp::Delete => { @@ -101,23 +110,26 @@ fn apply_event(cache: &mut PermissionCache, event: &WriteEvent) { val.get("grantee").and_then(|v| v.as_str()), ) { - invalidation::on_grant_delete(cache, tenant_id, resource_id, grantee); + invalidation::on_grant_delete(cache, scope, resource_id, grantee); } if is_resource_graph && let Some(ref val) = old_val && let Some(child_id) = val.get("id").and_then(|v| v.as_str()) { - invalidation::on_edge_delete(cache, tenant_id, child_id); + invalidation::on_edge_delete(cache, scope, child_id); } } crate::event::types::WriteOp::BulkInsert { .. } | crate::event::types::WriteOp::BulkDelete { .. } - | crate::event::types::WriteOp::Heartbeat => {} + | crate::event::types::WriteOp::Heartbeat + | crate::event::types::WriteOp::Publish => {} } debug!( - tenant_id, - collection, "permission_tree: cache updated from a write event" + database_id = scope.database_id.as_u64(), + tenant_id = scope.tenant_id, + collection = %key.collection, + "permission_tree: cache updated from a write event" ); } @@ -144,23 +156,38 @@ mod tests { use std::sync::Arc; use super::*; + use crate::control::security::permission_tree::scope::TreeScope; use crate::event::types::{EventSource, RowId, WriteOp}; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; const TENANT: u64 = 1; + const SCOPE: TreeScope = TreeScope { + database_id: DatabaseId::DEFAULT, + tenant_id: TENANT, + }; - fn cache_with_tree() -> Arc> { + fn cache_with_tree_in(scope: TreeScope) -> Arc> { let def = sonic_rs::from_str( r#"{"resource_column":"id","graph_index":"docs_tree","permission_table":"grants"}"#, ) .expect("tree def"); let mut cache = PermissionCache::new(); - cache.register_tree_def(TENANT, "docs", def); + cache.register_tree_def(scope.collection("docs"), def); cache.progress_mut().install_reload(&[0]); Arc::new(tokio::sync::RwLock::new(cache)) } + fn cache_with_tree() -> Arc> { + cache_with_tree_in(SCOPE) + } + fn grant_event(sequence: u64, op: WriteOp) -> WriteEvent { + grant_event_in(DatabaseId::DEFAULT, sequence, op) + } + + /// A grant row written to `grants` of `database_id`, as the Data Plane + /// emits it: the collection carries the database-qualified name. + fn grant_event_in(database_id: DatabaseId, sequence: u64, op: WriteOp) -> WriteEvent { let row = serde_json::json!({ "resource_id": "d1", "grantee": "role_a", @@ -176,12 +203,14 @@ mod tests { }; WriteEvent { sequence, - collection: Arc::from("grants"), + collection: Arc::from( + nodedb_types::QualifiedCollection::new(database_id, "grants").as_str(), + ), op, row_id: RowId::row(nodedb_types::RowIdentity::from_user_key("g1")), lsn: Lsn::new(sequence), record: None, - database_id: DatabaseId::DEFAULT, + database_id, tenant_id: TenantId::new(TENANT), vshard_id: VShardId::new(0), source: EventSource::User, @@ -191,6 +220,7 @@ mod tests { valid_time_ms: None, user_id: None, statement_digest: None, + commit_hlc: Some(crate::event::test_utils::test_commit_hlc()), } } @@ -215,14 +245,14 @@ mod tests { let guard = cache.read().await; assert_eq!( - guard.get_grant(TENANT, "d1", "role_a"), + guard.get_grant(SCOPE, "d1", "role_a"), Some(("viewer", false)) ); assert_eq!(guard.tenant_version(TENANT), version_before + 1); assert!(guard.progress().caught_up(&[1])); } - /// An event a reload already covers is not applied again: the ring may + /// An event a reload already covers is not applied again: the ring can /// have dropped a later write to the same row. #[tokio::test] async fn an_event_covered_by_a_reload_is_not_applied_again() { @@ -235,13 +265,13 @@ mod tests { cache .read() .await - .get_grant(TENANT, "d1", "role_a") + .get_grant(SCOPE, "d1", "role_a") .is_none() ); apply_ring_events(0, &[grant_event(3, WriteOp::Insert)], &cache, ¬ify).await; let guard = cache.read().await; - assert!(guard.get_grant(TENANT, "d1", "role_a").is_some()); + assert!(guard.get_grant(SCOPE, "d1", "role_a").is_some()); assert!(guard.progress().caught_up(&[3])); } @@ -262,8 +292,43 @@ mod tests { ) .await; let guard = cache.read().await; - assert!(guard.get_grant(TENANT, "d1", "role_a").is_none()); + assert!(guard.get_grant(SCOPE, "d1", "role_a").is_none()); assert!(!guard.progress().caught_up(&[3])); assert!(guard.progress().needs_reload_for(&[3])); } + + /// A grant written in a named database updates that database's tree + /// only, and the same table name in another database updates nothing. + #[tokio::test] + async fn a_grant_event_applies_to_its_own_database_only() { + let db1 = TreeScope::new(DatabaseId::new(7), TENANT); + let db2 = TreeScope::new(DatabaseId::new(8), TENANT); + let cache = cache_with_tree_in(db1); + let notify = tokio::sync::Notify::new(); + + apply_ring_events( + 0, + &[grant_event_in(db2.database_id, 1, WriteOp::Insert)], + &cache, + ¬ify, + ) + .await; + assert!(cache.read().await.get_grant(db1, "d1", "role_a").is_none()); + assert!(cache.read().await.get_grant(db2, "d1", "role_a").is_none()); + + apply_ring_events( + 0, + &[grant_event_in(db1.database_id, 2, WriteOp::Insert)], + &cache, + ¬ify, + ) + .await; + let guard = cache.read().await; + assert_eq!( + guard.get_grant(db1, "d1", "role_a"), + Some(("viewer", false)) + ); + assert!(guard.get_grant(db2, "d1", "role_a").is_none()); + assert!(guard.get_grant(SCOPE, "d1", "role_a").is_none()); + } } diff --git a/nodedb/src/control/security/permission_tree/invalidation.rs b/nodedb/src/control/security/permission_tree/invalidation.rs index a57c6c217..c333c7ac1 100644 --- a/nodedb/src/control/security/permission_tree/invalidation.rs +++ b/nodedb/src/control/security/permission_tree/invalidation.rs @@ -9,17 +9,19 @@ use tracing::debug; use super::cache::PermissionCache; +use super::scope::TreeScope; use super::types::PermissionGrant; /// Handle an INSERT/UPDATE on the permission grant table. /// /// Upserts the grant into the cache. The grant's `resource_id` and `grantee` /// determine the cache key; the `level` and `inherited` are the new values. -pub fn on_grant_upsert(cache: &mut PermissionCache, tenant_id: u64, grant: &PermissionGrant) { - cache.put_grant(tenant_id, grant); - cache.bump_tenant_version(tenant_id); +pub fn on_grant_upsert(cache: &mut PermissionCache, scope: TreeScope, grant: &PermissionGrant) { + cache.put_grant(scope, grant); + cache.bump_tenant_version(scope.tenant_id); debug!( - tenant_id, + tenant_id = scope.tenant_id, + database_id = scope.database_id.as_u64(), resource_id = %grant.resource_id, grantee = %grant.grantee, level = %grant.level, @@ -33,15 +35,18 @@ pub fn on_grant_upsert(cache: &mut PermissionCache, tenant_id: u64, grant: &Perm /// inherited permissions from ancestors. pub fn on_grant_delete( cache: &mut PermissionCache, - tenant_id: u64, + scope: TreeScope, resource_id: &str, grantee: &str, ) { - cache.remove_grant(tenant_id, resource_id, grantee); - cache.bump_tenant_version(tenant_id); + cache.remove_grant(scope, resource_id, grantee); + cache.bump_tenant_version(scope.tenant_id); debug!( - tenant_id, - resource_id, grantee, "permission_tree: grant deleted" + tenant_id = scope.tenant_id, + database_id = scope.database_id.as_u64(), + resource_id, + grantee, + "permission_tree: grant deleted" ); } @@ -51,17 +56,20 @@ pub fn on_grant_delete( /// and children map accordingly. pub fn on_edge_upsert( cache: &mut PermissionCache, - tenant_id: u64, + scope: TreeScope, child_id: &str, parent_id: &str, ) { // Remove old edge first (if child was previously under a different parent). - cache.remove_edge(tenant_id, child_id); - cache.put_edge(tenant_id, child_id, parent_id); - cache.bump_tenant_version(tenant_id); + cache.remove_edge(scope, child_id); + cache.put_edge(scope, child_id, parent_id); + cache.bump_tenant_version(scope.tenant_id); debug!( - tenant_id, - child_id, parent_id, "permission_tree: edge upserted" + tenant_id = scope.tenant_id, + database_id = scope.database_id.as_u64(), + child_id, + parent_id, + "permission_tree: edge upserted" ); } @@ -69,27 +77,33 @@ pub fn on_edge_upsert( /// /// The resource `child_id` is no longer under any parent (becomes a root or /// is being deleted entirely). -pub fn on_edge_delete(cache: &mut PermissionCache, tenant_id: u64, child_id: &str) { - cache.remove_edge(tenant_id, child_id); - cache.bump_tenant_version(tenant_id); - debug!(tenant_id, child_id, "permission_tree: edge deleted"); +pub fn on_edge_delete(cache: &mut PermissionCache, scope: TreeScope, child_id: &str) { + cache.remove_edge(scope, child_id); + cache.bump_tenant_version(scope.tenant_id); + debug!( + tenant_id = scope.tenant_id, + database_id = scope.database_id.as_u64(), + child_id, + "permission_tree: edge deleted" + ); } -/// Bulk reload all edges and grants for a tenant. +/// Bulk reload all edges and grants for a scope. /// /// Called on startup or when the cache is suspected to be stale. -/// Clears existing state for the tenant before loading. +/// Adds to the scope's existing state. pub fn full_reload( cache: &mut PermissionCache, - tenant_id: u64, + scope: TreeScope, edges: &[(String, String)], grants: &[PermissionGrant], ) { - cache.load_edges(tenant_id, edges); - cache.load_grants(tenant_id, grants); - cache.bump_tenant_version(tenant_id); + cache.load_edges(scope, edges); + cache.load_grants(scope, grants); + cache.bump_tenant_version(scope.tenant_id); debug!( - tenant_id, + tenant_id = scope.tenant_id, + database_id = scope.database_id.as_u64(), edges = edges.len(), grants = grants.len(), "permission_tree: full reload complete" @@ -99,6 +113,12 @@ pub fn full_reload( #[cfg(test)] mod tests { use super::*; + use crate::types::DatabaseId; + + const S1: TreeScope = TreeScope { + database_id: DatabaseId::DEFAULT, + tenant_id: 1, + }; #[test] fn every_mutator_bumps_the_tenant_version() { @@ -111,19 +131,19 @@ mod tests { level: "editor".into(), inherited: false, }; - on_grant_upsert(&mut cache, 1, &grant); + on_grant_upsert(&mut cache, S1, &grant); assert_eq!(cache.tenant_version(1), 1); - on_grant_delete(&mut cache, 1, "doc-1", "user-1"); + on_grant_delete(&mut cache, S1, "doc-1", "user-1"); assert_eq!(cache.tenant_version(1), 2); - on_edge_upsert(&mut cache, 1, "doc-1", "folder-a"); + on_edge_upsert(&mut cache, S1, "doc-1", "folder-a"); assert_eq!(cache.tenant_version(1), 3); - on_edge_delete(&mut cache, 1, "doc-1"); + on_edge_delete(&mut cache, S1, "doc-1"); assert_eq!(cache.tenant_version(1), 4); - full_reload(&mut cache, 1, &[], &[]); + full_reload(&mut cache, S1, &[], &[]); assert_eq!(cache.tenant_version(1), 5); // An unrelated tenant is never touched. @@ -140,28 +160,28 @@ mod tests { level: "editor".into(), inherited: false, }; - on_grant_upsert(&mut cache, 1, &grant); + on_grant_upsert(&mut cache, S1, &grant); assert_eq!( - cache.get_grant(1, "doc-1", "user-1").map(|(l, _)| l), + cache.get_grant(S1, "doc-1", "user-1").map(|(l, _)| l), Some("editor") ); - on_grant_delete(&mut cache, 1, "doc-1", "user-1"); - assert!(cache.get_grant(1, "doc-1", "user-1").is_none()); + on_grant_delete(&mut cache, S1, "doc-1", "user-1"); + assert!(cache.get_grant(S1, "doc-1", "user-1").is_none()); } #[test] fn edge_upsert_replaces_old_parent() { let mut cache = PermissionCache::new(); - on_edge_upsert(&mut cache, 1, "doc-1", "folder-a"); - assert_eq!(cache.get_parent(1, "doc-1"), Some("folder-a")); + on_edge_upsert(&mut cache, S1, "doc-1", "folder-a"); + assert_eq!(cache.get_parent(S1, "doc-1"), Some("folder-a")); // Move doc-1 to folder-b. - on_edge_upsert(&mut cache, 1, "doc-1", "folder-b"); - assert_eq!(cache.get_parent(1, "doc-1"), Some("folder-b")); - // Old parent should no longer have doc-1 as child. - assert!(cache.get_children(1, "folder-a").is_empty()); + on_edge_upsert(&mut cache, S1, "doc-1", "folder-b"); + assert_eq!(cache.get_parent(S1, "doc-1"), Some("folder-b")); + // Old parent no longer has doc-1 as child. + assert!(cache.get_children(S1, "folder-a").is_empty()); } #[test] @@ -179,12 +199,12 @@ mod tests { inherited: false, }]; - full_reload(&mut cache, 1, &edges, &grants); + full_reload(&mut cache, S1, &edges, &grants); - assert_eq!(cache.get_parent(1, "doc-1"), Some("folder-1")); - assert_eq!(cache.get_parent(1, "folder-1"), Some("workspace")); + assert_eq!(cache.get_parent(S1, "doc-1"), Some("folder-1")); + assert_eq!(cache.get_parent(S1, "folder-1"), Some("workspace")); assert_eq!( - cache.get_grant(1, "workspace", "user-1").map(|(l, _)| l), + cache.get_grant(S1, "workspace", "user-1").map(|(l, _)| l), Some("owner") ); } diff --git a/nodedb/src/control/security/permission_tree/mod.rs b/nodedb/src/control/security/permission_tree/mod.rs index 8baa47db6..dd13f84e4 100644 --- a/nodedb/src/control/security/permission_tree/mod.rs +++ b/nodedb/src/control/security/permission_tree/mod.rs @@ -5,10 +5,12 @@ pub mod event_handler; pub mod invalidation; pub mod reload; pub mod resolver; +pub mod scope; pub mod sources; pub mod sync_state; pub mod types; pub use cache::{PermissionCache, TreeSource, TreeSourceKind}; +pub use scope::{TreeKey, TreeScope}; pub use sources::SourceIndex; pub use types::PermissionTreeDef; diff --git a/nodedb/src/control/security/permission_tree/reload.rs b/nodedb/src/control/security/permission_tree/reload.rs index 8cbf152a1..e5dfea53e 100644 --- a/nodedb/src/control/security/permission_tree/reload.rs +++ b/nodedb/src/control/security/permission_tree/reload.rs @@ -13,21 +13,21 @@ //! node's own apply position. The scans dispatch through the read-only local //! path, never the write funnel. -use std::collections::HashMap; +use std::collections::{HashMap, HashSet}; -use nodedb_types::{QualifiedCollection, TenantId}; +use nodedb_types::TenantId; use crate::control::local_dispatch::{LocalRead, dispatch_local_read, reject_data_plane_error}; use crate::control::state::SharedState; -use crate::types::DatabaseId; use super::cache::{PermissionCache, TreeSource, TreeSourceKind}; use super::event_handler::{extract_edge, extract_grant}; +use super::scope::TreeScope; use super::types::PermissionGrant; -/// Edges and grants a reload read for one tenant. +/// Edges and grants a reload read for one scope. #[derive(Default)] -struct TenantRows { +struct ScopeRows { edges: Vec<(String, String)>, grants: Vec, } @@ -74,19 +74,20 @@ async fn reload_locked(state: &SharedState, cache: &mut PermissionCache) -> crat // counted, and none it applies later can be. let emitted = fence.emitted_snapshot().unwrap_or_default(); - let mut rows: HashMap = HashMap::new(); + let mut rows: HashMap = HashMap::new(); for source in cache.tree_sources() { let docs = scan_source(state, &source).await?; - let tenant = rows.entry(source.tenant_id).or_default(); + let scope = rows.entry(source.key.scope).or_default(); match source.kind { - TreeSourceKind::Hierarchy => tenant.edges.extend(docs.iter().filter_map(|doc| { + TreeSourceKind::Hierarchy => scope.edges.extend(docs.iter().filter_map(|doc| { extract_edge(doc).map(|(child, parent)| (child.to_owned(), parent.to_owned())) })), - TreeSourceKind::Grants => tenant.grants.extend(docs.iter().filter_map(extract_grant)), + TreeSourceKind::Grants => scope.grants.extend(docs.iter().filter_map(extract_grant)), } } - for (tenant_id, tenant) in rows { - cache.replace_tenant_state(tenant_id, &tenant.edges, &tenant.grants); + cache.retain_scopes(&rows.keys().copied().collect::>()); + for (scope, loaded) in rows { + cache.replace_scope_state(scope, &loaded.edges, &loaded.grants); } cache.progress_mut().install_reload(&emitted); fence.permission_applied().notify_waiters(); @@ -97,14 +98,15 @@ async fn reload_locked(state: &SharedState, cache: &mut PermissionCache) -> crat /// /// A source that no longer exists holds no rows. A source that is not a /// document collection cannot carry edges or grants, and refuses the reload: -/// planning with it would silently grant or deny nothing. +/// planning with it silently grants or denies nothing. async fn scan_source( state: &SharedState, source: &TreeSource, ) -> crate::Result> { - let database_id = DatabaseId::DEFAULT; + let key = &source.key; + let database_id = key.scope.database_id; let catalog = state.credentials.catalog(); - let stored = catalog.get_collection(database_id, source.tenant_id, &source.collection)?; + let stored = catalog.get_collection(database_id, key.scope.tenant_id, &key.collection)?; let Some(stored) = stored.filter(|collection| collection.is_active) else { return Ok(Vec::new()); }; @@ -113,7 +115,7 @@ async fn scan_source( detail: format!( "permission tree source '{}' is a {} collection; the hierarchy and the \ permission table must be document collections", - source.collection, stored.collection_type + key.collection, stored.collection_type ), }); } @@ -123,11 +125,11 @@ async fn scan_source( // path. let response = dispatch_local_read( state, - TenantId::new(source.tenant_id), + TenantId::new(key.scope.tenant_id), database_id, - nodedb_types::CollectionKey::from_bare(database_id, &source.collection).vshard(), + key.vshard(), LocalRead::DocumentScan { - collection: QualifiedCollection::new(database_id, &source.collection), + collection: key.qualified(), }, ) .await?; diff --git a/nodedb/src/control/security/permission_tree/resolver.rs b/nodedb/src/control/security/permission_tree/resolver.rs index b3153b6a7..7c0dae478 100644 --- a/nodedb/src/control/security/permission_tree/resolver.rs +++ b/nodedb/src/control/security/permission_tree/resolver.rs @@ -8,6 +8,7 @@ //! grant, access is denied (returns the lowest level, typically "none"). use super::cache::PermissionCache; +use super::scope::TreeScope; use super::types::PermissionTreeDef; /// Resolve the effective permission level for a user on a resource. @@ -19,7 +20,7 @@ use super::types::PermissionTreeDef; /// # Arguments /// * `cache` — In-memory permission cache (read-only reference). /// * `def` — Permission tree definition for the collection. -/// * `tenant_id` — Tenant isolation scope. +/// * `scope` — The database and tenant the tree lives in. /// * `user_id` — The user to check. /// * `user_roles` — Roles the user belongs to (also checked as grantees). /// * `resource_id` — The resource to check access on. @@ -29,7 +30,7 @@ use super::types::PermissionTreeDef; pub fn resolve_permission( cache: &PermissionCache, def: &PermissionTreeDef, - tenant_id: u64, + scope: TreeScope, user_id: &str, user_roles: &[String], resource_id: &str, @@ -46,19 +47,19 @@ pub fn resolve_permission( for _ in 0..max_depth { // Check for explicit grant at this node for the user. - if let Some((level, _inherited)) = cache.get_grant(tenant_id, ¤t, user_id) { + if let Some((level, _inherited)) = cache.get_grant(scope, ¤t, user_id) { return level.to_owned(); } // Check for grants via roles. for role in user_roles { - if let Some((level, _inherited)) = cache.get_grant(tenant_id, ¤t, role) { + if let Some((level, _inherited)) = cache.get_grant(scope, ¤t, role) { return level.to_owned(); } } // Walk to parent. - match cache.get_parent(tenant_id, ¤t) { + match cache.get_parent(scope, ¤t) { Some(parent) => current = parent.to_owned(), None => break, // Reached root. } @@ -69,7 +70,7 @@ pub fn resolve_permission( /// Find all resource IDs that a user can access at or above a required level. /// -/// Iterates all known resources for the tenant and resolves each one. +/// Iterates all known resources of the scope and resolves each one. /// Returns the set of resource IDs where the effective permission meets /// the required threshold. /// @@ -83,7 +84,7 @@ pub fn resolve_permission( pub fn accessible_resources( cache: &PermissionCache, def: &PermissionTreeDef, - tenant_id: u64, + scope: TreeScope, user_id: &str, user_roles: &[String], required_level: &str, @@ -93,11 +94,11 @@ pub fn accessible_resources( None => return Vec::new(), // Unknown level = deny all. }; - let all_ids = cache.all_resource_ids(tenant_id); + let all_ids = cache.all_resource_ids(scope); let mut accessible = Vec::new(); for rid in all_ids { - let effective = resolve_permission(cache, def, tenant_id, user_id, user_roles, rid); + let effective = resolve_permission(cache, def, scope, user_id, user_roles, rid); if let Some(effective_ordinal) = def.level_ordinal(&effective) && effective_ordinal >= required_ordinal { @@ -114,13 +115,13 @@ pub fn accessible_resources( pub fn check_permission( cache: &PermissionCache, def: &PermissionTreeDef, - tenant_id: u64, + scope: TreeScope, user_id: &str, user_roles: &[String], resource_id: &str, required_level: &str, ) -> bool { - let effective = resolve_permission(cache, def, tenant_id, user_id, user_roles, resource_id); + let effective = resolve_permission(cache, def, scope, user_id, user_roles, resource_id); def.level_meets_requirement(&effective, required_level) } @@ -128,6 +129,12 @@ pub fn check_permission( mod tests { use super::*; use crate::control::security::permission_tree::types::PermissionGrant; + use crate::types::DatabaseId; + + const S1: TreeScope = TreeScope { + database_id: DatabaseId::DEFAULT, + tenant_id: 1, + }; fn setup() -> (PermissionCache, PermissionTreeDef) { let mut cache = PermissionCache::new(); @@ -148,12 +155,12 @@ mod tests { }; // Hierarchy: workspace → folder-design → doc-mockup - cache.put_edge(1, "folder-design", "workspace-acme"); - cache.put_edge(1, "doc-mockup", "folder-design"); + cache.put_edge(S1, "folder-design", "workspace-acme"); + cache.put_edge(S1, "doc-mockup", "folder-design"); // Grant: user-42 is editor on folder-design. cache.put_grant( - 1, + S1, &PermissionGrant { resource_id: "folder-design".into(), grantee: "user-42".into(), @@ -164,7 +171,7 @@ mod tests { // Grant: user-99 is viewer on workspace-acme. cache.put_grant( - 1, + S1, &PermissionGrant { resource_id: "workspace-acme".into(), grantee: "user-99".into(), @@ -181,7 +188,7 @@ mod tests { let (cache, def) = setup(); // user-42 has editor on folder-design → doc-mockup inherits editor. - let level = resolve_permission(&cache, &def, 1, "user-42", &[], "doc-mockup"); + let level = resolve_permission(&cache, &def, S1, "user-42", &[], "doc-mockup"); assert_eq!(level, "editor"); } @@ -190,7 +197,7 @@ mod tests { let (cache, def) = setup(); // user-42 has direct editor on folder-design. - let level = resolve_permission(&cache, &def, 1, "user-42", &[], "folder-design"); + let level = resolve_permission(&cache, &def, S1, "user-42", &[], "folder-design"); assert_eq!(level, "editor"); } @@ -199,7 +206,7 @@ mod tests { let (cache, def) = setup(); // user-unknown has no grants anywhere. - let level = resolve_permission(&cache, &def, 1, "user-unknown", &[], "doc-mockup"); + let level = resolve_permission(&cache, &def, S1, "user-unknown", &[], "doc-mockup"); assert_eq!(level, "none"); } @@ -209,7 +216,7 @@ mod tests { // Grant role "design-team" editor on folder-design. cache.put_grant( - 1, + S1, &PermissionGrant { resource_id: "folder-design".into(), grantee: "design-team".into(), @@ -222,7 +229,7 @@ mod tests { let level = resolve_permission( &cache, &def, - 1, + S1, "user-77", &["design-team".into()], "doc-mockup", @@ -235,7 +242,7 @@ mod tests { let (cache, def) = setup(); // user-99 has viewer on workspace → inherits to folder and doc. - let level = resolve_permission(&cache, &def, 1, "user-99", &[], "doc-mockup"); + let level = resolve_permission(&cache, &def, S1, "user-99", &[], "doc-mockup"); assert_eq!(level, "viewer"); } @@ -245,7 +252,7 @@ mod tests { // user-42 can access folder-design (editor) and doc-mockup (inherited editor). // user-42 cannot access workspace-acme (no grant). - let accessible = accessible_resources(&cache, &def, 1, "user-42", &[], "viewer"); + let accessible = accessible_resources(&cache, &def, S1, "user-42", &[], "viewer"); assert!(accessible.contains(&"folder-design".to_owned())); assert!(accessible.contains(&"doc-mockup".to_owned())); assert!(!accessible.contains(&"workspace-acme".to_owned())); @@ -259,7 +266,7 @@ mod tests { assert!(check_permission( &cache, &def, - 1, + S1, "user-42", &[], "doc-mockup", @@ -268,7 +275,7 @@ mod tests { assert!(check_permission( &cache, &def, - 1, + S1, "user-42", &[], "doc-mockup", @@ -277,7 +284,7 @@ mod tests { assert!(!check_permission( &cache, &def, - 1, + S1, "user-42", &[], "doc-mockup", @@ -291,7 +298,7 @@ mod tests { // Override: user-99 is viewer at workspace, but editor at doc-mockup. cache.put_grant( - 1, + S1, &PermissionGrant { resource_id: "doc-mockup".into(), grantee: "user-99".into(), @@ -301,7 +308,23 @@ mod tests { ); // Direct grant at doc-mockup takes precedence (short-circuits before walking up). - let level = resolve_permission(&cache, &def, 1, "user-99", &[], "doc-mockup"); + let level = resolve_permission(&cache, &def, S1, "user-99", &[], "doc-mockup"); assert_eq!(level, "editor"); } + + /// A tree and its grants in one database resolve nothing in another + /// database of the same tenant. + #[test] + fn grants_in_one_database_do_not_resolve_in_another() { + let (cache, def) = setup(); + let other = TreeScope::new(DatabaseId::new(9), 1); + + let level = resolve_permission(&cache, &def, other, "user-42", &[], "doc-mockup"); + assert_eq!(level, "none"); + assert!(accessible_resources(&cache, &def, other, "user-42", &[], "viewer").is_empty()); + assert_eq!( + resolve_permission(&cache, &def, S1, "user-42", &[], "doc-mockup"), + "editor" + ); + } } diff --git a/nodedb/src/control/security/permission_tree/scope.rs b/nodedb/src/control/security/permission_tree/scope.rs new file mode 100644 index 000000000..9ff033eee --- /dev/null +++ b/nodedb/src/control/security/permission_tree/scope.rs @@ -0,0 +1,97 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Keys of the permission-tree state. +//! +//! A tree governs one collection of one tenant in one database. Its +//! permission table and hierarchy rows live in that same database, so the +//! resource hierarchy and the grants are held per `(database, tenant)`. + +use nodedb_types::id::VShardId; +use nodedb_types::{CollectionKey, DatabaseId, QualifiedCollection}; + +/// The resource hierarchy and grants of one tenant in one database. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub struct TreeScope { + pub database_id: DatabaseId, + pub tenant_id: u64, +} + +impl TreeScope { + pub fn new(database_id: DatabaseId, tenant_id: u64) -> Self { + Self { + database_id, + tenant_id, + } + } + + /// The key of `collection`, a bare catalog name in this scope. + pub fn collection(self, collection: impl Into) -> TreeKey { + TreeKey { + scope: self, + collection: collection.into(), + } + } +} + +/// One collection in a [`TreeScope`], by its bare catalog name. +#[derive(Debug, Clone, PartialEq, Eq, Hash)] +pub struct TreeKey { + pub scope: TreeScope, + pub collection: String, +} + +impl TreeKey { + pub fn new(database_id: DatabaseId, tenant_id: u64, collection: impl Into) -> Self { + TreeScope::new(database_id, tenant_id).collection(collection) + } + + /// The key of `qualified`, a database-qualified name a plan or a write + /// event carries for `database_id`. + pub fn from_qualified( + database_id: DatabaseId, + tenant_id: u64, + qualified: &str, + ) -> crate::Result { + let key = CollectionKey::from_qualified_str(database_id, qualified)?; + Ok(Self::new(database_id, tenant_id, key.name())) + } + + /// The name storage engines and plans key this collection by. + pub fn qualified(&self) -> QualifiedCollection { + QualifiedCollection::new(self.scope.database_id, &self.collection) + } + + /// The vShard this collection homes to. + pub fn vshard(&self) -> VShardId { + CollectionKey::from_bare(self.scope.database_id, &self.collection).vshard() + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_qualified_name_resolves_to_its_database_scope() { + let key = TreeKey::from_qualified(DatabaseId::new(7), 1, "7/docs").expect("qualified"); + assert_eq!(key, TreeKey::new(DatabaseId::new(7), 1, "docs")); + assert_eq!(key.qualified().as_str(), "7/docs"); + + let key = TreeKey::from_qualified(DatabaseId::DEFAULT, 1, "docs").expect("bare"); + assert_eq!(key, TreeKey::new(DatabaseId::DEFAULT, 1, "docs")); + } + + #[test] + fn a_name_qualified_for_another_database_is_rejected() { + assert!(TreeKey::from_qualified(DatabaseId::new(7), 1, "8/docs").is_err()); + assert!(TreeKey::from_qualified(DatabaseId::new(7), 1, "docs").is_err()); + } + + #[test] + fn the_same_name_in_two_databases_is_two_keys() { + let a = TreeKey::new(DatabaseId::new(7), 1, "docs"); + let b = TreeKey::new(DatabaseId::new(8), 1, "docs"); + assert_ne!(a, b); + assert_ne!(a.qualified(), b.qualified()); + } +} diff --git a/nodedb/src/control/security/permission_tree/sources.rs b/nodedb/src/control/security/permission_tree/sources.rs index ef27af7f1..4223d36ba 100644 --- a/nodedb/src/control/security/permission_tree/sources.rs +++ b/nodedb/src/control/security/permission_tree/sources.rs @@ -15,17 +15,17 @@ //! before the cache takes it from the queue. //! //! A write therefore counts as an authorization change from the moment its -//! tree definition committed on this node. A removed tree may count a little +//! tree definition committed on this node. A removed tree can count a little //! longer, until both sets drop it, which only adds a barrier. //! -//! Tree sources live in the default database, as the permission-tree DDL -//! writes them, so each source collection homes on one vShard. +//! A source lives in its tree's database. Collections are held by the +//! database-qualified name plans carry, so the same name in two databases is +//! two sources, each homing on its own vShard. use std::collections::{HashMap, HashSet}; use std::sync::RwLock; -use crate::types::DatabaseId; - +use super::scope::TreeKey; use super::types::PermissionTreeDef; #[derive(Debug, Default)] @@ -35,18 +35,14 @@ struct SourceSet { } impl SourceSet { - fn from_defs<'a>( - defs: impl IntoIterator, - ) -> Self { + fn from_defs<'a>(defs: impl IntoIterator) -> Self { let mut set = Self::default(); - for ((_, governed), def) in defs { - for collection in [governed.as_str(), def.permission_table.as_str()] { - set.collections.insert(collection.to_owned()); - set.vshards.insert( - nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, collection) - .vshard() - .as_u32(), - ); + for (governed, def) in defs { + let table = governed.scope.collection(def.permission_table.clone()); + for source in [governed, &table] { + set.collections + .insert(source.qualified().as_str().to_owned()); + set.vshards.insert(source.vshard().as_u32()); } } set @@ -55,7 +51,7 @@ impl SourceSet { #[derive(Debug, Default)] struct Committed { - defs: HashMap<(u64, String), PermissionTreeDef>, + defs: HashMap, set: SourceSet, } @@ -68,26 +64,20 @@ pub struct SourceIndex { impl SourceIndex { /// Rebuild the cached set from the cache's tree definitions. - pub(super) fn rebuild(&self, tree_defs: &HashMap<(u64, String), PermissionTreeDef>) { + pub(super) fn rebuild(&self, tree_defs: &HashMap) { *self.cached.write().unwrap_or_else(|p| p.into_inner()) = SourceSet::from_defs(tree_defs); } /// Record a tree definition the metadata applier committed. `None` - /// removes the tree of `(tenant_id, collection)`. - pub fn note_committed( - &self, - tenant_id: u64, - collection: &str, - def: Option<&PermissionTreeDef>, - ) { + /// removes the tree of `key`. + pub fn note_committed(&self, key: &TreeKey, def: Option<&PermissionTreeDef>) { let mut committed = self.committed.write().unwrap_or_else(|p| p.into_inner()); - let key = (tenant_id, collection.to_owned()); match def { Some(def) => { - committed.defs.insert(key, def.clone()); + committed.defs.insert(key.clone(), def.clone()); } None => { - committed.defs.remove(&key); + committed.defs.remove(key); } } committed.set = SourceSet::from_defs(&committed.defs); @@ -103,7 +93,8 @@ impl SourceIndex { !self.any(|set| !set.collections.is_empty()) } - /// Whether `collection` feeds a tree. + /// Whether `collection`, a database-qualified name as a plan carries it, + /// feeds a tree. pub fn is_source_collection(&self, collection: &str) -> bool { self.any(|set| set.collections.contains(collection)) } @@ -137,22 +128,26 @@ impl SourceIndex { #[cfg(test)] mod tests { use super::*; + use crate::types::DatabaseId; - #[test] - fn the_index_names_governed_collections_and_permission_tables() { - let def: PermissionTreeDef = sonic_rs::from_str( + fn def() -> PermissionTreeDef { + sonic_rs::from_str( r#"{"resource_column":"id","graph_index":"tree","permission_table":"grants"}"#, ) - .expect("tree def"); + .expect("tree def") + } + + #[test] + fn the_index_names_governed_collections_and_permission_tables() { let mut defs = HashMap::new(); - defs.insert((1, "docs".to_owned()), def); + defs.insert(TreeKey::new(DatabaseId::DEFAULT, 1, "docs"), def()); let index = SourceIndex::default(); assert!(index.is_empty()); index.rebuild(&defs); assert!(index.is_source_collection("docs")); assert!(index.is_source_collection("grants")); assert!(!index.is_source_collection("other")); - let grants_vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, "grants") + let grants_vshard = TreeKey::new(DatabaseId::DEFAULT, 1, "grants") .vshard() .as_u32(); assert!(index.is_source_vshard(grants_vshard)); @@ -162,17 +157,32 @@ mod tests { assert!(!index.is_source_vshard(grants_vshard)); } + /// A tree in a named database names its sources by their qualified + /// names and their own vShards, never the default database's. + #[test] + fn a_named_database_tree_names_its_own_sources() { + let db = DatabaseId::new(7); + let mut defs = HashMap::new(); + defs.insert(TreeKey::new(db, 1, "docs"), def()); + let index = SourceIndex::default(); + index.rebuild(&defs); + assert!(index.is_source_collection("7/docs")); + assert!(index.is_source_collection("7/grants")); + assert!(!index.is_source_collection("docs")); + assert!(!index.is_source_collection("grants")); + let vshards = index.source_vshards(); + assert!(vshards.contains(&TreeKey::new(db, 1, "grants").vshard().as_u32())); + assert!(vshards.contains(&TreeKey::new(db, 1, "docs").vshard().as_u32())); + } + #[test] fn a_committed_tree_counts_before_the_cache_takes_it() { - let def: PermissionTreeDef = sonic_rs::from_str( - r#"{"resource_column":"id","graph_index":"tree","permission_table":"grants"}"#, - ) - .expect("tree def"); + let key = TreeKey::new(DatabaseId::DEFAULT, 1, "docs"); let index = SourceIndex::default(); - index.note_committed(1, "docs", Some(&def)); + index.note_committed(&key, Some(&def())); assert!(index.is_source_collection("grants")); assert!(index.is_source_collection("docs")); - index.note_committed(1, "docs", None); + index.note_committed(&key, None); assert!(index.is_empty()); } } diff --git a/nodedb/src/control/security/random.rs b/nodedb/src/control/security/random.rs index d29119b35..d4b684f87 100644 --- a/nodedb/src/control/security/random.rs +++ b/nodedb/src/control/security/random.rs @@ -13,7 +13,6 @@ //! and enable enumeration. use argon2::password_hash::rand_core::{OsRng, RngCore}; -use std::fmt::Write; /// Size in bytes of the random payload (128 bits). const RANDOM_BYTES: usize = 16; @@ -28,12 +27,7 @@ const RANDOM_BYTES: usize = 16; pub fn generate_tagged_random_hex(prefix: &str) -> String { let mut bytes = [0u8; RANDOM_BYTES]; OsRng.fill_bytes(&mut bytes); - let mut s = String::with_capacity(prefix.len() + RANDOM_BYTES * 2); - s.push_str(prefix); - for b in bytes { - let _ = write!(s, "{b:02x}"); - } - s + format!("{prefix}{}", hex::encode(bytes)) } #[cfg(test)] @@ -42,7 +36,7 @@ mod tests { /// The identifier must not embed the wall-clock second. An attacker that /// knows roughly when it was issued (from HTTP `Date:` headers, TLS - /// handshake timestamps, or response timing) would otherwise recover a + /// handshake timestamps, or response timing) otherwise recovers a /// timestamp component directly. #[test] fn does_not_leak_wall_clock_second() { @@ -114,7 +108,7 @@ mod tests { /// A batch of identifiers must all be distinct AND must not share any /// common prefix beyond the caller-chosen tag — a shared runtime prefix - /// would indicate a deterministic (timestamp/counter) component. + /// indicates a deterministic (timestamp/counter) component. #[test] fn batch_ids_have_no_shared_deterministic_prefix() { let prefix = "tag_"; diff --git a/nodedb/src/control/security/scope/expiry.rs b/nodedb/src/control/security/scope/expiry.rs index f59cb1bd6..3370deda3 100644 --- a/nodedb/src/control/security/scope/expiry.rs +++ b/nodedb/src/control/security/scope/expiry.rs @@ -20,12 +20,11 @@ use super::grant::{ScopeGrantParams, ScopeStatus}; /// Spawn the periodic scope-grant expiry sweep. /// /// The sweep must run exactly once cluster-wide — each pass proposes catalog -/// mutations, so every node running it would duplicate them — but it must -/// still run on a standalone node, which has no metadata group and therefore -/// no leader; [`SharedState::is_singleton_worker`] covers both. +/// mutations, so every node running it duplicates them. The metadata +/// leader runs it ([`SharedState::is_singleton_worker`]). A one-node cluster +/// leads its own metadata group. /// -/// The pass itself is synchronous and writes redb, so it runs on a blocking -/// thread rather than the reactor. +/// The pass awaits each replicated `ON EXPIRE` action on the loop's task. pub fn spawn_expiry_task(shared: Arc, interval_secs: u64) { // Below ~10s the sweep costs more than the resolution it buys: expiry is // already enforced on every read by `ScopeGrant::is_effective`, and this @@ -54,14 +53,7 @@ pub fn spawn_expiry_task(shared: Arc, interval_secs: u64) { if !loop_shared.is_singleton_worker() { continue; } - let state_for_sweep = Arc::clone(&loop_shared); - let result = tokio::task::spawn_blocking(move || { - process_expired_grants(&state_for_sweep); - }) - .await; - if let Err(e) = result { - warn!(error = %e, "scope expiry sweep task panicked"); - } + process_expired_grants(&loop_shared).await; } }, ); @@ -78,9 +70,9 @@ pub struct ScopeEvent { } /// Process all expired and grace-period grants. Returns emitted events. -pub fn process_expired_grants_with_events(state: &SharedState) -> Vec { +pub async fn process_expired_grants_with_events(state: &SharedState) -> Vec { let mut events = Vec::new(); - process_expired_grants_inner(state, &mut events); + process_expired_grants_inner(state, &mut events).await; events } @@ -89,9 +81,9 @@ pub fn process_expired_grants_with_events(state: &SharedState) -> Vec) { +async fn process_expired_grants_inner(state: &SharedState, events: &mut Vec) { let all_grants = state.scope_grants.list(None); let mut expired_count = 0u32; let mut grace_count = 0u32; @@ -141,7 +133,7 @@ fn process_expired_grants_inner(state: &SharedState, events: &mut Vec crate::Result<()> { +async fn execute_on_expire( + state: &SharedState, + grant: &super::grant::ScopeGrant, +) -> crate::Result<()> { let action = &grant.on_expire_action; if action.is_empty() { - // No action configured — just let it stay expired. + // No action configured — let it stay expired. // The grant is already filtered out of effective_scopes(). return Ok(()); } @@ -195,7 +190,8 @@ fn execute_on_expire(state: &SharedState, grant: &super::grant::ScopeGrant) -> c &grant.scope_name, &grant.grantee_type, &grant.grantee_id, - )?; + ) + .await?; info!( scope = %grant.scope_name, grantee = %grant.grantee_id, @@ -223,13 +219,14 @@ fn execute_on_expire(state: &SharedState, grant: &super::grant::ScopeGrant) -> c // idempotent upserts, so a failure here leaves the expired (and // therefore ineffective) original in place for the next sweep to // retry — rather than dropping the grantee to no scope at all. - propose_grant(state, &stored)?; + propose_grant(state, &stored).await?; propose_revoke( state, &grant.scope_name, &grant.grantee_type, &grant.grantee_id, - )?; + ) + .await?; info!( old_scope = %grant.scope_name, new_scope = %downgrade_scope, @@ -243,9 +240,8 @@ fn execute_on_expire(state: &SharedState, grant: &super::grant::ScopeGrant) -> c #[cfg(test)] mod tests { - use std::sync::Arc; - use super::*; + use crate::control::cluster::test_one_node; fn now_secs() -> u64 { std::time::SystemTime::now() @@ -254,20 +250,15 @@ mod tests { .as_secs() } - fn test_state(dir: &tempfile::TempDir) -> Arc { - let (_, _, state, _, _) = crate::event::test_utils::event_test_deps(dir); - state - } - /// Install a grant the way a `GRANT SCOPE` statement does — through the /// replicated propose path, so the catalog row exists too and the tests /// can assert on durable state rather than only the in-memory map. - fn install(state: &SharedState, params: ScopeGrantParams<'_>) { + async fn install(state: &SharedState, params: ScopeGrantParams<'_>) { let stored = state .scope_grants .prepare_grant(params) .expect("prepare grant"); - propose_grant(state, &stored).expect("propose grant"); + propose_grant(state, &stored).await.expect("propose grant"); } fn catalog_scopes(state: &SharedState) -> Vec { @@ -281,13 +272,13 @@ mod tests { .collect() } - #[tokio::test] + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn expired_grant_with_revoke_all() { - let dir = tempfile::tempdir().expect("tempdir"); - let state = test_state(&dir); + let cluster = test_one_node::boot().await; + let state = &cluster.state; let past = now_secs() - 100; install( - &state, + state, ScopeGrantParams { scope_name: "pro:all", grantee_type: "org", @@ -298,7 +289,8 @@ mod tests { on_expire_action: "revoke_all", conditions: Vec::new(), }, - ); + ) + .await; // Grant exists but is expired. assert!( @@ -307,23 +299,24 @@ mod tests { .has_scope("u1", &["acme".into()], "pro:all") ); - // Process expiry — should revoke. - process_expired_grants(&state); + // Process expiry — must revoke. + process_expired_grants(state).await; - // Grant should be gone. + // Grant is gone. assert_eq!(state.scope_grants.count(), 0); + cluster.shutdown().await; } - /// The whole point of routing the action through the propose path: a - /// revoke that only cleared the in-memory map would come back at the next + /// Why the action goes through the propose path: a + /// revoke that only cleared the in-memory map comes back at the next /// restart, re-granting an expired scope. - #[tokio::test] + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn expiry_removes_the_durable_grant() { - let dir = tempfile::tempdir().expect("tempdir"); - let state = test_state(&dir); + let cluster = test_one_node::boot().await; + let state = &cluster.state; let past = now_secs() - 100; install( - &state, + state, ScopeGrantParams { scope_name: "pro:all", grantee_type: "org", @@ -334,24 +327,26 @@ mod tests { on_expire_action: "revoke_all", conditions: Vec::new(), }, - ); - assert_eq!(catalog_scopes(&state), vec!["pro:all".to_string()]); + ) + .await; + assert_eq!(catalog_scopes(state), vec!["pro:all".to_string()]); - process_expired_grants(&state); + process_expired_grants(state).await; assert!( - catalog_scopes(&state).is_empty(), + catalog_scopes(state).is_empty(), "expired grant survived in the catalog" ); + cluster.shutdown().await; } - #[tokio::test] + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn expired_grant_with_downgrade() { - let dir = tempfile::tempdir().expect("tempdir"); - let state = test_state(&dir); + let cluster = test_one_node::boot().await; + let state = &cluster.state; let past = now_secs() - 100; install( - &state, + state, ScopeGrantParams { scope_name: "pro:all", grantee_type: "org", @@ -362,11 +357,12 @@ mod tests { on_expire_action: "grant:free:basic", conditions: Vec::new(), }, - ); + ) + .await; - process_expired_grants(&state); + process_expired_grants(state).await; - // pro:all should be gone, free:basic should exist. + // pro:all is gone, free:basic exists. assert!( !state .scope_grants @@ -377,18 +373,19 @@ mod tests { .scope_grants .has_scope("u1", &["acme".into()], "free:basic") ); - // …and the swap is durable, not just in memory. - assert_eq!(catalog_scopes(&state), vec!["free:basic".to_string()]); + // …and the swap is durable, not only in memory. + assert_eq!(catalog_scopes(state), vec!["free:basic".to_string()]); + cluster.shutdown().await; } - #[tokio::test] + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn grace_period_still_effective() { - let dir = tempfile::tempdir().expect("tempdir"); - let state = test_state(&dir); + let cluster = test_one_node::boot().await; + let state = &cluster.state; // Expired 10s ago but grace is 60s. let past = now_secs() - 10; install( - &state, + state, ScopeGrantParams { scope_name: "pro:all", grantee_type: "org", @@ -399,7 +396,8 @@ mod tests { on_expire_action: "revoke_all", conditions: Vec::new(), }, - ); + ) + .await; // In grace period — still effective. assert!( @@ -408,9 +406,10 @@ mod tests { .has_scope("u1", &["acme".into()], "pro:all") ); - // Process expiry — should NOT revoke (still in grace). - process_expired_grants(&state); + // Process expiry — does NOT revoke (still in grace). + process_expired_grants(state).await; assert_eq!(state.scope_grants.count(), 1); // Still there. - assert_eq!(catalog_scopes(&state), vec!["pro:all".to_string()]); + assert_eq!(catalog_scopes(state), vec!["pro:all".to_string()]); + cluster.shutdown().await; } } diff --git a/nodedb/src/control/security/scope/grant/replication.rs b/nodedb/src/control/security/scope/grant/replication.rs index 3ab99b066..c8f22a899 100644 --- a/nodedb/src/control/security/scope/grant/replication.rs +++ b/nodedb/src/control/security/scope/grant/replication.rs @@ -15,7 +15,7 @@ //! so there is exactly one place that knows how a grant reaches durable state. use crate::control::catalog_entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::catalog::StoredScopeGrant; use crate::control::security::time::now_secs; use crate::control::state::SharedState; @@ -27,22 +27,19 @@ use super::types::{ScopeGrant, ScopeGrantParams, grant_key}; /// the same upsert with a later expiry, and the automatic downgrade the expiry /// sweep issues). /// -/// A grant that only reached the node that decided on it would authorize there +/// A grant that only reached the node that decided on it authorizes there /// and nowhere else, so the catalog write is the applier's job on every node. -/// The `LocalOnly` branch is the standalone-origin path, where there is -/// no raft group to apply the entry. -pub(crate) fn propose_grant(state: &SharedState, stored: &StoredScopeGrant) -> crate::Result<()> { +pub(crate) async fn propose_grant( + state: &SharedState, + stored: &StoredScopeGrant, +) -> crate::Result<()> { let entry = CatalogEntry::PutScopeGrant(Box::new(stored.clone())); - let outcome = propose_catalog_entry(state, &entry)?; - if outcome.needs_local_apply() { - state.credentials.catalog().put_scope_grant(stored)?; - state.scope_grants.install_replicated_grant(stored); - } + propose_catalog_entry_async(state, &entry).await?; Ok(()) } -/// Replicate a scope-grant removal. Same dual path as [`propose_grant`]. -pub(crate) fn propose_revoke( +/// Replicate a scope-grant removal, as [`propose_grant`] does an upsert. +pub(crate) async fn propose_revoke( state: &SharedState, scope_name: &str, grantee_type: &str, @@ -53,16 +50,7 @@ pub(crate) fn propose_revoke( grantee_type: grantee_type.to_string(), grantee_id: grantee_id.to_string(), }; - let outcome = propose_catalog_entry(state, &entry)?; - if outcome.needs_local_apply() { - state - .credentials - .catalog() - .delete_scope_grant(scope_name, grantee_type, grantee_id)?; - state - .scope_grants - .install_replicated_revoke(scope_name, grantee_type, grantee_id); - } + propose_catalog_entry_async(state, &entry).await?; Ok(()) } @@ -258,7 +246,7 @@ mod tests { } /// `prepare_grant` is the proposal builder: it must not make the grant - /// visible, or a failed propose would leave the proposing node + /// visible, or a failed propose leaves the proposing node /// authorizing on a grant no other node has. #[test] fn prepare_grant_does_not_install() { diff --git a/nodedb/src/control/security/scope/grant/store.rs b/nodedb/src/control/security/scope/grant/store.rs index ba7bea511..0f5798487 100644 --- a/nodedb/src/control/security/scope/grant/store.rs +++ b/nodedb/src/control/security/scope/grant/store.rs @@ -8,8 +8,8 @@ //! The store is deliberately read-only with respect to durable state: it is //! seeded once from the catalog by [`ScopeGrantStore::open`] and thereafter //! mutated only by the replicated installers in [`super::replication`]. A -//! scope grant that reached redb without passing through raft would authorize -//! on one node and be invisible on every other, so this store offers no way to +//! scope grant that reached redb without passing through raft authorizes +//! on one node and is invisible on every other, so this store offers no way to //! write one. use std::collections::{HashMap, HashSet}; @@ -37,34 +37,18 @@ impl ScopeGrantStore { /// Seed the in-memory map from the catalog at startup. pub fn open(catalog: &SystemCatalog) -> crate::Result { - let stored = catalog.load_all_scope_grants()?; - let mut grants = HashMap::with_capacity(stored.len()); - for s in &stored { - // A grant whose stored conditions cannot be decoded is dropped, - // not loaded unconditionally: an unreadable restriction has to - // deny, never widen. - match ScopeGrant::from_stored(s) { - Ok(grant) => { - let key = grant_key(&s.scope_name, &s.grantee_type, &s.grantee_id); - grants.insert(key, grant); - } - Err(e) => warn!( - scope = %s.scope_name, - grantee_type = %s.grantee_type, - grantee_id = %s.grantee_id, - error = %e, - "scope grant dropped at load: conditions could not be decoded" - ), - } - } - if !grants.is_empty() { - info!(count = grants.len(), "scope grants loaded from catalog"); - } Ok(Self { - grants: RwLock::new(grants), + grants: RwLock::new(load_grants(catalog)?), }) } + /// Replace the in-memory map with the catalog's grants. + pub fn reload_from_catalog(&self, catalog: &SystemCatalog) -> crate::Result<()> { + let grants = load_grants(catalog)?; + *self.grants.write().unwrap_or_else(|p| p.into_inner()) = grants; + Ok(()) + } + /// Get all effective scope names granted to a specific grantee. /// Filters out expired grants. pub fn scopes_for(&self, grantee_type: &str, grantee_id: &str) -> Vec { @@ -172,3 +156,31 @@ impl Default for ScopeGrantStore { Self::new() } } + +/// Every decodable grant in `catalog`, keyed by grant key. +fn load_grants(catalog: &SystemCatalog) -> crate::Result> { + let stored = catalog.load_all_scope_grants()?; + let mut grants = HashMap::with_capacity(stored.len()); + for s in &stored { + // A grant whose stored conditions cannot be decoded is dropped, + // not loaded unconditionally: an unreadable restriction has to + // deny, never widen. + match ScopeGrant::from_stored(s) { + Ok(grant) => { + let key = grant_key(&s.scope_name, &s.grantee_type, &s.grantee_id); + grants.insert(key, grant); + } + Err(e) => warn!( + scope = %s.scope_name, + grantee_type = %s.grantee_type, + grantee_id = %s.grantee_id, + error = %e, + "scope grant dropped at load: conditions could not be decoded" + ), + } + } + if !grants.is_empty() { + info!(count = grants.len(), "scope grants loaded from catalog"); + } + Ok(grants) +} diff --git a/nodedb/src/control/security/siem/exporter.rs b/nodedb/src/control/security/siem/exporter.rs index ca936f7dd..10a83c82d 100644 --- a/nodedb/src/control/security/siem/exporter.rs +++ b/nodedb/src/control/security/siem/exporter.rs @@ -257,12 +257,7 @@ pub(super) fn compute_hmac(secret: &str, message: &str) -> String { return String::new(); }; mac.update(message.as_bytes()); - let result = mac.finalize(); - result - .into_bytes() - .iter() - .map(|b| format!("{b:02x}")) - .collect() + hex::encode(mac.finalize().into_bytes()) } #[cfg(test)] diff --git a/nodedb/src/control/sequence/ddl_overlay.rs b/nodedb/src/control/sequence/ddl_overlay.rs index bbea0afa2..b897bae07 100644 --- a/nodedb/src/control/sequence/ddl_overlay.rs +++ b/nodedb/src/control/sequence/ddl_overlay.rs @@ -4,8 +4,8 @@ //! [`super::registry::SequenceRegistry`] map. //! //! `CREATE SEQUENCE` inside an open transaction is buffered and only reaches -//! the shared registry at COMMIT (`post_apply` runs only for -//! `ProposeOutcome::needs_local_apply()`, which `Buffered` never satisfies). +//! the shared registry at COMMIT (`post_apply` never runs for a +//! `ProposeOutcome::Buffered` entry). //! Without this fallback, `NEXTVAL` / `CURRVAL` / `SETVAL` on a sequence //! created earlier in the same transaction resolve as missing. //! @@ -42,6 +42,7 @@ fn buffered_def(database_id: u64, tenant_id: u64, name: &str) -> Option @@ -93,6 +94,8 @@ mod tests { database_id, tenant_id, name: name.to_owned(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, } } diff --git a/nodedb/src/control/sequence/mod.rs b/nodedb/src/control/sequence/mod.rs index 2d617f6b6..fb108fcc0 100644 --- a/nodedb/src/control/sequence/mod.rs +++ b/nodedb/src/control/sequence/mod.rs @@ -6,7 +6,6 @@ pub mod error_map; pub mod format; pub mod gap_free; pub mod log; -pub mod range_alloc; pub mod registry; pub mod session_values; pub mod types; @@ -14,7 +13,6 @@ pub mod types; pub use self::access::{SequenceAccess, SessionSequenceAccess}; pub use self::format::{FormatToken, ResetScope}; pub use self::gap_free::GapFreeManager; -pub use self::range_alloc::RangeAllocator; pub use self::registry::SequenceRegistry; pub use self::session_values::SessionSequenceValues; pub use self::types::SequenceError; diff --git a/nodedb/src/control/sequence/range_alloc.rs b/nodedb/src/control/sequence/range_alloc.rs deleted file mode 100644 index f469eacc9..000000000 --- a/nodedb/src/control/sequence/range_alloc.rs +++ /dev/null @@ -1,236 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! Raft-coordinated range allocation for distributed sequences. -//! -//! Each node (Origin shard or Lite instance) requests a chunk of IDs from -//! the Raft leader. Local `nextval` advances within the chunk without -//! network round-trips. When the chunk is exhausted, a new chunk is allocated. -//! -//! Uses the existing `RaftProposer` on SharedState — proposes serialized -//! `RangeAllocationRequest` messages to the Raft group, which are applied -//! by the state machine to update the global sequence counter. - -use serde::{Deserialize, Serialize}; - -/// A request to allocate a range of sequence values via Raft consensus. -#[derive( - Debug, Clone, Serialize, Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack, -)] -pub struct RangeAllocationRequest { - pub database_id: u64, - pub tenant_id: u64, - pub sequence_name: String, - /// How many values to allocate in this chunk. - pub chunk_size: i64, - /// Expected epoch — allocation rejected if epoch mismatch (stale node). - pub epoch: u64, -} - -/// A GAP_FREE counter advance proposed to Raft for cluster-safe serialization. -/// -/// In cluster mode, each gap-free nextval is proposed as a Raft log entry. -/// On leader failover, the new leader replays the log and has the exact counter. -#[derive( - Debug, Clone, Serialize, Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack, -)] -pub struct GapFreeAdvanceRequest { - pub database_id: u64, - pub tenant_id: u64, - pub sequence_name: String, - /// The value being reserved (must match the local counter advance). - pub reserved_value: i64, - pub epoch: u64, -} - -/// Response from the Raft leader after a range allocation. -#[derive( - Debug, Clone, Serialize, Deserialize, zerompk::ToMessagePack, zerompk::FromMessagePack, -)] -pub struct RangeAllocationResponse { - /// First value in the allocated range (inclusive). - pub range_start: i64, - /// Last value in the allocated range (inclusive). - pub range_end: i64, - /// Epoch of this allocation. - pub epoch: u64, -} - -/// Allocates sequence ranges via the Raft proposer. -/// -/// In single-node mode (no Raft), ranges are allocated locally with no -/// coordination. In cluster mode, each allocation is proposed to the -/// Raft leader and committed across the group. -pub struct RangeAllocator { - /// Default chunk size for allocations. - pub default_chunk_size: i64, -} - -impl RangeAllocator { - pub fn new(default_chunk_size: i64) -> Self { - Self { default_chunk_size } - } - - /// Allocate a chunk of sequence values. - /// - /// In cluster mode: proposes to Raft leader, awaits commit. - /// In single-node mode: directly advances the catalog counter. - pub fn allocate_chunk( - &self, - state: &crate::control::state::SharedState, - database_id: u64, - tenant_id: u64, - sequence_name: &str, - increment: i64, - epoch: u64, - ) -> Result { - let chunk_size = self.default_chunk_size; - - // In cluster mode, propose through Raft for distributed uniqueness. - if let Some(proposer) = state.raft_proposer.get() { - let request = RangeAllocationRequest { - database_id, - tenant_id, - sequence_name: sequence_name.to_string(), - chunk_size, - epoch, - }; - let payload = - zerompk::to_msgpack_vec(&request).map_err(|e| crate::Error::Serialization { - format: "msgpack".into(), - detail: format!("range allocation request: {e}"), - })?; - - // Propose to vshard 0 (system shard for metadata operations). - let (_group_id, _log_index) = - proposer(0, payload).map_err(|e| crate::Error::Dispatch { - detail: format!("sequence range allocation raft propose: {e}"), - })?; - - // Compute the allocated range based on current state. - // The Raft commit handler will advance the global counter. - let current = state - .sequence_registry - .get_def(database_id, tenant_id, sequence_name) - .map(|d| d.start_value) - .unwrap_or(1); - - let range_start = current; - let range_end = if increment > 0 { - current + chunk_size * increment - increment - } else { - current + chunk_size * increment + increment.abs() - }; - - return Ok(RangeAllocationResponse { - range_start, - range_end, - epoch, - }); - } - - // Single-node mode: allocate directly from the local counter. - // No Raft needed — just advance the counter by chunk_size. - let handle_exists = state - .sequence_registry - .exists(database_id, tenant_id, sequence_name); - if !handle_exists { - return Err(crate::Error::BadRequest { - detail: format!("sequence \"{sequence_name}\" does not exist"), - }); - } - - let current_val = state - .sequence_registry - .node_current_value(database_id, tenant_id, sequence_name) - .unwrap_or(0); - - let range_start = current_val + increment; - let range_end = range_start + (chunk_size - 1) * increment; - - Ok(RangeAllocationResponse { - range_start, - range_end, - epoch, - }) - } - - /// Propose a GAP_FREE counter advance to Raft. - /// - /// In cluster mode: serializes the advance as a Raft log entry so that - /// on leader failover, the new leader has the exact counter value. - /// In single-node mode: no-op (local counter is authoritative). - pub fn propose_gap_free_advance( - &self, - state: &crate::control::state::SharedState, - database_id: u64, - tenant_id: u64, - sequence_name: &str, - reserved_value: i64, - epoch: u64, - ) -> Result<(), crate::Error> { - let Some(proposer) = state.raft_proposer.get() else { - // Single-node mode — local counter is authoritative, no Raft needed. - return Ok(()); - }; - - let request = GapFreeAdvanceRequest { - database_id, - tenant_id, - sequence_name: sequence_name.to_string(), - reserved_value, - epoch, - }; - let payload = - zerompk::to_msgpack_vec(&request).map_err(|e| crate::Error::Serialization { - format: "msgpack".into(), - detail: format!("gap-free advance request: {e}"), - })?; - - // Propose to vshard 0 (system shard). - proposer(0, payload).map_err(|e| crate::Error::Dispatch { - detail: format!("gap-free advance raft propose: {e}"), - })?; - - Ok(()) - } -} - -impl Default for RangeAllocator { - fn default() -> Self { - Self::new(10_000) - } -} - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn request_roundtrip() { - let req = RangeAllocationRequest { - database_id: 7, - tenant_id: 1, - sequence_name: "order_seq".into(), - chunk_size: 10_000, - epoch: 1, - }; - let bytes = zerompk::to_msgpack_vec(&req).unwrap(); - let decoded: RangeAllocationRequest = zerompk::from_msgpack(&bytes).unwrap(); - assert_eq!(decoded.database_id, 7); - assert_eq!(decoded.sequence_name, "order_seq"); - assert_eq!(decoded.chunk_size, 10_000); - } - - #[test] - fn response_roundtrip() { - let resp = RangeAllocationResponse { - range_start: 1, - range_end: 10_000, - epoch: 1, - }; - let bytes = zerompk::to_msgpack_vec(&resp).unwrap(); - let decoded: RangeAllocationResponse = zerompk::from_msgpack(&bytes).unwrap(); - assert_eq!(decoded.range_start, 1); - assert_eq!(decoded.range_end, 10_000); - } -} diff --git a/nodedb/src/control/sequence/registry.rs b/nodedb/src/control/sequence/registry.rs index 1fdb83ebf..1933df889 100644 --- a/nodedb/src/control/sequence/registry.rs +++ b/nodedb/src/control/sequence/registry.rs @@ -43,6 +43,36 @@ impl SequenceRegistry { &self.gap_free } + /// Replace every sequence with the catalog's definitions and states. + /// + /// `unchanged(database_id, tenant_id, name)` reports a sequence whose + /// definition and state the replace left as they were. Its live handle + /// stays: the counter holds values this node issued that the catalog + /// never saw, and rebuilding it from the catalog reissues them. + pub fn reload_from_catalog( + &self, + catalog: &SystemCatalog, + unchanged: impl Fn(u64, u64, &str) -> bool, + ) -> crate::Result<()> { + let mut loaded = Vec::new(); + for def in catalog.load_all_sequences()? { + let keep = unchanged(def.database_id, def.tenant_id, &def.name); + let state = catalog.get_sequence_state(def.database_id, def.tenant_id, &def.name)?; + loaded.push((def, keep, state)); + } + let mut map = self.sequences.write().unwrap_or_else(|p| p.into_inner()); + let mut live = std::mem::take(&mut *map); + for (def, keep, state) in loaded { + let key = registry_key(def.database_id, def.tenant_id, &def.name); + let handle = match live.remove(&key) { + Some(handle) if keep => handle, + _ => SequenceHandle::new(def, state), + }; + map.insert(key, handle); + } + Ok(()) + } + /// Load all sequences from the catalog on startup. pub fn load_from_catalog(&self, catalog: &SystemCatalog) { let all_defs = match catalog.load_all_sequences() { @@ -330,7 +360,7 @@ impl SequenceRegistry { database_id: handle.def.database_id, tenant_id: handle.def.tenant_id, name: handle.def.name.clone(), - current_value: handle.current_value(), + current_value: handle.persisted_value(), is_called: handle.is_called(), epoch: handle.def.epoch, period_key: handle.period_key(), diff --git a/nodedb/src/control/sequence/types.rs b/nodedb/src/control/sequence/types.rs index e4ddf85e4..331a1a3ca 100644 --- a/nodedb/src/control/sequence/types.rs +++ b/nodedb/src/control/sequence/types.rs @@ -24,6 +24,9 @@ pub struct SequenceHandle { impl SequenceHandle { /// Create a new handle from a sequence definition and optional persisted state. + /// + /// A called state holds the last value `nextval` returned. An uncalled + /// state holds the value the next `nextval` returns, as `RESTART` sets it. pub fn new( def: crate::control::security::catalog::sequence_types::StoredSequence, state: Option, @@ -32,7 +35,11 @@ impl SequenceHandle { if s.is_called { (s.current_value, true, s.period_key) } else { - (def.start_value - def.increment, false, s.period_key) + ( + s.current_value.saturating_sub(def.increment), + false, + s.period_key, + ) } } else { (def.start_value - def.increment, false, String::new()) @@ -243,6 +250,17 @@ impl SequenceHandle { self.counter.load(Ordering::Relaxed) } + /// The value a persisted state stores: the last value returned once + /// called, else the value the next `nextval` returns. + pub fn persisted_value(&self) -> i64 { + let counter = self.current_value(); + if self.is_called() { + counter + } else { + counter.saturating_add(self.def.increment) + } + } + /// Whether nextval has been called. pub fn is_called(&self) -> bool { self.called.load(Ordering::Relaxed) @@ -350,6 +368,25 @@ mod tests { assert_eq!(h.nextval().unwrap(), 3); } + #[test] + fn restarted_state_survives_a_reload() { + use crate::control::security::catalog::sequence_types::SequenceState; + let mut def = StoredSequence::new(4, 1, "test".into(), "admin".into()); + def.increment = 5; + let restarted = SequenceState::new(4, 1, "test".into(), 100, def.epoch); + let h = SequenceHandle::new(def.clone(), Some(restarted)); + assert_eq!(h.persisted_value(), 100); + assert_eq!(h.nextval().unwrap(), 100); + + let persisted = SequenceState { + current_value: h.persisted_value(), + is_called: h.is_called(), + ..SequenceState::new(4, 1, "test".into(), 0, def.epoch) + }; + let reloaded = SequenceHandle::new(def, Some(persisted)); + assert_eq!(reloaded.nextval().unwrap(), 105); + } + #[test] fn currval_before_nextval() { let h = make_handle(1, 1, 1, 100, false); diff --git a/nodedb/src/control/server/broadcast.rs b/nodedb/src/control/server/broadcast.rs index a6603f2fd..06b7053f8 100644 --- a/nodedb/src/control/server/broadcast.rs +++ b/nodedb/src/control/server/broadcast.rs @@ -135,6 +135,31 @@ pub async fn broadcast_count_to_all_cores( plan: PhysicalPlan, trace_id: TraceId, count_key: &str, +) -> crate::Result { + // One statement, one deadline instant — every core waits to the same one. + let deadline = statement_deadline(shared.tuning.network.default_deadline_secs); + broadcast_count_to_all_cores_until( + shared, + tenant_id, + database_id, + plan, + trace_id, + count_key, + deadline, + ) + .await +} + +/// [`broadcast_count_to_all_cores`] with every core bounded by `deadline`. +/// A caller outside the statement's task picks its own deadline. +pub(crate) async fn broadcast_count_to_all_cores_until( + shared: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + plan: PhysicalPlan, + trace_id: TraceId, + count_key: &str, + deadline: std::time::Instant, ) -> crate::Result { BROADCAST_CALLS.fetch_add(1, Ordering::Relaxed); let num_cores = shared @@ -154,7 +179,7 @@ pub async fn broadcast_count_to_all_cores( database_id, vshard_id, plan: plan.clone(), - deadline: statement_deadline(shared.tuning.network.default_deadline_secs), + deadline, priority: Priority::Normal, trace_id, consistency: ReadConsistency::Strong, @@ -166,15 +191,21 @@ pub async fn broadcast_count_to_all_cores( txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission, }; let rx = shared.tracker.register(request_id); - shared + let dispatched = shared .dispatcher .lock() .unwrap_or_else(|p| p.into_inner()) - .dispatch_to_core(core_id, request)?; + .dispatch_to_core(core_id, request); + if let Err(error) = dispatched { + // No response ever arrives for a refused request. + shared.tracker.cancel(&request_id); + return Err(error); + } receivers.push((request_id, rx)); } @@ -184,8 +215,6 @@ pub async fn broadcast_count_to_all_cores( // out of time reports the statement's deadline rather than collapsing into // a generic dispatch failure. let mut first_error: Option = None; - // One statement, one deadline instant — every core waits to the same one. - let deadline = statement_deadline(shared.tuning.network.default_deadline_secs); for (request_id, mut rx) in receivers { let resp = match tokio::time::timeout_at( @@ -223,7 +252,7 @@ pub async fn broadcast_count_to_all_cores( } // A broadcast is an all-core barrier. Returning success after even one - // error would let callers finalize control-plane state while that core + // error will let callers finalize control-plane state while that core // still retains the old Array store. if let Some(error) = first_error { return Err(error); @@ -293,6 +322,7 @@ pub async fn broadcast_register_to_all_cores( txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission, }; diff --git a/nodedb/src/control/server/calvin_submit/inbox_hook.rs b/nodedb/src/control/server/calvin_submit/inbox_hook.rs index b585f12e4..cf4578d59 100644 --- a/nodedb/src/control/server/calvin_submit/inbox_hook.rs +++ b/nodedb/src/control/server/calvin_submit/inbox_hook.rs @@ -39,8 +39,12 @@ use std::sync::Arc; use std::time::Duration; -use nodedb_cluster::calvin::types::TxClass; -use nodedb_cluster::{SubmitCalvinInboxRequest, SubmitCalvinInboxResponse, TypedClusterError}; +use nodedb_cluster::calvin::PartsOfferStatus; +use nodedb_cluster::calvin::types::{PartStreamId, StreamedPart, TxClass}; +use nodedb_cluster::{ + CalvinPartsRequest, CalvinPartsResponse, SubmitCalvinInboxRequest, SubmitCalvinInboxResponse, + TypedClusterError, +}; use crate::control::planner::calvin::submit::submit_local_assign; use crate::control::state::SharedState; @@ -120,4 +124,33 @@ impl nodedb_cluster::CalvinSubmitInbox for RegistryCalvinSubmitInbox { }, } } + + async fn on_calvin_parts(&self, req: CalvinPartsRequest) -> CalvinPartsResponse { + let refused = |detail: String| CalvinPartsResponse { + status: PartsOfferStatus::Rejected as u8, + next_index: 0, + detail: Some(detail), + }; + let parts: Vec = match zerompk::from_msgpack(&req.parts_bytes) { + Ok(parts) => parts, + Err(e) => return refused(format!("calvin-parts: failed to decode parts: {e}")), + }; + let Some(inbox) = self.state.sequencer_inbox.get() else { + return CalvinPartsResponse { + status: PartsOfferStatus::Unknown as u8, + next_index: 0, + detail: Some("calvin-parts: no sequencer inbox on this node".into()), + }; + }; + let stream = PartStreamId { + node: req.stream_node, + seq: req.stream_seq, + }; + let offer = inbox.offer_parts(stream, parts); + CalvinPartsResponse { + status: offer.status as u8, + next_index: offer.next_index, + detail: offer.detail, + } + } } diff --git a/nodedb/src/control/server/dispatch_utils/change_events/cluster_array.rs b/nodedb/src/control/server/dispatch_utils/change_events/cluster_array.rs new file mode 100644 index 000000000..acd2bfad2 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/change_events/cluster_array.rs @@ -0,0 +1,23 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The change metadata a `ClusterArrayOp` names. + +use nodedb_physical::physical_plan::ClusterArrayOp; + +use crate::control::change_stream::ChangeOperation; + +use super::extract::{WriteChangeMeta, every_row}; + +/// Map a `ClusterArrayOp` to its CDC change metadata. +pub(super) fn cluster_array_change_meta(op: &ClusterArrayOp) -> Vec { + match op { + ClusterArrayOp::Put { array_id, .. } => { + vec![(array_id.name.clone(), every_row(), ChangeOperation::Insert)] + } + ClusterArrayOp::Delete { array_id, .. } => { + vec![(array_id.name.clone(), every_row(), ChangeOperation::Delete)] + } + // Slice/Agg are reads — no row changed. + ClusterArrayOp::Slice { .. } | ClusterArrayOp::Agg { .. } => Vec::new(), + } +} diff --git a/nodedb/src/control/server/dispatch_utils/change_events/extract.rs b/nodedb/src/control/server/dispatch_utils/change_events/extract.rs index 96f120f1b..aa7fd992b 100644 --- a/nodedb/src/control/server/dispatch_utils/change_events/extract.rs +++ b/nodedb/src/control/server/dispatch_utils/change_events/extract.rs @@ -7,17 +7,20 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::change_stream::ChangeOperation; use crate::types::TenantId; use nodedb_physical::physical_plan::{ - ArrayOp, ClusterArrayOp, ColumnarOp, CrdtOp, DocumentOp, DocumentResolvedMutation, KvOp, - KvResolvedMutation, MetaOp, TimeseriesOp, VectorOp, VectorResolvedMutation, VectorWriteTargets, + ArrayOp, ColumnarOp, CrdtOp, DocumentOp, DocumentResolvedMutation, KvOp, KvResolvedMutation, + MetaOp, TimeseriesOp, VectorOp, VectorResolvedMutation, VectorWriteTargets, }; use nodedb_types::{RowIdentity, StorageKey}; +use super::cluster_array::cluster_array_change_meta; +use super::redo::redo_change_meta; + /// One row change a plan yields: `(collection, row identity, op)`. pub(super) type WriteChangeMeta = (String, RowIdentity, ChangeOperation); /// The identity a batch or predicate write reports: every row in the /// collection, not one addressable row. A subscriber sees `"*"`. -fn every_row() -> RowIdentity { +pub(super) fn every_row() -> RowIdentity { RowIdentity::from_user_key("*") } @@ -45,6 +48,33 @@ fn vector_target_events( } } +/// One event per resolved vector-primary mutation, naming every row touched — +/// never collapsed to "*". An upsert's pre-image is what the resolve found +/// stored: absent is an insert, present is an update. +fn vector_resolved_change_meta( + collection: &str, + mutations: &[VectorResolvedMutation], +) -> Vec { + mutations + .iter() + .map(|mutation| { + let operation = match mutation { + VectorResolvedMutation::Delete { .. } => ChangeOperation::Delete, + VectorResolvedMutation::Update { .. } => ChangeOperation::Update, + VectorResolvedMutation::Upsert { old_payload, .. } => match old_payload { + Some(_) => ChangeOperation::Update, + None => ChangeOperation::Insert, + }, + }; + ( + collection.to_owned(), + StorageKey::for_surrogate(mutation.surrogate()).to_identity(), + operation, + ) + }) + .collect() +} + /// A KV row's identity is its key bytes, rendered as text for the subscriber. fn kv_identity(key: &[u8]) -> RowIdentity { RowIdentity::from_user_key(String::from_utf8_lossy(key)) @@ -166,7 +196,7 @@ pub(super) fn extract_write_metadata( PhysicalPlan::Document(_) => Vec::new(), // Batch write and truncate: document_id="*" names every row. Per-row - // events would flood the bus — subscribe via collection_filter. + // events will flood the bus — subscribe via collection_filter. PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection, .. }) => { vec![(collection.to_string(), every_row(), ChangeOperation::Insert)] } @@ -259,7 +289,7 @@ pub(super) fn extract_write_metadata( ChangeOperation::Insert, ), ], - // Reports one event per mutation, naming every collection/key touched — may + // Reports one event per mutation, naming every collection/key touched — can // span two collections (a resolved `TransferItem`). PhysicalPlan::Kv(KvOp::ResolvedWrite { mutations, .. }) => mutations .iter() @@ -271,9 +301,9 @@ pub(super) fn extract_write_metadata( Some(_) => ChangeOperation::Update, None => ChangeOperation::Insert, }, - KvResolvedMutation::Expire { .. } | KvResolvedMutation::Persist { .. } => { - ChangeOperation::Update - } + KvResolvedMutation::Rewrite { .. } + | KvResolvedMutation::Expire { .. } + | KvResolvedMutation::Persist { .. } => ChangeOperation::Update, }; ( mutation.collection().to_string(), @@ -318,7 +348,7 @@ pub(super) fn extract_write_metadata( // `GRAPH INSERT EDGE` and `SetNodeLabels`/`RemoveNodeLabels` are known CDC gaps. PhysicalPlan::Graph(_) => Vec::new(), - // Vector is normally a Document secondary index — publishing here would duplicate. + // Vector is normally a Document secondary index — publishing here will duplicate. // The direct write family is the exception: the sole writes for a vector-primary // collection. The row's identity is its PK-bound surrogate, rendered as the decimal // surrogate, which is what a sidecar read reports as the row id. @@ -356,30 +386,11 @@ pub(super) fn extract_write_metadata( targets, .. }) => vector_target_events(collection, targets, ChangeOperation::Update), - // Reports one event per mutation, naming every row touched — never collapses to "*". - // An upsert's pre-image is what the resolve found stored: absent = insert, present = update. PhysicalPlan::Vector(VectorOp::ResolvedDirectWrite { collection, mutations, .. - }) => mutations - .iter() - .map(|mutation| { - let operation = match mutation { - VectorResolvedMutation::Delete { .. } => ChangeOperation::Delete, - VectorResolvedMutation::Update { .. } => ChangeOperation::Update, - VectorResolvedMutation::Upsert { old_payload, .. } => match old_payload { - Some(_) => ChangeOperation::Update, - None => ChangeOperation::Insert, - }, - }; - ( - collection.to_string(), - StorageKey::for_surrogate(mutation.surrogate()).to_identity(), - operation, - ) - }) - .collect(), + }) => vector_resolved_change_meta(&collection.to_string(), mutations), // The resolve pass reads; the rows it decides are published by the // resolved write that applies them. PhysicalPlan::Vector(_) => Vec::new(), @@ -467,40 +478,27 @@ pub(super) fn extract_write_metadata( // Query: joins, aggregates, coordinator Exchange nodes are read-only. PhysicalPlan::Query(_) => Vec::new(), - // Publishes the same events its constituent writes would emit under autocommit. + // Publishes the same events its constituent writes will emit under autocommit. PhysicalPlan::Meta(MetaOp::TransactionBatch { plans, .. }) => plans .iter() .flat_map(|plan| extract_write_metadata(plan, _tenant_id)) .collect(), + // A committed transaction's redo publishes the rows it installs. + PhysicalPlan::Meta(MetaOp::ApplyTransactionRedo { redo, .. }) => redo_change_meta(redo), PhysicalPlan::Meta(_) => Vec::new(), - // Never reached via normal dispatch — `routing/cluster_array.rs` calls this - // directly via `publish_cluster_array_change_events`, so it IS load-bearing there. + // A coordinator op publishes nothing itself: each shard's committed + // array write publishes as its replicas apply it. PhysicalPlan::ClusterArray(op) => cluster_array_change_meta(op), PhysicalPlan::ClusterEvent(_) => Vec::new(), } } -/// Map a `ClusterArrayOp` to its CDC change metadata. Shared by the -/// `PhysicalPlan::ClusterArray` arm and `publish_cluster_array_change_events`, -/// which holds the op by reference to avoid cloning the write batch. -pub(crate) fn cluster_array_change_meta(op: &ClusterArrayOp) -> Vec { - match op { - ClusterArrayOp::Put { array_id, .. } => { - vec![(array_id.name.clone(), every_row(), ChangeOperation::Insert)] - } - ClusterArrayOp::Delete { array_id, .. } => { - vec![(array_id.name.clone(), every_row(), ChangeOperation::Delete)] - } - // Slice/Agg are reads — no row changed. - ClusterArrayOp::Slice { .. } | ClusterArrayOp::Agg { .. } => Vec::new(), - } -} - #[cfg(test)] mod tests { use super::*; use nodedb_array::types::ArrayId; + use nodedb_physical::physical_plan::ClusterArrayOp; use nodedb_physical::physical_plan::{ColumnarInsertIntent, GraphOp}; use nodedb_types::{ DatabaseId, QualifiedCollection, Surrogate, VectorQuantization, VectorStorageDtype, @@ -525,7 +523,7 @@ mod tests { PhysicalPlan::Document(DocumentOp::PointDelete { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "users"), document_id: "u2".into(), - surrogate: Surrogate::new(2), + surrogate: Some(Surrogate::new(2)), pk_bytes: Vec::new(), returning: None, rls_filters: Vec::new(), @@ -597,6 +595,7 @@ mod tests { cells_msgpack: Vec::new(), wal_lsn: 0, provenance: None, + vshard_id: 0, }); let meta = extract_write_metadata(&plan, TenantId::new(1)); assert_eq!( @@ -612,6 +611,7 @@ mod tests { coords_msgpack: Vec::new(), wal_lsn: 0, provenance: None, + vshard_id: 0, }); let meta = extract_write_metadata(&plan, TenantId::new(1)); assert_eq!( @@ -680,7 +680,7 @@ mod tests { } // Implicit edges mirror into a separate `GraphOp::EdgePut`; the underlying - // `DocumentOp` already published the event — emitting here would double-publish. + // `DocumentOp` already published the event — emitting here will double-publish. #[test] fn graph_edge_put_emits_no_change_event() { let plan = PhysicalPlan::Graph(GraphOp::EdgePut { diff --git a/nodedb/src/control/server/dispatch_utils/change_events/mod.rs b/nodedb/src/control/server/dispatch_utils/change_events/mod.rs index 62546be8d..3d30d142e 100644 --- a/nodedb/src/control/server/dispatch_utils/change_events/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/change_events/mod.rs @@ -3,10 +3,12 @@ //! Write-metadata extraction and CDC change-event publishing for dispatched //! writes. +mod cluster_array; mod extract; mod publish; +mod redo; pub(crate) use publish::{ - WriteChangeSet, extract_write_change_set, publish_change_set, publish_change_set_with_lsn, - publish_cluster_array_change_events, publish_origin_change_events, + CalvinApply, PendingChanges, WriteChangeSet, extract_write_change_set, + publish_calvin_change_sets, publish_settled_changes, redo_change_set, }; diff --git a/nodedb/src/control/server/dispatch_utils/change_events/publish.rs b/nodedb/src/control/server/dispatch_utils/change_events/publish.rs index 0a32f0dcb..4a8c4db9e 100644 --- a/nodedb/src/control/server/dispatch_utils/change_events/publish.rs +++ b/nodedb/src/control/server/dispatch_utils/change_events/publish.rs @@ -1,25 +1,26 @@ // SPDX-License-Identifier: BUSL-1.1 //! CDC change-event publishing for dispatched writes: turning the metadata -//! [`super::extract`] derived from a plan into `ChangeEvent`s on the local -//! change stream plus the cluster-wide NOTIFY fan-out. +//! [`super::extract`] derived from a plan into `ChangeEvent`s on the +//! change stream, at the position of the write in its partition's feed. + +use std::sync::Arc; use crate::bridge::envelope::{PhysicalPlan, Response}; +use crate::control::change_stream::{ChangeEvent, ChangePartition, ChangeRun, PositionedChange}; use crate::control::state::SharedState; +use crate::event::cdc::CdcOffset; use crate::types::{DatabaseId, TenantId}; -use nodedb_physical::physical_plan::ClusterArrayOp; -use super::extract::{WriteChangeMeta, cluster_array_change_meta, extract_write_metadata}; +use super::extract::{WriteChangeMeta, extract_write_metadata}; -/// Current wall-clock time as milliseconds since Unix epoch. -/// -/// Returns 0 if the system clock is before the epoch (should never happen -/// on correctly configured systems). -fn current_timestamp_ms() -> u64 { - std::time::SystemTime::now() - .duration_since(std::time::UNIX_EPOCH) - .map(|d| d.as_millis() as u64) - .unwrap_or(0) +const NANOS_PER_MS: u64 = 1_000_000; + +/// The change-event timestamp of a write that committed at `commit_hlc` +/// (HLC wall time, nanoseconds). Every replica reads the same commit HLC +/// from the committed entry, so every node stamps the same value. +fn commit_timestamp_ms(commit_hlc: u64) -> u64 { + commit_hlc / NANOS_PER_MS } /// Check if a timeseries collection has CDC enabled. @@ -49,58 +50,53 @@ fn is_timeseries_cdc_enabled( true } -/// Publish a change event (and cluster-wide NOTIFY) for a successful write. +/// The change event of one row change, or `None` when its collection +/// publishes none: a timeseries collection publishes only with `cdc` +/// enabled. /// -/// CDC opt-in check for timeseries: skip publishing unless `cdc_enabled`. -/// Document collections always publish (backward compatible). -fn publish_change_event( +/// The plan names the collection database-qualified. The event names it +/// bare, as the catalog and every subscriber filter do; its database travels +/// beside it. +fn change_event( shared: &SharedState, tenant_id: TenantId, database_id: DatabaseId, change_meta: WriteChangeMeta, lsn: nodedb_types::Lsn, -) { - let (collection, document_id, op) = change_meta; + commit_hlc: u64, +) -> Option { + let (qualified, document_id, op) = change_meta; + let collection = match nodedb_types::CollectionKey::from_qualified_str(database_id, &qualified) + { + Ok(key) => key.name().to_owned(), + Err(error) => { + tracing::error!( + %error, + collection = %qualified, + "a row change names a collection outside its database; it publishes no event" + ); + return None; + } + }; if !is_timeseries_cdc_enabled(shared, database_id, tenant_id, &collection) { - return; + return None; } - - use crate::control::change_stream::ChangeEvent; - let event = ChangeEvent { + Some(ChangeEvent { lsn, tenant_id, collection, document_id, operation: op, - timestamp_ms: current_timestamp_ms(), + timestamp_ms: commit_timestamp_ms(commit_hlc), after: None, - }; - - // Cluster-wide NOTIFY: broadcast to all peers via QUIC. - if let (Some(transport), Some(topology)) = (&shared.cluster_transport, &shared.cluster_topology) - { - use std::sync::atomic::Ordering; - static NOTIFY_SEQ: std::sync::atomic::AtomicU64 = std::sync::atomic::AtomicU64::new(1); - let seq = NOTIFY_SEQ.fetch_add(1, Ordering::Relaxed); - crate::control::change_stream::broadcast_notify_to_cluster( - database_id, - &event, - shared.node_id, - seq, - transport, - topology, - ); - } - - shared.change_stream.publish_in_database(database_id, event); + }) } /// The Control-Plane change events one write plan yields. /// /// Extraction is split from publishing because the write funnel consumes the /// plan — it moves into the `Request` — long before the `Response` the event's -/// LSN comes from exists. A caller that still owns its plan at publish time -/// uses [`publish_origin_change_events`] and never names this type. +/// LSN comes from exists. pub(crate) struct WriteChangeSet { /// One tuple per logical row change — see `extract_write_metadata`. metas: Vec, @@ -115,91 +111,167 @@ pub(crate) fn extract_write_change_set(plan: &PhysicalPlan, tenant_id: TenantId) } } -/// Publish an already-extracted change set at an explicit LSN. Almost every -/// write plan yields exactly one event; a handful of multi-row / -/// multi-collection ops yield more than one, and reads / DDL / index -/// maintenance yield none. -pub(crate) fn publish_change_set_with_lsn( +/// The change events of a committed transaction's redo record `redo`: one per +/// row it installs. A Calvin commit publishes these, as the data-group apply +/// of a `TransactionRedo` entry does. +pub(crate) fn redo_change_set(redo: &[u8]) -> WriteChangeSet { + WriteChangeSet { + metas: super::redo::redo_change_meta(redo), + } +} + +fn change_events( shared: &SharedState, tenant_id: TenantId, database_id: DatabaseId, change_set: WriteChangeSet, lsn: nodedb_types::Lsn, -) { - let WriteChangeSet { metas } = change_set; - for meta in metas { - publish_change_event(shared, tenant_id, database_id, meta, lsn); + commit_hlc: u64, +) -> Vec { + change_set + .metas + .into_iter() + .filter_map(|meta| change_event(shared, tenant_id, database_id, meta, lsn, commit_hlc)) + .collect() +} + +impl WriteChangeSet { + /// Whether the write yields no change event. + pub(crate) fn is_empty(&self) -> bool { + self.metas.is_empty() } } -/// Publish an already-extracted change set, taking the LSN from a Data-Plane -/// [`Response`]'s watermark. -pub(crate) fn publish_change_set( - shared: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, +/// A replicated write's change set, staged under its data-group entry once +/// the write applied. +pub(crate) struct PendingChanges { change_set: WriteChangeSet, - response: &Response, -) { - publish_change_set_with_lsn( - shared, - tenant_id, - database_id, - change_set, - response.watermark_lsn, - ); + /// The data group of the committed entry the write applies. + group_id: u64, + /// The entry's log index in its group. + log_index: u64, + /// HLC wall time, in nanoseconds, at which the write committed. + commit_hlc: u64, +} + +impl PendingChanges { + /// Staged under the data-group entry `(group_id, log_index)` once it + /// applies, and published when the apply loop settles the entry in log + /// order ([`publish_settled_changes`]). + pub(crate) fn staged(change_set: WriteChangeSet, group_id: u64, log_index: u64) -> Self { + Self { + change_set, + group_id, + log_index, + commit_hlc: 0, + } + } + + /// Stamp the write's commit HLC, which dates its events. A replicated + /// write takes the HLC its proposer stamped on the entry. + pub(crate) fn committed_at(mut self, commit_hlc: u64) -> Self { + self.commit_hlc = commit_hlc; + self + } + + /// Stage the write's events, at the LSN of its Data-Plane [`Response`], + /// under its entry. Almost every write plan yields exactly one event; a + /// handful of multi-row / multi-collection ops yield more than one, and + /// reads / DDL / index maintenance yield none. + pub(crate) fn publish( + self, + shared: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + response: &Response, + ) { + let events = change_events( + shared, + tenant_id, + database_id, + self.change_set, + response.watermark_lsn, + self.commit_hlc, + ); + shared + .change_stream + .stage(self.group_id, self.log_index, database_id, events); + } } -/// Publish the Control-Plane change event(s) for a write this node originated -/// and has already had committed and applied. +/// Publish the staged changes of `group_id`'s entries `first..=last`, which +/// the apply loop settled in log order, and forward them to the nodes that +/// do not replicate the group when this node leads it. /// -/// The cluster Raft path cannot let the write funnel own its change feed: the -/// proposing node never reaches `submit_write` — it proposes, and every -/// replica's apply loop submits the committed entry independently. Publishing -/// from the apply loop would emit one event per replica plus a full NOTIFY -/// fan-out from each, so a subscriber would see the write once per replica. The -/// proposing node is the one node that handled the write exactly once, so it is -/// the one that publishes. -pub(crate) fn publish_origin_change_events( - shared: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - plan: &PhysicalPlan, - response: &Response, +/// Every replica publishes the group's writes at their log positions, so +/// every node serves the same feed. +pub(crate) fn publish_settled_changes( + shared: &Arc, + group_id: u64, + first: u64, + last: u64, ) { - publish_change_set( - shared, - tenant_id, - database_id, - extract_write_change_set(plan, tenant_id), - response, - ); + shared.change_stream.settle_group(group_id, first, last); + shared + .change_stream + .forward(shared, ChangePartition::Group(group_id)); } -/// Publish the Control-Plane change event(s) for a `ClusterArray` write. +/// Publish the changes of the Calvin transaction at `(sequencer_epoch, +/// position)` that vShard `vshard` applied, and forward them when this node +/// leads the vShard's data group. /// -/// `ClusterArrayOp` never reaches the SPSC bridge / Data-Plane `Response` -/// path (see `PhysicalPlan::ClusterArray`'s own doc comment) — the coordinator -/// dispatch loop in `routing/cluster_array.rs` executes the op directly via -/// `ClusterArrayExecutor` and has no `Response::watermark_lsn` to read, so it -/// calls this entry point with the `wal_lsn` the op itself carries (allocated -/// by the Control Plane for the write) instead of going through -/// [`publish_origin_change_events`]. -pub(crate) fn publish_cluster_array_change_events( - shared: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - op: &ClusterArrayOp, - lsn: u64, +/// Every replica's scheduler applies the vShard's transactions in sequencer +/// order, so every replica publishes them at the same positions. +pub(crate) fn publish_calvin_change_sets( + shared: &Arc, + calvin: CalvinApply, + change_sets: Vec, + lsn: nodedb_types::Lsn, ) { - let change_set = WriteChangeSet { - metas: cluster_array_change_meta(op), - }; - publish_change_set_with_lsn( - shared, + let CalvinApply { tenant_id, database_id, - change_set, - nodedb_types::Lsn::new(lsn), - ); + vshard, + sequencer_epoch, + position, + commit_hlc, + } = calvin; + let base = u64::from(position); + let changes: Vec = change_sets + .into_iter() + .flat_map(|change_set| { + change_events(shared, tenant_id, database_id, change_set, lsn, commit_hlc) + }) + .enumerate() + .map(|(ordinal, event)| PositionedChange { + position: CdcOffset::data_event_in(0, sequencer_epoch, base, ordinal as u64 + 1), + database_id, + event, + }) + .collect(); + if changes.is_empty() { + return; + } + shared.change_stream.publish_run(ChangeRun { + partition: ChangePartition::Calvin(vshard), + after: None, + through: CdcOffset::data_event_in(0, sequencer_epoch, base, u64::MAX).correction(), + changes, + }); + shared + .change_stream + .forward(shared, ChangePartition::Calvin(vshard)); +} + +/// The Calvin transaction a published change set belongs to. +#[derive(Debug, Clone, Copy)] +pub(crate) struct CalvinApply { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + pub vshard: u32, + pub sequencer_epoch: u64, + pub position: u32, + /// The transaction's commit HLC, which every replica derives alike. + pub commit_hlc: u64, } diff --git a/nodedb/src/control/server/dispatch_utils/change_events/redo.rs b/nodedb/src/control/server/dispatch_utils/change_events/redo.rs new file mode 100644 index 000000000..be59310f3 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/change_events/redo.rs @@ -0,0 +1,155 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The change events of a committed transaction: the net change of every row +//! its redo record writes. +//! +//! Every producer of a committed redo record names each row it changes in the +//! record's `row_changes` (see [`crate::wal::redo::row_changes`]): the +//! transaction resolve for every engine, and the restore and clone re-issues +//! for the rows they install. The events come from those entries alone, so an +//! update publishes `Update`, several writes to one row publish once with +//! their net kind, and no kind is inferred from a sub-op's record type. + +use nodedb_types::RowIdentity; + +use crate::control::change_stream::ChangeOperation; +use crate::wal::{EVERY_ROW, RedoRowChange, RedoRowKind}; + +use super::extract::{WriteChangeMeta, every_row}; + +/// One change per row the `TransactionRedo` payload `redo` names. A payload +/// that does not decode yields none: the Data Plane refuses it, so it +/// installs nothing. +pub(super) fn redo_change_meta(redo: &[u8]) -> Vec { + let Ok(record) = crate::wal::RedoRecord::from_bytes(redo) else { + return Vec::new(); + }; + record + .row_changes + .into_iter() + .filter_map(named_row_change) + .collect() +} + +/// The change a named row publishes. A row with no net change publishes +/// nothing. +fn named_row_change(change: RedoRowChange) -> Option { + let operation = match change.kind { + RedoRowKind::Insert => ChangeOperation::Insert, + RedoRowKind::Update => ChangeOperation::Update, + RedoRowKind::Delete => ChangeOperation::Delete, + RedoRowKind::NoChange => return None, + }; + let identity = if change.row == EVERY_ROW { + every_row() + } else { + RowIdentity::from_user_key(change.row) + }; + Some((change.collection, identity, operation)) +} + +#[cfg(test)] +mod tests { + use nodedb_types::sync::wire::SyncProvenance; + use nodedb_wal::record::RecordType; + + use super::*; + use crate::wal::{RedoRecord, RedoSubRecord}; + + fn named(collection: &str, row: &str, kind: RedoRowKind) -> RedoRowChange { + RedoRowChange { + collection: collection.to_owned(), + row: row.to_owned(), + kind, + } + } + + fn record(ops: Vec, row_changes: Vec) -> Vec { + RedoRecord { + version: 1, + ops, + calvin_stamp: None, + cross_shard_applied: None, + row_sources: Vec::new(), + publishes: Vec::new(), + row_changes, + } + .to_bytes() + .expect("encode redo") + } + + fn doc_put(id: &str) -> RedoSubRecord { + RedoSubRecord { + record_type: RecordType::Put as u32, + payload: zerompk::to_msgpack_vec(&( + "orders", + id, + vec![0x80u8], + None::, + 7u32, + )) + .expect("encode"), + } + } + + #[test] + fn named_rows_publish_their_net_kind_once() { + // The redo holds final images: `o1` inserted then updated, `o2` + // updated, `o3` updated then deleted, `o4` inserted then deleted. + let changes = vec![ + named("orders", "o1", RedoRowKind::Insert), + named("orders", "o2", RedoRowKind::Update), + named("orders", "o3", RedoRowKind::Delete), + named("orders", "o4", RedoRowKind::NoChange), + ]; + let meta = redo_change_meta(&record(vec![doc_put("o1"), doc_put("o2")], changes)); + assert_eq!( + meta, + vec![ + ( + "orders".to_owned(), + RowIdentity::from_user_key("o1"), + ChangeOperation::Insert + ), + ( + "orders".to_owned(), + RowIdentity::from_user_key("o2"), + ChangeOperation::Update + ), + ( + "orders".to_owned(), + RowIdentity::from_user_key("o3"), + ChangeOperation::Delete + ), + ] + ); + } + + #[test] + fn a_whole_collection_change_names_every_row() { + let changes = vec![ + named("metrics", EVERY_ROW, RedoRowKind::Delete), + named("metrics", EVERY_ROW, RedoRowKind::Insert), + ]; + let meta = redo_change_meta(&record(Vec::new(), changes)); + assert_eq!( + meta, + vec![ + ("metrics".to_owned(), every_row(), ChangeOperation::Delete), + ("metrics".to_owned(), every_row(), ChangeOperation::Insert), + ] + ); + } + + #[test] + fn a_sub_op_no_entry_names_publishes_nothing() { + // A put the record does not name is never read as an insert. + let meta = redo_change_meta(&record(vec![doc_put("o1")], Vec::new())); + assert!(meta.is_empty()); + } + + #[test] + fn a_malformed_redo_publishes_nothing() { + assert!(redo_change_meta(b"not a redo").is_empty()); + } +} diff --git a/nodedb/src/control/server/dispatch_utils/dispatch.rs b/nodedb/src/control/server/dispatch_utils/dispatch.rs index 7a69a081c..f6583661e 100644 --- a/nodedb/src/control/server/dispatch_utils/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/dispatch.rs @@ -4,16 +4,18 @@ //! plan to the shared Control-Plane write funnel (`submit_write`), which owns //! write admission, the WAL append, the enqueue, and the response collect. +use futures::future::BoxFuture; + use crate::bridge::envelope::{PhysicalPlan, Response}; use crate::control::server::shared::clone_write::CloneCheckedTask; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; -use super::minted::{MintedRecords, RecordOwner}; +use super::minted::RecordOwner; use super::submit_write::{ ChangeFeedOwner, SubmitWrite, WalDurability, WriteOrdering, submit_write, }; -use super::types::{AutocommitWrite, DataPlaneDispatch, WriteDispatch}; +use super::types::{AutocommitWrite, DataPlaneDispatch, ReadRoute, WriteDispatch}; /// Dispatch a clone-checked, capability-bearing external task to the Data Plane. /// @@ -25,7 +27,9 @@ pub async fn dispatch_authorized_to_data_plane( checked: CloneCheckedTask, trace_id: TraceId, ) -> crate::Result { - let task = checked.into_authorized().into_physical_task(); + // The write lease lives until the dispatch returns its outcome. + let (authorized, _lease) = checked.into_parts(); + let task = authorized.into_physical_task(); dispatch_to_data_plane_inner( shared, DataPlaneDispatch { @@ -41,37 +45,8 @@ pub async fn dispatch_authorized_to_data_plane( resolved_now_ms: None, minted: None, }, - }, - ) - .await -} - -/// Dispatch a clone-checked task whose records the caller already appended -/// under `minted`. The request carries the highest of their LSNs, so the -/// write is durable before it is acknowledged. The funnel closes their -/// outcome-floor window from the task's outcome. -pub(crate) async fn dispatch_authorized_minted_to_data_plane( - shared: &SharedState, - checked: CloneCheckedTask, - trace_id: TraceId, - minted: MintedRecords, -) -> crate::Result { - let task = checked.into_authorized().into_physical_task(); - dispatch_to_data_plane_inner( - shared, - DataPlaneDispatch { - tenant_id: task.tenant_id, - database_id: task.database_id, - vshard_id: task.vshard_id, - plan: task.plan, - trace_id, - event_source: crate::event::EventSource::User, - txn_id: task.txn_id, - durability: WalDurability::CallerSupplied { - wal_lsn: minted.highest(), - resolved_now_ms: None, - minted: Some(minted), - }, + change_feed: ChangeFeedOwner::LocalApply, + read_route: ReadRoute::Owned, }, ) .await @@ -83,7 +58,9 @@ pub async fn dispatch_authorized_autocommit_write( checked: CloneCheckedTask, trace_id: TraceId, ) -> crate::Result { - let task = checked.into_authorized().into_physical_task(); + // The write lease lives until the dispatch returns its outcome. + let (authorized, _lease) = checked.into_parts(); + let task = authorized.into_physical_task(); dispatch_to_data_plane_inner( shared, DataPlaneDispatch { @@ -98,7 +75,10 @@ pub async fn dispatch_authorized_autocommit_write( now_override: None, apply_key: 0, commit_hlc: None, + change_position: None, }, + change_feed: ChangeFeedOwner::LocalApply, + read_route: ReadRoute::Owned, }, ) .await @@ -117,7 +97,9 @@ pub(crate) async fn dispatch_authorized_autocommit_write_with_source( trace_id: TraceId, event_source: crate::event::EventSource, ) -> crate::Result { - let task = checked.into_authorized().into_physical_task(); + // The write lease lives until the dispatch returns its outcome. + let (authorized, _lease) = checked.into_parts(); + let task = authorized.into_physical_task(); dispatch_to_data_plane_inner( shared, DataPlaneDispatch { @@ -132,7 +114,10 @@ pub(crate) async fn dispatch_authorized_autocommit_write_with_source( now_override: None, apply_key: 0, commit_hlc: None, + change_position: None, }, + change_feed: ChangeFeedOwner::LocalApply, + read_route: ReadRoute::Owned, }, ) .await @@ -193,6 +178,10 @@ pub(crate) async fn dispatch_to_data_plane_with_source( resolved_now_ms: None, minted: None, }, + // Trusted internal plumbing: reads, maintenance, and node-local + // state. It publishes no change event. + change_feed: ChangeFeedOwner::Unowned, + read_route: ReadRoute::Owned, }, ) .await @@ -241,6 +230,49 @@ pub(crate) async fn dispatch_trusted_internal_write_to_data_plane( resolved_now_ms, minted, }, + change_feed: ChangeFeedOwner::LocalApply, + read_route: ReadRoute::Owned, + }, + ) + .await +} + +/// Dispatch a write that replays a WAL record this node already applied +/// the effects of once: the record is durable, and its change events were +/// published when it first applied. It publishes none. +pub(crate) async fn dispatch_replayed_write_to_data_plane( + shared: &SharedState, + write: WriteDispatch, +) -> crate::Result { + let WriteDispatch { + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + event_source, + txn_id, + wal_lsn, + resolved_now_ms, + minted, + } = write; + dispatch_to_data_plane_inner( + shared, + DataPlaneDispatch { + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + event_source, + txn_id, + durability: WalDurability::CallerSupplied { + wal_lsn, + resolved_now_ms, + minted, + }, + read_route: ReadRoute::Owned, + change_feed: ChangeFeedOwner::Unowned, }, ) .await @@ -252,7 +284,7 @@ pub(crate) async fn dispatch_trusted_internal_write_to_data_plane( /// This is the entry point for single-node local writes that own their own /// autocommit durability (the native SQL / direct-op boot path, HTTP query, /// RESP KV write, protocol-neutral INSERT/UPSERT). The WAL LSN must be minted -/// after admission and just before the dispatcher enqueue so that WAL-LSN order +/// after admission and right before the dispatcher enqueue so that WAL-LSN order /// equals Data-Plane apply order per key; performing the append inside the /// funnel (rather than at the caller, before admission) is what closes that /// ordering gap. `wal_lsn` / `resolved_now_ms` are therefore *not* caller @@ -280,62 +312,37 @@ pub(crate) async fn dispatch_autocommit_write( trace_id, event_source, txn_id, - // The funnel appends the WAL record under the admission guard just + // The funnel appends the WAL record under the admission guard right // before enqueue and stamps the minted LSN onto the `Request`. durability: WalDurability::AppendHere { now_override: None, apply_key: 0, commit_hlc: None, + change_position: None, }, + change_feed: ChangeFeedOwner::LocalApply, + read_route: ReadRoute::Owned, }, ) .await } -/// Dispatch a physical plan to the Data Plane carrying an explicit transaction -/// id so the Data Plane can resolve this transaction's staging overlay -/// (read-your-own-writes) and route `StageWrite`. Used by the native endpoint, -/// whose in-transaction tasks flow through this shared path. +/// [`dispatch_step`] behind a boxed future. /// -/// It appends no WAL record, so it refuses a write that only the funnel's -/// `AppendHere` route logs. A staged write is not such a write: COMMIT logs it. -pub(crate) async fn dispatch_to_data_plane_with_txn( +/// Every Data Plane dispatch passes this funnel: protocol writes, the +/// transaction route, clone copy-ups, the gateway's local leg, and the sync +/// and array inbound handlers. Under it sit Exchange resolution, the +/// cross-node gather, the array coordinator and the write funnel. Unboxed, +/// that chain nests inside every caller's future and overflows the +/// compiler's layout depth limit. The box ends it at this shared node. +pub(super) fn dispatch_to_data_plane_inner( shared: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - vshard_id: VShardId, - plan: PhysicalPlan, - trace_id: TraceId, - txn_id: Option, -) -> crate::Result { - super::durability_barrier::refuse_unlogged_write(&plan)?; - dispatch_to_data_plane_inner( - shared, - DataPlaneDispatch { - tenant_id, - database_id, - vshard_id, - plan, - trace_id, - event_source: crate::event::EventSource::User, - txn_id, - // Staged in-transaction writes are not yet durably committed; the - // committed write version is recorded at COMMIT via the batch funnel, - // so durability is not the funnel's to append here. - durability: WalDurability::CallerSupplied { - wal_lsn: None, - resolved_now_ms: None, - minted: None, - }, - }, - ) - .await + params: DataPlaneDispatch, +) -> BoxFuture<'_, crate::Result> { + Box::pin(dispatch_step(shared, params)) } -async fn dispatch_to_data_plane_inner( - shared: &SharedState, - params: DataPlaneDispatch, -) -> crate::Result { +async fn dispatch_step(shared: &SharedState, params: DataPlaneDispatch) -> crate::Result { let DataPlaneDispatch { tenant_id, database_id, @@ -345,6 +352,8 @@ async fn dispatch_to_data_plane_inner( event_source, txn_id, mut durability, + read_route, + change_feed, } = params; let owner = RecordOwner { tenant_id, @@ -353,7 +362,7 @@ async fn dispatch_to_data_plane_inner( }; // A write that carries its own records is never a query. Only a query // plan holds Exchange nodes, and resolving one fans it out to the cores, - // so a record-carrying query would reach the cores before any close. + // so a record-carrying query will reach the cores before any close. let plan = if durability.has_minted() { if matches!(plan, PhysicalPlan::Query(_)) { if let Some(minted) = durability.take_minted() { @@ -374,17 +383,37 @@ async fn dispatch_to_data_plane_inner( // node pass through unchanged. Catalog materialization is // identity-scoped and already done upstream on the pgwire and native // paths. The internal funnel is not session-transaction-scoped, so the - // transaction id is `None`. + // transaction id is `None`. An owned read (COPY, cursors, view refresh, + // constraint subqueries) is strong, so every leg confirms. let resolved = crate::control::server::exchange::resolve_exchange_in_plan( shared, - database_id, - tenant_id, plan, - trace_id, - None, + crate::control::server::exchange::ReadScope { + database_id, + tenant_id, + trace_id, + txn_id: None, + linearizable: read_route == ReadRoute::Owned, + }, ) .await?; match resolved { + crate::control::server::exchange::Resolved::Plan(p) + if read_route == ReadRoute::Owned => + { + let scope = super::owner_read::OwnedReadScope { + tenant_id, + database_id, + vshard_id, + trace_id, + txn_id, + linearizable: true, + }; + match super::owner_read::route_owned_read(shared, scope, *p).await? { + super::owner_read::OwnedRead::Local(plan) => *plan, + super::owner_read::OwnedRead::Served(response) => return Ok(response), + } + } crate::control::server::exchange::Resolved::Plan(p) => *p, crate::control::server::exchange::Resolved::Gathered( resp, @@ -413,10 +442,8 @@ async fn dispatch_to_data_plane_inner( user_id: None, durability, ordering: WriteOrdering::Gate, - // The autocommit / internal funnel is the path that feeds `/cdc` - // and WS-RPC subscribers; every other caller of `submit_write` is - // `Unowned`. - change_feed: ChangeFeedOwner::Funnel, + // Each entry point declares whether its writes publish. + change_feed, }, ) .await @@ -432,9 +459,10 @@ mod tests { use nodedb_physical::physical_plan::ArrayOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; - use super::dispatch_authorized_autocommit_write_with_source; + use super::SubmitWrite; use crate::bridge::dispatch::{BridgeResponse, CoreChannelDataSide, Dispatcher}; use crate::bridge::envelope::{Payload, Status}; + use crate::control::server::dispatch_utils::{MintedRecords, enqueue_write}; use crate::control::state::SharedState; use crate::engine::array::wal::ArrayPutCell; use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; @@ -443,17 +471,60 @@ mod tests { const ARRAY: &str = "grid"; fn fixture() -> (Arc, CoreChannelDataSide, tempfile::TempDir) { + fixture_with_capacity(64) + } + + /// One core whose queue, and each tenant's in-flight cap, hold `capacity` + /// requests. + fn fixture_with_capacity( + capacity: usize, + ) -> (Arc, CoreChannelDataSide, tempfile::TempDir) { let directory = tempfile::tempdir().expect("temporary WAL directory"); let wal = Arc::new( WalManager::open_for_testing(&directory.path().join("autocommit.wal")) .expect("test WAL"), ); - let (dispatcher, mut sides) = Dispatcher::new(1, 64); + let (dispatcher, mut sides) = Dispatcher::new(1, capacity); let side = sides.pop().expect("one data side"); let state = SharedState::new(dispatcher, wal).expect("shared state"); (state, side, directory) } + /// The `CREATE ARRAY` task for the test array: one Int64 dimension and + /// one Int64 attribute. + fn array_create_task(tenant_id: TenantId) -> PhysicalTask { + use nodedb_array::schema::ArraySchemaBuilder; + use nodedb_array::schema::attr_spec::{AttrSpec, AttrType}; + use nodedb_array::schema::dim_spec::{DimSpec, DimType}; + use nodedb_array::types::domain::{Domain, DomainBound}; + + let schema = ArraySchemaBuilder::new(ARRAY) + .dim(DimSpec::new( + "x", + DimType::Int64, + Domain::new(DomainBound::Int64(0), DomainBound::Int64(15)), + )) + .attr(AttrSpec::new("v", AttrType::Int64, true)) + .tile_extents(vec![4]) + .build() + .expect("build test array schema"); + PhysicalTask { + tenant_id, + database_id: DatabaseId::DEFAULT, + vshard_id: nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, ARRAY).vshard(), + plan: crate::bridge::envelope::PhysicalPlan::Array(ArrayOp::OpenArray { + array_id: ArrayId::in_database(tenant_id, DatabaseId::DEFAULT, ARRAY), + schema_msgpack: zerompk::to_msgpack_vec(&schema).expect("encode schema"), + schema_hash: 0xA11CE, + prefix_bits: 8, + audit_retain_ms: None, + minimum_audit_retain_ms: None, + }), + post_set_op: PostSetOp::None, + txn_id: None, + } + } + fn array_put_task(tenant_id: TenantId) -> PhysicalTask { // An empty cell batch is a valid encoding; what this exercises is the // durability handling of the plan shape, not the cells. @@ -467,27 +538,29 @@ mod tests { cells_msgpack: zerompk::to_msgpack_vec(&cells).expect("encode cells"), wal_lsn: 0, provenance: None, + vshard_id: nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, ARRAY) + .vshard() + .as_u32(), }), post_set_op: PostSetOp::None, txn_id: None, } } - /// Answer one request, returning the plan's stamped LSN to the caller. - async fn respond_once_capturing_lsn( + /// Answer every request with `Ok`, the Raft applies of the one-node + /// cluster included. Records the stamped LSN of each array put. + async fn answer_capturing_array_lsn( state: Arc, mut side: CoreChannelDataSide, stamped: Arc>>, ) { - let deadline = Instant::now() + Duration::from_secs(5); - let mut handled = false; - while !handled && Instant::now() < deadline { - if let Ok(request) = side.request_rx.try_pop() { + loop { + while let Ok(request) = side.request_rx.try_pop() { if let crate::bridge::envelope::PhysicalPlan::Array(ArrayOp::Put { wal_lsn, .. }) = &request.inner.plan { - *stamped.lock().expect("stamped lock") = Some(*wal_lsn); + *stamped.lock().unwrap_or_else(|p| p.into_inner()) = Some(*wal_lsn); } side.response_tx .try_push(BridgeResponse { @@ -505,25 +578,37 @@ mod tests { }, }) .expect("fake data-plane response queue has capacity"); - handled = true; } state.poll_and_route_responses(); tokio::task::yield_now().await; } - assert!(handled, "fake data plane received the dispatched request"); - state.poll_and_route_responses(); } - /// The array sync inbound path acks its peer off this dispatch and nothing - /// upstream appends a redo for it, so the funnel must own the record: mint - /// it, stamp it into the plan (the array engine versions its tiles from the - /// LSN carried there, and replay stamps the same version off the record - /// header — a zero would make the two disagree), and hold the reply behind - /// the durable-at-ack barrier. - #[tokio::test] + /// A synced array put applies through its replicated entry, and nothing + /// upstream appends a redo for it, so the apply's funnel must own the + /// record: mint it, stamp it into the plan (the array engine versions its + /// tiles from the LSN carried there, and replay stamps the same version + /// off the record header — a zero will make the two disagree), and hold + /// the reply behind the durable-at-ack barrier. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn an_autocommit_write_mints_stamps_and_fsyncs_its_own_redo() { - let (state, side, _directory) = fixture(); + let stamped = Arc::new(std::sync::Mutex::new(None)); + let core_stamped = Arc::clone(&stamped); + let cluster = crate::control::cluster::test_one_node::boot_with_core( + |_| {}, + move |state, side| tokio::spawn(answer_capturing_array_lsn(state, side, core_stamped)), + ) + .await; + let state = Arc::clone(&cluster.state); let tenant_id = TenantId::new(1); + // The replicated apply routes a cell write by the array's catalog + // incarnation, and an array absent from the catalog is superseded. + crate::control::array_catalog::ddl::run_trusted_array_ddl( + &state, + array_create_task(tenant_id), + ) + .await + .expect("create the test array"); let task = array_put_task(tenant_id); let identity = crate::control::security::identity::AuthenticatedIdentity::new_internal_service( @@ -557,24 +642,23 @@ mod tests { } }; - let stamped = Arc::new(std::sync::Mutex::new(None)); - let responder = tokio::spawn(respond_once_capturing_lsn( - Arc::clone(&state), - side, - Arc::clone(&stamped), - )); - let response = dispatch_authorized_autocommit_write_with_source( - &state, - checked, - crate::types::TraceId::ZERO, - crate::event::EventSource::CrdtSync, - ) - .await - .expect("autocommit array write succeeds"); - responder.await.expect("responder completes"); + // An array put yields change events, so it applies through its + // replicated entry: the apply's funnel mints and stamps the redo. + let response = + crate::control::server::dispatch_utils::dispatch_authorized_durable_write_with_source( + &state, + checked, + crate::types::TraceId::ZERO, + crate::event::EventSource::CrdtSync, + ) + .await + .expect("replicated array write succeeds"); assert_eq!(response.status, Status::Ok); - let stamped = stamped.lock().expect("stamped lock").expect("an array put"); + let stamped = stamped + .lock() + .unwrap_or_else(|p| p.into_inner()) + .expect("the core saw an array put"); assert!( stamped > 0, "the plan the Data Plane executes must carry the minted LSN, not a zero" @@ -583,6 +667,8 @@ mod tests { state.wal.durable_through() >= stamped, "the minted redo must be fsync-durable before the write is acknowledged" ); + drop(state); + cluster.shutdown().await; } // --- Caller records under the outcome floor --- @@ -594,7 +680,7 @@ mod tests { nodedb_physical::physical_plan::DocumentOp::PointGet { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "users"), document_id: "u1".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: None, pk_bytes: Vec::new(), rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, @@ -603,8 +689,8 @@ mod tests { ) } - fn minted_record(state: &SharedState) -> (super::MintedRecords, Lsn) { - let minted = super::MintedRecords::open(&state.outcome_floor); + fn minted_record(state: &SharedState) -> (MintedRecords, Lsn) { + let minted = MintedRecords::open(&state.outcome_floor); let lsn = minted .appender(&state.wal, crate::wal::manager::NO_APPLY_KEY) .with_event_source(crate::event::EventSource::User) @@ -618,7 +704,7 @@ mod tests { (minted, lsn) } - fn write_with(minted: super::MintedRecords, lsn: Lsn) -> super::WriteDispatch { + fn write_with(minted: MintedRecords, lsn: Lsn) -> super::WriteDispatch { super::WriteDispatch { tenant_id: TenantId::new(1), database_id: DatabaseId::DEFAULT, @@ -700,6 +786,37 @@ mod tests { assert_eq!(state.outcome_floor.leaked_windows(), 0); } + /// A row write whose caller appended its records outside the funnel is + /// refused, and the records are cancelled: a row write's records are + /// appended inside the funnel, under its vShard's write-order fence. + #[tokio::test] + async fn a_row_write_the_caller_journalled_is_refused_and_cancelled() { + let (state, _side, _directory) = fixture(); + let (minted, lsn) = minted_record(&state); + let mut write = write_with(minted, lsn); + write.plan = crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::PointPut { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "users"), + document_id: "u1".into(), + value: Vec::new(), + surrogate: nodedb_types::Surrogate::new(1), + pk_bytes: Vec::new(), + returning: None, + rls_filters: Vec::new(), + resolved_sum_targets: Vec::new(), + }, + ); + + let result = super::dispatch_trusted_internal_write_to_data_plane(&state, write).await; + + assert!( + matches!(result, Err(crate::Error::Internal { .. })), + "a row write the caller journalled must be refused with an internal error" + ); + assert!(!replayed(&state).contains(&lsn.as_u64())); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + } + #[tokio::test] async fn a_refusal_that_applied_nothing_cancels_the_callers_records() { let (state, side, _directory) = fixture(); @@ -748,17 +865,24 @@ mod tests { assert_eq!(state.outcome_floor.leaked_windows(), 0); } - /// A point write the admission gate serializes on its key. - fn incr_plan() -> crate::bridge::envelope::PhysicalPlan { - crate::bridge::envelope::PhysicalPlan::Kv(nodedb_physical::physical_plan::KvOp::Incr { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "counters"), - key: b"k1".to_vec(), - delta: 1, - ttl_ms: 0, - surrogate: nodedb_types::Surrogate::new(1), - rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), - shape: nodedb_physical::physical_plan::KvCounterShape::Raw, - }) + /// A point write the admission gate serializes on its key. A vector index + /// insert yields no change event, so the funnel admits it on a route no + /// other replica applies. + fn vector_insert_plan() -> crate::bridge::envelope::PhysicalPlan { + crate::bridge::envelope::PhysicalPlan::Vector( + nodedb_physical::physical_plan::VectorOp::Insert { + collection: nodedb_types::QualifiedCollection::new( + DatabaseId::DEFAULT, + "embeddings", + ), + vector: vec![1.0, 0.0], + dim: 2, + field_name: String::new(), + surrogate: nodedb_types::Surrogate::new(1), + pk_bytes: None, + provenance: None, + }, + ) } /// Wait until the floor passes `lsn`, routing responses meanwhile. @@ -775,7 +899,7 @@ mod tests { async fn a_caller_dropped_while_waiting_for_admission_cancels_its_records() { let (state, _side, _directory) = fixture(); let (minted, lsn) = minted_record(&state); - let plan = incr_plan(); + let plan = vector_insert_plan(); let (_, keys) = crate::control::server::shared::write_admission::lock_keys::plan_lock_keys(&plan) .expect("a point write has a lock key"); @@ -882,21 +1006,47 @@ mod tests { /// A committed proposal refused for good is refused on every replica, so /// its abort marker carries the proposal key and a redelivered copy finds /// the refusal in the ledger rebuilt after a restart. + /// A KV put of `key` with surrogate `surrogate`. + fn kv_put_plan(key: &[u8], surrogate: u32) -> crate::bridge::envelope::PhysicalPlan { + crate::bridge::envelope::PhysicalPlan::Kv(nodedb_physical::physical_plan::KvOp::Put { + collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), + key: key.to_vec(), + value: b"v1".to_vec(), + ttl_ms: 0, + surrogate: nodedb_types::Surrogate::new(surrogate), + returning: None, + rls_filters: Vec::new(), + provenance: None, + }) + } + + /// A committed write the funnel appends under `apply_key` and enqueues + /// without the admission gate. + fn ordered_write(plan: crate::bridge::envelope::PhysicalPlan, apply_key: u64) -> SubmitWrite { + SubmitWrite { + tenant_id: TenantId::new(1), + database_id: DatabaseId::DEFAULT, + vshard_id: VShardId::new(0), + plan, + trace_id: crate::types::TraceId::ZERO, + event_source: crate::event::EventSource::User, + txn_id: None, + user_id: None, + durability: super::WalDurability::AppendHere { + now_override: None, + apply_key, + commit_hlc: None, + change_position: None, + }, + ordering: super::WriteOrdering::AlreadyOrdered, + change_feed: super::ChangeFeedOwner::Unowned, + } + } + #[tokio::test] async fn a_final_refusal_of_a_keyed_proposal_is_its_ledger_outcome() { const KEY: u64 = 0xC0FF_EE01; let (state, side, _directory) = fixture(); - let plan = - crate::bridge::envelope::PhysicalPlan::Kv(nodedb_physical::physical_plan::KvOp::Put { - collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, "cache"), - key: b"k1".to_vec(), - value: b"v1".to_vec(), - ttl_ms: 0, - surrogate: nodedb_types::Surrogate::new(1), - returning: None, - rls_filters: Vec::new(), - provenance: None, - }); let responder = tokio::spawn(respond_once_with( Arc::clone(&state), side, @@ -907,28 +1057,9 @@ mod tests { }), )); - let outcome = super::submit_write( - &state, - super::SubmitWrite { - tenant_id: TenantId::new(1), - database_id: DatabaseId::DEFAULT, - vshard_id: VShardId::new(0), - plan, - trace_id: crate::types::TraceId::ZERO, - event_source: crate::event::EventSource::User, - txn_id: None, - user_id: None, - durability: super::WalDurability::AppendHere { - now_override: None, - apply_key: KEY, - commit_hlc: None, - }, - ordering: super::WriteOrdering::AlreadyOrdered, - change_feed: super::ChangeFeedOwner::Unowned, - }, - ) - .await - .expect("the refusal is a response"); + let outcome = super::submit_write(&state, ordered_write(kv_put_plan(b"k1", 1), KEY)) + .await + .expect("the refusal is a response"); responder.await.expect("responder completes"); assert_eq!(outcome.response.status, Status::Error); @@ -942,4 +1073,119 @@ mod tests { "the refusal's marker names the proposal" ); } + + /// The enqueue returned, and the `PendingWrite` is dropped before its + /// response phase runs. The task the enqueue handed the records to still + /// settles them from the answer. + #[tokio::test] + async fn a_pending_write_dropped_before_its_response_phase_closes_its_records() { + let (state, side, _directory) = fixture(); + let pending = enqueue_write(&state, ordered_write(kv_put_plan(b"k1", 1), 0)) + .await + .expect("the write is enqueued"); + let lsn = Lsn::new( + replayed(&state) + .into_iter() + .max() + .expect("the funnel appended the record"), + ); + assert!( + state.outcome_floor.floor() < lsn, + "the core holds the record" + ); + + drop(pending); + respond_once_with(Arc::clone(&state), side, Status::Ok, None).await; + + assert!( + floor_passes(&state, lsn).await, + "the answer closed the window" + ); + assert!( + replayed(&state).contains(&lsn.as_u64()), + "no marker names it" + ); + assert_eq!(state.outcome_floor.leaked_windows(), 0); + assert_eq!(state.outcome_floor.held_windows(), 0); + } + + /// A committed write waits for a free dispatch slot. Its caller is dropped + /// during that wait: no core holds the write, so its record is cancelled + /// and its response-tracker entry is removed. + #[tokio::test] + async fn a_write_dropped_while_waiting_for_capacity_leaves_nothing_open() { + let (state, _side, _directory) = fixture_with_capacity(2); + let first = enqueue_write(&state, ordered_write(kv_put_plan(b"k1", 1), 0)) + .await + .expect("the first write is enqueued"); + let second = enqueue_write(&state, ordered_write(kv_put_plan(b"k2", 2), 0)) + .await + .expect("the second write is enqueued"); + let in_flight = state.tracker.in_flight(); + let leaked = state.outcome_floor.leaked_windows(); + let lsn = state.wal.next_lsn(); + + let waited = tokio::time::timeout( + Duration::from_millis(50), + enqueue_write(&state, ordered_write(kv_put_plan(b"k3", 3), 0)), + ) + .await; + + assert!(waited.is_err(), "the write waits for a free slot"); + assert!(state.wal.next_lsn() > lsn, "the write appended its record"); + assert!( + !replayed(&state).contains(&lsn.as_u64()), + "a marker names it" + ); + assert_eq!( + state.tracker.in_flight(), + in_flight, + "no entry waits for a response that never comes" + ); + assert_eq!(state.outcome_floor.leaked_windows(), leaked); + assert_eq!(state.outcome_floor.held_windows(), 0); + drop((first, second)); + } + + // --- Array DDL never reaches a core through the funnel --- + + /// Array DDL runs through the replicated catalog. The funnel refuses it + /// before any record is minted, any catalog row is written, or any core + /// holds it. + #[tokio::test] + async fn the_funnel_refuses_array_ddl() { + let (state, mut side, _directory) = fixture(); + let array_id = ArrayId::in_database(TenantId::new(1), DatabaseId::DEFAULT, "refused"); + let plan = crate::bridge::envelope::PhysicalPlan::Array(ArrayOp::OpenArray { + array_id: array_id.clone(), + schema_msgpack: vec![0x90], + schema_hash: 7, + prefix_bits: 8, + audit_retain_ms: None, + minimum_audit_retain_ms: None, + }); + let lsn = state.wal.next_lsn(); + + let refused = enqueue_write(&state, ordered_write(plan, 0)).await; + + assert!(refused.is_err(), "array DDL must not enter the funnel"); + assert!(side.request_rx.try_pop().is_err(), "no core holds it"); + assert_eq!(state.wal.next_lsn(), lsn, "no record was minted"); + assert!( + state + .credentials + .catalog() + .get_array_in_database(TenantId::new(1), DatabaseId::DEFAULT, &array_id.name) + .expect("catalog read") + .is_none() + ); + assert!( + state + .array_catalog + .read() + .expect("array catalog") + .lookup_by_id(&array_id) + .is_none() + ); + } } diff --git a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs index fc6d66b3e..699eccf1e 100644 --- a/nodedb/src/control/server/dispatch_utils/durability_barrier.rs +++ b/nodedb/src/control/server/dispatch_utils/durability_barrier.rs @@ -6,9 +6,9 @@ //! write-class plan reaching it with no LSN is NOT by itself a bug: most //! engines have write ops whose durability is owned somewhere other than this //! funnel's WAL append, and those arms are deliberate and documented. A naive -//! "write-class plan with no LSN" assertion would fire on every document -//! INSERT (the row is redb-synchronous-durable) and be switched off within a -//! day, which is strictly worse than no check. +//! "write-class plan with no LSN" assertion will fire on every document +//! UPDATE that matched no row (it journals nothing) and be switched off within +//! a day, which is strictly worse than no check. //! //! What IS an invariant is narrower and checkable: for the engines below, //! EVERY write-class op mints a WAL redo record on this path. If one of them @@ -32,7 +32,7 @@ use nodedb_physical::physical_plan::{GraphOp, MetaOp}; /// to an engine whose every write-class op mints one on this path. /// /// Stays at zero by construction. A non-zero value names a write op that was -/// classified as needing no durable record and is now acknowledged before it +/// classified as needing no durable record and is acknowledged before it /// is recoverable. static WRITES_ACKED_WITHOUT_DURABILITY: AtomicU64 = AtomicU64::new(0); @@ -50,9 +50,9 @@ pub fn writes_acked_without_durability() -> u64 { /// * the plan is not a base-state write at all — reads, control ops, and the /// per-transaction overlay ops (`StageWrite`, savepoint mark / rollback), /// all excluded by [`plan_is_write`]; -/// * `Document` — every document write op is documented as redb-synchronous- -/// durable, and the one restart-fidelity gap (a secondary vector index) is -/// covered by the post-apply write-set redo, not by a forward record; +/// * `Document` — a document write journals the rows its apply decides after +/// apply, from `Response::write_set`, and a write whose apply stores no row +/// (an update or delete that matched nothing) journals nothing at all; /// * `Crdt` — constraint installs are Raft-log-replay durable and /// `RestoreToVersion` only computes a forward delta that a follow-up /// `Apply` logs; @@ -146,7 +146,7 @@ mod tests { key: b"k".to_vec(), value: b"v".to_vec(), ttl_ms: 0, - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), returning: None, rls_filters: Vec::new(), provenance: None, @@ -184,8 +184,8 @@ mod tests { } /// The false-positive guard that decides whether this check is usable at - /// all: document writes are redb-synchronous-durable and legitimately - /// reach the barrier with no LSN, on the hottest write path there is. + /// all: a document write that stores no row journals nothing and + /// legitimately reaches the barrier with no LSN. #[test] fn document_write_is_not_held_to_the_barrier() { let plan = PhysicalPlan::Document(DocumentOp::Truncate { diff --git a/nodedb/src/control/server/dispatch_utils/durable_write.rs b/nodedb/src/control/server/dispatch_utils/durable_write.rs index 4b4390d0a..b86295ec1 100644 --- a/nodedb/src/control/server/dispatch_utils/durable_write.rs +++ b/nodedb/src/control/server/dispatch_utils/durable_write.rs @@ -4,12 +4,15 @@ //! //! A planned autocommit write reaches its engine one of two ways: //! -//! - Cluster mode, replicable write: the write is proposed through Raft. Every -//! replica applies the committed entry through the funnel, which appends the -//! redo record. The proposing node publishes the change event. -//! - Otherwise: the write enters the funnel with `WalDurability::AppendHere`. -//! The funnel appends the redo record under the write-admission guard, inside -//! the write's outcome-floor window. +//! - Replicable write: the write is proposed through Raft. Every replica +//! applies the committed entry through the funnel, which appends the redo +//! record. The proposing node publishes the change event. An edge write +//! runs as a Calvin transaction at that seam instead +//! (`planner::calvin::edge_sequencing`). +//! - A write inside a transaction block, or a plan with no replicated form: +//! the write enters the funnel with `WalDurability::AppendHere`. The funnel +//! appends the redo record under the write-admission guard, inside the +//! write's outcome-floor window. //! //! A write dispatched any other way applies with no WAL record. A crash loses //! it, and no replica sees it. Every caller that holds an autocommit write @@ -26,7 +29,6 @@ use crate::control::wal_replication::{ }; use crate::types::{DatabaseId, Lsn, RequestId, TenantId, TraceId, VShardId}; -use super::change_events::{extract_write_change_set, publish_change_set_with_lsn}; use super::dispatch::{ dispatch_authorized_autocommit_write, dispatch_authorized_to_data_plane, dispatch_autocommit_write, @@ -65,24 +67,22 @@ pub(crate) async fn dispatch_durable_autocommit_write( /// Same routing as [`dispatch_durable_autocommit_write`], for a task that /// passed the clone-write gate and authorization. /// -/// In cluster mode a write whose RLS write policy is decided per row cannot -/// be proposed bare: a follower has no writing identity to decide the policy -/// against. It resolves to a concrete row set first, on this node, and the -/// resolved write is proposed (`control::write_resolve`), as the planned -/// pgwire and native writes do. +/// A write whose RLS write policy is decided per row cannot be proposed +/// bare: a follower has no writing identity to decide the policy against. It +/// resolves to a concrete row set first, on this node, and the resolved write +/// is proposed (`control::write_resolve`), as the planned pgwire and native +/// writes do. pub(crate) async fn dispatch_authorized_durable_write( shared: &SharedState, checked: CloneCheckedTask, trace_id: TraceId, ) -> crate::Result { if checked.txn_id().is_none() - && shared.async_raft_proposer().is_some() && let Some(resolver) = crate::control::write_resolve::resolver_for_plan(checked.plan()) { + let (authorized, _lease) = checked.into_parts(); return crate::control::write_resolve::run_authorized_write_resolve( - shared, - checked.into_authorized(), - resolver, + shared, authorized, resolver, ) .await; } @@ -112,9 +112,9 @@ pub(crate) async fn dispatch_authorized_durable_write( /// replica stamps the source on the write's events, so a synced write does /// not re-fire AFTER triggers. /// -/// A clustered write whose RLS write policy must resolve against current rows -/// before it is proposed is refused: the resolved write carries none of the -/// plan's sync provenance, so the idempotency gate could not run. +/// A write whose RLS write policy must resolve against current rows before it +/// is proposed is refused: the resolved write carries none of the plan's sync +/// provenance, so the idempotency gate cannot run. pub(crate) async fn dispatch_authorized_durable_write_with_source( shared: &SharedState, checked: CloneCheckedTask, @@ -122,7 +122,6 @@ pub(crate) async fn dispatch_authorized_durable_write_with_source( event_source: crate::event::EventSource, ) -> crate::Result { if checked.txn_id().is_none() - && shared.async_raft_proposer().is_some() && crate::control::write_resolve::resolver_for_plan(checked.plan()).is_some() { return Err(crate::Error::PlanError { @@ -160,8 +159,8 @@ pub(crate) async fn dispatch_authorized_durable_write_with_source( /// Dispatch one authorized task by its class: a write on the durable route, /// anything else on the read route. /// -/// For a transport fallback that dispatches reads and writes through one call -/// site with no gateway installed. +/// For a call site that carries both reads and writes on this node's cores: +/// the transaction meta-ops and the DML statement dispatch. pub(crate) async fn dispatch_authorized_task_by_class( shared: &SharedState, checked: CloneCheckedTask, @@ -181,8 +180,8 @@ struct WriteTarget { vshard_id: VShardId, } -/// Propose `plan` through Raft when this node runs a proposer and the plan -/// encodes to a replicated entry. `None` means the write takes the local route. +/// Propose `plan` through Raft when the plan encodes to a replicated entry. +/// `None` means the plan has no replicated form and takes the local route. /// /// A Data-Plane verdict comes back as `Err(Error::DataPlane(code))`, the shape /// the pgwire replicated path returns. @@ -192,29 +191,31 @@ async fn propose_if_replicable( plan: &PhysicalPlan, event_source: crate::event::EventSource, ) -> crate::Result> { - let Some(proposer) = shared.async_raft_proposer() else { - return Ok(None); - }; + let proposer = shared.async_raft_proposer()?; let WriteTarget { tenant_id, database_id, vshard_id, } = target; + // The entry carries resolved rows: a timeseries ingest resolves here, + // on the proposer, before the entry exists. + let resolved = crate::control::write_resolve::resolve_for_log( + shared, + crate::control::write_resolve::WriteResolveContext { + tenant_id, + database_id, + }, + vshard_id, + plan, + ) + .await?; + let plan = resolved.as_ref().unwrap_or(plan); let replicable = ReplicableWrite::decide_for_replication(plan)?; let Some(entry) = to_replicated_entry(tenant_id, database_id, vshard_id, &replicable)? else { return Ok(None); }; let entry = entry.with_event_source(event_source); let (payload, write_version) = propose_replicated_entry(shared, proposer, entry).await?; - // Replicas apply with `ChangeFeedOwner::Unowned`. The proposing node - // handled the write once, so it publishes the change event. - publish_change_set_with_lsn( - shared, - tenant_id, - database_id, - extract_write_change_set(plan, tenant_id), - write_version, - ); Ok(Some(replicated_write_response( shared, payload, diff --git a/nodedb/src/control/server/dispatch_utils/minted/mod.rs b/nodedb/src/control/server/dispatch_utils/minted/mod.rs index 847da8f72..e89198a84 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/mod.rs @@ -6,6 +6,8 @@ mod owned; mod records; mod resolve; +mod sent; -pub(crate) use owned::{Collect, OwnedResponse, OwnedWait, await_response_owned}; +pub(crate) use owned::{Collect, OwnedReport, OwnedResponse, OwnedWait, spawn_owned_wait}; pub(crate) use records::{MintedRecords, RecordOwner}; +pub(crate) use sent::SentRecords; diff --git a/nodedb/src/control/server/dispatch_utils/minted/owned.rs b/nodedb/src/control/server/dispatch_utils/minted/owned.rs index df10d0e91..fa40cf37f 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/owned.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/owned.rs @@ -4,7 +4,7 @@ //! own. //! //! Once a write is enqueued, its records close from the core's final -//! response. A caller future dropped mid-wait would drop the records with it +//! response. A caller future dropped mid-wait will drop the records with it //! and leak their window. The wait and the close run in a spawned task //! instead. The caller awaits what the task reports, and a dropped caller //! leaves the task running until the records close. @@ -27,9 +27,6 @@ use super::resolve::{resolve_at_final, resolve_on_response}; pub(crate) enum Collect { /// Every frame up to the final one, merged, under a byte budget. Merged { max_result_bytes: usize }, - /// The first frame. A partial first frame leaves the task closing the - /// records from the final one. - First, } /// Where the owned task waits, and how the records close. @@ -42,6 +39,9 @@ pub(crate) struct OwnedWait { /// The instant the caller stops waiting. pub deadline: Instant, pub collect: Collect, + /// Where the final response goes when it arrives after the caller + /// stopped waiting. + pub late: Option>, } /// What the owned task reports to the caller. @@ -64,20 +64,38 @@ pub(crate) enum OwnedResponse { ChannelClosed, } -/// Wait for `rx`'s response in a spawned task that owns `minted`, and -/// return what it reports. -pub(crate) async fn await_response_owned( +/// What a task that owns a dispatched write's records reports. Dropping it +/// leaves the task to close the records. +pub(crate) struct OwnedReport { + rx: oneshot::Receiver, +} + +impl OwnedReport { + /// Wait for what the task reports. + pub(crate) async fn recv(self) -> crate::Result { + self.rx.await.map_err(|_| crate::Error::Internal { + detail: "the task waiting for a dispatched write's response ended without \ + reporting" + .into(), + }) + } +} + +/// Mark `minted` sent and hand it to a spawned task that waits for `rx`'s +/// response and closes the records from it. +/// +/// Call it in the same synchronous step as the enqueue of the request that +/// carries the records. No caller future owns them after that, so no drop +/// at a later await leaks their window. +pub(crate) fn spawn_owned_wait( wait: OwnedWait, rx: ResponseReceiver, minted: MintedRecords, -) -> crate::Result { +) -> OwnedReport { + minted.mark_sent(); let (report_tx, report_rx) = oneshot::channel(); tokio::spawn(wait_and_close(wait, rx, minted, report_tx)); - report_rx.await.map_err(|_| crate::Error::Internal { - detail: "the task waiting for a dispatched write's response ended without \ - reporting" - .into(), - }) + OwnedReport { rx: report_rx } } async fn wait_and_close( @@ -92,19 +110,19 @@ async fn wait_and_close( final_refusal_key, deadline, collect, + late, } = wait; + let forward = |final_response: Option| { + if let (Some(late), Some(response)) = (late, final_response) { + let _ = late.send(response); + } + }; let until = tokio::time::Instant::from_std(deadline); let collected = match collect { Collect::Merged { max_result_bytes } => { tokio::time::timeout_at(until, collect_bounded_response(&mut rx, max_result_bytes)) .await } - Collect::First => { - tokio::time::timeout_at(until, async { - rx.recv().await.ok_or(DispatchCollectError::ChannelClosed) - }) - .await - } }; match collected { Ok(Ok(response)) if response.partial => { @@ -112,7 +130,7 @@ async fn wait_and_close( response, closed: Ok(()), }); - resolve_at_final(&wal, owner, final_refusal_key, rx, minted).await; + forward(resolve_at_final(&wal, owner, final_refusal_key, rx, minted).await); } Ok(Ok(response)) => { let closed = @@ -121,7 +139,7 @@ async fn wait_and_close( } Ok(Err(DispatchCollectError::OverBudget { bytes })) => { let _ = report.send(OwnedResponse::OverBudget { bytes }); - resolve_at_final(&wal, owner, final_refusal_key, rx, minted).await; + forward(resolve_at_final(&wal, owner, final_refusal_key, rx, minted).await); } Ok(Err(DispatchCollectError::ChannelClosed)) => { minted.hold(); @@ -129,7 +147,7 @@ async fn wait_and_close( } Err(_) => { let _ = report.send(OwnedResponse::DeadlineExceeded); - resolve_at_final(&wal, owner, final_refusal_key, rx, minted).await; + forward(resolve_at_final(&wal, owner, final_refusal_key, rx, minted).await); } } } @@ -195,9 +213,40 @@ mod tests { collect: Collect::Merged { max_result_bytes: 1 << 20, }, + late: None, } } + /// A final response that arrives after the deadline goes to the late + /// receiver, once the records closed from it. + #[tokio::test] + async fn a_late_final_response_is_forwarded() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let tracker = RequestTracker::new(); + let rx = tracker.register(RequestId::new(3)); + let (minted, _lsn) = minted_record(&wal, &floor); + let (late_tx, late_rx) = oneshot::channel(); + let report = spawn_owned_wait( + OwnedWait { + late: Some(late_tx), + ..wait(&wal, Instant::now() + Duration::from_millis(20)) + }, + rx, + minted, + ); + let outcome = report.recv().await.expect("report"); + assert!(matches!(outcome, OwnedResponse::DeadlineExceeded)); + + let mut ok = refusal(3); + ok.status = Status::Ok; + ok.error_code = None; + assert!(tracker.complete(ok)); + let late = late_rx.await.expect("the late response is forwarded"); + assert_eq!(late.status, Status::Ok); + } + /// The caller future is dropped while it waits. The task still closes /// the records from the refusal that arrives afterwards. #[tokio::test] @@ -212,7 +261,7 @@ mod tests { let caller = tokio::time::timeout( Duration::from_millis(10), - await_response_owned(wait(&wal, deadline), rx, minted), + spawn_owned_wait(wait(&wal, deadline), rx, minted).recv(), ) .await; assert!( @@ -243,7 +292,8 @@ mod tests { let tracker = RequestTracker::new(); let rx = tracker.register(RequestId::new(2)); - let outcome = await_response_owned(wait(&wal, Instant::now()), rx, minted) + let outcome = spawn_owned_wait(wait(&wal, Instant::now()), rx, minted) + .recv() .await .expect("report"); assert!(matches!(outcome, OwnedResponse::DeadlineExceeded)); @@ -258,4 +308,42 @@ mod tests { } assert!(floor.floor() >= lsn); } + + /// The report is dropped before anyone polls it. The task owns the + /// records from the spawn, so the answer still settles them, once. + #[tokio::test] + async fn a_report_dropped_unpolled_still_settles_the_records_once() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (minted, lsn) = minted_record(&wal, &floor); + let tracker = RequestTracker::new(); + let rx = tracker.register(RequestId::new(3)); + let deadline = Instant::now() + Duration::from_secs(30); + + drop(spawn_owned_wait(wait(&wal, deadline), rx, minted)); + assert!(floor.floor() < lsn, "the window holds while the core works"); + + let mut applied = refusal(3); + applied.status = Status::Ok; + applied.error_code = None; + assert!(tracker.complete(applied)); + for _ in 0..200 { + if floor.floor() >= lsn { + break; + } + tokio::time::sleep(Duration::from_millis(5)).await; + } + assert!(floor.floor() >= lsn, "the answer settled the records"); + wal.sync().expect("sync"); + let replayed: Vec = wal + .replay() + .expect("replay") + .iter() + .map(|record| record.header.lsn) + .collect(); + assert!(replayed.contains(&lsn.as_u64()), "no marker names it"); + assert_eq!(floor.leaked_windows(), 0); + assert_eq!(floor.held_windows(), 0); + } } diff --git a/nodedb/src/control/server/dispatch_utils/minted/records.rs b/nodedb/src/control/server/dispatch_utils/minted/records.rs index d38746087..d0fd26aa8 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/records.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/records.rs @@ -82,7 +82,7 @@ impl MintedRecords { /// Hold an existing record at `lsn` that is sent to a core again. /// Refused, with the reason, when its outcome is final or a live or held - /// window carries it to one: a second apply would land below the floor, + /// window carries it to one: a second apply will land below the floor, /// or apply the record twice. pub(crate) fn resend(floor: &Arc, lsn: Lsn) -> Result { let window = floor.open_existing(lsn)?; @@ -141,13 +141,9 @@ impl MintedRecords { self.sent.store(true, Ordering::Release); } - /// The highest appended or resent LSN, or `None` when there is none. - pub(crate) fn highest(&self) -> Option { - self.recorded() - .iter() - .map(|record| record.lsn) - .chain(self.resent) - .max() + /// The highest LSN appended under this window, `None` before any append. + pub(crate) fn last_lsn(&self) -> Option { + self.recorded().iter().map(|record| record.lsn).max() } /// Every appended LSN, in append order. @@ -466,7 +462,6 @@ mod tests { let first = append(&wal, &minted, b"a"); let second = append(&wal, &minted, b"b"); assert_eq!(minted.lsns(), vec![first, second]); - assert_eq!(minted.highest(), Some(second)); minted.cancel(&wal, owner(), 0).await.expect("cancel"); assert!( wal.durable_through() > second.as_u64(), diff --git a/nodedb/src/control/server/dispatch_utils/minted/resolve.rs b/nodedb/src/control/server/dispatch_utils/minted/resolve.rs index 7bbc7104e..32d400fcf 100644 --- a/nodedb/src/control/server/dispatch_utils/minted/resolve.rs +++ b/nodedb/src/control/server/dispatch_utils/minted/resolve.rs @@ -66,14 +66,14 @@ pub(crate) async fn resolve_on_response( /// A refusal that arrives after the caller timed out still cancels its /// records, so a restart cannot apply a write the caller never saw applied. A /// channel that closes before a final response leaves the outcome unknown: -/// the window holds. +/// the window holds. Returns the final response, when one arrived. pub(crate) async fn resolve_at_final( wal: &Arc, owner: RecordOwner, final_refusal_key: u64, mut rx: crate::control::ResponseReceiver, minted: MintedRecords, -) { +) -> Option { loop { match rx.recv().await { Some(response) if response.partial => continue, @@ -86,11 +86,11 @@ pub(crate) async fn resolve_at_final( "a late refusal's abort marker failed; its records stay replayable" ); } - return; + return Some(response); } None => { minted.hold(); - return; + return None; } } } diff --git a/nodedb/src/control/server/dispatch_utils/minted/sent.rs b/nodedb/src/control/server/dispatch_utils/minted/sent.rs new file mode 100644 index 000000000..131ac4e58 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/minted/sent.rs @@ -0,0 +1,114 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Records a task sent to cores and closes from their answers itself. +//! +//! [`super::spawn_owned_wait`] closes one request's records from its final +//! response. A task that fans one record out to several cores and decides the +//! close from all their answers owns the records instead. Once they are sent, +//! a drop of that task (an abort, or a runtime shutdown) leaves their outcome +//! unknown. [`SentRecords`] then holds the window: restart replay reaches the +//! records, and the floor files no leak. + +use super::records::MintedRecords; + +/// Records sent to at least one core, closed by the task that holds them. +#[must_use = "sent records hold the outcome floor until they settle or hold"] +pub(crate) struct SentRecords { + /// `None` once a close took the records. + records: Option, +} + +impl SentRecords { + /// Mark `records` sent. Call it in the same synchronous step as the + /// enqueue of the requests that carry them. + pub(crate) fn sent(records: MintedRecords) -> Self { + records.mark_sent(); + Self { + records: Some(records), + } + } + + /// Every core's outcome is final. + pub(crate) fn settle(mut self) { + if let Some(records) = self.records.take() { + records.settle(); + } + } + + /// Some core has no final outcome in this process. + pub(crate) fn hold(mut self) { + if let Some(records) = self.records.take() { + records.hold(); + } + } +} + +impl Drop for SentRecords { + fn drop(&mut self) { + if let Some(records) = self.records.take() { + records.hold(); + } + } +} + +#[cfg(test)] +mod tests { + use std::sync::Arc; + + use super::*; + use crate::bridge::dispatch::OutcomeFloor; + use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + use crate::wal::WalManager; + use crate::wal::manager::NO_APPLY_KEY; + + fn sent_record(wal: &Arc, floor: &Arc) -> (SentRecords, Lsn) { + let minted = MintedRecords::open(floor); + let lsn = minted + .appender(wal, NO_APPLY_KEY) + .with_event_source(crate::event::EventSource::User) + .append_put( + TenantId::new(1), + VShardId::new(0), + DatabaseId::DEFAULT, + b"x", + ) + .expect("append"); + (SentRecords::sent(minted), lsn) + } + + /// The task that owns sent records is aborted mid-wait. The window + /// holds: nothing leaks, and the floor stays below the record. + #[tokio::test] + async fn an_aborted_owner_holds_its_sent_records() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (sent, lsn) = sent_record(&wal, &floor); + + let owner = tokio::spawn(async move { + std::future::pending::<()>().await; + sent.settle(); + }); + tokio::task::yield_now().await; + owner.abort(); + assert!(owner.await.is_err(), "the owner was aborted"); + + assert!(floor.floor() < lsn, "restart replay must reach the record"); + assert_eq!(floor.leaked_windows(), 0); + assert_eq!(floor.held_windows(), 1); + } + + #[test] + fn settled_sent_records_release_the_floor_once() { + let dir = tempfile::tempdir().expect("tempdir"); + let wal = Arc::new(WalManager::open_for_testing(&dir.path().join("wal")).expect("wal")); + let floor = OutcomeFloor::new(); + let (sent, lsn) = sent_record(&wal, &floor); + + sent.settle(); + + assert!(floor.floor() >= lsn); + assert_eq!(floor.leaked_windows(), 0); + assert_eq!(floor.held_windows(), 0); + } +} diff --git a/nodedb/src/control/server/dispatch_utils/mod.rs b/nodedb/src/control/server/dispatch_utils/mod.rs index 224638b30..6582a3fe4 100644 --- a/nodedb/src/control/server/dispatch_utils/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/mod.rs @@ -7,18 +7,19 @@ mod dispatch; mod durability_barrier; mod durable_write; mod minted; +mod owner_read; mod submit_write; mod types; +mod unlogged_dispatch; mod write_abort; pub(crate) use change_events::{ - WriteChangeSet, extract_write_change_set, publish_change_set_with_lsn, - publish_cluster_array_change_events, publish_origin_change_events, + CalvinApply, WriteChangeSet, publish_calvin_change_sets, publish_settled_changes, + redo_change_set, }; pub use dispatch::{dispatch_authorized_autocommit_write, dispatch_authorized_to_data_plane}; pub(crate) use dispatch::{ - dispatch_authorized_autocommit_write_with_source, dispatch_authorized_minted_to_data_plane, - dispatch_autocommit_write, dispatch_to_data_plane, dispatch_to_data_plane_with_txn, + dispatch_autocommit_write, dispatch_replayed_write_to_data_plane, dispatch_to_data_plane, dispatch_trusted_internal_write_to_data_plane, }; pub use durability_barrier::writes_acked_without_durability; @@ -26,14 +27,19 @@ pub(crate) use durable_write::{ dispatch_authorized_durable_write, dispatch_authorized_durable_write_with_source, dispatch_authorized_task_by_class, dispatch_durable_autocommit_write, }; -pub(crate) use minted::{ - Collect, MintedRecords, OwnedResponse, OwnedWait, RecordOwner, await_response_owned, +pub(crate) use minted::{MintedRecords, RecordOwner, SentRecords}; +pub(crate) use owner_read::{ + OwnedRead, OwnedReadScope, ReadPlacement, not_found_response, ok_payload_response, + owner_response, prepare_local_pass, read_placement, route_owned_read, }; pub(crate) use submit_write::{ ChangeFeedOwner, PendingWrite, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, - enqueue_write, submit_write, + dispatch_when_capacity_frees, enqueue_write, submit_write, }; pub(crate) use types::{AutocommitWrite, WriteDispatch}; +pub(crate) use unlogged_dispatch::{ + dispatch_routed_read_to_data_plane, dispatch_to_data_plane_with_txn, +}; pub(crate) use write_abort::{ error_is_final_refusal, refusal_is_final, write_definitely_not_applied, }; diff --git a/nodedb/src/control/server/dispatch_utils/owner_read.rs b/nodedb/src/control/server/dispatch_utils/owner_read.rs new file mode 100644 index 000000000..c14997490 --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/owner_read.rs @@ -0,0 +1,270 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Serving a read from the group that owns its rows. +//! +//! A read that runs on this node's cores sees only the groups this node +//! replicates. A collection whose home group lives on other nodes has no rows +//! here, so a local read of it returns nothing. [`route_owned_read`] picks +//! where the read runs: +//! +//! - This node replicates the home group: the read runs here. A linearizable +//! read confirms the group first. +//! - A read inside a transaction runs on the group's leader. The +//! transaction's staging overlay lives there +//! (`shared/session/leader_forward.rs`). +//! - Otherwise the read goes through the gateway to the group's leader. The +//! gateway carries the same consistency, and the serving node confirms it. +//! +//! A graph or array plan spreads its rows across every hosted group. It runs +//! here and confirms the groups it reads (`graph_dispatch::read_groups`). + +use futures::future::{BoxFuture, Either, Ready, ready}; + +use crate::bridge::envelope::{ErrorCode, Payload, PhysicalPlan, Response, Status}; +use crate::control::cluster::linearizable_read::{ + confirm_linearizable_read, hosted_groups_of_vshards, statement_read_deadline, +}; +use crate::control::gateway::RouteDecision; +use crate::control::gateway::core::QueryContext; +use crate::control::gateway::live_leaders::resolve_live_decision; +use crate::control::security::identity::{Permission, required_permission}; +use crate::control::server::payload_merge::merge_msgpack_arrays; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, Lsn, RequestId, TenantId, TraceId, TxnId, VShardId}; + +/// Where one read runs. +pub(crate) struct OwnedReadScope { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + /// The vShard that holds the rows the plan reads. + pub vshard_id: VShardId, + pub trace_id: TraceId, + pub txn_id: Option, + /// The read must observe every write committed before it began. + pub linearizable: bool, +} + +/// The outcome of [`route_owned_read`]. +pub(crate) enum OwnedRead { + /// The plan runs on this node's cores. Any confirm already ran. Boxed: + /// a plan is far larger than a response. + Local(Box), + /// The owning group's leader answered the read. + Served(Response), +} + +/// Decide where `plan` runs, and run it at the owner when that is another +/// node. A plan that reads no user rows always runs here. +/// +/// The return type is a named future, never an opaque `impl Future`. The +/// owner path enters the gateway, and the gateway dispatches back through +/// the funnel that awaits this function. An opaque type here puts that cycle +/// in the compiler's layout and `Send` checks, which then overflow or fail. +/// A plan that runs here without a confirm answers without a heap allocation. +pub(crate) fn route_owned_read( + shared: &SharedState, + scope: OwnedReadScope, + plan: PhysicalPlan, +) -> Either>, BoxFuture<'_, crate::Result>> { + if shared.cluster_routing.is_none() || !matches!(required_permission(&plan), Permission::Read) { + return Either::Left(ready(Ok(OwnedRead::Local(Box::new(plan))))); + } + Either::Right(Box::pin(route_cluster_read(shared, scope, plan))) +} + +/// [`route_owned_read`] for a read on a cluster node. +async fn route_cluster_read( + shared: &SharedState, + scope: OwnedReadScope, + plan: PhysicalPlan, +) -> crate::Result { + let deadline = statement_read_deadline(shared); + if nodedb_physical::physical_plan::plan_contains_cluster_partitioned_leaf(&plan) { + if scope.linearizable { + let groups = crate::control::server::graph_dispatch::graph_read_groups( + shared, + scope.database_id, + &plan, + )?; + confirm_linearizable_read(shared, &groups, deadline).await?; + } + return Ok(OwnedRead::Local(Box::new(plan))); + } + match read_placement(shared, scope.vshard_id, scope.txn_id)? { + ReadPlacement::Here(hosted) => { + if scope.linearizable { + confirm_linearizable_read(shared, &hosted, deadline).await?; + } + // The read's versions are this node's WAL positions. + crate::control::server::shared::session::served_reads::note( + scope.vshard_id.as_u32(), + shared.node_id, + ); + Ok(OwnedRead::Local(Box::new(plan))) + } + ReadPlacement::Owner => { + let gateway = shared.installed_gateway()?; + let ctx = QueryContext { + tenant_id: scope.tenant_id, + trace_id: scope.trace_id, + database_id: scope.database_id, + txn_id: scope.txn_id, + linearizable: scope.linearizable, + }; + owner_response(gateway.execute_internal_with_watermarks(&ctx, plan).await) + .map(OwnedRead::Served) + } + } +} + +/// Where a read of `vshard_id`'s rows runs. +pub(crate) enum ReadPlacement { + /// On this node, which replicates these groups of the read. + Here(Vec), + /// On the group's leader, through the gateway. + Owner, +} + +/// Where a read of `vshard_id` runs: here when this node replicates the group +/// (a transaction's read also needs this node to lead it, since the staging +/// overlay lives on the leader), on the owner otherwise. Always here without +/// a cluster. +pub(crate) fn read_placement( + shared: &SharedState, + vshard_id: VShardId, + txn_id: Option, +) -> crate::Result { + if shared.cluster_routing.is_none() { + return Ok(ReadPlacement::Here(Vec::new())); + } + let hosted = hosted_groups_of_vshards(shared, [vshard_id.as_u32()])?; + let serve_here = if txn_id.is_some() { + leads_vshard(shared, vshard_id) + } else { + !hosted.is_empty() + }; + Ok(if serve_here { + ReadPlacement::Here(hosted) + } else { + ReadPlacement::Owner + }) +} + +/// The owner's answer to a read, in the shape a local dispatch returns. +pub(crate) fn owner_response( + outcome: crate::Result, +) -> crate::Result { + match outcome { + Ok((payloads, watermarks, read_version_lsn)) => { + let watermark_lsn = watermarks + .iter() + .map(|(_, lsn)| *lsn) + .max() + .unwrap_or(Lsn::ZERO); + let payload = match payloads.len() { + 0 => Vec::new(), + 1 => payloads.into_iter().next().unwrap_or_default(), + _ => merge_msgpack_arrays(&payloads), + }; + Ok(response( + Status::Ok, + payload, + watermark_lsn, + read_version_lsn, + None, + )) + } + // The local path answers a missing row with a `NotFound` status, not an + // error. The owner's answer keeps that shape. + Err(crate::Error::DataPlane(ErrorCode::NotFound)) => Ok(response( + Status::Error, + Vec::new(), + Lsn::ZERO, + Lsn::ZERO, + Some(ErrorCode::NotFound), + )), + Err(error) => Err(error), + } +} + +/// Prepare a read-only pass that must run on this node's cores. +/// +/// A columnar DML resolve pass (`ResolveDml`) reads rows but needs the +/// `Write` grant, so the gateway will carry it as a write. It runs +/// here, which is sound only where [`read_placement`] places the read: on a +/// node that replicates `vshard_id`'s group, and that leads it for a +/// transaction's pass. Otherwise the pass is refused with `NotLeader` naming +/// the leader, and no pass reads a replica that lacks its rows or overlay. A +/// served pass confirms the group first. +pub(crate) async fn prepare_local_pass( + shared: &SharedState, + vshard_id: VShardId, + txn_id: Option, +) -> crate::Result<()> { + match read_placement(shared, vshard_id, txn_id)? { + ReadPlacement::Here(hosted) => { + confirm_linearizable_read(shared, &hosted, statement_read_deadline(shared)).await + } + ReadPlacement::Owner => { + let (leader_node, leader_term) = shared + .cluster_routing + .as_ref() + .map(|lock| lock.read().unwrap_or_else(|p| p.into_inner())) + .and_then(|routing| routing.leader_at_term_for_vshard(vshard_id.as_u32()).ok()) + .unwrap_or((0, 0)); + Err(crate::Error::NotLeader { + vshard_id, + leader_node, + leader_addr: String::new(), + leader_term, + }) + } + } +} + +/// Whether live Raft names this node the leader of `vshard_id`'s group. +fn leads_vshard(shared: &SharedState, vshard_id: VShardId) -> bool { + matches!( + resolve_live_decision(shared, vshard_id.as_u32()), + RouteDecision::Local + ) +} + +/// A successful response carrying `payload`, in the shape a local dispatch +/// returns. +pub(crate) fn ok_payload_response(payload: Payload) -> Response { + response(Status::Ok, payload.to_vec(), Lsn::ZERO, Lsn::ZERO, None) +} + +/// A `NotFound` refusal, the shape a local dispatch returns for a read that +/// found nothing. +pub(crate) fn not_found_response() -> Response { + response( + Status::Error, + Vec::new(), + Lsn::ZERO, + Lsn::ZERO, + Some(ErrorCode::NotFound), + ) +} + +fn response( + status: Status, + payload: Vec, + watermark_lsn: Lsn, + read_version_lsn: Lsn, + error_code: Option, +) -> Response { + Response { + request_id: RequestId::new(0), + status, + attempt: 1, + partial: false, + payload: Payload::from_vec(payload), + watermark_lsn, + error_code: error_code.map(Box::new), + read_set_valid: None, + read_version_lsn, + write_set: Vec::new(), + } +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/ambiguous_ddl.rs b/nodedb/src/control/server/dispatch_utils/submit_write/ambiguous_ddl.rs deleted file mode 100644 index f65216b9d..000000000 --- a/nodedb/src/control/server/dispatch_utils/submit_write/ambiguous_ddl.rs +++ /dev/null @@ -1,27 +0,0 @@ -// SPDX-License-Identifier: BUSL-1.1 - -//! The ambiguous-outcome guard for an enqueued Array DDL, split out of -//! `funnel.rs` because it is a self-contained decision with no ordering -//! dependency on the guard/append/enqueue/await/durability sequence — it only -//! runs from the funnel's post-enqueue error branches. - -use crate::control::state::SharedState; - -/// An enqueued Array CREATE/ALTER has an ambiguous outcome if the response -/// times out, closes, or overflows collection. Its Data-Plane mutation may -/// already exist, so rolling back the catalog would manufacture a ghost engine. -/// Preserve the transition and stop the node through the canonical watch; boot -/// recovery will reconcile from the durable catalog and WAL rather than serving -/// a potentially split Control/Data-plane view. -pub(super) fn preserve_ambiguous_array_ddl( - shared: &SharedState, - transition: &crate::control::array_catalog::ddl::AuthorizedDdlTransition, -) { - if !transition.preserves_on_ambiguous_apply() { - return; - } - if let Err(error) = transition.finalize(shared) { - tracing::error!(error = %error, "array DDL ambiguous after enqueue; finalization failed before fail-stop"); - } - shared.shutdown.signal(); -} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/answer.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/answer.rs new file mode 100644 index 000000000..01468577c --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/answer.rs @@ -0,0 +1,54 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Where the response phase reads a dispatched write's answer. + +use std::time::Instant; + +use crate::control::ResponseReceiver; +use crate::control::local_dispatch::{DispatchCollectError, collect_bounded_response}; +use crate::control::server::dispatch_utils::minted::{OwnedReport, OwnedResponse}; + +/// A dispatched write's answer, before anyone waits for it. +pub(super) enum Answer { + /// The write carries no records. The waiter reads the channel, up to + /// `deadline` and under the byte budget. The waiter drops the receiver + /// once it stops reading, which removes the request's tracker entry. + Unminted { + rx: ResponseReceiver, + deadline: Instant, + max_result_bytes: usize, + }, + /// A spawned task owns the write's records, closes them from the final + /// response, and reports the answer. + Owned(OwnedReport), +} + +impl Answer { + pub(super) async fn wait(self) -> crate::Result { + match self { + Self::Owned(report) => report.recv().await, + Self::Unminted { + mut rx, + deadline, + max_result_bytes, + } => { + let collected = tokio::time::timeout_at( + tokio::time::Instant::from_std(deadline), + collect_bounded_response(&mut rx, max_result_bytes), + ) + .await; + Ok(match collected { + Ok(Ok(response)) => OwnedResponse::Answered { + response, + closed: Ok(()), + }, + Ok(Err(DispatchCollectError::OverBudget { bytes })) => { + OwnedResponse::OverBudget { bytes } + } + Ok(Err(DispatchCollectError::ChannelClosed)) => OwnedResponse::ChannelClosed, + Err(_) => OwnedResponse::DeadlineExceeded, + }) + } + } + } +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs index 70bccb67b..f289df5a6 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/dispatch.rs @@ -6,20 +6,27 @@ use std::time::Instant; use tokio::sync::OwnedMutexGuard; +use crate::bridge::dispatch::JournalGroup; use crate::bridge::envelope::{Admission, PhysicalPlan, Priority, Request}; -use crate::control::array_catalog::ddl::AuthorizedDdlTransition; -use crate::control::server::shared::write_admission::WriteAdmissionGuard; +use crate::control::server::shared::write_admission::{WriteAdmissionGuard, WriteOrder}; use crate::control::state::SharedState; use crate::types::{ DatabaseId, Lsn, ReadConsistency, RequestId, TenantId, TraceId, TxnId, VShardId, }; -use super::wal_append::rollback_on_err; +/// The guards a dispatched write holds before its enqueue, held for their +/// `Drop`, which releases them. +pub(super) struct HeldGuards { + pub _admission: Option, + pub _order: Option>, + /// The document write order (see `write_order_fence`). + pub _write_order: WriteOrder, +} -/// The write-admission guards a dispatched write must hold across the -/// response await — deferred (rather than dropped at enqueue) only when the -/// write's redo is minted post-apply. -pub(super) type DeferredGuards = Option<(Option, Option>)>; +/// The guards a dispatched write must hold across the response await — +/// deferred (rather than dropped at enqueue) only when the write's redo is +/// minted post-apply. +pub(super) type DeferredGuards = Option; /// Everything [`dispatch_to_data_plane`] needs to build the wire `Request`. pub(super) struct DispatchTarget { @@ -34,7 +41,11 @@ pub(super) struct DispatchTarget { pub txn_id: Option, pub wal_lsn: Option, pub resolved_now_ms: Option, + /// The write's commit HLC, which dates every event it emits. + pub commit_hlc: u64, pub admission: Admission, + /// The record group the write journals its write set into. + pub journal: Option, } /// What the dispatch phase produced: the id the response is tracked under, @@ -61,10 +72,8 @@ pub(super) struct DispatchOutcome { /// retry, up to the request's deadline. Every other refusal returns at once. pub(super) async fn dispatch_to_data_plane( shared: &SharedState, - ddl_transition: &AuthorizedDdlTransition, target: DispatchTarget, - admission_guard: Option, - order_guard: Option>, + guards: HeldGuards, post_apply_pending: bool, waits_for_capacity: bool, ) -> crate::Result { @@ -89,44 +98,46 @@ pub(super) async fn dispatch_to_data_plane( txn_id: target.txn_id, wal_lsn: target.wal_lsn, resolved_now_ms: target.resolved_now_ms, + commit_hlc: Some(target.commit_hlc), admission: target.admission, }; + // A refusal, or a caller dropped mid-wait, drops `rx`, which removes the + // tracker entry of the request no core holds. let rx = shared.tracker.register(request_id); + let journal = target.journal; let dispatched = if waits_for_capacity { - dispatch_when_capacity_frees(shared, request).await + dispatch_when_capacity_frees(shared, request, journal).await } else { - match shared.dispatcher.lock() { - Ok(mut d) => d.dispatch(request), - Err(poisoned) => poisoned.into_inner().dispatch(request), - } + let attempt = match shared.dispatcher.lock() { + Ok(mut d) => d.try_dispatch_journalled(request, journal), + Err(poisoned) => poisoned + .into_inner() + .try_dispatch_journalled(request, journal), + }; + attempt.map_err(|refusal| refusal.error) }; - if dispatched.is_err() { - // No response will ever arrive for a refused request. - shared.tracker.cancel(&request_id); - } - rollback_on_err(shared, ddl_transition, dispatched)?; + dispatched?; // Release the write-admission guards immediately after the enqueue, before // the Data-Plane round-trip. The per-database WFQ is strict FIFO, so once LSN // order equals enqueue order the apply order follows from the queue alone; - // holding the guards across the response await would only serialize same-key + // holding the guards across the response await will only serialize same-key // throughput needlessly. // // EXCEPTION — a post-apply-redo write mints its durable redo AFTER apply, - // from the write-set on the response; the guards MUST stay held across the - // response collect + that append so two concurrent same-surrogate writes - // cannot reorder their redo appends. Both guard types are `Send`, so - // holding them across the `.await` is sound. Moved into an `Option` so the - // release is a single, unconditional `drop` below regardless of which path - // took it. (`None` guard slots when no lock manager was registered / for - // the exempt-read / Calvin / already-ordered cases.) + // from the write-set on the response; the guards and the document write + // order MUST stay held across the response collect + that append so no + // other write touching its rows mints a record in between. Every guard is + // `Send`, so holding them across the `.await` is sound. Moved into an + // `Option` so the release is a single, unconditional `drop` in the + // response phase. (`None` guard slots when no lock manager was registered / + // for the exempt-read / Calvin / already-ordered cases.) let deferred_guards = if post_apply_pending { - Some((admission_guard, order_guard)) + Some(guards) } else { - drop(admission_guard); - drop(order_guard); + drop(guards); None }; @@ -141,7 +152,15 @@ pub(super) async fn dispatch_to_data_plane( /// Dispatch `request`, waiting for freed capacity after each capacity /// refusal, until the request's deadline. Returns the last refusal once the /// deadline passes. -async fn dispatch_when_capacity_frees(shared: &SharedState, request: Request) -> crate::Result<()> { +/// +/// A caller that fans one statement out into more requests than the tenant's +/// in-flight cap dispatches each through here: the fan-out then advances as +/// earlier requests answer, instead of refusing the statement. +pub(crate) async fn dispatch_when_capacity_frees( + shared: &SharedState, + request: Request, + journal: Option, +) -> crate::Result<()> { let deadline = tokio::time::Instant::from_std(request.deadline); let capacity_freed = match shared.dispatcher.lock() { Ok(d) => d.capacity_freed(), @@ -155,8 +174,10 @@ async fn dispatch_when_capacity_frees(shared: &SharedState, request: Request) -> tokio::pin!(freed); freed.as_mut().enable(); let attempt = match shared.dispatcher.lock() { - Ok(mut d) => d.try_dispatch(request), - Err(poisoned) => poisoned.into_inner().try_dispatch(request), + Ok(mut d) => d.try_dispatch_journalled(request, journal.clone()), + Err(poisoned) => poisoned + .into_inner() + .try_dispatch_journalled(request, journal.clone()), }; let refusal = match attempt { Ok(()) => return Ok(()), diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs index 7f18bc5dc..8a4b92400 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/driver.rs @@ -4,22 +4,34 @@ //! that fixed order for one write. [`super::pending::PendingWrite::finish`] //! runs the response phase. -use crate::control::server::dispatch_utils::change_events::extract_write_change_set; +use std::sync::Arc; + +use crate::bridge::dispatch::JournalGroup; +use crate::control::server::dispatch_utils::change_events::{ + PendingChanges, extract_write_change_set, +}; use crate::control::server::dispatch_utils::durability_barrier::funnel_minted_redo_engine; -use crate::control::server::dispatch_utils::minted::{MintedRecords, RecordOwner}; +use crate::control::server::dispatch_utils::minted::{ + Collect, MintedRecords, OwnedWait, RecordOwner, spawn_owned_wait, +}; use crate::control::server::shared::session::statement_deadline; -use crate::control::server::shared::write_admission::{bare_ok_response, route_write_to_calvin}; +use crate::control::server::shared::write_admission::{ + bare_ok_response, order_row_write, route_write_to_calvin, +}; use crate::control::server::wal_dispatch; use crate::control::state::SharedState; +use crate::engine::timeseries::resolved_ingest::TsDriftPolicy; use super::super::params::{ ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, }; use super::admission::{AdmissionOutcome, admit_write}; -use super::dispatch::{DispatchTarget, dispatch_to_data_plane}; +use super::answer::Answer; +use super::dispatch::{DispatchTarget, HeldGuards, dispatch_to_data_plane}; +use super::late_parts::LateParts; use super::pending::PendingWrite; use super::response::{ResponsePhaseInput, UserWriteMark}; -use super::wal_append::authorize_and_append; +use super::wal_append::{AppendScope, authorize_and_append}; /// Admit, make durable, and enqueue one write on its core. /// @@ -48,16 +60,6 @@ pub(crate) async fn enqueue_write( database_id, vshard_id, }; - // On a single node no lease exists, so a write to a permission-tree source - // is acknowledged only once the local permission cache holds it. In a - // cluster the lease barrier on the Raft proposal path covers it instead. - let binds_authorization = shared.authorization_fence.timing().is_none() - && plan.named_collections().iter().any(|collection| { - shared - .authorization_fence - .sources() - .is_source_collection(collection) - }); // Only a user data write advances the tenant's observed write-HLC, which // the RESTORE staleness gate compares envelopes against. A schema install // such as a constraint set writes no row, so it records no mark. The mark @@ -72,18 +74,26 @@ pub(crate) async fn enqueue_write( let collection = plan.named_collections().first().map(|c| (*c).to_owned()); (site, collection) }); - // The instant the write committed, which is the value its mark carries: - // - a replicated entry carries its proposer's stamp; - // - a caller that appended upstream committed before this call, so the - // instant this call starts bounds it from above; - // - otherwise the append below is the commit, stamped once it lands. - let upstream_commit_hlc = match &durability { - WalDurability::AppendHere { commit_hlc, .. } => *commit_hlc, - WalDurability::CallerSupplied { .. } => Some(shared.hlc_clock.now().wall_ns), - }; // Records the caller appended for this write, under their outcome-floor // window. Every path below closes the window. - let caller_minted = durability.take_minted(); + let mut caller_minted = durability.take_minted(); + // A document row or edge write mints its records under its vShard's + // write-order fence, which this funnel takes below, so WAL order equals + // the order its rows reach storage. Records a caller appended before the + // call were minted outside that fence. + if caller_minted.is_some() && wal_dispatch::plan_post_apply_redo(&plan).is_some() { + if let Some(minted) = caller_minted.take() { + minted.cancel(&shared.wal, owner, 0).await?; + } + return Err(crate::Error::Internal { + detail: format!( + "internal invariant break: a caller appended the records of a row write on \ + '{}' before the write funnel; a row write's records are appended inside the \ + funnel, under its vShard's write-order fence", + plan.collection().unwrap_or("") + ), + }); + } // The running statement's deadline, pinned once at the session boundary and // shared by every request the statement fans out into. Used for both the @@ -98,19 +108,51 @@ pub(crate) async fn enqueue_write( // identity, so a caller whose change feed is `Unowned` skips it rather than // allocating tuples nothing will read. let change_set = match change_feed { - ChangeFeedOwner::Funnel => Some(extract_write_change_set(&plan, tenant_id)), + // A staged write's rows are not committed: it yields no event yet. + ChangeFeedOwner::LocalApply if txn_id.is_some() => None, + // Every committed user write applies through a replicated entry, so a + // write that yields change events on a route no other replica applies + // is refused before any record of it lands. + ChangeFeedOwner::LocalApply => { + if !extract_write_change_set(&plan, tenant_id).is_empty() { + if let Some(minted) = caller_minted { + minted.cancel(&shared.wal, owner, 0).await?; + } + return Err( + crate::control::change_stream::ChangeStreamError::UnreplicatedChange { + collection: plan + .named_collections() + .first() + .map(|collection| (*collection).to_owned()) + .unwrap_or_default(), + } + .into(), + ); + } + None + } + ChangeFeedOwner::Replicated { + group_id, + log_index, + } => Some(PendingChanges::staged( + extract_write_change_set(&plan, tenant_id), + group_id, + log_index, + )), ChangeFeedOwner::Unowned => None, }; // Post-apply redo classification, computed before `plan` is moved (the - // RouteToCalvin admit arm moves it). For a write whose autocommit WAL path - // mints no redo of its own but whose effect must survive a WAL-only restart - // (a document PointUpdate on a collection carrying a secondary vector - // index), the durable redo is minted AFTER apply from the surrogate + + // RouteToCalvin admit arm moves it). A document write whose stored rows + // its apply decides journals them AFTER apply, from the surrogate + // post-image the Data Plane returns in `Response::write_set`. // `Some(collection)` for such a write, else `None`. let post_apply = wal_dispatch::plan_post_apply_redo(&plan); let appends_here = matches!(&durability, WalDurability::AppendHere { .. }); + // A write that appends its own records journals the rows its apply + // decides as its record group. A write whose records live elsewhere, + // such as a WAL record replayed again, journals nothing here. + let groups = appends_here && post_apply.is_some(); let apply_key = match &durability { WalDurability::AppendHere { apply_key, .. } => *apply_key, WalDurability::CallerSupplied { .. } => 0, @@ -126,6 +168,20 @@ pub(crate) async fn enqueue_write( // rather than failing: a committed entry that fails for local load leaves // this replica without a write every other replica applied. let waits_for_capacity = matches!(ordering, WriteOrdering::AlreadyOrdered); + // A committed entry carries the rows its proposer resolved. Resolving it + // again here will let this replica accept other lines than its peers. + if waits_for_capacity && crate::control::write_resolve::is_unresolved_ingest(&plan) { + if let Some(minted) = caller_minted { + minted.cancel(&shared.wal, owner, 0).await?; + } + return Err(crate::Error::Internal { + detail: format!( + "internal invariant break: a committed timeseries ingest on '{}' carries \ + unresolved lines; its proposer resolves it before the entry exists", + plan.collection().unwrap_or("") + ), + }); + } // Durable-at-ack obligation, also computed before `plan` moves. `Some` only // for a write whose redo record THIS funnel is required to mint; a caller @@ -153,6 +209,7 @@ pub(crate) async fn enqueue_write( order_guard, } => (admission, admission_guard, order_guard), AdmissionOutcome::RouteToCalvin => { + // The scheduler journals the write in its own sequenced record. // The scheduler applies the write from its own records, so // the caller's records never apply. let superseded = caller_minted.map(|minted| { @@ -167,30 +224,60 @@ pub(crate) async fn enqueue_write( return Ok(PendingWrite::done(SubmitOutcome { response: routed .unwrap_or_else(|| bare_ok_response(crate::types::RequestId::new(0))), - wal_lsn: None, })); } }; - // On a server with no Raft groups a user write that mints its own record - // takes its commit stamp now, and its mark is durable before the mint: a - // crash after the record reaches disk keeps the mark. The stamp stays open - // until the mint, so a backup cut at or above it waits for the record. - let local_stamp = match &user_write_origin { - Some((_, collection)) - if appends_here - && caller_minted.is_none() - && upstream_commit_hlc.is_none() - && shared.async_raft_proposer().is_none() => - { - Some(shared.tenant_marks.stamp_local_write( - &shared.hlc_clock, - shared.credentials.catalog(), - tenant_id.as_u64(), - collection.as_deref(), - )?) - } - _ => None, + // Order this write's row records (document rows and edges) against every + // other row write of its vShard, from before the append below through its + // last post-apply record. Admission's guard, when it holds one, already + // names the row. A staged write stores nothing until its COMMIT, which is + // ordered then. + let write_order = if txn_id.is_none() { + order_row_write( + shared, + vshard_id, + &plan, + admission_guard.is_some() || order_guard.is_some(), + ) + .await + } else { + crate::control::server::shared::write_admission::WriteOrder::default() + }; + + // A gate-admitted timeseries ingest resolves its rows on its core before + // its record is appended, so the record logs exactly the rows the install + // stores. The resolve queues on the core behind every write enqueued + // before this one. Its install refuses a schema a concurrent write + // changed, and the writer resolves again: the record is cancelled then. + let plan = if appends_here + && txn_id.is_none() + && !waits_for_capacity + && crate::control::write_resolve::is_unresolved_ingest(&plan) + { + crate::control::write_resolve::resolve_ingest_plan( + shared, + crate::control::write_resolve::WriteResolveContext { + tenant_id, + database_id, + }, + vshard_id, + &plan, + TsDriftPolicy::Refuse, + ) + .await? + } else { + plan + }; + + // The instant the write committed, which is the value its mark carries: + // - a replicated entry carries its proposer's stamp; + // - a caller that appended upstream committed before this call, so this + // instant bounds it from above; + // - otherwise the append below is the commit, stamped once it lands. + let upstream_commit_hlc = match &durability { + WalDurability::AppendHere { commit_hlc, .. } => *commit_hlc, + WalDurability::CallerSupplied { .. } => Some(shared.hlc_clock.now().wall_ns), }; // A write that mints its own LSN opens its outcome-floor window before the @@ -200,15 +287,21 @@ pub(crate) async fn enqueue_write( None => appends_here.then(|| MintedRecords::open(&shared.outcome_floor)), }; - // Array DDL authorization + durability, under the admission guard, - // immediately before the enqueue below. + // Durability, under the admission guard, immediately before the enqueue + // below. + // The records carry the commit instant when it is known before the + // append; otherwise each takes the node's HLC as it lands. let wal_append_outcome = match authorize_and_append( shared, - owner, plan, durability, - minted.as_ref(), - event_source, + AppendScope { + owner, + minted: minted.as_ref(), + event_source, + commit_hlc: upstream_commit_hlc, + groups, + }, ) { Ok(outcome) => outcome, Err(error) => { @@ -219,35 +312,52 @@ pub(crate) async fn enqueue_write( return Err(error); } }; - let commit_hlc = local_stamp - .as_ref() - .map(|stamp| stamp.hlc()) - .or(upstream_commit_hlc) - .unwrap_or_else(|| shared.hlc_clock.now().wall_ns); - // The record is minted: a backup cut now waits for it through the outcome - // floor. - drop(local_stamp); + let commit_hlc = upstream_commit_hlc.unwrap_or_else(|| shared.hlc_clock.now().wall_ns); + // The write's events date by its commit HLC. A replicated write's is the + // HLC its proposer stamped on the entry, which every replica shares: + // every proposal path stamps one. + let change_set = change_set.map(|changes| changes.committed_at(commit_hlc)); let user_write = user_write_origin.map(|(site, collection)| UserWriteMark { site, collection, commit_hlc, }); - let ddl_transition = wal_append_outcome.ddl_transition; let plan = wal_append_outcome.plan; let wal_lsn = wal_append_outcome.wal_lsn; + // The write's change events date by its commit HLC. Recorded before the + // enqueue, so it exists before any event of the write reaches the Event + // Plane. + if let Some(lsn) = wal_lsn { + shared + .cdc_router + .positions() + .record_commit_hlc(lsn.as_u64(), commit_hlc); + } let resolved_now_ms = wal_append_outcome.resolved_now_ms; + let group_origin = wal_append_outcome.group_origin; + // The core stores the write set of a grouped write beside its effects. + let journal = match (&post_apply, group_origin) { + (Some(collection), Some(origin)) => Some(JournalGroup { + origin: origin.lsn, + collection: collection.clone(), + apply_key, + commit_hlc: upstream_commit_hlc, + change_position: wal_append_outcome.change_position, + }), + _ => None, + }; // A crash test parks one collection's logged write here: its LSN is - // minted, no core holds it, and only this task waits, so a later write - // with a higher LSN applies first. The write still holds its own per-key - // admission guards, which no write to another key contends on. + // minted and no core holds it. A replicated write parks inside its apply + // entry's enqueue, so no later entry of its Raft group starts until the + // gate opens. A write of another group applies meanwhile, with a higher + // LSN. The parked write also holds its own per-key admission guards. #[cfg(feature = "failpoints")] crate::control::fail_gate::before_dispatch(shared.node_id, &plan, wal_lsn).await; // Build the wire request and hand it to the Data-Plane dispatcher. let dispatched = dispatch_to_data_plane( shared, - &ddl_transition, DispatchTarget { tenant_id, database_id, @@ -260,23 +370,21 @@ pub(crate) async fn enqueue_write( txn_id, wal_lsn, resolved_now_ms, + commit_hlc, admission, + journal, + }, + HeldGuards { + _admission: admission_guard, + _order: order_guard, + _write_order: write_order, }, - admission_guard, - order_guard, post_apply.is_some(), waits_for_capacity, ) .await; let dispatch_outcome = match dispatched { - Ok(outcome) => { - // A core holds the request now. From here the records close - // from its final response, never from a drop. - if let Some(minted) = &minted { - minted.mark_sent(); - } - outcome - } + Ok(outcome) => outcome, Err(error) => { // The dispatcher refused the request, so no core applied it. A // dispatch refusal depends on this node's load, so the markers @@ -288,30 +396,72 @@ pub(crate) async fn enqueue_write( } }; - // The response phase collects the outcome and runs the post-apply steps a - // successful write still owes. - Ok(PendingWrite::dispatched( - ResponsePhaseInput { - request_id: dispatch_outcome.request_id, + // A core holds the request now. Its records move to the task that closes + // them from the final response, with no await since the enqueue. A drop of this future's caller or of the + // `PendingWrite` leaves those tasks running. + let max_result_bytes = shared.tuning.network.max_query_result_bytes as usize; + // A grouped write whose final response arrives late journals its parts + // from it. + let (late_tx, late_parts) = match (&post_apply, group_origin, &minted) { + (Some(collection), Some(origin), Some(_)) => { + let (tx, rx) = tokio::sync::oneshot::channel(); + let late = LateParts { + response: rx, + wal: Arc::clone(&shared.wal), + tenant_id, + vshard_id, + database_id, + collection: collection.clone(), + origin, + apply_key, + event_source, + commit_hlc: upstream_commit_hlc, + }; + (Some(tx), Some(late)) + } + _ => (None, None), + }; + let answer = match minted { + Some(minted) => Answer::Owned(spawn_owned_wait( + OwnedWait { + wal: Arc::clone(&shared.wal), + owner, + final_refusal_key, + deadline, + collect: Collect::Merged { max_result_bytes }, + late: late_tx, + }, + dispatch_outcome.rx, + minted, + )), + None => Answer::Unminted { rx: dispatch_outcome.rx, deadline, - dispatch_started: dispatch_outcome.dispatch_started, - tenant_id, - database_id, - vshard_id, - wal_lsn, - appends_here, - final_refusal_key, - apply_key, - event_source, - post_apply, - funnel_redo_engine, - change_set, - ddl_transition, - deferred_guards: dispatch_outcome.deferred_guards, - minted, - user_write, + max_result_bytes, }, - binds_authorization, - )) + }; + + // The response phase collects the outcome and runs the post-apply steps a + // successful write still owes. + Ok(PendingWrite::dispatched(ResponsePhaseInput { + request_id: dispatch_outcome.request_id, + answer, + max_result_bytes, + deadline, + dispatch_started: dispatch_outcome.dispatch_started, + tenant_id, + database_id, + vshard_id, + wal_lsn, + group_origin, + upstream_commit_hlc, + apply_key, + event_source, + post_apply, + funnel_redo_engine, + change_set, + deferred_guards: dispatch_outcome.deferred_guards, + user_write, + late_parts, + })) } diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/late_parts.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/late_parts.rs new file mode 100644 index 000000000..59a351d1a --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/late_parts.rs @@ -0,0 +1,78 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The parts of a grouped write whose final response arrives after its +//! caller stopped waiting. +//! +//! The core can still apply the write. Its parts must then journal in the +//! order its rows reached storage, so the write keeps its guards and its +//! order fence until the final response arrives and the parts are appended. + +use std::sync::Arc; + +use tokio::sync::oneshot; + +use crate::bridge::envelope::{Response, Status}; +use crate::control::server::wal_dispatch::{self, GroupOrigin, WriteSetTarget}; +use crate::types::{DatabaseId, TenantId, VShardId}; +use crate::wal::WalManager; + +use super::dispatch::DeferredGuards; + +/// Where a grouped write journals its parts from a late final response. +pub(super) struct LateParts { + pub response: oneshot::Receiver, + pub wal: Arc, + pub tenant_id: TenantId, + pub vshard_id: VShardId, + pub database_id: DatabaseId, + pub collection: String, + pub origin: GroupOrigin, + pub apply_key: u64, + pub event_source: crate::event::EventSource, + pub commit_hlc: Option, +} + +impl LateParts { + /// Hold `guards` until the final response arrives, journal the parts it + /// reports, then release them. A channel that closes with no final + /// response leaves the outcome unknown: the write set the core stored + /// settles at boot. + pub(super) fn spawn(self, guards: DeferredGuards) { + tokio::spawn(async move { + let Ok(response) = self.response.await else { + drop(guards); + return; + }; + if response.status == Status::Ok || !response.write_set.is_empty() { + let appender = self + .wal + .appender(self.apply_key) + .with_event_source(self.event_source); + let appender = match self.commit_hlc { + Some(hlc) => appender.with_commit_hlc(hlc), + None => appender, + }; + let appended = wal_dispatch::append_group_parts( + appender, + WriteSetTarget { + tenant_id: self.tenant_id, + vshard_id: self.vshard_id, + database_id: self.database_id, + collection: &self.collection, + origin: self.origin, + }, + &response.write_set, + ); + if let Err(error) = appended { + tracing::error!( + %error, + origin = self.origin.lsn.as_u64(), + "a late write's parts failed to journal; the write set its core \ + stored settles at boot" + ); + } + } + drop(guards); + }); + } +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs index 9e710788a..0b2b62f8d 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/mod.rs @@ -26,19 +26,23 @@ //! //! Split by concern, run in this fixed order by [`driver::submit_write`]: //! - [`admission`]: the write-admission gate. -//! - [`wal_append`]: Array DDL authorization and the WAL redo append/stamp. +//! - [`wal_append`]: the array DDL refusal and the WAL redo append/stamp. //! - [`dispatch`]: building the wire `Request` and handing it to the Data //! Plane. +//! - [`answer`]: where the response phase reads the answer. //! - [`response`]: collecting the response, classifying the outcome, and the //! post-apply steps a successful write still owes. //! - [`pending`]: the write between its enqueue and its response phase. mod admission; +mod answer; mod dispatch; mod driver; +mod late_parts; mod pending; mod response; mod wal_append; +pub(crate) use dispatch::dispatch_when_capacity_frees; pub(crate) use driver::enqueue_write; pub(crate) use pending::{PendingWrite, submit_write}; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/pending.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/pending.rs index 5dc04cf15..6c3ff7ecc 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/pending.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/pending.rs @@ -9,9 +9,13 @@ //! order and collects each outcome independently, so one parked write never //! holds back the writes behind it. +use crate::bridge::envelope::{ErrorCode, Status}; use crate::control::state::SharedState; +use crate::control::write_resolve::MAX_WRITE_RESOLVE_RETRIES; -use super::super::params::{SubmitOutcome, SubmitWrite}; +use super::super::params::{ + ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, +}; use super::driver::enqueue_write; use super::response::{ResponsePhaseInput, collect_classify_and_finish}; @@ -24,12 +28,7 @@ enum Stage { /// The write has its outcome already: the Calvin scheduler applied it. Done(SubmitOutcome), /// A core holds the write. The response phase collects its outcome. - Dispatched { - input: Box, - /// The write changes a permission-tree source on a node with no - /// lease, so its ack waits until the local permission cache holds it. - binds_authorization: bool, - }, + Dispatched(Box), } impl PendingWrite { @@ -39,12 +38,9 @@ impl PendingWrite { } } - pub(super) fn dispatched(input: ResponsePhaseInput, binds_authorization: bool) -> Self { + pub(super) fn dispatched(input: ResponsePhaseInput) -> Self { Self { - stage: Stage::Dispatched { - input: Box::new(input), - binds_authorization, - }, + stage: Stage::Dispatched(Box::new(input)), } } @@ -53,33 +49,88 @@ impl PendingWrite { /// /// See [`SubmitOutcome`] for what comes back. pub(crate) async fn finish(self, shared: &SharedState) -> crate::Result { - let (input, binds_authorization) = match self.stage { - Stage::Done(outcome) => return Ok(outcome), - Stage::Dispatched { - input, - binds_authorization, - } => (input, binds_authorization), - }; - let max_result_bytes = shared.tuning.network.max_query_result_bytes as usize; - let outcome = collect_classify_and_finish(shared, max_result_bytes, *input).await?; - if binds_authorization { - crate::control::security::auth_lease::await_local_coverage( - shared, - std::time::Instant::now() - + std::time::Duration::from_secs(shared.tuning.network.default_deadline_secs), - ) - .await?; + match self.stage { + Stage::Done(outcome) => Ok(outcome), + Stage::Dispatched(input) => collect_classify_and_finish(shared, *input).await, } - Ok(outcome) } } /// Admit, make durable, enqueue, collect, and publish one write. /// +/// A live timeseries ingest resolves its rows before its record is appended +/// (`driver`). Its install refuses with `OllpRetryRequired`, cancelling the +/// record, when a concurrent write changed the collection schema since the +/// resolve. It is then submitted again, up to +/// [`MAX_WRITE_RESOLVE_RETRIES`] times, and resolves against the new schema. +/// /// See [`SubmitOutcome`] for what comes back. pub(crate) async fn submit_write( shared: &SharedState, params: SubmitWrite, ) -> crate::Result { - enqueue_write(shared, params).await?.finish(shared).await + let mut params = params; + let mut attempt: u32 = 0; + loop { + let retry = timeseries_retry(¶ms); + let outcome = enqueue_write(shared, params).await?.finish(shared).await?; + match retry { + Some(next) if refused_for_drift(&outcome) && attempt < MAX_WRITE_RESOLVE_RETRIES => { + attempt += 1; + params = next; + } + _ => return Ok(outcome), + } + } +} + +/// A copy of `params` to submit again when its install refuses for schema +/// drift: a live, autocommit, unresolved timeseries ingest the funnel +/// appends. `None` for every other write. +fn timeseries_retry(params: &SubmitWrite) -> Option { + let WalDurability::AppendHere { + now_override, + apply_key: 0, + commit_hlc, + change_position: None, + } = ¶ms.durability + else { + return None; + }; + if params.txn_id.is_some() + || !matches!(params.ordering, WriteOrdering::Gate) + || !crate::control::write_resolve::is_unresolved_ingest(¶ms.plan) + { + return None; + } + let change_feed = match ¶ms.change_feed { + ChangeFeedOwner::LocalApply => ChangeFeedOwner::LocalApply, + ChangeFeedOwner::Unowned => ChangeFeedOwner::Unowned, + ChangeFeedOwner::Replicated { .. } => return None, + }; + Some(SubmitWrite { + tenant_id: params.tenant_id, + database_id: params.database_id, + vshard_id: params.vshard_id, + plan: params.plan.clone(), + trace_id: params.trace_id, + event_source: params.event_source, + txn_id: None, + user_id: params.user_id.clone(), + durability: WalDurability::AppendHere { + now_override: *now_override, + apply_key: 0, + commit_hlc: *commit_hlc, + change_position: None, + }, + ordering: WriteOrdering::Gate, + change_feed, + }) +} + +/// Whether the core refused the write because the collection schema moved +/// between its resolve and its install. +fn refused_for_drift(outcome: &SubmitOutcome) -> bool { + outcome.response.status == Status::Error + && outcome.response.error_code.as_deref() == Some(&ErrorCode::OllpRetryRequired) } diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs index dced5d147..d4ce8d8df 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/response.rs @@ -2,59 +2,59 @@ //! Collect the Data Plane's response, classify the outcome, and run the //! post-apply steps a successful write still owes: the post-apply redo, the -//! durable-at-ack barrier, DDL finalization, and the change-event publish. +//! durable-at-ack barrier, and the change-event publish. -use std::sync::Arc; use std::time::Instant; use crate::bridge::envelope::Status; -use crate::control::array_catalog::ddl::AuthorizedDdlTransition; -use crate::control::local_dispatch::{DispatchCollectError, collect_bounded_response}; -use crate::control::server::dispatch_utils::change_events::{WriteChangeSet, publish_change_set}; +use crate::control::server::dispatch_utils::change_events::PendingChanges; use crate::control::server::dispatch_utils::durability_barrier::assert_durable_before_ack; -use crate::control::server::dispatch_utils::minted::{ - Collect, MintedRecords, OwnedResponse, OwnedWait, RecordOwner, await_response_owned, -}; -use crate::control::server::dispatch_utils::submit_write::ambiguous_ddl::preserve_ambiguous_array_ddl; +use crate::control::server::dispatch_utils::minted::OwnedResponse; use crate::control::server::wal_dispatch; use crate::control::state::SharedState; use crate::types::{DatabaseId, Lsn, RequestId, TenantId, VShardId}; use super::super::params::SubmitOutcome; -use super::wal_append::rollback_on_err; +use super::answer::Answer; /// Everything the response phase needs, gathered from the admission, WAL /// append, and dispatch phases that ran before it. pub(super) struct ResponsePhaseInput { pub request_id: RequestId, - pub rx: crate::control::ResponseReceiver, + /// Where the response phase reads the answer. + pub answer: Answer, + /// The byte budget of the merged response. + pub max_result_bytes: usize, pub deadline: Instant, pub dispatch_started: Instant, pub tenant_id: TenantId, pub database_id: DatabaseId, pub vshard_id: VShardId, pub wal_lsn: Option, - pub appends_here: bool, + /// The origin of the write's record group, `Some` for a write this + /// funnel journalled that stores rows its apply decides. Such a write owes + /// the parts of its group. A write whose records live elsewhere, such as + /// a WAL record replayed again, journals nothing here. + pub group_origin: Option, + /// The commit instant every record of the write carries, when it was + /// known before the append. See `AppendScope::commit_hlc`. + pub upstream_commit_hlc: Option, /// The idempotency key every record this write appends carries (see /// `WalDurability::AppendHere`). pub apply_key: u64, /// The event source the write runs with. Its post-apply redo records /// carry it. pub event_source: crate::event::EventSource, - /// The key a final refusal's abort marker carries, `0` when this write's - /// refusals are not final. A final refusal is the proposal's outcome: the - /// proposal ledger rebuilt at boot counts the key as applied. - pub final_refusal_key: u64, pub post_apply: Option, pub funnel_redo_engine: Option<&'static str>, - pub change_set: Option, - pub ddl_transition: AuthorizedDdlTransition, + pub change_set: Option, pub deferred_guards: super::dispatch::DeferredGuards, - /// The records minted for this write, under their outcome-floor window. - pub minted: Option, /// `Some` when the plan writes user data, so a success advances the /// tenant's observed write-HLC. Reads and system operations never do. pub user_write: Option, + /// Where a grouped write journals its parts when its final response + /// arrives after the caller stopped waiting. + pub late_parts: Option, } /// The origin a successful user data write records on the tenant's observed @@ -80,35 +80,29 @@ pub(super) struct UserWriteMark { /// `ExecutionLimitExceeded` error. pub(super) async fn collect_classify_and_finish( shared: &SharedState, - max_result_bytes: usize, input: ResponsePhaseInput, ) -> crate::Result { let ResponsePhaseInput { request_id, - rx, + answer, + max_result_bytes, deadline, dispatch_started, tenant_id, database_id, vshard_id, wal_lsn, - appends_here, + group_origin, + upstream_commit_hlc, apply_key, event_source, - final_refusal_key, post_apply, funnel_redo_engine, change_set, - ddl_transition, deferred_guards, - minted, user_write, + late_parts, } = input; - let owner = RecordOwner { - tenant_id, - database_id, - vshard_id, - }; let vshard_u32 = vshard_id.as_u32(); let observe = |shared: &SharedState| { @@ -119,54 +113,30 @@ pub(super) async fn collect_classify_and_finish( // Wait to the same instant the envelope carries. The Data Plane normally // answers with `DeadlineExceeded` first; this bounds the wait when it is // inside a stage that carries no safe point yet. A write's records close - // in a task this future does not own, so a caller dropped mid-wait still + // in the task the enqueue handed them to, so a caller dropped here still // closes them. A refusal that arrives after the deadline still cancels // them. - let outcome = match minted { - Some(minted) => { - await_response_owned( - OwnedWait { - wal: Arc::clone(&shared.wal), - owner, - final_refusal_key, - deadline, - collect: Collect::Merged { max_result_bytes }, - }, - rx, - minted, - ) - .await? - } - None => collect_unminted(shared, request_id, rx, deadline, max_result_bytes).await, - }; + let outcome = answer.wait().await?; let response = match outcome { OwnedResponse::Answered { response, closed } => { - if response.status != Status::Ok { - let _ = ddl_transition.rollback(shared); - } // A failed cancel holds the window and fails the write here. closed?; response } OwnedResponse::DeadlineExceeded => { observe(shared); - // Dispatch completed, but the Data Plane may have applied CREATE - // or ALTER before this deadline. Never roll that catalog state - // back on an ambiguous post-enqueue outcome. - preserve_ambiguous_array_ddl(shared, &ddl_transition); - if !ddl_transition.preserves_on_ambiguous_apply() { - let _ = ddl_transition.rollback(shared); + // The core can still apply the write: its guards stay held until + // its parts are journalled. + if let Some(late) = late_parts { + late.spawn(deferred_guards); } return Err(crate::Error::DeadlineExceeded { request_id }); } OwnedResponse::OverBudget { bytes } => { observe(shared); - // A partial response proves dispatch began but not whether an - // Array DDL completed; preserve CREATE/ALTER and fail-stop. - preserve_ambiguous_array_ddl(shared, &ddl_transition); - if !ddl_transition.preserves_on_ambiguous_apply() { - let _ = ddl_transition.rollback(shared); + if let Some(late) = late_parts { + late.spawn(deferred_guards); } return Err(crate::Error::ExecutionLimitExceeded { detail: format!( @@ -177,14 +147,8 @@ pub(super) async fn collect_classify_and_finish( } OwnedResponse::ChannelClosed => { observe(shared); - // The producer can close after applying but before sending its - // response. CREATE/ALTER must remain catalog-finalized here. - preserve_ambiguous_array_ddl(shared, &ddl_transition); - if !ddl_transition.preserves_on_ambiguous_apply() { - let _ = ddl_transition.rollback(shared); - } // A producer that stopped after the deadline stopped because the - // statement ran out of time. Reporting the closure would hand the + // statement ran out of time. Reporting the closure will hand the // client an internal error for its own timeout. if std::time::Instant::now() >= deadline { return Err(crate::Error::DeadlineExceeded { request_id }); @@ -195,29 +159,35 @@ pub(super) async fn collect_classify_and_finish( } }; - // Mint the post-apply redo record while the guards are still held, then - // release them. A PointUpdate whose collection carries a secondary vector - // index returns its surrogate + post-image in `write_set`; without this - // durable `Put` a WAL-only restart rebuilds the HNSW from the pre-update body - // and resurrects the old embedding. + // Journal every row the write reported, as the parts of its record + // group, while its guards and its order fence are still held, then + // release them. The rows' records then sit in the WAL in the order the + // rows reached storage, so replay in LSN order rebuilds exactly the state + // the core holds. A refusal that follows committed rows reports them too, + // and they are journalled the same way. A refusal that applied nothing + // has its origin cancelled by the window, so its group needs no part. let post_apply_lsn = if let Some(collection) = &post_apply - && appends_here - && response.status == Status::Ok + && let Some(origin) = group_origin + && (response.status == Status::Ok || !response.write_set.is_empty()) { - rollback_on_err( - shared, - &ddl_transition, - wal_dispatch::append_write_set_redo( - shared - .wal - .appender(apply_key) - .with_event_source(event_source), + let appender = shared + .wal + .appender(apply_key) + .with_event_source(event_source); + let appender = match upstream_commit_hlc { + Some(hlc) => appender.with_commit_hlc(hlc), + None => appender, + }; + wal_dispatch::append_group_parts( + appender, + wal_dispatch::WriteSetTarget { tenant_id, vshard_id, database_id, collection, - &response.write_set, - ), + origin, + }, + &response.write_set, )? } else { None @@ -235,60 +205,51 @@ pub(super) async fn collect_classify_and_finish( // under the admission guard for a write that owns its durability, or supplied // by a caller that appended upstream (procedural batch flush, // interactive-COMMIT transaction redo). `post_apply_lsn` covers the - // post-apply redo appended just above. Both records are already buffered in + // post-apply redo appended right above. Both records are already buffered in // the shared WAL; one group-commit fsync coalesces concurrent writers (see // `WalManager::wait_durable`), and it runs here — after the admission guards // are released — so it never serializes same-key throughput. Reads / control // ops / trigger / staged-write dispatch carry no LSN and skip the barrier; // `durability_barrier` decides which of those skips are legitimate and makes // the rest loud instead of letting them ack a write nothing can recover. - // On a server with no Raft groups the write's mark lives in the catalog. - // It is durable before the ack. A write that minted its own record made it - // durable before the mint, so this costs no catalog commit there. - if response.status == Status::Ok - && let Some(mark) = &user_write - && shared.async_raft_proposer().is_none() - { - rollback_on_err( - shared, - &ddl_transition, - shared.tenant_marks.record_local_write( - shared.credentials.catalog(), - tenant_id.as_u64(), - mark.commit_hlc, - mark.collection.as_deref(), - ), - )?; - } - + let restore_write = event_source == crate::event::EventSource::Restore; if response.status == Status::Ok { let durable_target = match (wal_lsn, post_apply_lsn) { (Some(a), Some(b)) => Some(a.max(b)), (a, b) => a.or(b), }; match durable_target { - Some(lsn) => { - rollback_on_err(shared, &ddl_transition, shared.wal.wait_durable(lsn).await)? - } + Some(lsn) => shared.wal.wait_durable(lsn).await?, // Nothing to fsync. Legitimate for most plans, but if this funnel // was the one required to mint the record, the ack below promises // durability the engine cannot deliver — silent until a `kill -9` // proves it, hence the check. None => assert_durable_before_ack(funnel_redo_engine), } + } else if let Some(lsn) = post_apply_lsn { + // The refusal reports rows that stay committed: their records are + // durable before the refusal returns. + shared.wal.wait_durable(lsn).await?; } - - if response.status == Status::Ok { - ddl_transition.finalize(shared)?; + // The group's parts are durable, or its origin was cancelled: the core + // drops the write set it stored. + if post_apply.is_some() + && let Some(origin) = group_origin + { + match shared.dispatcher.lock() { + Ok(mut d) => d.note_write_set_settled(vshard_id, origin.lsn), + Err(poisoned) => poisoned + .into_inner() + .note_write_set_settled(vshard_id, origin.lsn), + } } - // Publish change events for successful writes whose change feed this funnel - // owns. `None` is a caller whose change feed is `Unowned` — see - // [`super::super::params::ChangeFeedOwner`] for why the node that applies those - // writes is not the node that publishes them. + // Stage for the apply loop the change events of a successful replicated + // write. `None` is a write with no replicated entry — see + // [`super::super::params::ChangeFeedOwner`]. if response.status == Status::Ok { if let Some(change_set) = change_set { - publish_change_set(shared, tenant_id, database_id, change_set, &response); + change_set.publish(shared, tenant_id, database_id, &response); } // Record the write's commit HLC on the tenant's observed high-water @@ -296,44 +257,26 @@ pub(super) async fn collect_classify_and_finish( // refuses an envelope older than the mark, so the mark is the instant // the write committed, never the instant this bookkeeping ran: a // backup taken after the ack then always carries a newer watermark. + // + // A write RESTORE re-issued raises its group's restore mark instead, + // under its restore id. Its commit HLC still folds into this node's + // clock, so a later backup here stamps a newer watermark. if let Some(mark) = &user_write { - shared.advance_tenant_write_hlc( - tenant_id.as_u64(), - mark.commit_hlc, - mark.site, - mark.collection.as_deref(), - ); + if restore_write { + shared + .hlc_clock + .update(nodedb_types::Hlc::new(mark.commit_hlc, 0)); + } else { + shared.advance_tenant_write_hlc( + tenant_id.as_u64(), + mark.commit_hlc, + mark.site, + mark.collection.as_deref(), + ); + } } } observe(shared); - Ok(SubmitOutcome { response, wal_lsn }) -} - -/// Collect a response that carries no records, under the same deadline and -/// byte budget a write's owned wait applies. -async fn collect_unminted( - shared: &SharedState, - request_id: RequestId, - mut rx: crate::control::ResponseReceiver, - deadline: Instant, - max_result_bytes: usize, -) -> OwnedResponse { - let collected = tokio::time::timeout_at( - tokio::time::Instant::from_std(deadline), - collect_bounded_response(&mut rx, max_result_bytes), - ) - .await; - match collected { - Ok(Ok(response)) => OwnedResponse::Answered { - response, - closed: Ok(()), - }, - Ok(Err(DispatchCollectError::OverBudget { bytes })) => { - shared.tracker.cancel(&request_id); - OwnedResponse::OverBudget { bytes } - } - Ok(Err(DispatchCollectError::ChannelClosed)) => OwnedResponse::ChannelClosed, - Err(_) => OwnedResponse::DeadlineExceeded, - } + Ok(SubmitOutcome { response }) } diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs index 0f3900c94..b2afe6892 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/funnel/wal_append.rs @@ -1,54 +1,46 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Array DDL authorization and WAL redo append/stamp for the funnel. +//! WAL redo append/stamp for the funnel. //! -//! Array DDL conversion is intentionally read-only. Once a task has passed -//! authorization and admission, its durable catalog state installs -//! immediately before the Data-Plane dispatch; the mirror changes only after -//! the redb transaction commits. The DDL transition this creates must be -//! rolled back by every later phase that can fail before the Data Plane has -//! proved it applied the write — see [`rollback_on_err`]. +//! Array DDL never reaches a core as a write: it runs through the replicated +//! catalog (`array_catalog::ddl`). The funnel refuses it. use crate::bridge::envelope::PhysicalPlan; -use crate::control::array_catalog::ddl::AuthorizedDdlTransition; use crate::control::server::dispatch_utils::minted::{MintedRecords, RecordOwner}; use crate::control::server::wal_dispatch::{self, WalAppendRequest}; use crate::control::state::SharedState; +use crate::event::cdc::position::ChangePositionMarker; use crate::types::Lsn; +use crate::wal::manager::NO_APPLY_KEY; use super::super::params::WalDurability; -/// What the authorize-and-append phase produced: the DDL transition every -/// later phase must roll back on failure, the plan (stamped with its minted -/// LSN, if any), and the resolved durability values. +/// What the authorize-and-append phase produced: the plan (stamped with its +/// minted LSN, if any) and the resolved durability values. pub(super) struct WalAppendOutcome { - pub ddl_transition: AuthorizedDdlTransition, pub plan: PhysicalPlan, pub wal_lsn: Option, pub resolved_now_ms: Option, + /// The origin of the write's record group, when it journals rows after + /// apply. `wal_lsn` is its LSN. + pub group_origin: Option, + /// The replicated position whose marker precedes the write's records. + pub change_position: Option, } -/// Roll `ddl_transition` back and convert `result`'s error, or pass a success -/// through unchanged. Every phase after DDL authorization that can fail calls -/// this instead of returning its error directly, so an authorized Array -/// CREATE/DROP/ALTER never survives a failure later in the sequence. -pub(super) fn rollback_on_err( - shared: &SharedState, - ddl_transition: &AuthorizedDdlTransition, - result: crate::Result, -) -> crate::Result { - match result { - Ok(value) => Ok(value), - Err(error) => { - let _ = ddl_transition.rollback(shared); - Err(error) - } - } +/// Where and how [`authorize_and_append`] journals one write. +pub(super) struct AppendScope<'a> { + pub owner: RecordOwner, + /// The window the write's records join. + pub minted: Option<&'a MintedRecords>, + pub event_source: crate::event::EventSource, + pub commit_hlc: Option, + pub groups: bool, } -/// Install the plan's Array DDL catalog transition, then make the write -/// durable: append its WAL redo record here (under the write-admission guard -/// the caller already holds) or take the LSN a caller supplied upstream. +/// Refuse array DDL, then make the write durable: append its WAL redo record +/// here (under the write-admission guard the caller already holds) or take +/// the LSN a caller supplied upstream. /// /// Durability, under the guard, immediately before the enqueue: the LSN is /// minted in the same order the request is about to be enqueued. @@ -59,60 +51,109 @@ pub(super) fn rollback_on_err( /// while replay stamps them from the record header — so the plan the Data /// Plane is about to execute must name the record that reproduces it. This is /// the only place that knows both, and it knows them for every caller: no -/// upstream path may allocate an LSN of its own and hope it matches. +/// upstream path can allocate an LSN of its own and hope it matches. /// /// An `AppendHere` write appends through `minted`, so every record it writes /// joins the write's outcome-floor window. +/// +/// `commit_hlc` is the write's commit instant when it is known before the +/// append. Every record the write appends carries it. `None` stamps each +/// record with the node's HLC at its append. +/// +/// `groups` is whether the write journals rows after apply. Such a write +/// appends its group's origin (see `wal_dispatch::append_group_origin`). pub(super) fn authorize_and_append( shared: &SharedState, - owner: RecordOwner, mut plan: PhysicalPlan, durability: WalDurability, - minted: Option<&MintedRecords>, - event_source: crate::event::EventSource, + scope: AppendScope<'_>, ) -> crate::Result { + let AppendScope { + owner, + minted, + event_source, + commit_hlc, + groups, + } = scope; let RecordOwner { tenant_id, database_id, vshard_id, } = owner; - let ddl_transition = crate::control::array_catalog::ddl::apply_authorized_ddl( - shared, - tenant_id, - database_id, - &plan, - )?; + if crate::control::array_catalog::ddl::is_array_ddl(&plan) { + return Err(crate::Error::Internal { + detail: "array DDL reached the write funnel; it runs through the replicated \ + array catalog" + .into(), + }); + } - let (wal_lsn, resolved_now_ms) = match durability { + let marked_position = match &durability { + WalDurability::AppendHere { + change_position, .. + } => *change_position, + WalDurability::CallerSupplied { .. } => None, + }; + let (wal_lsn, resolved_now_ms, group_origin) = match durability { WalDurability::AppendHere { now_override, apply_key, + change_position, .. } => { - let outcome = rollback_on_err( - shared, - &ddl_transition, - wal_dispatch::wal_append(WalAppendRequest { - wal: match minted { - Some(minted) => minted.appender(&shared.wal, apply_key), - None => shared.wal.appender(apply_key), - }, - event_source, + // The marker precedes the redo in the WAL, so the fsync that makes + // the redo durable makes the marker durable too. + if let Some(position) = change_position { + let marker = ChangePositionMarker { + apply_key, + position, + }; + stamped(shared.wal.appender(NO_APPLY_KEY), commit_hlc).append_change_position( tenant_id, vshard_id, database_id, - plan: &plan, - credentials: None, - now_override, - }), - )?; - (outcome.lsn, outcome.resolved_now_ms) + &marker.to_bytes(), + )?; + } + let request = WalAppendRequest { + wal: stamped( + match minted { + Some(minted) => minted.appender(&shared.wal, apply_key), + None => shared.wal.appender(apply_key), + }, + commit_hlc, + ), + event_source, + tenant_id, + vshard_id, + database_id, + plan: &plan, + credentials: None, + now_override, + }; + let (lsn, resolved_now_ms, group_origin) = if groups { + let appended = wal_dispatch::append_group_origin(request)?; + ( + Some(appended.origin.lsn), + appended.resolved_now_ms, + Some(appended.origin), + ) + } else { + let outcome = wal_dispatch::wal_append(request)?; + (outcome.lsn, outcome.resolved_now_ms, None) + }; + // Recorded before the enqueue, so it exists before any event of + // the write reaches the Event Plane. + if let (Some(position), Some(lsn)) = (change_position, lsn) { + shared.cdc_router.positions().record(lsn.as_u64(), position); + } + (lsn, resolved_now_ms, group_origin) } WalDurability::CallerSupplied { wal_lsn, resolved_now_ms, .. - } => (wal_lsn, resolved_now_ms), + } => (wal_lsn, resolved_now_ms, None), }; if let Some(lsn) = wal_lsn { @@ -120,9 +161,21 @@ pub(super) fn authorize_and_append( } Ok(WalAppendOutcome { - ddl_transition, plan, wal_lsn, resolved_now_ms, + group_origin, + change_position: marked_position, }) } + +/// `appender`, carrying `commit_hlc` when it is known. +fn stamped( + appender: crate::wal::manager::WalAppender<'_>, + commit_hlc: Option, +) -> crate::wal::manager::WalAppender<'_> { + match commit_hlc { + Some(hlc) => appender.with_commit_hlc(hlc), + None => appender, + } +} diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/mod.rs b/nodedb/src/control/server/dispatch_utils/submit_write/mod.rs index e6cdd4c6c..3fb8300cf 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/mod.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/mod.rs @@ -1,10 +1,9 @@ // SPDX-License-Identifier: BUSL-1.1 -mod ambiguous_ddl; mod funnel; mod params; -pub(crate) use funnel::{PendingWrite, enqueue_write, submit_write}; +pub(crate) use funnel::{PendingWrite, dispatch_when_capacity_frees, enqueue_write, submit_write}; pub(crate) use params::{ ChangeFeedOwner, SubmitOutcome, SubmitWrite, WalDurability, WriteOrdering, }; diff --git a/nodedb/src/control/server/dispatch_utils/submit_write/params.rs b/nodedb/src/control/server/dispatch_utils/submit_write/params.rs index 7bdaab306..6a205b363 100644 --- a/nodedb/src/control/server/dispatch_utils/submit_write/params.rs +++ b/nodedb/src/control/server/dispatch_utils/submit_write/params.rs @@ -17,7 +17,7 @@ use crate::types::{DatabaseId, Lsn, TenantId, TraceId, TxnId, VShardId}; pub(crate) enum WalDurability { /// The funnel appends the redo itself — under the write-admission guard, /// immediately before the enqueue — and stamps the minted LSN onto the - /// `Request`. Minting the LSN after admission and just before the enqueue + /// `Request`. Minting the LSN after admission and right before the enqueue /// is what makes WAL-LSN order equal dispatcher-enqueue order per key; the /// strict-FIFO per-database WFQ then makes apply order follow enqueue /// order, so restart replay (in LSN order) cannot diverge from live state. @@ -31,10 +31,16 @@ pub(crate) enum WalDurability { /// committed upstream: the proposer's stamp on a replicated entry. `None` /// when this append is the commit, so the funnel stamps the instant of the /// append itself. + /// + /// `change_position` is the Raft log position of the data-group entry this + /// write applies, `None` for a write no entry carries. The funnel makes it + /// durable ahead of the redo and records it for the CDC router, so every + /// replica positions the write's change events alike. AppendHere { now_override: Option, apply_key: u64, commit_hlc: Option, + change_position: Option, }, /// The caller already recorded this write's durability elsewhere — COMMIT's /// single `Transaction` record, the procedural batch flush, a trigger / @@ -82,49 +88,35 @@ pub(crate) enum WriteOrdering { Gate, /// Ordering was decided upstream and must not be re-decided. The Raft data /// group committed this entry at a fixed log index and every replica - /// applies it in exactly that order; re-entering the gate could route it + /// applies it in exactly that order; re-entering the gate can route it /// back through Calvin or block it behind a lock it does not need. AlreadyOrdered, } /// Who owns emitting this write's Control-Plane change event. pub(crate) enum ChangeFeedOwner { - /// The funnel extracts the write's change metadata from the plan and - /// publishes it once the apply succeeds. This is the route for every write - /// this node both handles and applies itself — the autocommit / internal - /// funnel, the pgwire SQL path's local dispatch, and the array executor's - /// single-node write — and it is what carries those writes to `/cdc` and - /// WS-RPC subscribers. - Funnel, - /// The funnel emits no change event for this write, because the node that - /// handled the write already emitted it. - /// - /// This is the route for a submit that applies a Raft-committed entry (the - /// data-group apply loop and the array apply path). Those run on EVERY - /// replica: publishing here would emit one event per replica, each with its - /// own cluster-wide NOTIFY fan-out to every peer, and no dedup exists on - /// either side — a subscriber would silently see the write once per - /// replica, multiplied again by the fan-out. The proposing node handled the - /// write exactly once and publishes there instead, after commit + apply - /// (see `publish_origin_change_events`). + /// The write applies on this node alone, outside any replicated entry: + /// a staged write inside a transaction, or a plan with no replicated + /// form. It has no position on any feed, so the funnel refuses one that + /// yields change events. + LocalApply, + /// The write applies the committed data-group entry at `(group_id, + /// log_index)`. Every replica applies it, and every replica stages its + /// change events under that entry once the apply succeeds. The apply + /// loop publishes them when it settles the entry in log order, so every + /// replica emits the group's feed at the same positions. + Replicated { group_id: u64, log_index: u64 }, + /// The funnel emits no change event for this write: its rows reach + /// subscribers another way. Unowned, } -/// What [`super::submit_write`] produced: the Data Plane's answer, and the LSN of the -/// record that reproduces this write on replay. +/// What [`super::submit_write`] produced: the Data Plane's answer. pub(crate) struct SubmitOutcome { /// The Data Plane's `Response` verbatim — including one whose `status` is /// `Error`. Callers that need an error status surfaced as a typed error /// check `status` themselves. pub response: Response, - /// The forward write's redo LSN: minted here for `AppendHere`, echoed from - /// the caller for `CallerSupplied`. `None` when this write mints no record - /// of its own — a read / control op, a plan whose variant appends nothing - /// (an array `Flush` reorganizes tiles already durable via their `Put` - /// records), or a Calvin-routed write whose durability the scheduler owns. - /// It is NOT a "no durability" signal, and no caller may substitute a - /// fabricated LSN for it. - pub wal_lsn: Option, } /// Inputs for [`super::submit_write`]. diff --git a/nodedb/src/control/server/dispatch_utils/types.rs b/nodedb/src/control/server/dispatch_utils/types.rs index 317b11d77..665b78b5c 100644 --- a/nodedb/src/control/server/dispatch_utils/types.rs +++ b/nodedb/src/control/server/dispatch_utils/types.rs @@ -59,4 +59,20 @@ pub(super) struct DataPlaneDispatch { /// the write-admission guard, or the caller already recorded durability /// elsewhere and supplies the LSN it minted. pub(super) durability: super::submit_write::WalDurability, + /// Where a read in this dispatch runs. Writes ignore it. + pub(super) read_route: ReadRoute, + /// Whether the write publishes its change events on this node's feed. + pub(super) change_feed: super::submit_write::ChangeFeedOwner, +} + +/// Where the funnel runs a read. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum ReadRoute { + /// The caller routed the read to this node and confirmed it as its + /// session requires (the gateway's local route, a received leg). It runs + /// here as is. + Routed, + /// No caller routed the read. It is a strong read, and the funnel serves + /// it from the group that owns its vShard (`owner_read`). + Owned, } diff --git a/nodedb/src/control/server/dispatch_utils/unlogged_dispatch.rs b/nodedb/src/control/server/dispatch_utils/unlogged_dispatch.rs new file mode 100644 index 000000000..3a61a0a8e --- /dev/null +++ b/nodedb/src/control/server/dispatch_utils/unlogged_dispatch.rs @@ -0,0 +1,122 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The funnel entries that append no WAL record of their own: in-transaction +//! tasks and routed reads. + +use crate::bridge::envelope::{PhysicalPlan, Response}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; + +use super::dispatch::dispatch_to_data_plane_inner; +use super::submit_write::WalDurability; +use super::types::{DataPlaneDispatch, ReadRoute}; + +/// Dispatch a physical plan to the Data Plane carrying an explicit transaction +/// id so the Data Plane can resolve this transaction's staging overlay +/// (read-your-own-writes) and route `StageWrite`. Used by the native endpoint, +/// whose in-transaction tasks flow through this shared path. +/// +/// A read is strong and runs where the group that owns `vshard_id` serves it +/// (`owner_read`). +/// +/// It appends no WAL record, so it refuses a write that only the funnel's +/// `AppendHere` route logs. A staged write is not such a write: COMMIT logs it. +pub(crate) async fn dispatch_to_data_plane_with_txn( + shared: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + plan: PhysicalPlan, + trace_id: TraceId, + txn_id: Option, +) -> crate::Result { + dispatch_unlogged( + shared, + UnloggedDispatch { + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + txn_id, + }, + ReadRoute::Owned, + ) + .await +} + +/// [`dispatch_to_data_plane_with_txn`] for a plan the caller already routed +/// to this node and confirmed as its session requires: the gateway's local +/// route, or a leg another node sent here. A read runs on this node as is. +pub(crate) async fn dispatch_routed_read_to_data_plane( + shared: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + plan: PhysicalPlan, + trace_id: TraceId, + txn_id: Option, +) -> crate::Result { + dispatch_unlogged( + shared, + UnloggedDispatch { + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + txn_id, + }, + ReadRoute::Routed, + ) + .await +} + +/// One dispatch that appends no WAL record of its own. +struct UnloggedDispatch { + tenant_id: TenantId, + database_id: DatabaseId, + vshard_id: VShardId, + plan: PhysicalPlan, + trace_id: TraceId, + txn_id: Option, +} + +async fn dispatch_unlogged( + shared: &SharedState, + dispatch: UnloggedDispatch, + read_route: ReadRoute, +) -> crate::Result { + let UnloggedDispatch { + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + txn_id, + } = dispatch; + super::durability_barrier::refuse_unlogged_write(&plan)?; + dispatch_to_data_plane_inner( + shared, + DataPlaneDispatch { + tenant_id, + database_id, + vshard_id, + plan, + trace_id, + event_source: crate::event::EventSource::User, + txn_id, + // Staged in-transaction writes are not yet durably committed; the + // committed write version is recorded at COMMIT via the batch funnel, + // so durability is not the funnel's to append here. + durability: WalDurability::CallerSupplied { + wal_lsn: None, + resolved_now_ms: None, + minted: None, + }, + read_route, + change_feed: super::submit_write::ChangeFeedOwner::LocalApply, + }, + ) + .await +} diff --git a/nodedb/src/control/server/dispatch_utils/write_abort.rs b/nodedb/src/control/server/dispatch_utils/write_abort.rs index d6ccc2f1f..a7e31abd9 100644 --- a/nodedb/src/control/server/dispatch_utils/write_abort.rs +++ b/nodedb/src/control/server/dispatch_utils/write_abort.rs @@ -1,19 +1,19 @@ // SPDX-License-Identifier: BUSL-1.1 //! Classification of a Data-Plane failure into "the write was refused and -//! nothing was applied" versus "the write may in fact have landed". +//! nothing was applied" versus "the write can in fact have landed". //! //! The write funnel appends a write's redo record BEFORE the Data Plane decides //! whether to accept it, so a refusal always arrives with the record already in //! the log. Left alone, restart replay re-applies it and a write the server told //! the client it refused comes back. The cure is a `WriteAborted` marker naming //! the forward record's LSN — but emitting one for a failure whose write -//! actually landed is strictly worse than the bug: recovery would then DELETE +//! actually landed is strictly worse than the bug: recovery will then DELETE //! committed data. //! //! So the predicate below is deliberately one-directional. A code earns an abort //! only when the code itself is proof that nothing was installed. Anything whose -//! outcome is ambiguous — the shard state is unknown, the failure could have +//! outcome is ambiguous — the shard state is unknown, the failure can have //! occurred part way through apply, or the code is an opaque catch-all — keeps //! the current behaviour and replays. That is the safe side of the trade: a //! refused write that survives a restart is a bug, a committed write erased by @@ -66,7 +66,6 @@ fn is_transient_verdict(code: &ErrorCode) -> bool { | ErrorCode::NotFound | ErrorCode::RejectedAuthz { .. } | ErrorCode::CrdtFrontierMismatch { .. } - | ErrorCode::FanOutExceeded | ErrorCode::ResourcesExhausted | ErrorCode::RejectedDanglingEdge { .. } | ErrorCode::DuplicateWrite @@ -160,10 +159,10 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { | ErrorCode::UndefinedColumn { .. } => true, // NOT established — every one of these can be reported by a request - // whose write reached, or may have reached, engine state. Emitting an + // whose write reached, or can have reached, engine state. Emitting an // abort for one risks deleting a committed write on recovery. // - // * `DeadlineExceeded` — the Data Plane may still be applying. + // * `DeadlineExceeded` — the Data Plane can still be applying. // * `RollbackFailed` — documented as leaving shard state unknown; the // forward record is precisely what recovery needs. // * `ResourcesExhausted` — memory can run out part way through apply. @@ -172,8 +171,8 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { // * `CrdtFrontierMismatch` — a mismatch detected against an applied // Loro frontier; whether the local doc absorbed the delta is not // decidable from the code. - // * `FanOutExceeded` / `RecursionDepthExceeded` — a limit tripped part - // way through a multi-step plan, which may already have written rows. + // * `RecursionDepthExceeded` — a limit tripped part + // way through a multi-step plan, which can already have written rows. // * `DuplicateWrite` — the idempotency gate fired because the write // ALREADY applied under the original request; nothing to undo, and // the duplicate record replays to the same state. @@ -185,7 +184,6 @@ pub(crate) fn write_definitely_not_applied(code: &ErrorCode) -> bool { | ErrorCode::ResourcesExhausted | ErrorCode::Internal { .. } | ErrorCode::CrdtFrontierMismatch { .. } - | ErrorCode::FanOutExceeded | ErrorCode::RecursionDepthExceeded { .. } | ErrorCode::DuplicateWrite => false, } @@ -239,7 +237,7 @@ mod tests { } /// The asymmetry that keeps this safe: an ambiguous outcome must never - /// produce an abort, because the write it would erase may have landed. + /// produce an abort, because the write it will erase can have landed. #[test] fn ambiguous_outcomes_never_abort_the_record() { assert!(!write_definitely_not_applied(&ErrorCode::DeadlineExceeded)); diff --git a/nodedb/src/control/server/exchange/all_cores/base_snapshot.rs b/nodedb/src/control/server/exchange/all_cores/base_snapshot.rs new file mode 100644 index 000000000..3c761b384 --- /dev/null +++ b/nodedb/src/control/server/exchange/all_cores/base_snapshot.rs @@ -0,0 +1,129 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Base-snapshot fan: dispatch `CreateSnapshot` to every local core and keep +//! each core's encoded `CoreSnapshot` beside its core id. + +use std::time::Instant; + +use futures::future::join_all; + +use crate::bridge::envelope::Response; +use crate::control::server::exchange::core_outcome::{ + CoreOutcome, classify_core_response, require_every_core, +}; +use crate::control::server::exchange::gather::eager_dispatch_to_all_cores_until; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, TraceId}; +use nodedb_physical::physical_plan::{MetaOp, PhysicalPlan}; + +/// Capture every local core for one base snapshot. +/// +/// Returns one `(core_id, encoded CoreSnapshot)` per core, in core order. +/// A base needs every core: a core that fails, refuses, or answers empty fails +/// the whole capture. Every core's request expires at `deadline`. +pub(crate) async fn capture_base_on_local_cores( + state: &SharedState, + deadline: Instant, +) -> crate::Result)>> { + crate::control::server::broadcast::broadcast_call_count_increment(); + let max_result_bytes = state.tuning.network.max_query_result_bytes as usize; + let receivers = eager_dispatch_to_all_cores_until( + state, + TenantId::new(0), + DatabaseId::DEFAULT, + TraceId::generate(), + None, + deadline, + |_| PhysicalPlan::Meta(MetaOp::CreateSnapshot), + )?; + + let collects = receivers + .into_iter() + .map(|(core_id, request_id, mut rx)| async move { + let context = format!("base snapshot on core {core_id}"); + // A core image is one terminal frame, which the byte ceiling + // applies to only when it streams. + let collected = crate::control::local_dispatch::collect_under_deadline( + &mut rx, + crate::control::local_dispatch::DeadlineCollect { + request_id, + deadline, + max_result_bytes, + context: &context, + }, + ) + .await; + (core_id, collected) + }); + let outcomes = join_all(collects) + .await + .into_iter() + .map(|(core_id, collected)| core_image(core_id, classify_core_response(collected))); + require_every_core(outcomes) +} + +/// One core's image, or the error that fails the capture. +fn core_image(core_id: usize, outcome: CoreOutcome) -> CoreOutcome<(usize, Vec)> { + let Some(response) = outcome? else { + return Err(crate::Error::Storage { + engine: "snapshot".into(), + detail: format!("core {core_id} refused CreateSnapshot; a base needs every core"), + }); + }; + if response.payload.is_empty() { + return Err(crate::Error::Storage { + engine: "snapshot".into(), + detail: format!("core {core_id} answered CreateSnapshot with no image"), + }); + } + Ok(Some((core_id, response.payload.as_ref().to_vec()))) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::{ErrorCode, Payload, Status}; + use crate::types::{Lsn, RequestId}; + + fn response(status: Status, payload: &[u8], code: Option) -> Response { + Response { + request_id: RequestId::new(1), + status, + attempt: 1, + partial: false, + payload: Payload::from_vec(payload.to_vec()), + watermark_lsn: Lsn::ZERO, + error_code: code.map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + #[test] + fn an_answering_core_keeps_its_id() { + let image = core_image(3, Ok(Some(response(Status::Ok, b"img", None)))); + assert_eq!(image.unwrap(), Some((3, b"img".to_vec()))); + } + + #[test] + fn a_refusing_core_fails_the_capture() { + let err = core_image(2, Ok(None)).unwrap_err(); + assert!(err.to_string().contains("core 2"), "{err}"); + } + + #[test] + fn an_empty_image_fails_the_capture() { + let err = core_image(1, Ok(Some(response(Status::Ok, b"", None)))).unwrap_err(); + assert!(err.to_string().contains("no image"), "{err}"); + } + + #[test] + fn one_failed_core_fails_every_core() { + let outcomes = vec![ + core_image(0, Ok(Some(response(Status::Ok, b"a", None)))), + core_image(1, Ok(None)), + ]; + assert!(require_every_core(outcomes).is_err()); + } +} diff --git a/nodedb/src/control/server/exchange/all_cores/bsp.rs b/nodedb/src/control/server/exchange/all_cores/bsp.rs index cd06bbf51..7ea8ab90b 100644 --- a/nodedb/src/control/server/exchange/all_cores/bsp.rs +++ b/nodedb/src/control/server/exchange/all_cores/bsp.rs @@ -8,12 +8,13 @@ use crate::types::{DatabaseId, Lsn, TenantId, TraceId}; use nodedb_physical::physical_plan::{BspSuperstepResult, PhysicalPlan}; use super::dispatch::NodeLevelResult; -use super::fanout::gather_graph_op_all_cores; +use super::fanout::gather_every_core; +use super::read_cut::resolve_plan_cut; /// BSP superstep fan: dispatch to all local cores, decode each core's /// [`BspSuperstepResult`], merge by field concatenation, and re-encode. /// -/// Owned-node sets are disjoint across cores because `gather_graph_op_all_cores` +/// Owned-node sets are disjoint across cores because `gather_every_core` /// scopes each core's `owned_vshards` to the vShards homed on that core, so each /// graph node is owned by exactly one core; concatenation therefore requires no /// dedup. @@ -21,15 +22,19 @@ pub(super) async fn fan_bsp_all_cores( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, - plan: PhysicalPlan, + mut plan: PhysicalPlan, trace_id: TraceId, ) -> crate::Result { - let responses = - gather_graph_op_all_cores(state, tenant_id, database_id, plan, trace_id, None, "bsp") - .await?; + // Every core reads the graph at the run's read cut. + let system_as_of = resolve_plan_cut(state, &mut plan).await?; + let responses = gather_every_core(state, tenant_id, database_id, plan, trace_id, "bsp").await?; let mut parts: Vec = Vec::with_capacity(responses.len()); + // The highest watermark any core served the superstep at, for the + // coordinator's transaction read-set. + let mut watermark_lsn = Lsn::ZERO; for resp in responses { + watermark_lsn = watermark_lsn.max(resp.watermark_lsn); // An empty payload decodes to BspSuperstepResult::default() (a // zero-vertex shard — contributes nothing to global_n or the ranks), // matching decode_single_result's contract. @@ -45,7 +50,8 @@ pub(super) async fn fan_bsp_all_cores( parts.push(part); } - let merged = merge_bsp_results(parts); + let mut merged = merge_bsp_results(parts); + merged.system_as_of = Some(system_as_of); let payload = zerompk::to_msgpack_vec(&merged).map_err(|e| crate::Error::Serialization { format: "msgpack".into(), detail: format!("bsp gather: merged result encode: {e}"), @@ -53,7 +59,7 @@ pub(super) async fn fan_bsp_all_cores( Ok(NodeLevelResult { payload, - watermark_lsn: Lsn::ZERO, + watermark_lsn, read_version_lsn: Lsn::ZERO, }) } diff --git a/nodedb/src/control/server/exchange/all_cores/dispatch.rs b/nodedb/src/control/server/exchange/all_cores/dispatch.rs index 37112a77b..5a720abef 100644 --- a/nodedb/src/control/server/exchange/all_cores/dispatch.rs +++ b/nodedb/src/control/server/exchange/all_cores/dispatch.rs @@ -9,7 +9,7 @@ use crate::types::{DatabaseId, Lsn, TenantId, TraceId, TxnId}; use nodedb_physical::physical_plan::{GraphOp, MetaOp, PhysicalPlan}; use super::bsp::fan_bsp_all_cores; -use super::fanout::gather_graph_op_all_cores; +use super::fanout::gather_every_core; use super::snapshot::fan_tenant_snapshot_all_cores; use super::wcc::fan_wcc_all_cores; @@ -47,13 +47,20 @@ pub(crate) async fn execute_plan_all_local_cores( use crate::data::executor::handlers::graph_match::encode_match_envelope_raw; // Forwarded `txn_id` resolves the staged overlay once present on this node. + // Not linearizable here: the callers confirm before they fan out. + // A leg received from another node was confirmed by + // `exec_receiver::read_leg` when it carried `read_groups`, and the + // graph superstep path confirms in `cluster_resolve`. let outcome = broadcast_match_to_all_cores( state, tenant_id, database_id, plan, trace_id, - txn_id, + crate::control::server::graph_dispatch::GraphRead { + txn_id, + linearizable: false, + }, ) .await?; @@ -65,9 +72,11 @@ pub(crate) async fn execute_plan_all_local_cores( &outcome.resume, )?; + // The served watermark rides back so the coordinator can put + // the read on a transaction's read-set at its true version. Ok(NodeLevelResult { payload: envelope, - watermark_lsn: Lsn::ZERO, + watermark_lsn: outcome.watermark_lsn, read_version_lsn: Lsn::ZERO, }) } @@ -99,7 +108,11 @@ pub(crate) async fn execute_plan_all_local_cores( | GraphOp::RemoveNodeLabels { .. } | GraphOp::TemporalNeighbors { .. } | GraphOp::TemporalAlgorithm { .. } - | GraphOp::Stats { .. } => { + | GraphOp::Stats { .. } + | GraphOp::NodeEdgeGuard { .. } + | GraphOp::NodePresenceGuard { .. } + | GraphOp::TruncateEdges { .. } + | GraphOp::NodePresenceRead { .. } => { generic_gather(state, tenant_id, database_id, plan, trace_id, txn_id).await } }, @@ -107,35 +120,53 @@ pub(crate) async fn execute_plan_all_local_cores( // Most Meta ops return row arrays (generic gather); per-node snapshot ops // return one opaque per-node blob and must not be array-wrapped. PhysicalPlan::Meta(meta) => match meta { - // Single `TenantDataSnapshot` map per core — the row gather would prepend a + // Single `TenantDataSnapshot` map per core — the row gather will prepend a // fixarray header, breaking restore's decode. MetaOp::CreateTenantSnapshot { .. } => { fan_tenant_snapshot_all_cores(state, tenant_id, database_id, plan, trace_id).await } - // Single JSON result object, not an array — same single-blob corruption - // class; return the lone core's payload verbatim. + // One JSON result object from the core that restores the snapshot. MetaOp::RestoreTenantSnapshot { .. } => { - single_blob_gather(state, tenant_id, database_id, plan, trace_id, None).await + let op = "RestoreTenantSnapshot"; + single_blob_gather(state, tenant_id, database_id, plan, trace_id, op).await } - // Single `RedoRecord` blob, never actually fans across cores, but routed - // through `single_blob_gather` so the payload returns verbatim. - MetaOp::ResolveTxn { .. } => { - single_blob_gather(state, tenant_id, database_id, plan, trace_id, None).await - } - // Same single `RedoRecord` blob shape as `ResolveTxn` (reuses it internally). + // One `RedoRecord` blob from the core that holds Calvin's staging state. MetaOp::CalvinResolve { .. } => { - single_blob_gather(state, tenant_id, database_id, plan, trace_id, None).await + let op = "CalvinResolve"; + single_blob_gather(state, tenant_id, database_id, plan, trace_id, op).await } - - // Row gather would array-wrap the affected-count blob, corrupting extraction. - MetaOp::StageWrite { .. } | MetaOp::DropTxnOverlay { .. } => { - single_blob_gather(state, tenant_id, database_id, plan, trace_id, txn_id).await + // A transaction meta-op runs on the one core that holds the + // transaction's staging overlay (`execute_received_plan`). Fanned to + // every core, it stages, resolves, or applies on each of them. + MetaOp::StageWrite { .. } => Err(fanned_scoped_plan("StageWrite")), + MetaOp::DropTxnOverlay { .. } => Err(fanned_scoped_plan("DropTxnOverlay")), + MetaOp::ResolveTxn { .. } => Err(fanned_scoped_plan("ResolveTxn")), + MetaOp::MarkSavepoint { .. } => Err(fanned_scoped_plan("MarkSavepoint")), + MetaOp::RollbackToSavepoint { .. } => Err(fanned_scoped_plan("RollbackToSavepoint")), + MetaOp::TransactionBatch { .. } => Err(fanned_scoped_plan("TransactionBatch")), + MetaOp::ApplyTransactionRedo { .. } => Err(fanned_scoped_plan("ApplyTransactionRedo")), + MetaOp::RestoreRedo(_) => Err(fanned_scoped_plan("RestoreRedo")), + // One image per core, kept apart by core id. A merged payload + // loses which core each image belongs to. + // Each core answers the probes of the vShards it owns. + MetaOp::HomeVersions { .. } => { + super::home_versions::fan_home_versions( + state, + tenant_id, + database_id, + plan, + trace_id, + ) + .await } + MetaOp::CreateSnapshot => Err(crate::Error::Internal { + detail: "CreateSnapshot reached the all-core fan; it runs through \ + capture_base_on_local_cores" + .into(), + }), // Enumerated exhaustively (no `_ =>`) so a new single-blob MetaOp forces a decision. MetaOp::WalAppend { .. } | MetaOp::Cancel { .. } - | MetaOp::TransactionBatch { .. } - | MetaOp::CreateSnapshot | MetaOp::Compact | MetaOp::Checkpoint | MetaOp::RegisterContinuousAggregate { .. } @@ -157,19 +188,16 @@ pub(crate) async fn execute_plan_all_local_cores( | MetaOp::QueryAggregateWatermark { .. } | MetaOp::QueryLastValues { .. } | MetaOp::QueryLastValue { .. } + | MetaOp::VerifyHashChain { .. } | MetaOp::CalvinExecuteStatic { .. } | MetaOp::CalvinExecutePassive { .. } | MetaOp::CalvinExecuteActive { .. } | MetaOp::RebuildIndex { .. } | MetaOp::PutSynonymGroup { .. } | MetaOp::DeleteSynonymGroup { .. } - | MetaOp::RenameCollection { .. } - | MetaOp::MarkSavepoint { .. } - | MetaOp::RollbackToSavepoint { .. } | MetaOp::RecordCalvinWriteVersions { .. } | MetaOp::CalvinFlush { .. } - | MetaOp::CalvinDrop { .. } - | MetaOp::ApplyTransactionRedo { .. } => { + | MetaOp::CalvinDrop { .. } => { generic_gather(state, tenant_id, database_id, plan, trace_id, txn_id).await } }, @@ -248,33 +276,34 @@ async fn generic_gather( }) } -/// Single-blob gather: fan `plan` across all local cores but return the lone -/// non-empty core's payload verbatim, with no row array-wrap. +/// The error for a vShard-scoped transaction meta-op that reached the +/// all-core fan instead of its one owning core. +fn fanned_scoped_plan(op: &'static str) -> crate::Error { + crate::Error::Internal { + detail: format!("{op} reached the all-core fan; it runs on its one owning core"), + } +} + +/// Single-blob gather: fan `plan` across all local cores and return the one +/// core's payload verbatim, with no row array-wrap. /// -/// The row gather would prepend a msgpack array header and corrupt a Meta op's -/// opaque blob. If more than one core returns non-empty, the first is kept. +/// The row gather will prepend a msgpack array header and corrupt a Meta op's +/// opaque blob. Every core must answer, and at most one answers with a +/// payload. `op` names the plan in the error when a second core does. async fn single_blob_gather( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, plan: PhysicalPlan, trace_id: TraceId, - txn_id: Option, + op: &'static str, ) -> crate::Result { - let responses = gather_graph_op_all_cores( - state, - tenant_id, - database_id, - plan, - trace_id, - txn_id, - "single-blob", - ) - .await?; + let responses = + gather_every_core(state, tenant_id, database_id, plan, trace_id, "single-blob").await?; let mut watermark_lsn = Lsn::ZERO; let mut read_version_lsn = Lsn::ZERO; - let mut payload: Option> = None; + let mut payloads: Vec> = Vec::with_capacity(responses.len()); for resp in responses { if resp.watermark_lsn > watermark_lsn { watermark_lsn = resp.watermark_lsn; @@ -282,14 +311,73 @@ async fn single_blob_gather( if resp.read_version_lsn > read_version_lsn { read_version_lsn = resp.read_version_lsn; } - if payload.is_none() && !resp.payload.is_empty() { - payload = Some(resp.payload.as_ref().to_vec()); - } + payloads.push(resp.payload.as_ref().to_vec()); } Ok(NodeLevelResult { - payload: payload.unwrap_or_default(), + payload: pick_single_blob(payloads, op)?, watermark_lsn, read_version_lsn, }) } + +/// Return the one non-empty payload, in core order, or an empty one when no +/// core answered with a payload. A second non-empty payload is an invariant +/// break named by `op`. +fn pick_single_blob(payloads: Vec>, op: &'static str) -> crate::Result> { + let mut non_empty = payloads.into_iter().filter(|p| !p.is_empty()); + let first = non_empty.next().unwrap_or_default(); + let extra = non_empty.count(); + if extra == 0 { + return Ok(first); + } + Err(crate::Error::Internal { + detail: format!( + "{op}: {} local cores answered with a payload, expected at most one", + extra + 1 + ), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn the_lone_payload_is_returned() { + let picked = pick_single_blob( + vec![Vec::new(), b"redo".to_vec(), Vec::new()], + "CalvinResolve", + ) + .unwrap(); + assert_eq!(picked, b"redo".to_vec()); + } + + #[test] + fn no_payload_is_empty() { + let picked = pick_single_blob(vec![Vec::new(), Vec::new()], "CalvinResolve").unwrap(); + assert!(picked.is_empty()); + } + + #[test] + fn a_second_payload_breaks_the_one_owner_invariant() { + match pick_single_blob( + vec![b"a".to_vec(), Vec::new(), b"b".to_vec()], + "RestoreTenantSnapshot", + ) { + Err(crate::Error::Internal { detail }) => { + assert!(detail.contains("RestoreTenantSnapshot"), "{detail}"); + assert!(detail.contains("2 local cores"), "{detail}"); + } + other => panic!("expected an invariant error, got {other:?}"), + } + } + + #[test] + fn a_fanned_scoped_plan_names_the_op() { + match fanned_scoped_plan("StageWrite") { + crate::Error::Internal { detail } => assert!(detail.contains("StageWrite")), + other => panic!("expected an internal error, got {other:?}"), + } + } +} diff --git a/nodedb/src/control/server/exchange/all_cores/fanout.rs b/nodedb/src/control/server/exchange/all_cores/fanout.rs index 117b8b76f..089a123c1 100644 --- a/nodedb/src/control/server/exchange/all_cores/fanout.rs +++ b/nodedb/src/control/server/exchange/all_cores/fanout.rs @@ -3,66 +3,25 @@ //! Shared per-core fan-out primitive for graph BSP/WCC superstep plans and //! single-blob Meta ops (tenant snapshot, restore result). Used by every //! single-blob merge path (`dispatch::single_blob_gather`, `snapshot`, `bsp`, -//! `wcc`). +//! `wcc`). Every core must answer: see `exchange::core_outcome`. use futures::future::join_all; -use crate::bridge::envelope::{Response, Status}; +use crate::bridge::envelope::Response; +use crate::control::server::exchange::core_outcome::{ + CoreOutcome, classify_core_response, require_every_core, +}; use crate::control::server::exchange::gather::eager_dispatch_to_all_cores; use crate::control::server::shared::session::statement_deadline; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, TraceId, TxnId}; +use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; use nodedb_physical::physical_plan::{GraphOp, PhysicalPlan}; -/// One core's outcome: its response, `None` for a `NotFound` refusal, or the -/// typed error that stopped it. -type CoreOutcome = crate::Result>; - -/// Shared per-core fan for a BSP/WCC superstep plan: dispatch to every local -/// core, gather bounded responses, drop `NotFound`/empty-CSR cores. -/// -/// A core that fails is dropped while any other core answers. The call fails -/// only when no core answers. -pub(super) async fn gather_graph_op_all_cores( - state: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - plan: PhysicalPlan, - trace_id: TraceId, - txn_id: Option, - label: &'static str, -) -> crate::Result> { - let outcomes = - dispatch_all_cores(state, tenant_id, database_id, plan, trace_id, txn_id, label).await?; - let mut out = Vec::with_capacity(outcomes.len()); - // First error seen across cores, kept as a TYPED error: a core cut short by - // the statement's deadline reports the deadline, and a constraint refusal - // keeps its own SQLSTATE. - let mut first_error: Option = None; - for outcome in outcomes { - match outcome { - Ok(Some(resp)) => out.push(resp), - Ok(None) => {} - Err(error) => { - if first_error.is_none() { - first_error = Some(error); - } - } - } - } - if out.is_empty() - && let Some(error) = first_error - { - return Err(error); - } - Ok(out) -} - /// Fan `plan` to every local core and require every core to answer. /// -/// Each core holds only the state its own vShards home to. A merge that -/// dropped a failed core returns that core's state as absent, so the first -/// core error fails the whole call. A `NotFound` refusal contributes nothing. +/// The first core error fails the whole call with that core's typed error. A +/// `NotFound` refusal contributes nothing. An empty payload is kept as an +/// answer. pub(super) async fn gather_every_core( state: &SharedState, tenant_id: TenantId, @@ -71,15 +30,8 @@ pub(super) async fn gather_every_core( trace_id: TraceId, label: &'static str, ) -> crate::Result> { - let outcomes = - dispatch_all_cores(state, tenant_id, database_id, plan, trace_id, None, label).await?; - let mut out = Vec::with_capacity(outcomes.len()); - for outcome in outcomes { - if let Some(resp) = outcome? { - out.push(resp); - } - } - Ok(out) + let outcomes = dispatch_all_cores(state, tenant_id, database_id, plan, trace_id, label).await?; + require_every_core(outcomes) } /// Dispatch `plan` to every local core and collect each core's outcome. @@ -92,9 +44,8 @@ async fn dispatch_all_cores( database_id: DatabaseId, plan: PhysicalPlan, trace_id: TraceId, - txn_id: Option, label: &'static str, -) -> crate::Result> { +) -> crate::Result>> { // Shared broadcast call counter (parity with gather_all_cores). crate::control::server::broadcast::broadcast_call_count_increment(); @@ -113,13 +64,19 @@ async fn dispatch_all_cores( // Eager dispatch: register + dispatch to each core before awaiting any response. // Scope owned_vshards to `vshard % num_cores == core_id` — see doc above. let receivers = - eager_dispatch_to_all_cores(state, tenant_id, database_id, trace_id, txn_id, |core_id| { + eager_dispatch_to_all_cores(state, tenant_id, database_id, trace_id, None, |core_id| { let mut core_plan = plan.clone(); match &mut core_plan { PhysicalPlan::Graph(g) => match g { + // A core receives only the contributions to the nodes it + // owns: its handler refuses any other. GraphOp::BspSuperstep(bsp) => { bsp.owned_vshards .retain(|v| (*v as usize) % num_cores == core_id); + bsp.incoming_contributions.retain(|(name, _)| { + (VShardId::from_key(name.as_bytes()).as_u32() as usize) % num_cores + == core_id + }); } GraphOp::WccSuperstep(wcc) => { wcc.owned_vshards @@ -146,7 +103,11 @@ async fn dispatch_all_cores( | GraphOp::RemoveNodeLabels { .. } | GraphOp::TemporalNeighbors { .. } | GraphOp::TemporalAlgorithm { .. } - | GraphOp::Stats { .. } => {} + | GraphOp::Stats { .. } + | GraphOp::NodeEdgeGuard { .. } + | GraphOp::NodePresenceGuard { .. } + | GraphOp::TruncateEdges { .. } + | GraphOp::NodePresenceRead { .. } => {} }, // Non-graph plans fanned verbatim. Exhaustive (no `_ =>`) to force a decision. PhysicalPlan::Vector(_) @@ -184,16 +145,5 @@ async fn dispatch_all_cores( let results: Vec> = join_all(response_futures).await; - Ok(results - .into_iter() - .map(|result| { - let resp = result?; - if resp.status == Status::Error { - // `NotFound` is an empty slice on this core, not an error. - crate::control::local_dispatch::reject_data_plane_error(&resp)?; - return Ok(None); - } - Ok(Some(resp)) - }) - .collect()) + Ok(results.into_iter().map(classify_core_response).collect()) } diff --git a/nodedb/src/control/server/exchange/all_cores/home_versions.rs b/nodedb/src/control/server/exchange/all_cores/home_versions.rs new file mode 100644 index 000000000..2b46f11f4 --- /dev/null +++ b/nodedb/src/control/server/exchange/all_cores/home_versions.rs @@ -0,0 +1,156 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `MetaOp::HomeVersions` on this node: each core answers the probes of the +//! vShards it owns, under this node's leader lease on each probe's group. +//! +//! A write to a vShard runs on the one core the dispatcher's router assigns +//! it, and raises that core's watermark and the collection's write floor +//! there. So a probe is answered by that core alone. A write that another core +//! took, to another vShard, never moves the answer. +//! +//! A node that lost leadership of a group can miss writes a newer leader +//! committed. So a probe is answered only while this node holds its group's +//! leader lease, after this node applied the group through the lease read +//! index (`cluster::leased_read`). A probe of a group whose lease this node +//! does not hold is answered `HomeAnswer::NotLeader` with the leader and term +//! the routing table names, and the committer asks that leader. + +use std::collections::{BTreeMap, HashMap}; + +use futures::future::try_join_all; +use nodedb_physical::physical_plan::{ + HomeAnswer, HomeVersion, HomeVersionProbe, MetaOp, PhysicalPlan, +}; + +use crate::control::cluster::leased_read::{LeaseRefusal, confirm_leased_read}; +use crate::control::cluster::linearizable_read::statement_read_deadline; +use crate::control::server::exchange::owning_core::dispatch_single_owning_core; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, Lsn, TenantId, TraceId, VShardId}; + +use super::dispatch::NodeLevelResult; + +/// Answer the probes of a `HomeVersions` plan: refuse those of groups this +/// node holds no lease on, split the rest by owning core, ask each core for +/// its own, and return every answer as one msgpack array of `HomeVersion`. +pub(super) async fn fan_home_versions( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + plan: PhysicalPlan, + trace_id: TraceId, +) -> crate::Result { + let PhysicalPlan::Meta(MetaOp::HomeVersions { probes }) = plan else { + return Err(crate::Error::Internal { + detail: "fan_home_versions received a plan that is not HomeVersions".into(), + }); + }; + let group_of = groups_of_probes(state, &probes)?; + let mut groups: Vec = group_of.values().copied().collect(); + groups.sort_unstable(); + groups.dedup(); + let refusals: HashMap = + confirm_leased_read(state, &groups, statement_read_deadline(state)) + .await? + .into_iter() + .map(|refusal| (refusal.group_id, refusal)) + .collect(); + + let mut answers = Vec::with_capacity(probes.len()); + let mut served = Vec::with_capacity(probes.len()); + for probe in probes { + match group_of.get(&probe.vshard).and_then(|g| refusals.get(g)) { + Some(refusal) => answers.push(HomeVersion { + probe, + answer: HomeAnswer::NotLeader { + leader_node: refusal.leader_node, + leader_term: refusal.leader_term, + }, + }), + None => served.push(probe), + } + } + + let by_core = probes_by_core(state, served)?; + let asks = + by_core.into_values().map(|probes| async move { + // Every probe of the group lives on one core, so any of their + // vShards routes the plan there. + let vshard = probes.first().map(|p| p.vshard).unwrap_or_default(); + let response = dispatch_single_owning_core( + state, + tenant_id, + database_id, + PhysicalPlan::Meta(MetaOp::HomeVersions { probes }), + VShardId::new(vshard), + trace_id, + None, + ) + .await?; + let answers: Vec = zerompk::from_msgpack(response.payload.as_ref()) + .map_err(|e| crate::Error::Internal { + detail: format!("home versions: a core's answer does not decode: {e}"), + })?; + Ok::<_, crate::Error>((answers, response.watermark_lsn)) + }); + let mut watermark_lsn = Lsn::ZERO; + for (core_answers, watermark) in try_join_all(asks).await? { + answers.extend(core_answers); + watermark_lsn = watermark_lsn.max(watermark); + } + let payload = zerompk::to_msgpack_vec(&answers).map_err(|e| crate::Error::Internal { + detail: format!("home versions: encoding the node's answer failed: {e}"), + })?; + Ok(NodeLevelResult { + payload, + watermark_lsn, + read_version_lsn: Lsn::ZERO, + }) +} + +/// The Raft group of each probed vShard. Empty on a node with no routing +/// table: a single node leads every vShard. +fn groups_of_probes( + state: &SharedState, + probes: &[HomeVersionProbe], +) -> crate::Result> { + let Some(routing) = state.cluster_routing.as_ref() else { + return Ok(HashMap::new()); + }; + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + let mut group_of = HashMap::with_capacity(probes.len()); + for probe in probes { + let group_id = + routing + .group_for_vshard(probe.vshard) + .map_err(|_| crate::Error::NoLeader { + vshard_id: VShardId::new(probe.vshard), + })?; + group_of.insert(probe.vshard, group_id); + } + Ok(group_of) +} + +/// The probes grouped by the core the router assigns each probe's vShard. +fn probes_by_core( + state: &SharedState, + probes: Vec, +) -> crate::Result>> { + let dispatcher = state.dispatcher.lock().unwrap_or_else(|p| p.into_inner()); + let router = dispatcher.router(); + let mut by_core: BTreeMap> = BTreeMap::new(); + for probe in probes { + let core = + router + .resolve(VShardId::new(probe.vshard)) + .ok_or_else(|| crate::Error::Internal { + detail: format!( + "home versions: no local core owns vShard {}; resend the check to the \ + vShard's leader", + probe.vshard + ), + })?; + by_core.entry(core).or_default().push(probe); + } + Ok(by_core) +} diff --git a/nodedb/src/control/server/exchange/all_cores/mod.rs b/nodedb/src/control/server/exchange/all_cores/mod.rs index f24c3a374..0fd000c60 100644 --- a/nodedb/src/control/server/exchange/all_cores/mod.rs +++ b/nodedb/src/control/server/exchange/all_cores/mod.rs @@ -7,9 +7,10 @@ //! single merged payload in exactly the same shape a single core's handler //! produces. It is called: //! -//! - by the remote `ExecuteRequest` receiver (`exec_receiver/executor.rs`) so -//! that an inbound plan from another node is transparently fanned across all -//! local cores before the merged result is returned, +//! - by `exchange::received::execute_received_plan`, so that an inbound plan +//! from another node is fanned across all local cores before the merged +//! result is returned. A vShard-scoped transaction meta-op never reaches the +//! fan: it runs on its one owning core, and the fan refuses it, //! - by the local BSP scatter path (`bsp_pagerank/scatter.rs`) so the //! coordinator's own node is treated identically to every remote node. //! @@ -20,14 +21,23 @@ //! gather variants (row-array and single-blob). `fanout` holds the shared //! per-core dispatch primitive used by every single-blob merge path. //! `snapshot`, `bsp`, and `wcc` each hold one field-concatenation merge for a -//! single-blob `PhysicalPlan` variant. +//! single-blob `PhysicalPlan` variant. `read_cut` resolves the read cut a BSP +//! or WCC run reads the graph at, before `bsp` and `wcc` fan the plan. +//! `base_snapshot` keeps every core's image apart, by core id, for a base +//! snapshot. +//! `home_versions` asks each core for the home versions of the vShards it +//! owns. +mod base_snapshot; mod bsp; mod dispatch; mod fanout; +mod home_versions; +mod read_cut; mod snapshot; mod wcc; +pub(crate) use base_snapshot::capture_base_on_local_cores; pub use dispatch::NodeLevelResult; pub(crate) use dispatch::execute_plan_all_local_cores; pub(crate) use snapshot::snapshot_tenant_on_local_cores; diff --git a/nodedb/src/control/server/exchange/all_cores/read_cut.rs b/nodedb/src/control/server/exchange/all_cores/read_cut.rs new file mode 100644 index 000000000..54a7bd1d8 --- /dev/null +++ b/nodedb/src/control/server/exchange/all_cores/read_cut.rs @@ -0,0 +1,107 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The read cut of a distributed graph run, resolved on this node. +//! +//! Every superstep of a distributed PageRank run, and every node of a WCC +//! round, reads the same graph. The run's first dispatch carries the +//! watermark of a Calvin cut marker. Each node proposes the marker (a second +//! copy is harmless), then waits until: +//! +//! - its sequencer replica applied the marker, recording the highest epoch +//! instant applied before it (`CalvinCuts::note_instant`), and +//! - every Calvin scheduler it runs passed the marker: every transaction +//! sequenced before it is installed here. +//! +//! The cut is that instant as a system-time ordinal. Epoch instants rise +//! strictly in log order, so every edge version sequenced before the marker +//! is at or below the cut, and every version sequenced after it is above. +//! Every node applies the same log, so every node resolves the same cut. + +use std::time::Duration; + +use nodedb_cluster::calvin::SequencerEntry; +use nodedb_physical::physical_plan::{GraphOp, PhysicalPlan}; + +use crate::control::state::SharedState; + +/// How long a node waits for the marker before it proposes it again. A +/// leader change can drop a proposed marker. +const MARKER_RETRY: Duration = Duration::from_secs(1); + +/// Resolve the read cut of a BSP or WCC `plan` on this node. A plan that +/// carries a cut marker gets its cut in `system_as_of`. Returns the cut the +/// plan's cores read at. A plan with neither a marker nor a cut is an error. +pub(super) async fn resolve_plan_cut( + state: &SharedState, + plan: &mut PhysicalPlan, +) -> crate::Result { + let (marker, system_as_of) = match plan { + PhysicalPlan::Graph(GraphOp::BspSuperstep(bsp)) => { + (&mut bsp.read_cut_marker, &mut bsp.system_as_of) + } + PhysicalPlan::Graph(GraphOp::WccSuperstep(wcc)) => { + (&mut wcc.read_cut_marker, &mut wcc.system_as_of) + } + _ => { + return Err(crate::Error::Internal { + detail: "graph read cut: only a BSP or WCC superstep carries a read cut".into(), + }); + } + }; + if *marker != 0 { + *system_as_of = Some(resolve_read_cut(state, *marker).await?); + *marker = 0; + } + system_as_of.ok_or_else(|| crate::Error::Internal { + detail: "graph read cut: a distributed graph superstep arrived with no read cut and no \ + cut marker" + .into(), + }) +} + +/// Propose the cut marker `marker` and wait until it applied here and every +/// local Calvin scheduler passed it. Returns the cut as a system-time +/// ordinal: `0` when no epoch applied before the marker, so no Calvin edge +/// version is visible. +async fn resolve_read_cut(state: &SharedState, marker: u64) -> crate::Result { + let proposer = state + .calvin + .sequencer_proposer + .get() + .ok_or_else(|| crate::Error::Internal { + detail: "graph read cut: no sequencer proposer is set on this node, so the run \ + cannot place its cut marker. Retry once the cluster finished starting" + .into(), + })?; + let entry = zerompk::to_msgpack_vec(&SequencerEntry::CutMarker { + hlc: marker, + restore_point: 0, + }) + .map_err(|error| crate::Error::Internal { + detail: format!("graph read cut: encode the cut marker: {error}"), + })?; + let timeout = Duration::from_secs(state.tuning.network.default_deadline_secs); + let deadline = tokio::time::Instant::now() + timeout; + let mut last_refusal = None; + loop { + if let Err(error) = proposer.propose(entry.clone()) { + last_refusal = Some(error.to_string()); + } + let attempt = deadline.min(tokio::time::Instant::now() + MARKER_RETRY); + if let Some(instant) = state.calvin.cuts.await_instant(marker, attempt).await { + return Ok(instant.map_or(0, nodedb_types::ms_to_ordinal_upper)); + } + if tokio::time::Instant::now() >= deadline { + return Err(crate::Error::Internal { + detail: format!( + "graph read cut: cut marker {marker} did not apply and pass every Calvin \ + scheduler of this node within {}s (vShards still before it: {:?}, last \ + marker refusal: {}). Retry the query", + timeout.as_secs(), + state.calvin.cuts.lagging(marker), + last_refusal.as_deref().unwrap_or("none"), + ), + }); + } + } +} diff --git a/nodedb/src/control/server/exchange/all_cores/snapshot.rs b/nodedb/src/control/server/exchange/all_cores/snapshot.rs index cc9184945..9c54e153c 100644 --- a/nodedb/src/control/server/exchange/all_cores/snapshot.rs +++ b/nodedb/src/control/server/exchange/all_cores/snapshot.rs @@ -13,7 +13,8 @@ use super::dispatch::NodeLevelResult; use super::fanout::gather_every_core; /// Snapshot `tenant_id` in `database_id` on every local core and return the -/// one merged `TenantDataSnapshot` blob. +/// one merged `TenantDataSnapshot` blob. `arrays` exports every array cell +/// version too. /// /// Every local caller that snapshots a tenant goes through this function. Each /// core stores only the collections its vShards home to, so a snapshot taken @@ -23,10 +24,13 @@ pub(crate) async fn snapshot_tenant_on_local_cores( tenant_id: TenantId, database_id: DatabaseId, timeout: Duration, + arrays: bool, ) -> crate::Result> { let plan = PhysicalPlan::Meta(MetaOp::CreateTenantSnapshot { tenant_id: tenant_id.as_u64(), cut_watermark: None, + cut_capture: None, + arrays, }); let fan = fan_tenant_snapshot_all_cores(state, tenant_id, database_id, plan, TraceId::generate()); @@ -47,9 +51,9 @@ pub(crate) async fn snapshot_tenant_on_local_cores( /// concatenation, and re-encode ONE snapshot blob. /// /// Each core scans only the engine state for the vShards homed on that core, so -/// the per-core snapshots cover DISJOINT key sets — concatenating every `Vec` -/// field requires no dedup, exactly like the BSP/WCC superstep merges. Every -/// core must answer: a core that fails would leave its collections out of the +/// the per-core snapshots cover disjoint key sets, except a cross-shard edge, +/// which both of its endpoint homes store ([`merge_core_snapshots`]). Every +/// core must answer: a core that fails will leave its collections out of the /// snapshot, so its error fails the snapshot. At 1 core/node this yields the /// lone core's snapshot unchanged. pub(super) async fn fan_tenant_snapshot_all_cores( @@ -59,8 +63,6 @@ pub(super) async fn fan_tenant_snapshot_all_cores( plan: PhysicalPlan, trace_id: TraceId, ) -> crate::Result { - use crate::types::TenantDataSnapshot; - let responses = gather_every_core( state, tenant_id, @@ -70,6 +72,17 @@ pub(super) async fn fan_tenant_snapshot_all_cores( "tenant-snapshot", ) .await?; + merge_core_snapshots(responses) +} + +/// Merge every core's partial [`TenantDataSnapshot`] of one tenant into one +/// snapshot blob. The cores cover disjoint key sets, so every field +/// concatenates, except the edges: a cross-shard edge lives on both +/// endpoint homes, so two cores can report one edge version, kept once. +pub(super) fn merge_core_snapshots( + responses: Vec, +) -> crate::Result { + use crate::types::TenantDataSnapshot; let mut merged = TenantDataSnapshot::default(); let mut watermark_lsn = Lsn::ZERO; @@ -104,11 +117,35 @@ pub(super) async fn fan_tenant_snapshot_all_cores( index_configs, surrogate_pk, tenant_edges, + edge_hidden, + edge_cuts, + edge_applied, + tenant_edge_cuts, + tenant_edge_applied, group_write_marks, + group_proposal_keys, + group_proposal_keys_complete_from, documents_versioned, indexes_versioned, + vector_multi_documents, + arrays, + metadata_floor, + group_cut_index, + // The group snapshot builder sets the lane state and the Calvin + // cut on the merged payload. A per-core part carries neither. + group_event_lane: _, + group_calvin: _, } = part; + // The group snapshot builder sets the cut on the merged payload. A + // per-core part carries none. + merged.group_cut_index = merged.group_cut_index.max(group_cut_index); + merged.edge_hidden.extend(edge_hidden); + merged.edge_cuts.extend(edge_cuts); + merged.edge_applied.extend(edge_applied); + merged.tenant_edge_cuts.extend(tenant_edge_cuts); + merged.tenant_edge_applied.extend(tenant_edge_applied); merged.documents.extend(documents); + merged.vector_multi_documents.extend(vector_multi_documents); merged.indexes.extend(indexes); merged.edges.extend(edges); merged.vectors.extend(vectors); @@ -123,9 +160,29 @@ pub(super) async fn fan_tenant_snapshot_all_cores( merged.surrogate_pk.extend(surrogate_pk); merged.tenant_edges.extend(tenant_edges); merged.group_write_marks.extend(group_write_marks); + merged.group_proposal_keys.extend(group_proposal_keys); + merged.group_proposal_keys_complete_from = merged + .group_proposal_keys_complete_from + .max(group_proposal_keys_complete_from); merged.documents_versioned.extend(documents_versioned); merged.indexes_versioned.extend(indexes_versioned); + merged.arrays.extend(arrays); + merged.metadata_floor = merged.metadata_floor.max(metadata_floor); } + // A cross-shard edge is stored on both endpoint homes under one version + // key. When the two homes sit on different cores, both cores report it. + dedup_edges(&mut merged.edges, |(key, _)| key.clone()); + dedup_edges(&mut merged.tenant_edges, |(db, tid, key, _)| { + (*db, *tid, key.clone()) + }); + dedup_edges(&mut merged.edge_hidden, |(key, _)| key.clone()); + dedup_edges(&mut merged.edge_applied, |(key, _)| key.clone()); + dedup_edges(&mut merged.tenant_edge_applied, |(db, tid, key, _)| { + (*db, *tid, key.clone()) + }); + // Every core that holds a vShard of the collection records the cut. + dedup_edges(&mut merged.edge_cuts, Clone::clone); + dedup_edges(&mut merged.tenant_edge_cuts, Clone::clone); let payload = zerompk::to_msgpack_vec(&merged).map_err(|e| crate::Error::Serialization { format: "msgpack".into(), @@ -138,3 +195,31 @@ pub(super) async fn fan_tenant_snapshot_all_cores( read_version_lsn: Lsn::ZERO, }) } + +/// Keep the first entry of every `identity` in `entries`, in order. +fn dedup_edges(entries: &mut Vec, identity: impl Fn(&T) -> K) { + let mut seen = std::collections::HashSet::new(); + entries.retain(|entry| seen.insert(identity(entry))); +} + +#[cfg(test)] +mod tests { + use super::dedup_edges; + + #[test] + fn an_edge_both_homes_report_is_kept_once() { + let mut edges = vec![ + ("g\0a\0L\0b\x007".to_string(), vec![1]), + ("g\0a\0L\0c\x007".to_string(), vec![2]), + ("g\0a\0L\0b\x007".to_string(), vec![1]), + ]; + dedup_edges(&mut edges, |(key, _)| key.clone()); + assert_eq!( + edges, + vec![ + ("g\0a\0L\0b\x007".to_string(), vec![1]), + ("g\0a\0L\0c\x007".to_string(), vec![2]), + ] + ); + } +} diff --git a/nodedb/src/control/server/exchange/all_cores/wcc.rs b/nodedb/src/control/server/exchange/all_cores/wcc.rs index 3ef054740..13f5285d6 100644 --- a/nodedb/src/control/server/exchange/all_cores/wcc.rs +++ b/nodedb/src/control/server/exchange/all_cores/wcc.rs @@ -8,7 +8,8 @@ use crate::types::{DatabaseId, Lsn, TenantId, TraceId}; use nodedb_physical::physical_plan::{PhysicalPlan, WccSuperstepResult}; use super::dispatch::NodeLevelResult; -use super::fanout::gather_graph_op_all_cores; +use super::fanout::gather_every_core; +use super::read_cut::resolve_plan_cut; /// WCC contraction-round fan: dispatch to all local cores, decode each core's /// [`WccSuperstepResult`], merge by field concatenation, and re-encode. @@ -21,15 +22,19 @@ pub(super) async fn fan_wcc_all_cores( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, - plan: PhysicalPlan, + mut plan: PhysicalPlan, trace_id: TraceId, ) -> crate::Result { - let responses = - gather_graph_op_all_cores(state, tenant_id, database_id, plan, trace_id, None, "wcc") - .await?; + // Every core reads the graph at the round's read cut. + let system_as_of = resolve_plan_cut(state, &mut plan).await?; + let responses = gather_every_core(state, tenant_id, database_id, plan, trace_id, "wcc").await?; let mut parts: Vec = Vec::with_capacity(responses.len()); + // The highest watermark any core served the round at, for the + // coordinator's transaction read-set. + let mut watermark_lsn = Lsn::ZERO; for resp in responses { + watermark_lsn = watermark_lsn.max(resp.watermark_lsn); // An empty payload decodes to WccSuperstepResult::default() (a // zero-vertex shard — contributes no labels or boundary edges). let part = if resp.payload.is_empty() { @@ -44,7 +49,8 @@ pub(super) async fn fan_wcc_all_cores( parts.push(part); } - let merged = merge_wcc_results(parts); + let mut merged = merge_wcc_results(parts); + merged.system_as_of = Some(system_as_of); let payload = zerompk::to_msgpack_vec(&merged).map_err(|e| crate::Error::Serialization { format: "msgpack".into(), detail: format!("wcc gather: merged result encode: {e}"), @@ -52,14 +58,14 @@ pub(super) async fn fan_wcc_all_cores( Ok(NodeLevelResult { payload, - watermark_lsn: Lsn::ZERO, + watermark_lsn, read_version_lsn: Lsn::ZERO, }) } /// Merge per-core [`WccSuperstepResult`] parts by field concatenation. /// -/// Owned-node sets are DISJOINT across cores because `gather_graph_op_all_cores` +/// Owned-node sets are DISJOINT across cores because `gather_every_core` /// scopes each core's `owned_vshards` to the vShards homed on that core, so each /// graph node is owned by exactly one core. Concatenation therefore requires no /// dedup; cross-core edges already appear as boundary edges (their destination diff --git a/nodedb/src/control/server/exchange/cluster_leaf.rs b/nodedb/src/control/server/exchange/cluster_leaf.rs new file mode 100644 index 000000000..5b67ff500 --- /dev/null +++ b/nodedb/src/control/server/exchange/cluster_leaf.rs @@ -0,0 +1,279 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A cross-node gather of a plan whose rows are spread by tile or node key. +//! +//! - A cluster array read (`ClusterArray` `Slice` / `Agg`) runs through the +//! array executor, which already knows the shard that owns each tile. A +//! leaf nested under a coordinator-local wrapper (aggregate, join input, +//! post-process, set operation, lateral outer plan) is read first and +//! inlined as a `ProviderScan`. The rest of the plan then gathers like any +//! plan over ordinary collections. A slice read below its retention horizon +//! raises the client notice its own shaping will report. +//! - Graph reads never reach a gather: the SQL planner emits no graph leaf, +//! and every graph read runs through `graph_dispatch`. A local `Array` plan +//! never reaches one in a cluster either, which plans arrays as +//! `ClusterArray`. Either one is refused rather than read from this node's +//! partitions alone. + +use std::future::Future; +use std::pin::Pin; +use std::sync::Arc; + +use nodedb_physical::physical_plan::{ClusterArrayOp, ExchangeOp, PhysicalPlan, QueryOp}; + +use crate::control::cluster::ClusterArrayExecutor; +use crate::control::server::payload_merge::{encode_msgpack_array, extract_msgpack_elements}; +use crate::control::state::SharedState; +use crate::data::executor::response_codec::{ArraySliceResponse, flatten_to_relational_rows}; +use crate::types::{Lsn, TxnId}; + +use super::gather::GatherOutcome; +use super::read_scope::ReadScope; +use super::resolve::exchange::provider_scan_of_rows; + +/// Gather `plan`, which holds a cluster-partitioned leaf, across the nodes +/// that own its rows. +pub(super) async fn gather_cluster_partitioned( + state: &SharedState, + plan: PhysicalPlan, + scope: ReadScope, +) -> crate::Result { + if let PhysicalPlan::ClusterArray(op) = &plan { + let rows = read_cluster_array(state, op, scope.txn_id).await?; + return Ok(outcome_of(flatten_to_relational_rows(&rows))); + } + let mut local = plan; + inline_cluster_arrays(state, &mut local, scope.txn_id).await?; + if nodedb_physical::physical_plan::plan_contains_cluster_partitioned_leaf(&local) { + return Err(crate::Error::Internal { + detail: "a graph or local array plan reached a cross-node gather; graph reads \ + run through the graph coordinators, and a cluster plans arrays as \ + ClusterArray" + .into(), + }); + } + // The rest of the plan reads ordinary collections, or nothing at all: it + // routes like any other gathered plan. `Box::pin` breaks the recursion + // back into `gather_all_vshards`, which cannot reach here again because + // the plan no longer holds a cluster-partitioned leaf. + Box::pin(super::gather::gather_all_vshards(state, local, scope)).await +} + +/// Read one cluster array op through the array executor and return its rows +/// as one msgpack array. +async fn read_cluster_array( + state: &SharedState, + op: &ClusterArrayOp, + txn_id: Option, +) -> crate::Result> { + if matches!( + op, + ClusterArrayOp::Put { .. } | ClusterArrayOp::Delete { .. } + ) { + return Err(crate::Error::Internal { + detail: "a cluster array write reached a cross-node gather".into(), + }); + } + let transport = state + .cluster_transport + .as_ref() + .ok_or_else(|| crate::Error::Internal { + detail: "cluster transport not available for a cluster array read".to_owned(), + })?; + let routing = state + .cluster_routing + .as_ref() + .ok_or_else(|| crate::Error::Internal { + detail: "cluster routing not available for a cluster array read".to_owned(), + })?; + let executor = ClusterArrayExecutor::new( + Arc::clone(transport), + Arc::clone(routing), + state.node_id, + state.self_arc()?, + ); + let payload = executor.execute(op, txn_id).await?; + cluster_array_rows(op, payload) +} + +/// The rows of a cluster array read's payload, as one msgpack array. +/// +/// A `Slice` answers an `ArraySliceResponse` envelope whose `rows_msgpack` is +/// the row array. An `Agg` answers the row array itself (`{group, result}` or +/// `{result}` maps). A slice whose `truncated_before_horizon` is set raises +/// the same client notice a slice shaped on its own reports +/// (`response_shape::compose::array_slice`), through the statement notice +/// slot (`session::statement_notice`). +fn cluster_array_rows(op: &ClusterArrayOp, payload: Vec) -> crate::Result> { + match op { + ClusterArrayOp::Slice { .. } => { + let response: ArraySliceResponse = + zerompk::from_msgpack(&payload).map_err(|e| crate::Error::Codec { + detail: format!("cluster array slice response decode: {e}"), + })?; + if response.truncated_before_horizon { + crate::control::server::shared::session::statement_notice::raise( + crate::control::server::response_shape::compose::array_slice::TRUNCATED_BEFORE_HORIZON_NOTICE + .to_string(), + ); + } + Ok(response.rows_msgpack) + } + ClusterArrayOp::Agg { .. } | ClusterArrayOp::Put { .. } | ClusterArrayOp::Delete { .. } => { + Ok(payload) + } + } +} + +/// Replace every cluster array leaf under a coordinator-local wrapper with a +/// `ProviderScan` of its rows. The wrappers walked are the ones +/// `plan_contains_cluster_partitioned_leaf` looks through. +fn inline_cluster_arrays<'a>( + state: &'a SharedState, + plan: &'a mut PhysicalPlan, + txn_id: Option, +) -> Pin> + Send + 'a>> { + Box::pin(async move { + match plan { + PhysicalPlan::ClusterArray(op) => { + let rows = read_cluster_array(state, op, txn_id).await?; + *plan = provider_scan_of_rows(flatten_to_relational_rows(&rows)); + } + PhysicalPlan::Query(QueryOp::Aggregate { input, .. }) => { + if let Some(child) = input.as_deref_mut() { + inline_cluster_arrays(state, child, txn_id).await?; + } + } + PhysicalPlan::Query(QueryOp::HashJoin { + left_input, + right_input, + left_bitmap, + right_bitmap, + .. + }) => { + for side in [left_input, right_input, left_bitmap, right_bitmap] { + if let Some(child) = side.as_deref_mut() { + inline_cluster_arrays(state, child, txn_id).await?; + } + } + } + PhysicalPlan::Query(QueryOp::Exchange(ExchangeOp { child, .. })) + | PhysicalPlan::Query(QueryOp::PostProcess { input: child, .. }) + | PhysicalPlan::Query(QueryOp::LateralTopK { + outer_plan: child, .. + }) + | PhysicalPlan::Query(QueryOp::LateralLoop { + outer_plan: child, .. + }) => { + inline_cluster_arrays(state, child, txn_id).await?; + } + PhysicalPlan::Query(QueryOp::SetOp { inputs, .. }) => { + for child in inputs.iter_mut() { + inline_cluster_arrays(state, child, txn_id).await?; + } + } + _ => {} + } + Ok(()) + }) +} + +/// A gather outcome holding one payload. +fn outcome_of(payload: Vec) -> GatherOutcome { + let merged_array = encode_msgpack_array(&extract_msgpack_elements(&payload)); + GatherOutcome { + raw: payload, + merged_array, + watermark_lsn: Lsn::ZERO, + read_version_lsn: Lsn::ZERO, + shard_watermarks: Vec::new(), + } +} + +#[cfg(test)] +mod tests { + use super::*; + + /// One flat row, the plain msgpack map `{"k": value}` a shard emits. + /// `nodedb_types::Value` encodes as a tagged array, not a plain map, so + /// the bytes are written directly: fixmap of 1, fixstr "k", positive + /// fixint `value`. + fn row(value: u8) -> Vec { + assert!(value < 0x80, "a positive fixint holds 0..=127"); + vec![0x81, 0xa1, b'k', value] + } + + fn array_id() -> nodedb_array::types::ArrayId { + nodedb_array::types::ArrayId::new(nodedb_types::TenantId::new(1), "grid") + } + + fn slice_op() -> ClusterArrayOp { + ClusterArrayOp::Slice { + array_id: array_id(), + slice_msgpack: Vec::new(), + attr_projection: Vec::new(), + limit: 0, + slice_hilbert_ranges: Vec::new(), + prefix_bits: 0, + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + } + } + + fn agg_op() -> ClusterArrayOp { + ClusterArrayOp::Agg { + array_id: array_id(), + attr_idx: 0, + reducer_msgpack: Vec::new(), + group_by_dim: -1, + slice_hilbert_ranges: Vec::new(), + prefix_bits: 0, + system_as_of: None, + valid_at_ms: None, + } + } + + #[test] + fn a_slice_envelope_flattens_to_its_rows() { + let rows = vec![row(1), row(2)]; + let envelope = zerompk::to_msgpack_vec(&ArraySliceResponse { + rows_msgpack: encode_msgpack_array(&rows), + truncated_before_horizon: false, + }) + .expect("encode slice envelope"); + let flat = flatten_to_relational_rows( + &cluster_array_rows(&slice_op(), envelope).expect("slice rows"), + ); + assert_eq!(extract_msgpack_elements(&flat), rows); + } + + #[tokio::test] + async fn a_truncated_slice_raises_the_horizon_notice() { + use crate::control::server::shared::session::{conn_scope, statement_notice}; + conn_scope::scoped(async { + let envelope = zerompk::to_msgpack_vec(&ArraySliceResponse { + rows_msgpack: encode_msgpack_array(&[row(1)]), + truncated_before_horizon: true, + }) + .expect("encode slice envelope"); + cluster_array_rows(&slice_op(), envelope).expect("slice rows"); + assert_eq!( + statement_notice::take(), + vec![ + crate::control::server::response_shape::compose::array_slice::TRUNCATED_BEFORE_HORIZON_NOTICE + .to_string() + ] + ); + }) + .await; + } + + #[test] + fn an_aggregate_payload_flattens_to_its_rows() { + let rows = vec![row(10), row(20), row(30)]; + let flat = flatten_to_relational_rows( + &cluster_array_rows(&agg_op(), encode_msgpack_array(&rows)).expect("agg rows"), + ); + assert_eq!(extract_msgpack_elements(&flat), rows); + } +} diff --git a/nodedb/src/control/server/exchange/core_outcome.rs b/nodedb/src/control/server/exchange/core_outcome.rs new file mode 100644 index 000000000..6b8746d42 --- /dev/null +++ b/nodedb/src/control/server/exchange/core_outcome.rs @@ -0,0 +1,171 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The one merge rule every all-core gather applies to its per-core answers. +//! +//! Each core holds only the state its own vShards home to. A gather that +//! merges the cores that answered and skips a core that failed returns a +//! partial answer as a complete one. Every all-core gather therefore +//! classifies each core's response here and merges through +//! [`require_every_core`]. + +use crate::bridge::envelope::{Response, Status}; + +/// One core's answer: `Some` when the core answered, `None` when it refused +/// with `NotFound`, `Err` when it failed. +pub(crate) type CoreOutcome = crate::Result>; + +/// Classify one core's collected response. +/// +/// - `NotFound` is `Ok(None)`: the core holds no slice of the target. +/// - An empty payload is `Ok(Some)`: the core answered with no rows. +/// - Every other error status is `Err` with the core's typed error. +pub(crate) fn classify_core_response(collected: crate::Result) -> CoreOutcome { + let resp = collected?; + if resp.status == Status::Error { + crate::control::local_dispatch::reject_data_plane_error(&resp)?; + return Ok(None); + } + Ok(Some(resp)) +} + +/// Require every core to answer, and return the answers in core order. +/// +/// The first error in core order fails the whole gather. It crosses as the +/// core's own typed error, so its SQLSTATE class survives. +pub(crate) fn require_every_core( + outcomes: impl IntoIterator>, +) -> crate::Result> { + let mut answered = Vec::new(); + for outcome in outcomes { + if let Some(answer) = outcome? { + answered.push(answer); + } + } + Ok(answered) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::bridge::envelope::{ErrorCode, Payload}; + use crate::types::{Lsn, RequestId}; + + fn response(status: Status, payload: &[u8], code: Option) -> Response { + Response { + request_id: RequestId::new(7), + status, + attempt: 1, + partial: false, + payload: Payload::from_vec(payload.to_vec()), + watermark_lsn: Lsn::ZERO, + error_code: code.map(Box::new), + read_set_valid: None, + read_version_lsn: Lsn::ZERO, + write_set: Vec::new(), + } + } + + fn answered(payload: &[u8]) -> crate::Result { + Ok(response(Status::Ok, payload, None)) + } + + fn refused(code: ErrorCode) -> crate::Result { + Ok(response(Status::Error, &[], Some(code))) + } + + fn payloads(merged: &[Response]) -> Vec> { + merged.iter().map(|r| r.payload.as_ref().to_vec()).collect() + } + + #[test] + fn every_core_answering_merges_in_core_order() { + let merged = require_every_core( + [answered(b"a"), answered(b"b"), answered(b"c")] + .into_iter() + .map(classify_core_response), + ) + .unwrap(); + assert_eq!( + payloads(&merged), + vec![b"a".to_vec(), b"b".to_vec(), b"c".to_vec()] + ); + } + + #[test] + fn one_failed_core_fails_the_gather_with_its_own_code() { + let result = require_every_core( + [ + answered(b"a"), + refused(ErrorCode::DivisionByZero), + answered(b"c"), + ] + .into_iter() + .map(classify_core_response), + ); + match result { + Err(crate::Error::DataPlane(ErrorCode::DivisionByZero)) => {} + other => panic!("expected the failed core's own code, got {other:?}"), + } + } + + #[test] + fn a_core_that_never_answered_fails_the_gather() { + let result = require_every_core( + [ + answered(b"a"), + Err(crate::Error::DeadlineExceeded { + request_id: RequestId::new(3), + }), + ] + .into_iter() + .map(classify_core_response), + ); + match result { + Err(crate::Error::DeadlineExceeded { request_id }) => { + assert_eq!(request_id, RequestId::new(3)); + } + other => panic!("expected the deadline, got {other:?}"), + } + } + + #[test] + fn the_first_error_in_core_order_wins() { + let result = require_every_core( + [ + refused(ErrorCode::DivisionByZero), + refused(ErrorCode::DeadlineExceeded), + ] + .into_iter() + .map(classify_core_response), + ); + assert!(matches!( + result, + Err(crate::Error::DataPlane(ErrorCode::DivisionByZero)) + )); + } + + #[test] + fn an_empty_answer_stays_distinct_from_a_missing_one() { + let merged = require_every_core( + [answered(b""), refused(ErrorCode::NotFound), answered(b"c")] + .into_iter() + .map(classify_core_response), + ) + .unwrap(); + // The empty answer is kept. The `NotFound` core contributes nothing. + assert_eq!(payloads(&merged), vec![Vec::new(), b"c".to_vec()]); + } + + #[test] + fn an_error_status_without_a_code_fails_closed() { + let result = require_every_core( + [answered(b"a"), Ok(response(Status::Error, &[], None))] + .into_iter() + .map(classify_core_response), + ); + assert!(matches!( + result, + Err(crate::Error::DataPlane(ErrorCode::Internal { .. })) + )); + } +} diff --git a/nodedb/src/control/server/exchange/gather.rs b/nodedb/src/control/server/exchange/gather.rs index 07b57d9b8..7a1ada834 100644 --- a/nodedb/src/control/server/exchange/gather.rs +++ b/nodedb/src/control/server/exchange/gather.rs @@ -17,9 +17,11 @@ use futures::future::join_all; -use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Response, Status}; +use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Response}; use crate::control::arrow_convert; use crate::control::gateway::core::QueryContext; +use crate::control::server::exchange::core_outcome::{classify_core_response, require_every_core}; +use crate::control::server::exchange::read_scope::ReadScope; use crate::control::server::payload_merge::{encode_msgpack_array, extract_msgpack_elements}; use crate::control::server::result_stream::ResultStream; use crate::control::server::shared::session::statement_deadline; @@ -44,7 +46,7 @@ pub(crate) use super::response::stream_to_response; /// task ran inside a transaction block); it is threaded onto every per-core /// `Request` so the Data-Plane scan handler can merge the transaction's /// staging overlay (read-your-own-writes). Autocommit / non-transactional -/// callers pass `None`, which reproduces prior behaviour exactly. +/// callers pass `None`, which merges no overlay. /// /// Each entry is `(core_id, request_id, receiver)`. The request id travels /// with the receiver so a collect that ends at the deadline can name the @@ -64,10 +66,37 @@ pub(crate) fn eager_dispatch_to_all_cores( )>, > { // Every core in this fan-out belongs to ONE statement, so all of them - // carry that statement's deadline. Resolving per core would give each core + // carry that statement's deadline. Resolving per core will give each core // its own budget and leave the statement unbounded in aggregate. let deadline = statement_deadline(state.tuning.network.default_deadline_secs); + eager_dispatch_to_all_cores_until( + state, + tenant_id, + database_id, + trace_id, + txn_id, + deadline, + plan_for_core, + ) +} +/// [`eager_dispatch_to_all_cores`] with an explicit deadline on every core's +/// request, for a background task that runs outside any statement. +pub(crate) fn eager_dispatch_to_all_cores_until( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + trace_id: TraceId, + txn_id: Option, + deadline: std::time::Instant, + plan_for_core: impl Fn(usize) -> PhysicalPlan, +) -> crate::Result< + Vec<( + usize, + crate::types::RequestId, + crate::control::ResponseReceiver, + )>, +> { let num_cores = state .dispatcher .lock() @@ -96,6 +125,7 @@ pub(crate) fn eager_dispatch_to_all_cores( txn_id, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: crate::bridge::envelope::Admission::Exempt( crate::bridge::envelope::ExemptReason::Read, ), @@ -142,10 +172,8 @@ pub struct GatherOutcome { /// /// All per-core sends are issued before any response is awaited (`join_all`). /// `NotFound` errors from individual cores are treated as "no rows" (the -/// collection shard simply has no matching data on that core). Any other -/// error status from a core fails the whole gather: a fan-out that lost one -/// core's contribution answers a different question than the one asked, so a -/// truncated row set must never reach the client as a success. +/// collection shard has no matching data on that core). Any other core +/// error fails the whole gather through `core_outcome::require_every_core`. pub(crate) async fn gather_all_cores( state: &SharedState, tenant_id: TenantId, @@ -168,9 +196,9 @@ pub(crate) async fn gather_all_cores( })?; // Await all responses in parallel using join_all. Each core's scan result - // may stream as several `Partial` frames before its terminal frame, so drain + // can stream as several `Partial` frames before its terminal frame, so drain // and concatenate the full bounded response per core — taking only the first - // frame would silently truncate that core's contribution to `stream_chunk_size` + // frame will silently truncate that core's contribution to `stream_chunk_size` // rows. let max_result_bytes = state.tuning.network.max_query_result_bytes as usize; let response_futures = receivers @@ -191,6 +219,9 @@ pub(crate) async fn gather_all_cores( }); let results: Vec<(usize, crate::Result)> = join_all(response_futures).await; + let answered = require_every_core(results.into_iter().map(|(core_id, result)| { + classify_core_response(result).map(|resp| resp.map(|resp| (core_id, resp))) + }))?; let mut raw = Vec::new(); let mut all_elements: Vec> = Vec::new(); @@ -201,31 +232,8 @@ pub(crate) async fn gather_all_cores( // the read plan targets one collection, so one non-zero value survives. let mut max_read_version = Lsn::ZERO; let mut shard_watermarks: Vec<(VShardId, Lsn)> = Vec::new(); - // First error seen across cores, kept as a TYPED `crate::Error` so a code - // like `DivisionByZero` surfaces as SQLSTATE 22012 rather than collapsing - // to a generic `Dispatch` (XX000). - let mut first_error: Option = None; - - for (core_id, result) in results { - let resp = match result { - Ok(r) => r, - Err(e) => { - if first_error.is_none() { - first_error = Some(e); - } - continue; - } - }; - - if resp.status == Status::Error { - if let Err(e) = crate::control::local_dispatch::reject_data_plane_error(&resp) - && first_error.is_none() - { - first_error = Some(e); - } - continue; - } + for (core_id, resp) in answered { // Record this core's own watermark as a participating-shard version, // even when its payload is empty — an empty scan slice is still a // validatable observation at that shard's version (phantom safety). @@ -247,10 +255,6 @@ pub(crate) async fn gather_all_cores( all_elements.extend(extract_msgpack_elements(payload_bytes)); } - if let Some(err) = first_error { - return Err(err); - } - let merged_array = encode_msgpack_array(&all_elements); Ok(GatherOutcome { @@ -272,7 +276,7 @@ pub(crate) async fn gather_all_cores( /// /// NotFound tolerance matches `gather_all_cores`: a per-core terminal /// `Status::Error` with `ErrorCode::NotFound` ends that core's stream cleanly -/// (the collection shard simply has no rows on that core) rather than failing +/// (the collection shard has no rows on that core) rather than failing /// the whole stream. Any other error status propagates as a stream `Err`. This /// is handled by passing `tolerate_not_found: true` to /// [`stream_response_channel`], which centralizes the NotFound-vs-error @@ -283,23 +287,6 @@ pub(crate) async fn gather_all_cores( /// connection-level deadline, and a streamed result has no single point at /// which to apply a fan-out timeout without buffering. The request deadline in /// each per-core `Request` envelope still bounds Data-Plane work. -/// Consume an authorized scan before entering the internal all-core fan-out. -pub fn gather_all_cores_stream_authorized( - state: &SharedState, - authorized: crate::control::server::shared::authorization::AuthorizedTask, - trace_id: TraceId, -) -> crate::Result { - let task = authorized.into_physical_task(); - gather_all_cores_stream( - state, - task.tenant_id, - task.database_id, - task.plan, - trace_id, - task.txn_id, - ) -} - pub(crate) fn gather_all_cores_stream( state: &SharedState, tenant_id: TenantId, @@ -331,13 +318,8 @@ pub(crate) fn gather_all_cores_stream( /// Cluster-wide gather with routing awareness. /// -/// # Single-node mode -/// -/// If `state.gateway` is `None`, routing is delegated to -/// [`super::owning_core::gather_single_node`] (same shape-based routing). -/// -/// # Cluster mode — single-vShard-homed sources (document, kv, columnar, -/// timeseries, spatial, vector, text) +/// # Single-vShard-homed sources (document, kv, columnar, timeseries, +/// spatial, vector, text) /// /// Standard collections are *single-vShard-homed*: all rows for a collection /// live on exactly one vShard determined by `vshard_for_collection` over the @@ -349,63 +331,46 @@ pub(crate) fn gather_all_cores_stream( /// `route_plan` `other` arm, which sends it directly to the single owning /// vShard (local or remote) and returns exactly the right rows. /// -/// # Cluster mode — cluster-partitioned sources (graph traversal, array) +/// # Cluster-partitioned sources (array, graph) /// -/// Graph traversal ops and Array ops distribute data across vShards by node-id -/// or tile-id. Cross-node gather for these sources requires a dedicated -/// scatter-gather path that does not yet exist. To avoid producing wrong -/// results we fall back to the local `gather_all_cores` path. +/// A cluster array read spreads its rows across shards by tile. It runs +/// through the array executor, which reads each tile from the shard that owns +/// it (`cluster_leaf`). A graph read never reaches a gather: the SQL planner +/// emits no graph leaf, and graph reads run through `graph_dispatch`. /// -/// TRACKED DEBT: cross-node gather for genuinely vShard-partitioned sources -/// (graph traversal / array) needs its own broadcast + vshard-scoped path. /// The Exchange{Gather} broadcast approach is NOT correct for single-vShard- /// homed collections and must not be reinstated for them. pub(crate) async fn gather_all_vshards( state: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, plan: PhysicalPlan, - trace_id: TraceId, - txn_id: Option, + scope: ReadScope, ) -> crate::Result { - let Some(gateway) = state.gateway.get() else { - // Single-node: route by plan shape (cluster-partitioned leaf → broadcast; - // single-vShard-homed collection → its one owning core; else broadcast - // fallback), mirroring the cluster branch below. - return super::owning_core::gather_single_node( - state, - tenant_id, - database_id, - plan, - trace_id, - txn_id, - ) - .await; - }; + let ReadScope { + database_id, + tenant_id, + trace_id, + txn_id, + linearizable, + } = scope; + let gateway = state.installed_gateway()?; if nodedb_physical::physical_plan::plan_contains_cluster_partitioned_leaf(&plan) { - // Graph node-id / array tile partitioning: cross-node gather via this - // primitive is NOT yet correct (these engines have dedicated scatter- - // gather paths). Fall back to the prior local fan to avoid introducing - // wrong results. - // TRACKED DEBT: cross-node gather for genuinely vShard-partitioned - // sources (graph traversal / array) needs its own broadcast + - // vshard-scoped path. Do not replace this fallback with Exchange{Gather} - // broadcasting — that path is only correct for single-vShard-homed - // collections. - return gather_all_cores(state, tenant_id, database_id, plan, trace_id, txn_id).await; + // Array rows are spread by tile: the array executor reads each tile + // from its owning shard (`cluster_leaf`). + return super::cluster_leaf::gather_cluster_partitioned(state, plan, scope).await; } // Single-vShard-homed source (document/kv/columnar/ts/spatial/vector/text): // the whole collection lives on ONE vShard. Route the BARE plan through the // gateway so route_plan's `other` arm sends it to that single owning vShard - // (local or remote). Do NOT wrap in Exchange{Gather} — broadcasting would + // (local or remote). Do NOT wrap in Exchange{Gather} — broadcasting will // duplicate rows because the data-plane scan is not vshard-scoped. let ctx = QueryContext { tenant_id, trace_id, database_id, txn_id, + linearizable, }; // `Box::pin` breaks an async-fn recursion cycle: the gateway dispatches the @@ -415,7 +380,7 @@ pub(crate) async fn gather_all_vshards( // resolve is a no-op), but the future must be heap-indirected so its size // is finite. // The gateway already fails with a typed `crate::Error` — a shard's - // `Error::DataPlane` code included. Re-wrapping it in `Dispatch` would + // `Error::DataPlane` code included. Re-wrapping it in `Dispatch` will // rewrite every such verdict as SQLSTATE XX000, so it passes through. let (payloads, shard_watermarks, read_version_lsn): (Vec>, Vec<(VShardId, Lsn)>, Lsn) = Box::pin(gateway.execute_internal_with_watermarks(&ctx, plan)).await?; diff --git a/nodedb/src/control/server/exchange/mod.rs b/nodedb/src/control/server/exchange/mod.rs index 09156cc84..c1b44ef4a 100644 --- a/nodedb/src/control/server/exchange/mod.rs +++ b/nodedb/src/control/server/exchange/mod.rs @@ -9,17 +9,25 @@ //! lazy query sinks (pgwire fast path, native protocol, HTTP-NDJSON). pub mod all_cores; +mod cluster_leaf; +pub mod core_outcome; pub mod full_scan; pub mod gather; pub mod owning_core; +pub mod read_scope; +pub mod received; pub mod resolve; pub mod response; pub mod streamable; pub use all_cores::NodeLevelResult; -pub(crate) use all_cores::{execute_plan_all_local_cores, snapshot_tenant_on_local_cores}; +pub(crate) use all_cores::{ + capture_base_on_local_cores, execute_plan_all_local_cores, snapshot_tenant_on_local_cores, +}; pub(crate) use gather::gather_all_cores; pub use gather::{GatherOutcome, finalize_aggregate}; +pub use read_scope::ReadScope; +pub(crate) use received::execute_received_plan; pub use resolve::{ DistributedReadCapture, Resolved, resolve_and_materialize, resolve_exchange_in_plan, }; diff --git a/nodedb/src/control/server/exchange/owning_core.rs b/nodedb/src/control/server/exchange/owning_core.rs index 698e9dddd..a90eaea45 100644 --- a/nodedb/src/control/server/exchange/owning_core.rs +++ b/nodedb/src/control/server/exchange/owning_core.rs @@ -2,69 +2,21 @@ //! Route a single-vShard-homed plan to its ONE owning Data-Plane core. //! -//! `gather_single_owning_core` is the single-node sibling of -//! [`super::gather::gather_all_cores`] for plans whose whole collection lives on -//! exactly one vShard (document / kv / columnar / timeseries / spatial / vector -//! / text). Instead of broadcasting the plan to every core — which seeds an -//! identity scalar-aggregate row on each empty non-owning core and returns one -//! row per core for a no-`GROUP BY` aggregate — it resolves the collection's -//! owning vShard and dispatches the bare plan to it alone, exactly mirroring -//! what the cluster branch of [`super::gather::gather_all_vshards`] does via the -//! gateway. The owning core already holds every row of a single-vShard-homed -//! collection, so this returns the identical row set a broadcast would have, -//! minus the empty cores' spurious contributions. +//! `gather_single_owning_core` is the one-core sibling of +//! [`super::gather::gather_all_cores`]. It dispatches the bare plan to the one +//! core that owns a vShard, the way [`super::gather::gather_all_vshards`] +//! routes a single-vShard-homed plan through the gateway. Broadcasting instead +//! will seed an identity scalar-aggregate row on each empty non-owning core, +//! and a no-`GROUP BY` aggregate will return one row per core. use crate::bridge::envelope::PhysicalPlan; use crate::control::local_dispatch::reject_data_plane_error; -use crate::control::server::dispatch_utils::dispatch_to_data_plane_with_txn; +use crate::control::server::dispatch_utils::dispatch_routed_read_to_data_plane; use crate::control::server::payload_merge::{encode_msgpack_array, extract_msgpack_elements}; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId, TxnId, VShardId}; -use super::gather::{GatherOutcome, gather_all_cores}; - -/// Single-node routing decision for a resolved `Exchange{Gather}` child plan. -/// -/// Mirrors the cluster branch of [`super::gather::gather_all_vshards`]: -/// -/// - A cluster-partitioned leaf (graph traversal / array) spreads its rows -/// across cores by node-id / tile-id, so it fans to every local core via -/// [`gather_all_cores`]. -/// - A single-vShard-homed collection (document / kv / columnar / timeseries / -/// spatial / vector / text) lives wholly on ONE core; the bare plan routes to -/// that owning core via [`gather_single_owning_core`]. Broadcasting would seed -/// a scalar-aggregate identity row on every empty non-owning core — so a -/// no-`GROUP BY` aggregate returns one row PER core instead of one merged row -/// — and would duplicate a plain scan's rows across cores. -/// - A plan with no resolvable collection (e.g. `ProviderScan` carrying embedded -/// rows) keeps the broadcast fallback unchanged. -pub async fn gather_single_node( - state: &SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - plan: PhysicalPlan, - trace_id: TraceId, - txn_id: Option, -) -> crate::Result { - if nodedb_physical::physical_plan::plan_contains_cluster_partitioned_leaf(&plan) { - return gather_all_cores(state, tenant_id, database_id, plan, trace_id, txn_id).await; - } - if let Some(collection) = plan.collection() { - let vshard_id = - nodedb_types::CollectionKey::from_qualified_str(database_id, collection)?.vshard(); - return gather_single_owning_core( - state, - tenant_id, - database_id, - plan, - vshard_id, - trace_id, - txn_id, - ) - .await; - } - gather_all_cores(state, tenant_id, database_id, plan, trace_id, txn_id).await -} +use super::gather::GatherOutcome; /// Dispatch `plan` to the single Data-Plane core that owns `vshard_id` and /// gather the one bounded response into a [`GatherOutcome`]. @@ -128,7 +80,9 @@ pub async fn dispatch_single_owning_core( // re-enters `resolve_exchange_in_plan`. The plan handed here is the bare, // Exchange-free child of the resolved Gather, so the re-entrant resolve is a // no-op — but the future must be heap-indirected so its size stays finite. - let resp = Box::pin(dispatch_to_data_plane_with_txn( + // Every caller already routed the read to this node and confirmed it: a + // single-node gather, or a leg another node sent here. + let resp = Box::pin(dispatch_routed_read_to_data_plane( state, tenant_id, database_id, diff --git a/nodedb/src/control/server/exchange/read_scope.rs b/nodedb/src/control/server/exchange/read_scope.rs new file mode 100644 index 000000000..35df71a58 --- /dev/null +++ b/nodedb/src/control/server/exchange/read_scope.rs @@ -0,0 +1,20 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Request scope threaded through plan resolution and every gather it runs. + +use crate::types::{DatabaseId, TenantId, TraceId, TxnId}; + +/// Who a distributed read runs for, and what it must observe. +#[derive(Debug, Clone, Copy)] +pub struct ReadScope { + pub database_id: DatabaseId, + pub tenant_id: TenantId, + pub trace_id: TraceId, + /// The session transaction the read runs in, for its staging overlay. + /// `None` for autocommit and internal reads. + pub txn_id: Option, + /// Every leg observes all writes committed before the read began: the + /// node serving each leg confirms the leg's group first (see + /// `control::cluster::linearizable_read`). + pub linearizable: bool, +} diff --git a/nodedb/src/control/server/exchange/received.rs b/nodedb/src/control/server/exchange/received.rs new file mode 100644 index 000000000..322071d70 --- /dev/null +++ b/nodedb/src/control/server/exchange/received.rs @@ -0,0 +1,151 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Run a plan that an `ExecuteRequest` carried to this node. +//! +//! A transaction meta-op (`StageWrite`, `ResolveTxn`, `DropTxnOverlay`, ...) +//! runs on the one core owning its vShard, because only that core holds the +//! transaction's staging overlay. Fanning it to every core stages the write on +//! each of them and returns a non-owning core's answer. Every other plan fans +//! across all local cores. + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::gateway::router::is_task_vshard_scoped; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, TraceId, TxnId, VShardId}; + +use super::all_cores::{NodeLevelResult, execute_plan_all_local_cores}; +use super::owning_core::dispatch_single_owning_core; + +/// Where a received plan runs on this node. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum ReceivedRoute { + /// The one core owning this vShard. + OwningCore(VShardId), + /// Every local core, merged. + AllCores, +} + +/// Decide where a received plan runs. +/// +/// The request's vShard must match the plan's scope. A scoped plan without a +/// vShard will fan to every core. A vShard on any other plan means the +/// sender and receiver disagree on the plan's scope. +fn route_received_plan( + plan: &PhysicalPlan, + scoped_vshard: Option, +) -> crate::Result { + match (is_task_vshard_scoped(plan), scoped_vshard) { + (true, Some(vshard_id)) => Ok(ReceivedRoute::OwningCore(vshard_id)), + (false, None) => Ok(ReceivedRoute::AllCores), + (true, None) => Err(crate::Error::Internal { + detail: "execute request: a vShard-scoped plan arrived without its vShard".into(), + }), + (false, Some(vshard_id)) => Err(crate::Error::Internal { + detail: format!( + "execute request: vShard {} arrived with a plan that is not vShard-scoped", + vshard_id.as_u32() + ), + }), + } +} + +/// Run a received plan on the core or cores [`route_received_plan`] picks. +pub(crate) async fn execute_received_plan( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + plan: PhysicalPlan, + trace_id: TraceId, + txn_id: Option, + scoped_vshard: Option, +) -> crate::Result { + match route_received_plan(&plan, scoped_vshard)? { + ReceivedRoute::OwningCore(vshard_id) => { + let resp = dispatch_single_owning_core( + state, + tenant_id, + database_id, + plan, + vshard_id, + trace_id, + txn_id, + ) + .await?; + Ok(NodeLevelResult { + payload: resp.payload.to_vec(), + watermark_lsn: resp.watermark_lsn, + read_version_lsn: resp.read_version_lsn, + }) + } + ReceivedRoute::AllCores => { + execute_plan_all_local_cores(state, tenant_id, database_id, plan, trace_id, txn_id) + .await + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_physical::physical_plan::{KvOp, MetaOp}; + use nodedb_types::QualifiedCollection; + + fn kv_put() -> PhysicalPlan { + PhysicalPlan::Kv(KvOp::Put { + collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), + key: b"k".to_vec(), + value: b"v".to_vec(), + ttl_ms: 0, + surrogate: nodedb_types::Surrogate::new(1), + returning: None, + rls_filters: Vec::new(), + provenance: None, + }) + } + + #[test] + fn a_scoped_request_reaches_exactly_its_owning_core() { + let stage = PhysicalPlan::Meta(MetaOp::StageWrite { + plan: Box::new(kv_put()), + }); + assert_eq!( + route_received_plan(&stage, Some(VShardId::new(41))).unwrap(), + ReceivedRoute::OwningCore(VShardId::new(41)) + ); + let resolve = PhysicalPlan::Meta(MetaOp::ResolveTxn { + txn_id: TxnId::new(3), + plans: vec![kv_put()], + }); + assert_eq!( + route_received_plan(&resolve, Some(VShardId::new(7))).unwrap(), + ReceivedRoute::OwningCore(VShardId::new(7)) + ); + } + + #[test] + fn an_unscoped_request_fans_across_every_core() { + assert_eq!( + route_received_plan(&kv_put(), None).unwrap(), + ReceivedRoute::AllCores + ); + } + + #[test] + fn a_scoped_plan_without_its_vshard_is_refused() { + let drop = PhysicalPlan::Meta(MetaOp::DropTxnOverlay { + txn_id: TxnId::new(3), + }); + assert!(matches!( + route_received_plan(&drop, None), + Err(crate::Error::Internal { .. }) + )); + } + + #[test] + fn a_vshard_on_an_unscoped_plan_is_refused() { + assert!(matches!( + route_received_plan(&kv_put(), Some(VShardId::new(1))), + Err(crate::Error::Internal { .. }) + )); + } +} diff --git a/nodedb/src/control/server/exchange/resolve/exchange/aggregate_input_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/aggregate_input_arm.rs index 63667c288..e7e82f3da 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/aggregate_input_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/aggregate_input_arm.rs @@ -12,9 +12,9 @@ use nodedb_types::QualifiedCollection; use crate::control::server::exchange::resolve::capture::DistributedReadCapture; use crate::control::state::SharedState; -use super::dispatch::ResolveCtx; use super::entry::Resolved; use super::post_process_arm::{ChildRows, materialize_child_rows, provider_scan_of_rows}; +use crate::control::server::exchange::read_scope::ReadScope; /// Fields of a `QueryOp::Aggregate { input: Some(_) }` plan node, carried /// through resolution as one value. @@ -42,7 +42,7 @@ pub(super) struct AggregateFields { /// full relation and no Exchange reaches a Data-Plane core. pub(super) async fn resolve_aggregate_input( state: &SharedState, - ctx: ResolveCtx, + ctx: ReadScope, captures: &mut Vec, fields: AggregateFields, ) -> crate::Result { diff --git a/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs b/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs index 29212c4a4..08df70dca 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/dispatch.rs @@ -5,9 +5,9 @@ use nodedb_physical::physical_plan::{ExchangeMode, ExchangeOp, PhysicalPlan, QueryOp}; +use crate::control::server::exchange::read_scope::ReadScope; use crate::control::server::exchange::resolve::capture::DistributedReadCapture; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, TraceId, TxnId}; use super::aggregate_input_arm::AggregateFields; use super::entry::Resolved; @@ -17,16 +17,6 @@ use super::{ aggregate_input_arm, gather_arm, hash_join_arm, post_process_arm, set_op_arm, shuffle_arm, }; -/// Request-scoped identifiers threaded through every arm resolver, bundled -/// to keep each resolver's argument list within the clippy default arity. -#[derive(Clone, Copy)] -pub(super) struct ResolveCtx { - pub database_id: DatabaseId, - pub tenant_id: TenantId, - pub trace_id: TraceId, - pub txn_id: Option, -} - /// Resolve any `Exchange` nodes in `plan`. /// /// - Root-level `Gather` → gather all vShards, return `Resolved::Gathered`. @@ -52,24 +42,15 @@ pub(super) struct ResolveCtx { /// already-taken captures up unchanged. pub(super) async fn resolve_exchange( state: &SharedState, - database_id: DatabaseId, - tenant_id: TenantId, + ctx: ReadScope, plan: PhysicalPlan, - trace_id: TraceId, - txn_id: Option, captures: &mut Vec, ) -> crate::Result { - let ctx = ResolveCtx { - database_id, - tenant_id, - trace_id, - txn_id, - }; match plan { // Root-level Gather: fan child to all vShards and merge. First resolve any // Exchange{Broadcast} nodes nested inside the child (e.g. a HashJoin's // build side) so the plan fanned to cores is self-contained — no - // Exchange node may reach a Data-Plane core. + // Exchange node can reach a Data-Plane core. PhysicalPlan::Query(QueryOp::Exchange(ExchangeOp { child, mode: ExchangeMode::Gather { as_aggregate }, diff --git a/nodedb/src/control/server/exchange/resolve/exchange/entry.rs b/nodedb/src/control/server/exchange/resolve/exchange/entry.rs index 355a63d66..5ea370be7 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/entry.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/entry.rs @@ -6,8 +6,9 @@ use nodedb_physical::physical_plan::PhysicalPlan; use crate::bridge::envelope::Response; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::exchange::read_scope::ReadScope; use crate::control::state::SharedState; -use crate::types::{DatabaseId, Lsn, TenantId, TraceId, TxnId, VShardId}; +use crate::types::{Lsn, VShardId}; use crate::control::server::exchange::resolve::capture::DistributedReadCapture; use crate::control::server::exchange::resolve::materialize::materialize_providers; @@ -31,12 +32,12 @@ pub enum Resolved { /// gathered `HashJoin`) and the SHUFFLE JOIN path (probe/left and /// build/right). The record seam records one read-set entry per capture, so /// EVERY participating collection's vshard is validated at commit rather than - /// just the plan's collapsed left collection. Empty when there is no + /// only the plan's collapsed left collection. Empty when there is no /// in-transaction base-collection capture (autocommit reads, and shuffle /// AGGREGATE which carries its single read version on the response scalar). Gathered(Response, Vec<(VShardId, Lsn)>, Vec), /// The plan (possibly mutated by catalog materialization or Broadcast - /// embedding) is self-contained and should be dispatched normally. + /// embedding) is self-contained and is dispatched normally. Plan(Box), /// The plan was a single-node, unordered, non-aggregate scan eligible for /// streaming. The coordinator has eagerly dispatched it to all cores; the @@ -50,19 +51,17 @@ pub enum Resolved { /// /// See module-level documentation for the two-pass behaviour. /// -/// `txn_id` is the originating session transaction id (if the dispatching -/// task ran inside a transaction block); it is threaded down to every -/// per-core `Request` built by the gather primitives so in-transaction scans -/// can merge the transaction's staging overlay (read-your-own-writes). -/// Autocommit / non-transactional callers pass `None`. +/// `scope.txn_id` is the originating session transaction id (if the +/// dispatching task ran inside a transaction block); it is threaded down to +/// every per-core `Request` built by the gather primitives so in-transaction +/// scans can merge the transaction's staging overlay (read-your-own-writes). +/// Autocommit / non-transactional callers pass `None`. `scope.linearizable` +/// makes every leg confirm its group where it is served. pub async fn resolve_and_materialize( state: &SharedState, identity: &AuthenticatedIdentity, - database_id: DatabaseId, - tenant_id: TenantId, plan: PhysicalPlan, - trace_id: TraceId, - txn_id: Option, + scope: ReadScope, ) -> crate::Result { // Pass 1: fill empty ProviderScan rows (identity-scoped, per-request). let plan = materialize_providers(state, identity, plan).await?; @@ -71,16 +70,7 @@ pub async fn resolve_and_materialize( // base-collection gather point beneath the plan root and consumed (taken) // once at the root arm that returns `Resolved::Gathered`. let mut captures = Vec::new(); - resolve_exchange( - state, - database_id, - tenant_id, - plan, - trace_id, - txn_id, - &mut captures, - ) - .await + resolve_exchange(state, scope, plan, &mut captures).await } /// Resolve only `Exchange` nodes (pass 2), without catalog provider @@ -92,24 +82,12 @@ pub async fn resolve_and_materialize( /// pgwire/native paths that own the request identity. A no-op for plans with no /// `Exchange` node. /// -/// See `resolve_and_materialize` for `txn_id` semantics. +/// See `resolve_and_materialize` for the `scope` semantics. pub async fn resolve_exchange_in_plan( state: &SharedState, - database_id: DatabaseId, - tenant_id: TenantId, plan: PhysicalPlan, - trace_id: TraceId, - txn_id: Option, + scope: ReadScope, ) -> crate::Result { let mut captures = Vec::new(); - resolve_exchange( - state, - database_id, - tenant_id, - plan, - trace_id, - txn_id, - &mut captures, - ) - .await + resolve_exchange(state, scope, plan, &mut captures).await } diff --git a/nodedb/src/control/server/exchange/resolve/exchange/gather_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/gather_arm.rs index f21b67e4f..cd31c918b 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/gather_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/gather_arm.rs @@ -7,40 +7,32 @@ use nodedb_physical::physical_plan::PhysicalPlan; use crate::control::server::exchange::full_scan::{ScanSide, full_scan_plan_for_collection}; use crate::control::server::exchange::gather::{ - GatherOutcome, finalize_aggregate, gather_all_cores_stream, gather_all_vshards, - outcome_to_response, + GatherOutcome, finalize_aggregate, gather_all_vshards, outcome_to_response, }; use crate::control::server::exchange::resolve::capture::DistributedReadCapture; use crate::control::state::SharedState; -use super::dispatch::{ResolveCtx, resolve_exchange}; +use crate::control::server::exchange::read_scope::ReadScope; + +use super::dispatch::resolve_exchange; use super::entry::Resolved; /// Resolve a root-level `Exchange{Gather}` node. pub(super) async fn resolve_gather( state: &SharedState, - ctx: ResolveCtx, + ctx: ReadScope, child: PhysicalPlan, as_aggregate: bool, captures: &mut Vec, ) -> crate::Result { - let ResolveCtx { + let ReadScope { database_id, tenant_id, trace_id, txn_id, + linearizable, } = ctx; - let child = match Box::pin(resolve_exchange( - state, - database_id, - tenant_id, - child, - trace_id, - txn_id, - captures, - )) - .await? - { + let child = match Box::pin(resolve_exchange(state, ctx, child, captures)).await? { Resolved::Plan(p) => *p, Resolved::Gathered(resp, wms, caps) => { return Ok(Resolved::Gathered(resp, wms, caps)); @@ -56,12 +48,10 @@ pub(super) async fn resolve_gather( // Streaming fast path: a non-aggregate, unordered scan can stream // straight to the client without coordinator-side materialization. // - // - Single-node (`gateway.is_none()`): fan to all local cores via - // `gather_all_cores_stream`. - // - Cluster (`gateway.is_some()`): `gateway.execute_stream` routes - // the scan to its owning vShard — local cores when this node owns - // it, or the remote owner over QUIC (L4 streaming transport) — - // and merges the per-route streams with the same `select_all`. + // `gateway.execute_stream_internal` routes the scan to its owning + // vShard — local cores when this node owns it, or the remote owner over + // QUIC (L4 streaming transport) — and merges the per-route streams with + // `select_all`. // // Aggregate gathers keep the materialize-then-merge behaviour. // @@ -72,21 +62,15 @@ pub(super) async fn resolve_gather( // `gather_all_vshards` branch below whose `GatherOutcome` preserves // `shard_watermarks`. if !as_aggregate && txn_id.is_none() && child.is_streamable_unordered_scan() { - let stream = if let Some(gw) = state.gateway.get() { - let ctx = crate::control::gateway::core::QueryContext { - tenant_id, - trace_id, - database_id, - txn_id: None, - }; - // NOTE: cluster mode does not yet thread `txn_id` through - // `gateway.execute_stream` — cross-node in-transaction - // read-your-own-writes is a tracked gap; single-node - // (`gather_all_cores_stream` below) is fixed. - gw.execute_stream_internal(&ctx, child).await? - } else { - gather_all_cores_stream(state, tenant_id, database_id, child, trace_id, txn_id)? + let gateway = state.installed_gateway()?; + let ctx = crate::control::gateway::core::QueryContext { + tenant_id, + trace_id, + database_id, + txn_id: None, + linearizable, }; + let stream = gateway.execute_stream_internal(&ctx, child).await?; return Ok(Resolved::Stream(stream)); } @@ -111,14 +95,13 @@ pub(super) async fn resolve_gather( None }; - let outcome: GatherOutcome = - gather_all_vshards(state, tenant_id, database_id, child, trace_id, txn_id).await?; + let outcome: GatherOutcome = gather_all_vshards(state, child, ctx).await?; // Record the probe/single-collection read at its OWN observed // read-version (the gathered collection's `coll_write_lsn`), scoped to // a bare single-collection scan so the commit-time OCC validator // re-homes and revalidates exactly that collection's vshard. A - // `HashJoin` plan would otherwise collapse to the left collection + // `HashJoin` plan will otherwise collapse to the left collection // alone via `extract_collection` and miss the build side (captured // separately in `join_input`). if let Some(coll) = probe_collection @@ -151,18 +134,11 @@ pub(super) async fn resolve_gather( /// Gather without merge. pub(super) async fn resolve_broadcast( state: &SharedState, - ctx: ResolveCtx, + ctx: ReadScope, child: PhysicalPlan, captures: &mut Vec, ) -> crate::Result { - let ResolveCtx { - database_id, - tenant_id, - trace_id, - txn_id, - } = ctx; - let outcome = - gather_all_vshards(state, tenant_id, database_id, child, trace_id, txn_id).await?; + let outcome = gather_all_vshards(state, child, ctx).await?; Ok(Resolved::Gathered( outcome_to_response( outcome.merged_array, diff --git a/nodedb/src/control/server/exchange/resolve/exchange/hash_join_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/hash_join_arm.rs index e773ae057..bd597aa6a 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/hash_join_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/hash_join_arm.rs @@ -12,8 +12,8 @@ use crate::control::server::exchange::resolve::join_input::{ }; use crate::control::state::SharedState; -use super::dispatch::ResolveCtx; use super::entry::Resolved; +use crate::control::server::exchange::read_scope::ReadScope; /// Fields of a `QueryOp::HashJoin` plan node, carried through resolution as /// one value instead of as individually threaded arguments. @@ -45,16 +45,10 @@ pub(super) struct HashJoinFields { /// in `left_input` / `right_input`, then cross-node gather the build side. pub(super) async fn resolve_hash_join( state: &SharedState, - ctx: ResolveCtx, + ctx: ReadScope, captures: &mut Vec, fields: HashJoinFields, ) -> crate::Result { - let ResolveCtx { - database_id, - tenant_id, - trace_id, - txn_id, - } = ctx; let HashJoinFields { left_collection, right_collection, @@ -79,37 +73,18 @@ pub(super) async fn resolve_hash_join( right_scan_filters, } = fields; - let left_input = resolve_join_input( - state, - database_id, - tenant_id, - left_input, - trace_id, - txn_id, - captures, - ) - .await?; - right_input = resolve_join_input( - state, - database_id, - tenant_id, - right_input, - trace_id, - txn_id, - captures, - ) - .await?; + let left_input = resolve_join_input(state, ctx, left_input, captures).await?; + right_input = resolve_join_input(state, ctx, right_input, captures).await?; // Cross-node build-side gather. // // The HashJoin task routes to the LEFT (probe) collection's owning // vShard, where the LEFT side is scanned locally. The RIGHT (build) // collection is otherwise scanned BY NAME from that same node — but - // a single-vShard-homed build collection may live on a DIFFERENT + // a single-vShard-homed build collection can live on a DIFFERENT // node, so the by-name scan returns nothing and the join drops rows. // - // When a gateway is installed (it always is, single node included), - // and the build side has not already been materialized by + // When the build side has not already been materialized by // `resolve_join_input` (`right_input` still `None`), and // `right_collection` names a real user collection (catalog sides // carry an empty name and are already embedded as a @@ -118,17 +93,14 @@ pub(super) async fn resolve_hash_join( // `ProviderScan`. The HashJoin shipped to the probe node is then // self-contained. Only the RIGHT/build side is gathered; the // LEFT/probe side stays local to the routed vShard. - if state.gateway.get().is_some() && right_input.is_none() && !right_collection.is_empty() { + if right_input.is_none() && !right_collection.is_empty() { right_input = gather_join_build_side( state, - database_id, - tenant_id, + ctx, // The side's own collection and its own injected policy, // taken as one value: a planner that swaps build and probe // swaps both together, never one without the other. ScanSide::join_side(&right_collection, &right_rls_filters, &right_scan_filters), - trace_id, - txn_id, captures, ) .await?; diff --git a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs index b71d33e08..4a6027f2c 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/post_process_arm.rs @@ -22,7 +22,9 @@ use crate::data::executor::response_codec::{ flatten_vector_hits_to_relational_rows, }; -use super::dispatch::{ResolveCtx, resolve_exchange}; +use crate::control::server::exchange::read_scope::ReadScope; + +use super::dispatch::resolve_exchange; use super::entry::Resolved; /// Fields of a `QueryOp::PostProcess` plan node, carried through resolution @@ -128,15 +130,16 @@ pub(super) enum ChildRows { /// child's base collection in `captures` at its observed read-version. pub(super) async fn materialize_child_rows( state: &SharedState, - ctx: ResolveCtx, + ctx: ReadScope, captures: &mut Vec, input: PhysicalPlan, ) -> crate::Result { - let ResolveCtx { + let ReadScope { database_id, tenant_id, trace_id, txn_id, + .. } = ctx; // The converter wraps a sharded body in `Exchange{Gather}`; unwrap @@ -152,18 +155,8 @@ pub(super) async fn materialize_child_rows( // Resolve any Exchange nested inside the child first (e.g. a // `HashJoin` build-side `Broadcast`) so the plan gathered below is - // self-contained — no Exchange may reach a Data-Plane core. - let child = match Box::pin(resolve_exchange( - state, - database_id, - tenant_id, - child, - trace_id, - txn_id, - captures, - )) - .await? - { + // self-contained — no Exchange can reach a Data-Plane core. + let child = match Box::pin(resolve_exchange(state, ctx, child, captures)).await? { Resolved::Plan(p) => *p, // The unwrapped body is not itself a root Gather / stream; // surface these without dropping the caller's tail. @@ -187,8 +180,8 @@ pub(super) async fn materialize_child_rows( // consumes `child`. let hit_kind = classify_hit_shape(&child); // Extract the collection from the hit op directly: `collection()` - // has no arm for sparse / multi-vector search, so it would yield - // `None` and the PK resolver would be handed an empty collection. + // has no arm for sparse / multi-vector search, so it will yield + // `None` and the PK resolver will be handed an empty collection. let hit_collection = hit_collection_name(&child); // Record the child's single base collection in the in-transaction @@ -219,7 +212,7 @@ pub(super) async fn materialize_child_rows( ) .await? } else { - gather_all_vshards(state, tenant_id, database_id, child, trace_id, txn_id).await? + gather_all_vshards(state, child, ctx).await? }; if let Some(coll) = probe_collection @@ -272,7 +265,7 @@ pub(super) async fn materialize_child_rows( /// gathered here, so the relational tail never runs per-shard. pub(super) async fn resolve_post_process( state: &SharedState, - ctx: ResolveCtx, + ctx: ReadScope, captures: &mut Vec, fields: PostProcessFields, ) -> crate::Result { diff --git a/nodedb/src/control/server/exchange/resolve/exchange/set_op_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/set_op_arm.rs index 814038084..af47bfb81 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/set_op_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/set_op_arm.rs @@ -12,9 +12,9 @@ use crate::control::server::set_op_merge::{ }; use crate::control::state::SharedState; -use super::dispatch::ResolveCtx; use super::entry::Resolved; use super::post_process_arm::{ChildRows, materialize_child_rows, provider_scan_of_rows}; +use crate::control::server::exchange::read_scope::ReadScope; /// Resolve a `QueryOp::SetOp` node. /// @@ -30,7 +30,7 @@ use super::post_process_arm::{ChildRows, materialize_child_rows, provider_scan_o /// aggregate supplies its own tail over these rows. pub(super) async fn resolve_set_op( state: &SharedState, - ctx: ResolveCtx, + ctx: ReadScope, captures: &mut Vec, inputs: Vec, op: SetOpKind, diff --git a/nodedb/src/control/server/exchange/resolve/exchange/shuffle_arm.rs b/nodedb/src/control/server/exchange/resolve/exchange/shuffle_arm.rs index 787ec3b47..4060fd253 100644 --- a/nodedb/src/control/server/exchange/resolve/exchange/shuffle_arm.rs +++ b/nodedb/src/control/server/exchange/resolve/exchange/shuffle_arm.rs @@ -6,11 +6,12 @@ use nodedb_physical::physical_plan::PhysicalPlan; +use crate::control::server::exchange::resolve::peers::ShuffleRead; use crate::control::server::exchange::resolve::{shuffle, shuffle_aggregate}; use crate::control::state::SharedState; -use super::dispatch::ResolveCtx; use super::entry::Resolved; +use crate::control::server::exchange::read_scope::ReadScope; /// Resolve a root-level `Exchange{Shuffle}` node: orchestrate a real /// cross-node grace hash join. The child must be a `QueryOp::HashJoin` @@ -19,7 +20,7 @@ use super::entry::Resolved; /// rows as `Resolved::Gathered`. pub(super) async fn resolve_shuffle( state: &SharedState, - ctx: ResolveCtx, + ctx: ReadScope, child: PhysicalPlan, keys: Vec<(String, String)>, num_parts: usize, @@ -31,7 +32,10 @@ pub(super) async fn resolve_shuffle( child, keys, num_parts, - ctx.trace_id, + ShuffleRead { + trace_id: ctx.trace_id, + linearizable: ctx.linearizable, + }, ) .await } @@ -44,7 +48,7 @@ pub(super) async fn resolve_shuffle( /// finalized rows as `Resolved::Gathered`. pub(super) async fn resolve_shuffle_aggregate( state: &SharedState, - ctx: ResolveCtx, + ctx: ReadScope, child: PhysicalPlan, keys: Vec, num_parts: usize, @@ -56,7 +60,10 @@ pub(super) async fn resolve_shuffle_aggregate( child, keys, num_parts, - ctx.trace_id, + ShuffleRead { + trace_id: ctx.trace_id, + linearizable: ctx.linearizable, + }, ) .await } diff --git a/nodedb/src/control/server/exchange/resolve/join_input.rs b/nodedb/src/control/server/exchange/resolve/join_input.rs index a0de9f7e9..515dea2f4 100644 --- a/nodedb/src/control/server/exchange/resolve/join_input.rs +++ b/nodedb/src/control/server/exchange/resolve/join_input.rs @@ -5,14 +5,12 @@ use nodedb_physical::physical_plan::{ExchangeMode, ExchangeOp, PhysicalPlan, QueryOp}; +use crate::control::server::exchange::read_scope::ReadScope; use crate::control::state::SharedState; use crate::data::executor::response_codec::flatten_to_relational_rows; -use crate::types::{DatabaseId, TenantId, TraceId, TxnId}; use crate::control::server::exchange::full_scan::{ScanSide, full_scan_plan_for_collection}; -use crate::control::server::exchange::gather::{ - finalize_aggregate, gather_all_cores, gather_all_vshards, -}; +use crate::control::server::exchange::gather::{finalize_aggregate, gather_all_vshards}; use super::capture::DistributedReadCapture; use super::exchange::provider_scan_of_rows; @@ -29,13 +27,16 @@ use super::exchange::provider_scan_of_rows; /// materialized side is validated at commit like every other distributed read. pub(super) async fn resolve_join_input( state: &SharedState, - database_id: DatabaseId, - tenant_id: TenantId, + scope: ReadScope, input: Option>, - trace_id: TraceId, - txn_id: Option, captures: &mut Vec, ) -> crate::Result>> { + let ReadScope { + database_id, + tenant_id, + txn_id, + .. + } = scope; let Some(boxed) = input else { return Ok(None); }; @@ -54,8 +55,10 @@ pub(super) async fn resolve_join_input( // producing a Response whose payload is exactly `merged_array`. // `decode_response_to_docs` in `hash_handlers.rs` then reads that // Response as a msgpack array — so the two shapes match. - let outcome = - gather_all_cores(state, tenant_id, database_id, *child, trace_id, txn_id).await?; + // The gather reads every vShard of the child from its owner, not + // only the groups this node replicates, and confirms each leg as + // `scope` requires. + let outcome = Box::pin(gather_all_vshards(state, *child, scope)).await?; let provider_scan = provider_scan_of_rows(flatten_to_relational_rows(&outcome.merged_array)); Ok(Some(Box::new(provider_scan))) @@ -102,8 +105,8 @@ pub(super) async fn resolve_join_input( } else { None }; - let outcome = - gather_all_cores(state, tenant_id, database_id, *child, trace_id, txn_id).await?; + // Read from every vShard's owner, as the Broadcast arm does. + let outcome = Box::pin(gather_all_vshards(state, *child, scope)).await?; if let Some(coll) = child_collection && let Some(scan_plan) = full_scan_plan_for_collection( state, @@ -142,7 +145,7 @@ pub(super) async fn resolve_join_input( /// /// `side` carries the collection together with the RLS filters injected for /// it, so the gathered rows are filtered per side *before* the join, exactly -/// as the local name-scan this gather replaces would have filtered them. +/// as the local name-scan this gather replaces filtered them. /// /// Returns `Ok(None)` (the name-scan fallback) when the catalog has no record /// for the collection. This is graceful degradation, never an error: a missing @@ -150,13 +153,16 @@ pub(super) async fn resolve_join_input( /// the executing node. pub(super) async fn gather_join_build_side( state: &SharedState, - database_id: DatabaseId, - tenant_id: TenantId, + scope: ReadScope, side: ScanSide<'_>, - trace_id: TraceId, - txn_id: Option, captures: &mut Vec, ) -> crate::Result>> { + let ReadScope { + database_id, + tenant_id, + txn_id, + .. + } = scope; // Build an unprojected full-collection scan for the engine via the shared // builder, carrying this side's read policy. `Ok(None)` (no catalog / unknown // collection) keeps the existing graceful name-scan fallback — never an @@ -198,15 +204,7 @@ pub(super) async fn gather_join_build_side( // → `resolve_exchange` → here. The cycle terminates at runtime (the scan // plan is Exchange-free), but the future must be heap-indirected so its size // is finite. - let outcome = Box::pin(gather_all_vshards( - state, - tenant_id, - database_id, - scan_plan, - trace_id, - txn_id, - )) - .await?; + let outcome = Box::pin(gather_all_vshards(state, scan_plan, scope)).await?; if let Some(scan_plan) = capture_plan { captures.push(DistributedReadCapture { diff --git a/nodedb/src/control/server/exchange/resolve/peers.rs b/nodedb/src/control/server/exchange/resolve/peers.rs index 6418081e7..36a5f90c7 100644 --- a/nodedb/src/control/server/exchange/resolve/peers.rs +++ b/nodedb/src/control/server/exchange/resolve/peers.rs @@ -15,7 +15,7 @@ use nodedb_cluster::{ METADATA_GROUP_ID, RaftRpc, RoutingTable, ShuffleProduceRequest, ShuffleProduceResponse, }; -use crate::types::DatabaseId; +use crate::types::{DatabaseId, TraceId}; /// Producer nodes that own `collection`'s data. `collection` is the plan's /// database-qualified name. Resolve its canonical key's vShard → owning @@ -26,14 +26,7 @@ pub(super) fn producer_nodes( database_id: DatabaseId, collection: &str, ) -> crate::Result> { - let vshard = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)? - .vshard() - .as_u32(); - let group = routing - .group_for_vshard(vshard) - .map_err(|e| crate::Error::Internal { - detail: format!("shuffle: no group for vshard {vshard} ({collection}): {e}"), - })?; + let group = collection_group(routing, database_id, collection)?; let leader = routing .group_info(group) .map(|g| g.leader) @@ -44,6 +37,44 @@ pub(super) fn producer_nodes( Ok(vec![leader]) } +/// The Raft group that homes `collection`. +fn collection_group( + routing: &RoutingTable, + database_id: DatabaseId, + collection: &str, +) -> crate::Result { + let vshard = nodedb_types::CollectionKey::from_qualified_str(database_id, collection)? + .vshard() + .as_u32(); + routing + .group_for_vshard(vshard) + .map_err(|e| crate::Error::Internal { + detail: format!("shuffle: no group for vshard {vshard} ({collection}): {e}"), + }) +} + +/// How a shuffle's producers read: the statement's trace, and whether each +/// producer scan is a linearizable read. +#[derive(Debug, Clone, Copy)] +pub struct ShuffleRead { + pub trace_id: TraceId, + pub linearizable: bool, +} + +/// Groups a producer scanning `collection` confirms before it reads: the +/// collection's group for a linearizable read, none otherwise. +pub(super) fn producer_read_groups( + routing: &RoutingTable, + database_id: DatabaseId, + collection: &str, + linearizable: bool, +) -> crate::Result> { + if !linearizable { + return Ok(Vec::new()); + } + collection_group(routing, database_id, collection).map(|group| vec![group]) +} + /// Count distinct data-group leaders (the cluster's data-node count), excluding /// the metadata group, which owns no vShards. pub(super) fn distinct_data_node_count(routing: &RoutingTable) -> usize { diff --git a/nodedb/src/control/server/exchange/resolve/shuffle.rs b/nodedb/src/control/server/exchange/resolve/shuffle.rs index 57c06f438..f44818d68 100644 --- a/nodedb/src/control/server/exchange/resolve/shuffle.rs +++ b/nodedb/src/control/server/exchange/resolve/shuffle.rs @@ -43,11 +43,13 @@ use crate::control::server::exchange::full_scan::{ScanSide, full_scan_plan_for_c use crate::control::server::exchange::gather::outcome_to_response; use crate::control::server::payload_merge::merge_msgpack_arrays; use crate::control::state::SharedState; -use crate::types::{DatabaseId, Lsn, TenantId, TraceId}; +use crate::types::{DatabaseId, Lsn, TenantId}; use super::capture::DistributedReadCapture; use super::exchange::Resolved; -use super::peers::{distinct_data_node_count, producer_nodes, send_produce}; +use super::peers::{ + ShuffleRead, distinct_data_node_count, producer_nodes, producer_read_groups, send_produce, +}; /// Orchestrate a distributed shuffle hash join. /// @@ -64,8 +66,12 @@ pub async fn resolve_shuffle_join( child: PhysicalPlan, _keys: Vec<(String, String)>, num_parts: usize, - trace_id: TraceId, + read: ShuffleRead, ) -> crate::Result { + let ShuffleRead { + trace_id, + linearizable, + } = read; // 1. The child MUST be a HashJoin — shuffle wraps a complete hash join. let PhysicalPlan::Query(QueryOp::HashJoin { left_collection, @@ -168,6 +174,18 @@ pub async fn resolve_shuffle_join( // leader each, but compute generally and dedup). let build_nodes = producer_nodes(&routing_snapshot, database_id, right_collection.as_str())?; let probe_nodes = producer_nodes(&routing_snapshot, database_id, left_collection.as_str())?; + let build_read_groups = producer_read_groups( + &routing_snapshot, + database_id, + right_collection.as_str(), + linearizable, + )?; + let probe_read_groups = producer_read_groups( + &routing_snapshot, + database_id, + left_collection.as_str(), + linearizable, + )?; let build_producer_count = build_nodes.len() as u32; let probe_producer_count = probe_nodes.len() as u32; if build_producer_count == 0 || probe_producer_count == 0 { @@ -179,7 +197,7 @@ pub async fn resolve_shuffle_join( // Ensure the transport knows every target node's address before dispatching. // In production the WarmPeers startup phase registers every peer from the // topology, but a node that joined after that phase — or a coordinator that - // never had to reach a given peer — may not yet have it in the transport's + // never had to reach a given peer — can still lack it in the transport's // address map, and `send_rpc` to an unregistered peer fails with // NodeUnreachable even though the topology knows the node. Resolve each // producer/consumer node's address from the live topology and register it @@ -252,6 +270,7 @@ pub async fn resolve_shuffle_join( deadline_remaining_ms, trace_id: trace_id.0, descriptor_versions: Vec::::new(), + read_groups: build_read_groups.clone(), }; build_produce_futures.push(send_produce(transport, node, req)); } @@ -270,6 +289,7 @@ pub async fn resolve_shuffle_join( deadline_remaining_ms, trace_id: trace_id.0, descriptor_versions: Vec::::new(), + read_groups: probe_read_groups.clone(), }; probe_produce_futures.push(send_produce(transport, node, req)); } @@ -347,9 +367,8 @@ pub async fn resolve_shuffle_join( // `ShuffleProduceResponse.read_version_lsn`. The record seam records one // read-set entry per capture, re-homing and revalidating each side's vshard // independently, so a concurrent write to EITHER side between the in-txn read - // and commit is detected (the build side was previously never recorded — the - // hole this closes). The response's own scalar stays `ZERO`: the captures - // carry the versions, and also setting the scalar would double-record the + // and commit is detected (the build side is recorded too). The response's own scalar stays `ZERO`: the captures + // carry the versions, and also setting the scalar will double-record the // left side. The core-global watermark is not threaded through the shuffle // transport and likewise stays `ZERO`. let captures = vec![ diff --git a/nodedb/src/control/server/exchange/resolve/shuffle_aggregate.rs b/nodedb/src/control/server/exchange/resolve/shuffle_aggregate.rs index a40d9b134..f6e94d43d 100644 --- a/nodedb/src/control/server/exchange/resolve/shuffle_aggregate.rs +++ b/nodedb/src/control/server/exchange/resolve/shuffle_aggregate.rs @@ -25,7 +25,7 @@ //! per-group, so a plain concat is the correct finalize. Eligibility forbids a //! global ORDER BY (per-part finalize can't honour a cross-part sort), so the //! cap is order-free — the same arbitrary-but-bounded semantics Gather has with -//! no ORDER BY. Consumers receive NO per-part cap (that would silently drop +//! no ORDER BY. Consumers receive NO per-part cap (that will silently drop //! rows per part); the cap is applied only here. //! //! # Plane discipline @@ -50,10 +50,12 @@ use crate::control::cluster::warm_peers::register_peers_from_topology; use crate::control::server::exchange::gather::outcome_to_response; use crate::control::server::payload_merge::{encode_msgpack_array, extract_msgpack_elements}; use crate::control::state::SharedState; -use crate::types::{DatabaseId, Lsn, TenantId, TraceId}; +use crate::types::{DatabaseId, Lsn, TenantId}; use super::exchange::Resolved; -use super::peers::{distinct_data_node_count, producer_nodes, send_produce}; +use super::peers::{ + ShuffleRead, distinct_data_node_count, producer_nodes, producer_read_groups, send_produce, +}; /// Orchestrate a distributed shuffle GROUP BY aggregate. /// @@ -71,8 +73,12 @@ pub async fn resolve_shuffle_aggregate( child: PhysicalPlan, keys: Vec, num_parts: usize, - trace_id: TraceId, + read: ShuffleRead, ) -> crate::Result { + let ShuffleRead { + trace_id, + linearizable, + } = read; // 1. The child MUST be a root Aggregate — shuffle-aggregate wraps a complete // GROUP BY aggregate. let PhysicalPlan::Query(QueryOp::Aggregate { @@ -103,7 +109,7 @@ pub async fn resolve_shuffle_aggregate( }); } // A non-GROUP-BY (scalar) aggregate has a single global group; there is no - // key to repartition on, so a shuffle would be pointless. The planner emit + // key to repartition on, so a shuffle will be pointless. The planner emit // already gates on a non-empty GROUP BY, but reject here too for robustness. if group_by.is_empty() { return Err(crate::Error::Internal { @@ -182,6 +188,12 @@ pub async fn resolve_shuffle_aggregate( // 4. Producer node set for the source collection (single-vShard-homed → one // leader, but compute generally and dedup). let producers = producer_nodes(&routing_snapshot, database_id, collection.as_str())?; + let read_groups = producer_read_groups( + &routing_snapshot, + database_id, + collection.as_str(), + linearizable, + )?; let producer_count = producers.len() as u32; if producer_count == 0 { return Err(crate::Error::Internal { @@ -190,7 +202,7 @@ pub async fn resolve_shuffle_aggregate( } // Ensure the transport knows every target node's address before dispatching. - // (See `register_peers_from_topology` — robust to a peer the transport has + // (See `register_peers_from_topology` — tolerant of a peer the transport has // not warmed yet.) { let mut targets: BTreeSet = BTreeSet::new(); @@ -240,6 +252,7 @@ pub async fn resolve_shuffle_aggregate( deadline_remaining_ms, trace_id: trace_id.0, descriptor_versions: Vec::::new(), + read_groups: read_groups.clone(), }; produce_futures.push(send_produce(transport, node, req)); } @@ -263,12 +276,12 @@ pub async fn resolve_shuffle_aggregate( })?; // Per-part consumers receive NO row cap: each must return ALL of its (disjoint) // groups so the coordinator can apply the aggregate's global result cap ONCE - // over the union. Pushing `limit` to each part would truncate parts + // over the union. Pushing `limit` to each part will truncate parts // independently — a silent per-part row drop, and a result that disagrees with // the single-node Gather path's GLOBAL cap. The cap is reapplied at step 8. // Post-aggregate ORDER BY is restricted to bare output columns by the // planner, which is what the shuffle wire form carries. A key that is not - // a column would have been rejected at plan time, so none reaches here. + // a column is rejected at plan time, so none reaches here. let wire_sort_keys: Vec = sort_keys .iter() .filter_map(|k| { @@ -325,7 +338,7 @@ pub async fn resolve_shuffle_aggregate( // in-transaction distributed aggregate records a sound read-set entry (the // aggregate is single-collection, so `record_read_set` attributes it to the // right collection). The core-global `watermark_lsn` is NOT threaded through - // the shuffle transport and stays `ZERO`; using it as the read version would + // the shuffle transport and stays `ZERO`; using it as the read version will // skip required aborts (it advances on writes to ANY collection). Ok(Resolved::Gathered( outcome_to_response(merged, Lsn::ZERO, Lsn::new(max_read_version_lsn)), diff --git a/nodedb/src/control/server/graph_dispatch/bfs.rs b/nodedb/src/control/server/graph_dispatch/bfs.rs index 44eb1d916..9c3a26261 100644 --- a/nodedb/src/control/server/graph_dispatch/bfs.rs +++ b/nodedb/src/control/server/graph_dispatch/bfs.rs @@ -20,24 +20,33 @@ use crate::types::{DatabaseId, TenantId}; use super::helpers::{encode_path, ok_response}; use super::hop::{NeighborHopParams, execute_neighbor_hop}; +use super::shard_reads::ShardReadLog; /// Parameters for [`cross_core_bfs_with_options`]. pub struct CrossCoreBfsParams<'a> { pub tenant_id: TenantId, pub database_id: DatabaseId, - /// Collection scope, or `None` for a label-only traversal. + /// Database-qualified collection scope, or `None` for a label-only + /// traversal. pub collection: Option<&'a str>, pub start_nodes: Vec, pub edge_label: Option, pub direction: crate::engine::graph::edge_store::Direction, pub max_depth: usize, pub options: &'a GraphTraversalOptions, + /// Each node that expands part of the walk confirms its groups first. + pub linearizable: bool, } -/// Cross-core BFS with explicit traversal options (fan-out limits, partial mode). +/// Cross-core BFS with explicit traversal options. /// /// This is the cluster-aware entry point. Callers pass /// `&GraphTraversalOptions::default()` for standard traversal. +/// +/// The walk runs level by level, as one core's BFS does +/// (`CsrIndex::traverse_bfs`): each hop fetches every neighbor of the +/// frontier, and the level's new nodes are admitted in node-name order until +/// the visit cap. A capped walk admits the same nodes here as on one core. pub async fn cross_core_bfs_with_options( shared: &SharedState, params: CrossCoreBfsParams<'_>, @@ -51,18 +60,20 @@ pub async fn cross_core_bfs_with_options( direction, max_depth, options, + linearizable, } = params; + let cap = walk_visit_cap(shared, options); + let whole = whole_hop(); let mut visited: HashSet = HashSet::new(); let mut all_discovered: Vec = Vec::new(); let mut frontier: Vec = start_nodes; + let mut reads = ShardReadLog::new(); - for node in &frontier { - visited.insert(node.clone()); - all_discovered.push(node.clone()); - } + frontier.retain(|node| visited.insert(node.clone())); + all_discovered.extend(frontier.iter().cloned()); for _depth in 0..max_depth { - if frontier.is_empty() { + if frontier.is_empty() || all_discovered.len() >= cap { break; } @@ -75,30 +86,80 @@ pub async fn cross_core_bfs_with_options( frontier: &frontier, edge_label: edge_label.as_deref(), direction, - options, + options: &whole, discovered_so_far: all_discovered.len(), + linearizable, }, ) .await?; - // Extend global visited set and compute next frontier. - let mut next_frontier: Vec = Vec::new(); - for node in hop.merged_destinations { - if visited.insert(node.clone()) { - next_frontier.push(node.clone()); - all_discovered.push(node); - if all_discovered.len() >= options.max_visited { - break; - } - } - } + reads.merge(hop.reads); + frontier = admit_by_name( + hop.merged_destinations, + &mut visited, + &mut all_discovered, + cap, + ); + } - frontier = next_frontier; + // Every vShard the walk expanded joins the transaction read-set. + reads.publish( + shared, + tenant_id, + database_id, + collection.map(str::to_owned), + ); + Ok(ok_response(encode_path(&all_discovered)?)) +} - if all_discovered.len() >= options.max_visited { +/// The visit cap of a hop or subgraph walk: the plan's cap, bounded by this +/// node's graph tuning, as one core bounds it (`CoreLoop::walk_visit_cap`). +pub(super) fn walk_visit_cap(shared: &SharedState, options: &GraphTraversalOptions) -> usize { + options.max_visited.min(shared.tuning.graph.max_visited) +} + +/// Hop options that return every neighbor of the frontier. A capped walk +/// admits each level's nodes in name order, so it reads the whole level. +pub(super) fn whole_hop() -> GraphTraversalOptions { + GraphTraversalOptions { + max_visited: usize::MAX, + } +} + +/// Admit the unvisited `candidates` in node-name order until `discovered` +/// holds `cap` nodes. Returns the admitted nodes: the next frontier. +pub(super) fn admit_by_name( + mut candidates: Vec, + visited: &mut HashSet, + discovered: &mut Vec, + cap: usize, +) -> Vec { + candidates.retain(|node| !visited.contains(node)); + candidates.sort(); + candidates.dedup(); + let mut admitted = Vec::with_capacity(candidates.len()); + for node in candidates { + if discovered.len() >= cap { break; } + visited.insert(node.clone()); + discovered.push(node.clone()); + admitted.push(node); } + admitted +} - Ok(ok_response(encode_path(&all_discovered)?)) +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_capped_level_is_admitted_in_name_order() { + let mut visited: HashSet = ["a".to_string()].into_iter().collect(); + let mut discovered = vec!["a".to_string()]; + let candidates = ["z", "m", "a", "b", "m"].map(str::to_string).to_vec(); + let next = admit_by_name(candidates, &mut visited, &mut discovered, 3); + assert_eq!(next, vec!["b", "m"]); + assert_eq!(discovered, vec!["a", "b", "m"]); + } } diff --git a/nodedb/src/control/server/graph_dispatch/bsp_pagerank/coord.rs b/nodedb/src/control/server/graph_dispatch/bsp_pagerank/coord.rs index 9bd91037f..807823fbf 100644 --- a/nodedb/src/control/server/graph_dispatch/bsp_pagerank/coord.rs +++ b/nodedb/src/control/server/graph_dispatch/bsp_pagerank/coord.rs @@ -1,42 +1,45 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Control-Plane coordinator for distributed BSP PageRank (F1d-4 Phase B). +//! Control-Plane coordinator for distributed BSP PageRank. //! //! Drives the superstep loop with one dispatch per DISTINCT OWNER NODE (each //! carrying that node's full owned-vShard set), NOT one dispatch per vShard, -//! using the Phase A `GraphOp::BspSuperstep` primitive; each node's dispatch is -//! then fanned across that node's cores below. The coordinator OWNS all durable -//! state — the per-node rank vectors and the routed cross-shard contributions; +//! using the `GraphOp::BspSuperstep` primitive; each node's dispatch is then +//! fanned across that node's cores. The coordinator OWNS all durable state — +//! the per-node rank vectors and the routed cross-shard contributions; //! [`BspCoordinator`] is used ONLY for convergence bookkeeping (`record_ack` / -//! `all_acked` / `totals` / `advance`). +//! `advance`). //! //! Each shard is one distinct owner node (the local node + each distinct //! non-local data-group leader), carrying that node's FULL set of owned vShards. -//! This is one dispatch per node per superstep — not one per vShard — mirroring -//! `match_scatter`'s per-owner-node scatter; the handler ranks every node homed -//! on that owner in a single CSR pass (see `enumerate.rs` for the rationale and -//! the multi-core caveat). +//! This mirrors `match_scatter`'s per-owner-node scatter; the handler ranks +//! every node homed on that owner in a single CSR pass (see `enumerate.rs`). //! -//! Phases: +//! Steps: //! //! 1. **Count.** Dispatch one `BspSuperstep` with `global_n == 0` (the -//! count-only sentinel) to every node. The handler short-circuits after -//! building its owned-node set and returns just `vertex_count` + `node_names`. -//! `global_n = Σ vertex_count` (each graph node is homed on exactly one owner -//! node, so summing per-node owned counts counts each graph node once). -//! 2. **Superstep loop.** For `s = 0,1,2,…`: dispatch `BspSuperstep` to every -//! node with `global_n`, the routed `incoming_contributions`, and that node's -//! current `rank_vec`; collect each node's new `rank_vec` + `outbound` + -//! `local_delta`. Record one `SuperstepAck` per node, then `advance()`; halt -//! when the global delta drops below tolerance or `max_iterations` is reached. -//! 3. **Redistribute.** After each superstep, route every node's `outbound` -//! `(target_vshard, dst_name, contrib)` to the node that OWNS `target_vshard` -//! (resolved via the `vShard → owner node` map), accumulating into that -//! node's `incoming_contributions` for the next superstep. -//! 4. **Assemble.** On halt, zip each node's final `rank_vec` with its +//! count-only sentinel) to every node. `global_n = Σ vertex_count`: each +//! graph node is homed on exactly one owner node. The count phase carries +//! a fresh cut marker. Every node resolves the run's read cut from it and +//! answers it, and every later superstep reads at that one cut (see +//! `graph_dispatch::run_cut`). A write during the run is above the cut, so +//! the rank set stays the nodes that existed at the cut. +//! 2. **Initial scatter.** Superstep 0 sets every node's initial rank and +//! returns the contributions it scatters. It changes no rank, so it is not +//! a convergence step. +//! 3. **Iterate.** Superstep `s >= 1` sends each node its current rank, the +//! dangling mass of that rank, and the contributions every node scattered +//! from it. Each node returns the next rank and its scatter. One superstep +//! is one power iteration of single-node PageRank, so the run halts under +//! the same tolerance and iteration budget, with no rank mass in flight. +//! 4. **Route.** Every outbound `(target_vshard, dst_name, contrib)` goes to +//! the node that owns `target_vshard` in the enumeration. The handlers use +//! that same enumeration as their owned sets, so a contribution always +//! reaches the node that ranks its destination. +//! 5. **Assemble.** On halt, zip each node's final `rank_vec` with its //! `node_names`, concatenate across nodes (each owns a disjoint graph-node -//! set, so no dedup is needed), and build an `AlgoResultBatch` serialized -//! exactly like the single-node path so the client output is byte-identical. +//! set), and build an `AlgoResultBatch` serialized exactly like the +//! single-node path. use std::collections::HashMap; @@ -49,7 +52,9 @@ use crate::engine::graph::algo::result::AlgoResultBatch; use crate::types::{DatabaseId, TenantId}; use super::enumerate::enumerate_shards; -use super::scatter::{ScatterSuperstepParams, ShardDispatch, scatter_superstep}; +use super::scatter::{ScatterSuperstepParams, ShardDispatch, ShardResult, scatter_superstep}; +use crate::control::server::graph_dispatch::run_cut::{agreed_cut, new_cut_marker}; +use crate::control::server::graph_dispatch::shard_reads::{ShardReadLog, qualified}; /// Default max supersteps when the query carries no explicit `ITERATIONS`. /// Mirrors the single-node PageRank default iteration budget. @@ -66,6 +71,14 @@ struct ShardRankState { rank_vec: Vec, } +/// The state one superstep hands to the next: the contributions routed to +/// each owner node, and the dangling mass of the rank they came from. +#[derive(Default)] +struct Carry { + incoming: HashMap>, + global_dangling: f64, +} + /// Run distributed BSP PageRank and return the bare `AlgoResultBatch` payload /// (the exact shape `algo_payload_to_query_response` consumes — identical to the /// single-node path). @@ -78,6 +91,7 @@ pub async fn run_bsp_pagerank( database_id: DatabaseId, params: AlgoParams, deadline_ms: u64, + linearizable: bool, ) -> crate::Result { let algorithm = GraphAlgorithm::PageRank; @@ -85,20 +99,17 @@ pub async fn run_bsp_pagerank( let enumeration = enumerate_shards(state)?; let targets = enumeration.targets; // `vShard → owner node` map: routes each outbound contribution's - // `target_vshard` to the node-shard that owns it during redistribution. + // `target_vshard` to the node-shard that owns it. let vshard_owner = enumeration.vshard_owner; if targets.is_empty() { - // No data shards (single-node would not reach here; an empty cluster - // routing table yields an empty result set). return empty_payload(); } // Personalized-PageRank global seed sum: `Σ max(w, 0.0)` over the seed map. - // The map already lives on the coordinator (no extra round trip). This is a - // PRE-CONDITION for personalization being active; the final activation - // decision also requires at least one seed name to exist somewhere in the - // cluster graph (checked via the count phase's `seed_hits` below), matching - // single-node `build_personalization` returning `None` for unknown seeds. + // Personalization is active only when this is positive AND at least one + // seed name exists somewhere in the cluster graph (the count phase's + // `seed_hits`), matching single-node `build_personalization` returning + // `None` for unknown seeds. let global_seed_sum: f64 = params .personalization_vector() .map(|seed| seed.values().map(|&w| w.max(0.0)).sum()) @@ -109,8 +120,7 @@ pub async fn run_bsp_pagerank( .map(|m| m.clamp(1, u32::MAX as usize) as u32) .unwrap_or(DEFAULT_MAX_ITERATIONS); let tolerance = params.convergence_tolerance(); - // Convergence is keyed by owner node now; `BspCoordinator` only needs stable - // shard ids, so node ids (cast to u32) serve as the per-shard ack keys. + // `BspCoordinator` needs stable shard ids: node ids (cast to u32) serve. let shard_ids: Vec = targets.iter().map(|t| t.node_id as u32).collect(); let mut bsp = BspCoordinator::new( algorithm.name().to_string(), @@ -119,7 +129,10 @@ pub async fn run_bsp_pagerank( shard_ids, ); - // ── Phase 1: count. global_n = 0 sentinel → handler returns owned counts. ── + // ── Count. global_n = 0 sentinel → handler returns owned counts. ── + // The count phase also pins the run's read cut: every node resolves it + // from one Calvin cut marker and reads every superstep at it. + let read_cut_marker = new_cut_marker(state); let count_dispatches: Vec = targets .iter() .map(|t| ShardDispatch { @@ -129,8 +142,10 @@ pub async fn run_bsp_pagerank( route_vshard: t.route_vshard(), incoming_contributions: Vec::new(), rank_seed: Vec::new(), - global_dangling: 0.0, // count phase: no previous superstep dangling sums. - personalization_sum: 0.0, // count phase runs no superstep — irrelevant here. + global_dangling: 0.0, + personalization_sum: 0.0, + read_cut_marker, + system_as_of: None, }) .collect(); let counts = scatter_superstep( @@ -144,25 +159,37 @@ pub async fn run_bsp_pagerank( global_n: 0, // count-only sentinel dispatches: count_dispatches, deadline_ms, + linearizable, }, ) .await?; - // Each graph node is homed on exactly one owner node, so summing per-node - // owned counts counts every graph node exactly once. + // The count phase reads every owner's partition first. A write after it + // on any owned vShard changes what later supersteps read, so the count + // phase's watermarks are the ones the transaction read-set keeps. + let mut reads = ShardReadLog::new(); + for count in &counts { + if let Some(target) = targets.iter().find(|t| t.node_id == count.node_id) { + reads.note( + target.owned_vshards.iter().copied(), + count.watermark_lsn, + target.node_id, + ); + } + } + reads.publish( + state, + tenant_id, + database_id, + Some(qualified(database_id, ¶ms.collection)), + ); + + let system_as_of = agreed_cut(counts.iter().map(|c| (c.node_id, c.result.system_as_of)))?; let global_n: usize = counts.iter().map(|c| c.result.vertex_count).sum(); if global_n == 0 { - // No nodes anywhere — empty result (same as single-node empty CSR). return empty_payload(); } - // Cluster-wide count of owned nodes that are positively-weighted seed keys. - // Combined with `global_seed_sum`, this is the exact single-node - // `build_personalization` activation test: personalization is active iff a - // seed map was supplied (`global_seed_sum > 0.0`) AND at least one seed name - // exists somewhere in the cluster graph (`global_seed_hits > 0`). Otherwise - // (unknown seed / empty / non-positive) every dispatch carries - // `personalization_sum = 0.0` and the supersteps run UNIFORM PageRank. let global_seed_hits: usize = counts.iter().map(|c| c.result.seed_hits).sum(); let personalization_sum = if global_seed_sum > 0.0 && global_seed_hits > 0 { global_seed_sum @@ -170,8 +197,6 @@ pub async fn run_bsp_pagerank( 0.0 }; - // Seed per-node rank state from the count phase. `rank_vec` starts empty so - // superstep 0 initializes each owned node to `1/global_n`. let mut shard_state: HashMap = HashMap::with_capacity(targets.len()); for (target, count) in targets.iter().zip(counts) { shard_state.insert( @@ -186,24 +211,10 @@ pub async fn run_bsp_pagerank( ); } - // Cross-shard contributions routed to each node for the NEXT superstep: - // owner node id → Vec<(dst_name, contrib)>. - let mut incoming: HashMap> = HashMap::new(); - - // Global dangling-node rank mass aggregated from all shards' previous - // superstep. Starts at 0.0 before superstep 0 (no previous superstep exists), - // which collapses the base to the plain teleport `(1−d)/n` — correct for - // initialization. After each superstep the coordinator sums each shard's - // returned `dangling_sum` here; the NEXT superstep's dispatches carry this - // value so dangling mass redistributes globally, not just within the shard - // that owns the dangling node. - let mut global_dangling: f64 = 0.0; - - // ── Phase 2/3: superstep loop. ── + // ── Initial scatter (superstep 0) and iterations. ── + let mut carry = Carry::default(); let mut superstep: u32 = 0; loop { - // Build this superstep's dispatches in a STABLE node order so the - // results zip back deterministically. let mut ordered_nodes: Vec = shard_state.keys().copied().collect(); ordered_nodes.sort_unstable(); @@ -216,19 +227,17 @@ pub async fn run_bsp_pagerank( is_local: st.is_local, owned_vshards: st.owned_vshards.clone(), route_vshard: st.route_vshard, - incoming_contributions: incoming.remove(&node_id).unwrap_or_default(), + incoming_contributions: carry.incoming.remove(&node_id).unwrap_or_default(), rank_seed: st .node_names .iter() .cloned() .zip(st.rank_vec.iter().copied()) .collect(), - // Pass the globally aggregated dangling mass from the PREVIOUS - // superstep. 0.0 on superstep 0 (no previous sums yet). - global_dangling, - // Cluster-wide PPR seed sum (0.0 = uniform). Constant across the - // whole run; each shard normalizes its owned seeds by it. + global_dangling: carry.global_dangling, personalization_sum, + read_cut_marker: 0, + system_as_of: Some(system_as_of), } }) .collect(); @@ -244,78 +253,91 @@ pub async fn run_bsp_pagerank( global_n, dispatches, deadline_ms, + linearizable, }, ) .await?; - // Store new rank vectors + node names, record ACKs, route outbound, and - // aggregate per-shard dangling sums into global_dangling for the NEXT step. - incoming.clear(); - global_dangling = 0.0; + carry = Carry::default(); for sr in results { - let node_id = sr.node_id; - let res = sr.result; - - bsp.record_ack(SuperstepAck { - shard_id: node_id as u32, - iteration: superstep + 1, - local_delta: res.local_delta, - vertex_count: res.vertex_count, - contributions_sent: res.outbound.len(), - }); - - // Aggregate each node's local dangling mass into the global total for - // the NEXT superstep. Each graph node is homed on exactly one owner - // node, so summing per-node dangling sums counts every dangling node - // exactly once. - global_dangling += res.dangling_sum; - - // Route this node's outbound contributions to the node that OWNS the - // target vShard. An unmapped target vShard is a routing - // inconsistency — surface it rather than silently dropping mass. - for (target_vshard, dst_name, contrib) in res.outbound { - let Some(&owner) = vshard_owner.get(&target_vshard) else { - return Err(crate::Error::Internal { - detail: format!( - "bsp pagerank: outbound contribution to unmapped target \ - vshard={target_vshard} (dst={dst_name})" - ), - }); - }; - if !shard_state.contains_key(&owner) { - return Err(crate::Error::Internal { - detail: format!( - "bsp pagerank: outbound contribution to unknown owner \ - node={owner} for vshard={target_vshard} (dst={dst_name})" - ), - }); - } - incoming.entry(owner).or_default().push((dst_name, contrib)); - } - - if let Some(st) = shard_state.get_mut(&node_id) { - st.node_names = res.node_names; - st.rank_vec = res.rank_vec; + // Superstep 0 changes no rank: it is not a convergence step. + if superstep > 0 { + bsp.record_ack(SuperstepAck { + shard_id: sr.node_id as u32, + iteration: superstep, + local_delta: sr.result.local_delta, + vertex_count: sr.result.vertex_count, + contributions_sent: sr.result.outbound.len(), + }); } + absorb(sr, &vshard_owner, &mut shard_state, &mut carry)?; } - // Convergence bookkeeping. Every shard ACKs exactly once per dispatch, so - // `advance` should never see a partial barrier — but it now REFUSES one - // instead of summing a subset, which a `debug_assert` could not do in a - // release build. - let keep_going = bsp.advance().map_err(|e| crate::Error::Internal { - detail: format!("bsp pagerank: not all shards acked after superstep dispatch ({e})"), - })?; - if !keep_going { - break; + if superstep > 0 { + // Every shard ACKs exactly once per dispatch; `advance` refuses a + // partial barrier rather than summing a subset. + let keep_going = bsp.advance().map_err(|e| crate::Error::Internal { + detail: format!( + "bsp pagerank: not all shards acked after superstep dispatch ({e})" + ), + })?; + if !keep_going { + break; + } } superstep += 1; } - // ── Phase 4: assemble final AlgoResultBatch (single-node-identical shape). ── assemble_result(&shard_state) } +/// Fold one node's superstep result into the coordinator state: store its +/// rank, add its dangling mass, and route its contributions to the nodes that +/// own their destinations. A contribution to a vShard no node owns is an +/// error: its mass will leave the graph. +fn absorb( + result: ShardResult, + vshard_owner: &HashMap, + shard_state: &mut HashMap, + carry: &mut Carry, +) -> crate::Result<()> { + let ShardResult { + node_id, result, .. + } = result; + carry.global_dangling += result.dangling_sum; + for (target_vshard, dst_name, contrib) in result.outbound { + let Some(&owner) = vshard_owner.get(&target_vshard) else { + return Err(crate::Error::Internal { + detail: format!( + "bsp pagerank: outbound contribution to unmapped target \ + vshard={target_vshard} (dst={dst_name})" + ), + }); + }; + if !shard_state.contains_key(&owner) { + return Err(crate::Error::Internal { + detail: format!( + "bsp pagerank: outbound contribution to unknown owner \ + node={owner} for vshard={target_vshard} (dst={dst_name})" + ), + }); + } + carry + .incoming + .entry(owner) + .or_default() + .push((dst_name, contrib)); + } + let Some(st) = shard_state.get_mut(&node_id) else { + return Err(crate::Error::Internal { + detail: format!("bsp pagerank: result from node {node_id}, which ranks no shard"), + }); + }; + st.node_names = result.node_names; + st.rank_vec = result.rank_vec; + Ok(()) +} + /// Concatenate every node's `(node_name, rank)` into an `AlgoResultBatch` using /// the same `push_node_f64` + `to_msgpack` seam as single-node PageRank, so /// `algo_payload_to_query_response` produces byte-identical client output. Each diff --git a/nodedb/src/control/server/graph_dispatch/bsp_pagerank/enumerate.rs b/nodedb/src/control/server/graph_dispatch/bsp_pagerank/enumerate.rs index 1760756ef..6fc711e77 100644 --- a/nodedb/src/control/server/graph_dispatch/bsp_pagerank/enumerate.rs +++ b/nodedb/src/control/server/graph_dispatch/bsp_pagerank/enumerate.rs @@ -3,35 +3,25 @@ //! Shard enumeration for distributed BSP PageRank. //! //! A "shard" here is one **owner NODE** (the local node + each distinct -//! non-local data-group leader), NOT one vShard. A remote `ExecuteRequest` runs -//! on a single core whose per-core `EdgeStore` holds that node's full graph -//! slice (one core / node — the single-core-coverage property the MATCH scatter -//! relies on; see the caveat below), and the receiver always dispatches with -//! `vshard_id = 0` rather than routing to a per-vShard core. Enumerating one -//! dispatch per *vShard* therefore lands hundreds of dispatches on the SAME core -//! of the SAME node, each rebuilding that node's full CSR — a massive waste where -//! most per-vShard dispatches own zero nodes. Enumerating one dispatch per -//! *node*, carrying that node's FULL set of owned vShards, rebuilds each node's -//! CSR exactly once per superstep and ranks every node it owns in one pass. +//! non-local data-group leader), NOT one vShard. One dispatch per node, +//! carrying that node's FULL set of owned vShards, builds each core's CSR +//! once per superstep and ranks every node it owns in one pass. A superstep +//! plan is not vShard-scoped, so the receiving node fans it across all its +//! cores (`exchange::received`), and each core ranks the owned vShards it +//! homes (`exchange::all_cores::fanout`). //! -//! This mirrors `match_scatter::round_zero::distinct_remote_owners`'s -//! one-dispatch-per-distinct-owner-node enumeration (live Raft leadership via the -//! `raft_status_fn` snapshot, falling back to the routing-table hint). The -//! metadata group (0) owns no vShards and is skipped. A data group with no -//! resolvable leader is a hard `NotLeader` error — never a silently-dropped -//! shard (same contract as the MATCH scatter). -//! -//! **Multi-core caveat (pre-existing, shared with MATCH — out of scope here):** a -//! remote `ExecuteRequest` executes on core 0 only, and the per-core `EdgeStore` -//! means core 0 holds a node's FULL graph slice only when that node runs a single -//! Data-Plane core (as the cluster tests do). On a multi-core node, cross-node -//! reads (BOTH this per-node BSP enumeration AND the existing cross-shard MATCH -//! scatter) would observe only core-0's partition. This per-node enumeration -//! assumes the same single-core-coverage property the MATCH scatter already -//! relies on; a multi-core cross-node fan-out is a separate concern. +//! The owner of a vShard is its data group's leader: the live Raft leadership +//! from the `raft_status_fn` snapshot, falling back to the routing-table hint. +//! The leader replicates the group, so it holds every edge homed on the +//! group's vShards. This one map is both each handler's owned set and the +//! coordinator's contribution routing, so a contribution always reaches the +//! node that ranks its destination. The metadata group (0) owns no vShards +//! and is skipped. A data group with no resolvable leader is a hard +//! `NotLeader` error — never a silently-dropped shard. use std::collections::HashMap; +use crate::control::gateway::live_leaders::LiveLeaders; use crate::control::state::SharedState; use crate::types::VShardId; @@ -86,18 +76,10 @@ pub(in crate::control::server::graph_dispatch) fn enumerate_shards( vshard_owner: HashMap::new(), }); }; + // Raft snapshot first, routing guard second: see `LiveLeaders`. + let live = LiveLeaders::snapshot(state); let routing = routing_lock.read().unwrap_or_else(|p| p.into_inner()); - let raft_snapshot: Vec = - state.raft_status_fn.get().map(|f| f()).unwrap_or_default(); - let live_leader = |group_id: u64| -> u64 { - raft_snapshot - .iter() - .find(|gs| gs.group_id == group_id) - .map(|gs| gs.leader_id) - .unwrap_or(0) - }; - // Accumulate each owner node's full vShard set (union over the data groups it // leads), preserving first-seen node order, and build the vShard → owner map. let mut owned_by_node: HashMap> = HashMap::new(); @@ -114,7 +96,7 @@ pub(in crate::control::server::graph_dispatch) fn enumerate_shards( continue; } // Prefer live Raft leadership; fall back to the routing-table hint. - let mut leader = live_leader(group_id); + let mut leader = live.leader_of(group_id); if leader == 0 { leader = routing.group_info(group_id).map(|g| g.leader).unwrap_or(0); } @@ -127,6 +109,7 @@ pub(in crate::control::server::graph_dispatch) fn enumerate_shards( vshard_id: VShardId::new(first), leader_node: 0, leader_addr: String::new(), + leader_term: 0, }); } if !owned_by_node.contains_key(&leader) { diff --git a/nodedb/src/control/server/graph_dispatch/bsp_pagerank/mod.rs b/nodedb/src/control/server/graph_dispatch/bsp_pagerank/mod.rs index 148687948..b98358b6f 100644 --- a/nodedb/src/control/server/graph_dispatch/bsp_pagerank/mod.rs +++ b/nodedb/src/control/server/graph_dispatch/bsp_pagerank/mod.rs @@ -1,11 +1,12 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Control-Plane coordinator for distributed BSP PageRank (F1d-4 Phase B). +//! Control-Plane coordinator for distributed BSP PageRank. //! -//! Drives the `GraphOp::BspSuperstep` Phase A primitive across all shards: a -//! count phase to compute `global_n`, then a superstep loop with cross-shard -//! contribution routing and `BspCoordinator`-based convergence, assembling the -//! final ranks into the same `AlgoResultBatch` shape as single-node PageRank. +//! Drives the `GraphOp::BspSuperstep` primitive across all shards: a count +//! phase to compute `global_n`, an initial scatter, then a superstep loop +//! with cross-shard contribution routing and `BspCoordinator`-based +//! convergence. It assembles the final ranks into the same `AlgoResultBatch` +//! shape as single-node PageRank. mod coord; /// Shard enumeration is shared with the distributed-WCC coordinator diff --git a/nodedb/src/control/server/graph_dispatch/bsp_pagerank/scatter.rs b/nodedb/src/control/server/graph_dispatch/bsp_pagerank/scatter.rs index 9cda7c154..61ff23148 100644 --- a/nodedb/src/control/server/graph_dispatch/bsp_pagerank/scatter.rs +++ b/nodedb/src/control/server/graph_dispatch/bsp_pagerank/scatter.rs @@ -48,12 +48,19 @@ pub(super) struct ShardDispatch { /// seed map. `0.0` = uniform PageRank; `> 0.0` activates PPR on every shard. /// Constant across all dispatches within a run (it is a cluster-wide scalar). pub(super) personalization_sum: f64, + /// The cut marker the node resolves the run's read cut from: set on the + /// count phase only, `0` after (see `BspSuperstepPlan::read_cut_marker`). + pub(super) read_cut_marker: u64, + /// The run's read cut, once the count phase resolved it. + pub(super) system_as_of: Option, } /// One node's decoded superstep result, tagged with its owner node id. pub(super) struct ShardResult { pub(super) node_id: u64, pub(super) result: BspSuperstepResult, + /// The highest watermark the node's cores served the superstep at. + pub(super) watermark_lsn: crate::types::Lsn, } /// Parameters for [`scatter_superstep`]. @@ -66,6 +73,7 @@ pub(super) struct ScatterSuperstepParams<'a> { pub(super) global_n: usize, pub(super) dispatches: Vec, pub(super) deadline_ms: u64, + pub(super) linearizable: bool, } /// Dispatch one `BspSuperstep` to every owner node concurrently and decode each @@ -84,6 +92,7 @@ pub(super) async fn scatter_superstep( global_n, dispatches, deadline_ms, + linearizable, } = args; let shared_arc = gateway_shared(state)?; let version_set = GatewayVersionSet::from_pairs(Vec::new()); @@ -101,6 +110,8 @@ pub(super) async fn scatter_superstep( rank_seed: d.rank_seed, global_dangling: d.global_dangling, personalization_sum: d.personalization_sum, + read_cut_marker: d.read_cut_marker, + system_as_of: d.system_as_of, }))); let version_set = version_set.clone(); let node_id = d.node_id; @@ -109,7 +120,7 @@ pub(super) async fn scatter_superstep( let shared_arc = shared_arc.clone(); Box::pin(async move { - let payload = dispatch_superstep_to_node( + let read = dispatch_superstep_to_node( &shared_arc, DispatchSuperstepParams { tenant_id, @@ -120,11 +131,16 @@ pub(super) async fn scatter_superstep( route_vshard, plan, version_set: &version_set, + linearizable, }, ) .await?; - let result = decode_single_result_from_payload(node_id, payload)?; - Ok::(ShardResult { node_id, result }) + let result = decode_single_result_from_payload(node_id, read.payload)?; + Ok::(ShardResult { + node_id, + result, + watermark_lsn: read.watermark_lsn, + }) }) }); diff --git a/nodedb/src/control/server/graph_dispatch/bsp_wcc/coord.rs b/nodedb/src/control/server/graph_dispatch/bsp_wcc/coord.rs index 13f0580ec..4657099b5 100644 --- a/nodedb/src/control/server/graph_dispatch/bsp_wcc/coord.rs +++ b/nodedb/src/control/server/graph_dispatch/bsp_wcc/coord.rs @@ -7,7 +7,9 @@ //! 1. **Enumerate.** One shard per distinct owner node (local + each distinct //! non-local data-group leader), each carrying that node's FULL owned-vShard //! set. Reuses `bsp_pagerank::enumerate::enumerate_shards`. -//! 2. **Contract.** Dispatch ONE `GraphOp::WccSuperstep` per owner node. Each +//! 2. **Contract.** Dispatch ONE `GraphOp::WccSuperstep` per owner node, with +//! one fresh cut marker. Every node resolves the same read cut from it and +//! reads the graph at it (see `graph_dispatch::run_cut`). Each //! shard contracts its OWNED nodes into local components and returns //! `node_labels` (`(name, local_component_root_name)`) plus `boundary_edges` //! (`(owned_name, ghost_name)` for every owned→ghost out-edge). @@ -28,7 +30,9 @@ use crate::control::state::SharedState; use crate::engine::graph::algo::result::AlgoResultBatch; use crate::types::{DatabaseId, TenantId}; -use super::scatter::scatter_wcc_round; +use super::scatter::{WccRound, scatter_wcc_round}; +use crate::control::server::graph_dispatch::run_cut::{agreed_cut, new_cut_marker}; +use crate::control::server::graph_dispatch::shard_reads::{ShardReadLog, qualified}; /// Run distributed WCC and return the bare `AlgoResultBatch` payload (the exact /// shape `algo_payload_to_query_response` consumes — identical to the @@ -42,6 +46,7 @@ pub async fn run_bsp_wcc( database_id: DatabaseId, params: AlgoParams, deadline_ms: u64, + linearizable: bool, ) -> crate::Result { // ── Enumerate shards (one per distinct owner node, local + remote). ── let enumeration = enumerate_shards(state)?; @@ -52,15 +57,43 @@ pub async fn run_bsp_wcc( } // ── Single contraction round across every owner node. ── + // Every node reads at one read cut, so every node reads the same graph. let results = scatter_wcc_round( state, - tenant_id, - database_id, - ¶ms, - &targets, - deadline_ms, + WccRound { + tenant_id, + database_id, + params: ¶ms, + targets: &targets, + read_cut_marker: new_cut_marker(state), + deadline_ms, + linearizable, + }, ) .await?; + agreed_cut( + results + .iter() + .map(|sr| (sr.node_id, sr.result.system_as_of)), + )?; + + // Every owner's partition was read once, at the watermark it served. + let mut reads = ShardReadLog::new(); + for sr in &results { + if let Some(target) = targets.iter().find(|t| t.node_id == sr.node_id) { + reads.note( + target.owned_vshards.iter().copied(), + sr.watermark_lsn, + target.node_id, + ); + } + } + reads.publish( + state, + tenant_id, + database_id, + Some(qualified(database_id, ¶ms.collection)), + ); // Concatenate every shard's local labels + boundary edges. Each owner node // holds a disjoint owned-node set, so the union of labels has one entry per diff --git a/nodedb/src/control/server/graph_dispatch/bsp_wcc/scatter.rs b/nodedb/src/control/server/graph_dispatch/bsp_wcc/scatter.rs index 93552a2ad..719dd50fa 100644 --- a/nodedb/src/control/server/graph_dispatch/bsp_wcc/scatter.rs +++ b/nodedb/src/control/server/graph_dispatch/bsp_wcc/scatter.rs @@ -26,7 +26,22 @@ use nodedb_physical::physical_plan::{GraphOp, WccSuperstepPlan, WccSuperstepResu /// One owner node's decoded WCC result. pub(super) struct ShardWccResult { + pub(super) node_id: u64, pub(super) result: WccSuperstepResult, + /// The highest watermark the node's cores served the round at. + pub(super) watermark_lsn: crate::types::Lsn, +} + +/// The inputs of one WCC round. +pub(super) struct WccRound<'a> { + pub(super) tenant_id: TenantId, + pub(super) database_id: DatabaseId, + pub(super) params: &'a AlgoParams, + pub(super) targets: &'a [ShardTarget], + /// The cut marker every node resolves the round's read cut from. + pub(super) read_cut_marker: u64, + pub(super) deadline_ms: u64, + pub(super) linearizable: bool, } /// Dispatch one `WccSuperstep` to every owner node concurrently and decode each @@ -34,12 +49,17 @@ pub(super) struct ShardWccResult { /// loop; the coordinator stitches the returned results globally. pub(super) async fn scatter_wcc_round( state: &crate::control::state::SharedState, - tenant_id: TenantId, - database_id: DatabaseId, - params: &AlgoParams, - targets: &[ShardTarget], - deadline_ms: u64, + round: WccRound<'_>, ) -> crate::Result> { + let WccRound { + tenant_id, + database_id, + params, + targets, + read_cut_marker, + deadline_ms, + linearizable, + } = round; let shared_arc = gateway_shared(state)?; let version_set = GatewayVersionSet::from_pairs(Vec::new()); @@ -50,6 +70,9 @@ pub(super) async fn scatter_wcc_round( // homed here in one pass and emits boundary edges only for dsts on // OTHER nodes. owned_vshards: t.owned_vshards.clone(), + // Every node resolves the same read cut from the marker. + read_cut_marker, + system_as_of: None, }))); let version_set = version_set.clone(); let node_id = t.node_id; @@ -58,7 +81,7 @@ pub(super) async fn scatter_wcc_round( let shared_arc = shared_arc.clone(); Box::pin(async move { - let payload = dispatch_superstep_to_node( + let read = dispatch_superstep_to_node( &shared_arc, DispatchSuperstepParams { tenant_id, @@ -69,11 +92,16 @@ pub(super) async fn scatter_wcc_round( route_vshard, plan, version_set: &version_set, + linearizable, }, ) .await?; - let result = decode_wcc_from_payload(node_id, payload)?; - Ok::(ShardWccResult { result }) + let result = decode_wcc_from_payload(node_id, read.payload)?; + Ok::(ShardWccResult { + node_id, + result, + watermark_lsn: read.watermark_lsn, + }) }) }); diff --git a/nodedb/src/control/server/graph_dispatch/cluster_resolve.rs b/nodedb/src/control/server/graph_dispatch/cluster_resolve.rs index a533c4687..5ce92b04c 100644 --- a/nodedb/src/control/server/graph_dispatch/cluster_resolve.rs +++ b/nodedb/src/control/server/graph_dispatch/cluster_resolve.rs @@ -1,59 +1,22 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Shared cluster-routing helpers for graph scatter paths (`match_scatter` and -//! `bsp_pagerank`): resolve a vShard to a live `RouteDecision`, and fetch the +//! Shared cluster-dispatch helpers for graph scatter paths (`match_scatter` and +//! `bsp_pagerank`): dispatch a superstep to one owner node, and fetch the //! gateway `Arc` used for remote dispatch. //! -//! Both helpers resolve against LIVE Raft leadership where available so a stale -//! routing-table hint cannot misdirect a scatter. Factored here so the MATCH -//! scatter and the BSP PageRank coordinator share one implementation instead of -//! duplicating the routing-lock + live-leader plumbing. +//! vShard resolution against live Raft leadership lives in +//! `crate::control::gateway::live_leaders`. use std::sync::Arc; use crate::bridge::envelope::{Payload, PhysicalPlan}; use crate::control::gateway::dispatcher::{DispatchRouteParams, dispatch_route}; -use crate::control::gateway::router::resolve_decision; use crate::control::gateway::version_set::GatewayVersionSet; use crate::control::gateway::{RouteDecision, TaskRoute}; use crate::control::server::exchange::execute_plan_all_local_cores; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId}; -/// Resolve a vShard to a `RouteDecision` against live Raft leadership, falling -/// back to the routing-table hint when no live snapshot is available. -/// -/// `pub(crate)` so the in-transaction staging choke points -/// (`session::leader_forward`) can resolve a staged write's / overlay drop's -/// target leader with the same live-leader semantics the graph scatter uses, -/// instead of duplicating the routing-lock + live-leader plumbing. -pub(crate) fn resolve_for_vshard(state: &SharedState, vshard_id: u32) -> RouteDecision { - let routing_guard = state - .cluster_routing - .as_ref() - .map(|rw| rw.read().unwrap_or_else(|p| p.into_inner())); - let raft_snapshot: Vec = - state.raft_status_fn.get().map(|f| f()).unwrap_or_default(); - let live_leader = move |group_id: u64| -> u64 { - raft_snapshot - .iter() - .find(|gs| gs.group_id == group_id) - .map(|gs| gs.leader_id) - .unwrap_or(0) - }; - let live_lookup: Option<&dyn Fn(u64) -> u64> = if state.raft_status_fn.get().is_some() { - Some(&live_leader) - } else { - None - }; - resolve_decision( - vshard_id, - state.node_id, - routing_guard.as_deref(), - live_lookup, - ) -} - /// Parameters for [`dispatch_superstep_to_node`]. pub(in crate::control::server::graph_dispatch) struct DispatchSuperstepParams<'a> { pub(in crate::control::server::graph_dispatch) tenant_id: TenantId, @@ -64,10 +27,19 @@ pub(in crate::control::server::graph_dispatch) struct DispatchSuperstepParams<'a pub(in crate::control::server::graph_dispatch) route_vshard: u32, pub(in crate::control::server::graph_dispatch) plan: PhysicalPlan, pub(in crate::control::server::graph_dispatch) version_set: &'a GatewayVersionSet, + /// The superstep reads linearizably: each serving node confirms first. + pub(in crate::control::server::graph_dispatch) linearizable: bool, +} + +/// One owner node's answer to a graph superstep: its payload and the highest +/// watermark its cores served the plan at. +pub(in crate::control::server::graph_dispatch) struct NodeRead { + pub(in crate::control::server::graph_dispatch) payload: Payload, + pub(in crate::control::server::graph_dispatch) watermark_lsn: crate::types::Lsn, } /// Dispatch a single already-built graph-superstep `plan` to one owner node and -/// return its node-level payload. The LOCAL node fans the plan across all its +/// return its node-level payload and served watermark. The LOCAL node fans the plan across all its /// Data-Plane cores via `execute_plan_all_local_cores` (per-core results merged /// into one payload); a REMOTE node gets one `RouteDecision::Remote` dispatch via /// `dispatch_route`. An empty payload denotes a zero-vertex shard — the caller's @@ -76,7 +48,7 @@ pub(in crate::control::server::graph_dispatch) struct DispatchSuperstepParams<'a pub(in crate::control::server::graph_dispatch) async fn dispatch_superstep_to_node( shared_arc: &Arc, args: DispatchSuperstepParams<'_>, -) -> crate::Result { +) -> crate::Result { let DispatchSuperstepParams { tenant_id, database_id, @@ -86,12 +58,17 @@ pub(in crate::control::server::graph_dispatch) async fn dispatch_superstep_to_no route_vshard, plan, version_set, + linearizable, } = args; if is_local { // Local node: fan across ALL local cores and merge. The per-core // owned-node sets are disjoint, so the merged result is correct without // dedup. At 1 core/node this is behaviour-identical to a single-core - // dispatch. + // dispatch. A linearizable graph read confirms the groups the plan + // reads first. + if linearizable { + super::read_groups::confirm_graph_read(shared_arc, database_id, &plan).await?; + } let node_result = execute_plan_all_local_cores( shared_arc.as_ref(), tenant_id, @@ -102,7 +79,10 @@ pub(in crate::control::server::graph_dispatch) async fn dispatch_superstep_to_no None, ) .await?; - Ok(Payload::from_vec(node_result.payload)) + Ok(NodeRead { + payload: Payload::from_vec(node_result.payload), + watermark_lsn: node_result.watermark_lsn, + }) } else { // Remote node: one dispatch via the gateway. let route = TaskRoute { @@ -113,7 +93,7 @@ pub(in crate::control::server::graph_dispatch) async fn dispatch_superstep_to_no }, vshard_id: route_vshard, }; - let payloads = dispatch_route(DispatchRouteParams { + let outcome = dispatch_route(DispatchRouteParams { route, shared: shared_arc, tenant_id, @@ -123,16 +103,27 @@ pub(in crate::control::server::graph_dispatch) async fn dispatch_superstep_to_no version_set, // This resolve path carries no session-transaction context. txn_id: None, + linearizable, }) - .await? - .payloads; - payloads + .await?; + let watermark_lsn = outcome + .shard_watermarks + .iter() + .map(|(_, lsn)| *lsn) + .max() + .unwrap_or(crate::types::Lsn::ZERO); + let payload = outcome + .payloads .into_iter() .next() .map(Payload::from_vec) .ok_or_else(|| crate::Error::Internal { detail: format!("graph superstep: node={node_id} returned no payload"), - }) + })?; + Ok(NodeRead { + payload, + watermark_lsn, + }) } } diff --git a/nodedb/src/control/server/graph_dispatch/hop.rs b/nodedb/src/control/server/graph_dispatch/hop.rs index 8624b7812..40791cc6c 100644 --- a/nodedb/src/control/server/graph_dispatch/hop.rs +++ b/nodedb/src/control/server/graph_dispatch/hop.rs @@ -27,7 +27,7 @@ //! fully-attributed for BOTH the local-shard and remote-shard portions. //! //! Ownership is resolved against LIVE Raft leadership (via -//! [`resolve_decision`] with a live-leader lookup), not the cached routing +//! [`LiveLeaders::resolve`]), not the cached routing //! table, so a stale routing hint cannot misroute a frontier node. use std::collections::HashMap; @@ -38,7 +38,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::gateway::dispatcher::{ DispatchRouteParams, dispatch_route, statement_deadline_ms, }; -use crate::control::gateway::router::resolve_decision; +use crate::control::gateway::live_leaders::LiveLeaders; use crate::control::gateway::version_set::GatewayVersionSet; use crate::control::gateway::{RouteDecision, TaskRoute}; use crate::control::state::SharedState; @@ -47,6 +47,8 @@ use crate::engine::graph::traversal_options::GraphTraversalOptions; use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; use nodedb_physical::physical_plan::GraphOp; +use super::shard_reads::{ShardReadLog, key_vshards}; + /// A fully-attributed edge crossed by the hop: `(src, label, dst)`. pub(super) type NeighborTriple = (String, String, String); @@ -58,6 +60,9 @@ pub(super) struct HopOutput { /// Deduplicated destination node IDs after merging local + remote /// expansion. Feeds the next frontier. pub merged_destinations: Vec, + /// The key vShard of every frontier node, at the watermark its expansion + /// served. + pub reads: ShardReadLog, } /// Parameters for one BFS hop. @@ -72,6 +77,8 @@ pub(super) struct NeighborHopParams<'a> { /// Data-Plane-side allocation under `options.max_visited` via /// `NeighborsMulti.max_results`. pub discovered_so_far: usize, + /// Each node that expands part of the frontier confirms its groups first. + pub linearizable: bool, } /// Execute one hop of BFS from `params.frontier`. @@ -88,6 +95,7 @@ pub(super) async fn execute_neighbor_hop( direction, options, discovered_so_far, + linearizable, } = params; // Cap this hop's handler-side allocation to the remaining budget under @@ -103,7 +111,7 @@ pub(super) async fn execute_neighbor_hop( // Single-node mode: no routing table — every frontier node is local. if shared.cluster_routing.is_none() { - let triples = expand_local( + let (triples, _watermark) = expand_local( shared, ExpandScope { tenant_id, @@ -112,6 +120,7 @@ pub(super) async fn execute_neighbor_hop( edge_label, direction, max_results: remaining_budget, + linearizable, }, frontier, ) @@ -120,6 +129,7 @@ pub(super) async fn execute_neighbor_hop( return Ok(HopOutput { local_triples: triples, merged_destinations: merged, + reads: ShardReadLog::new(), }); } @@ -127,11 +137,12 @@ pub(super) async fn execute_neighbor_hop( // LIVE Raft leadership (not the stale routing-table hint). let (local_nodes, remote_by_owner) = partition_frontier_by_owner(shared, frontier)?; + let mut reads = ShardReadLog::new(); // Local-owned subset: expand on local cores. let mut all_triples: Vec = if local_nodes.is_empty() { Vec::new() } else { - expand_local( + let (triples, watermark) = expand_local( shared, ExpandScope { tenant_id, @@ -140,10 +151,13 @@ pub(super) async fn execute_neighbor_hop( edge_label, direction, max_results: remaining_budget, + linearizable, }, &local_nodes, ) - .await? + .await?; + reads.note(key_vshards(&local_nodes), watermark, shared.node_id); + triples }; // Remote-owned subsets: ship a typed `NeighborsMulti` to each owner and @@ -159,8 +173,10 @@ pub(super) async fn execute_neighbor_hop( edge_label, direction, max_results: remaining_budget, + linearizable, }, remote_by_owner, + &mut reads, ) .await?; all_triples.extend(remote_triples); @@ -170,6 +186,7 @@ pub(super) async fn execute_neighbor_hop( Ok(HopOutput { local_triples: all_triples, merged_destinations: merged, + reads, }) } @@ -185,32 +202,20 @@ struct RemoteOwnerBatch { /// remote-owned subsets grouped by `(owner node, vShard)`. /// /// Ownership is resolved against LIVE Raft leadership via -/// [`resolve_decision`] with a live-leader lookup, so a stale routing hint +/// [`LiveLeaders::resolve`], so a stale routing hint /// cannot misroute a frontier node. A node whose owning vShard currently has /// no known leader (`LeaderUnknown`) is a hard error — we never silently -/// degrade to a local-only expansion that would return a partial set. +/// degrade to a local-only expansion that will return a partial set. fn partition_frontier_by_owner( shared: &SharedState, frontier: &[String], ) -> crate::Result<(Vec, Vec)> { + // Raft snapshot first, routing guard second: see `LiveLeaders`. + let live = LiveLeaders::snapshot(shared); let routing_guard = shared .cluster_routing .as_ref() .map(|rw| rw.read().unwrap_or_else(|p| p.into_inner())); - let raft_snapshot: Vec = - shared.raft_status_fn.get().map(|f| f()).unwrap_or_default(); - let live_leader = move |group_id: u64| -> u64 { - raft_snapshot - .iter() - .find(|gs| gs.group_id == group_id) - .map(|gs| gs.leader_id) - .unwrap_or(0) - }; - let live_lookup: Option<&dyn Fn(u64) -> u64> = if shared.raft_status_fn.get().is_some() { - Some(&live_leader) - } else { - None - }; let mut local: Vec = Vec::new(); // Group remote nodes by owning vShard so each owner gets one batched plan. @@ -218,12 +223,7 @@ fn partition_frontier_by_owner( for node in frontier { let vshard_id = VShardId::from_key(node.as_bytes()).as_u32(); - let decision = resolve_decision( - vshard_id, - shared.node_id, - routing_guard.as_deref(), - live_lookup, - ); + let decision = live.resolve(vshard_id, shared.node_id, routing_guard.as_deref()); match decision { RouteDecision::Local => local.push(node.clone()), RouteDecision::Remote { @@ -245,6 +245,7 @@ fn partition_frontier_by_owner( vshard_id: VShardId::new((vs % VShardId::COUNT as u64) as u32), leader_node: 0, leader_addr: String::new(), + leader_term: 0, }); } RouteDecision::Broadcast { .. } => { @@ -270,14 +271,16 @@ struct ExpandScope<'a> { edge_label: Option<&'a str>, direction: Direction, max_results: u32, + linearizable: bool, } -/// Expand a locally-owned subset on all local Data-Plane cores. +/// Expand a locally-owned subset on all local Data-Plane cores. Returns the +/// crossed edges and the highest watermark a core served them at. async fn expand_local( shared: &SharedState, scope: ExpandScope<'_>, node_ids: &[String], -) -> crate::Result> { +) -> crate::Result<(Vec, crate::types::Lsn)> { let ExpandScope { tenant_id, database_id, @@ -285,6 +288,7 @@ async fn expand_local( edge_label, direction, max_results, + linearizable, } = scope; let plan = PhysicalPlan::Graph(GraphOp::NeighborsMulti { collection: collection @@ -295,6 +299,10 @@ async fn expand_local( max_results, rls_filters: Vec::new(), }); + // A linearizable hop confirms the groups its plan reads first. + if linearizable { + super::read_groups::confirm_graph_read(shared, database_id, &plan).await?; + } let resp = crate::control::server::broadcast::broadcast_to_all_cores( shared, tenant_id, @@ -303,16 +311,18 @@ async fn expand_local( TraceId::ZERO, ) .await?; - Ok(decode_neighbor_triples(&resp.payload)) + Ok((decode_neighbor_triples(&resp.payload)?, resp.watermark_lsn)) } /// Expand the remote-owned subsets concurrently: ship a typed /// `NeighborsMulti` plan to each owning node via [`dispatch_route`] and /// decode every returned payload with the shared `{src,label,node}` decoder. +/// Notes each owner's vShard in `reads` at the watermark it served. async fn expand_remote( shared: &SharedState, scope: ExpandScope<'_>, owners: Vec, + reads: &mut ShardReadLog, ) -> crate::Result> { let ExpandScope { tenant_id, @@ -321,6 +331,7 @@ async fn expand_remote( edge_label, direction, max_results, + linearizable, } = scope; // The dispatcher's remote path needs an owned `Arc`. In // cluster mode the gateway is always wired; `gateway_shared` fails loudly @@ -358,6 +369,7 @@ async fn expand_remote( }; let version_set = version_set.clone(); let shared_arc = shared_arc.clone(); + let read_vshard = (vshard_id % VShardId::COUNT as u64) as u32; // Box::pin keeps the heterogeneous async dispatch futures uniform for // `join_all` and guards against any future async-recursion concerns. Box::pin(async move { @@ -371,8 +383,10 @@ async fn expand_remote( version_set: &version_set, // Graph hop traversal carries no session-transaction context. txn_id: None, + linearizable, }) .await + .map(|outcome| (read_vshard, node_id, outcome)) }) }); @@ -381,12 +395,11 @@ async fn expand_remote( let mut triples: Vec = Vec::new(); for result in results { // A remote dispatch error is fatal: a dropped owner means a partial - // reachable set, exactly the silent-degradation bug this path fixes. - // Graph hop traversal consumes payloads only; per-shard watermarks are - // not part of the neighbor-triple decode. - let payloads = result?.payloads; - for payload in payloads { - triples.extend(decode_neighbor_triples_bytes(&payload)); + // reachable set. + let (read_vshard, node_id, outcome) = result?; + reads.note_leg([read_vshard], &outcome.shard_watermarks, node_id); + for payload in outcome.payloads { + triples.extend(decode_neighbor_triples_bytes(&payload)?); } } Ok(triples) @@ -408,39 +421,102 @@ fn dedup_destinations(triples: &[NeighborTriple]) -> Vec { /// fully-typed triples. (`Payload` derefs to `[u8]`.) /// /// [`Payload`]: crate::bridge::envelope::Payload -fn decode_neighbor_triples(payload: &crate::bridge::envelope::Payload) -> Vec { +fn decode_neighbor_triples( + payload: &crate::bridge::envelope::Payload, +) -> crate::Result> { decode_neighbor_triples_bytes(payload) } /// Decode raw Data-Plane response bytes (the shape both a local broadcast and /// a remote `dispatch_route` return — the same `NeighborsMulti` op produces it /// on any node) into fully-typed triples. -fn decode_neighbor_triples_bytes(payload: &[u8]) -> Vec { +pub(super) fn decode_neighbor_triples_bytes(payload: &[u8]) -> crate::Result> { if payload.is_empty() { - return Vec::new(); + return Ok(Vec::new()); } let json_text = crate::data::executor::response_codec::decode_payload_to_json(payload); decode_neighbor_triples_json(&json_text) } /// Shared inner decode: parse the `{src,label,node}` JSON array into triples. -/// Malformed entries (missing or non-string `src`/`node`) are skipped; -/// `label` defaults to "" since label-less edges are a valid graph shape. -fn decode_neighbor_triples_json(json_text: &str) -> Vec { - let arr = match sonic_rs::from_str::>(json_text) { - Ok(arr) => arr, - Err(_) => return Vec::new(), - }; +/// +/// A payload that is not a JSON array, or a row without a non-empty string +/// `src` and `node`, fails with a `Codec` error naming the row. Dropping it +/// returns a partial neighbor set as a complete one. `label` defaults to "" +/// because a label-less edge is a valid graph shape. +fn decode_neighbor_triples_json(json_text: &str) -> crate::Result> { + let arr = sonic_rs::from_str::>(json_text).map_err(|e| { + crate::Error::Codec { + detail: format!("graph neighbor rows: payload is not a JSON array: {e}"), + } + })?; let mut out = Vec::with_capacity(arr.len()); - for item in arr { + for (index, item) in arr.iter().enumerate() { let src = item.get("src").and_then(|v| v.as_str()); let node = item.get("node").and_then(|v| v.as_str()); let (src, node) = match (src, node) { (Some(s), Some(n)) if !s.is_empty() && !n.is_empty() => (s, n), - _ => continue, + _ => { + return Err(crate::Error::Codec { + detail: format!( + "graph neighbor row {index} lacks a non-empty string `src` and `node`: {item}" + ), + }); + } }; let label = item.get("label").and_then(|v| v.as_str()).unwrap_or(""); out.push((src.to_string(), label.to_string(), node.to_string())); } - out + Ok(out) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn well_formed_rows_decode_with_an_optional_label() { + let triples = decode_neighbor_triples_json( + r#"[{"src":"a","label":"K","node":"b"},{"src":"b","node":"c"}]"#, + ) + .unwrap(); + assert_eq!( + triples, + vec![ + ("a".to_string(), "K".to_string(), "b".to_string()), + ("b".to_string(), String::new(), "c".to_string()), + ] + ); + } + + #[test] + fn a_payload_that_is_not_an_array_is_a_codec_error() { + assert!(matches!( + decode_neighbor_triples_json(r#"{"src":"a"}"#), + Err(crate::Error::Codec { .. }) + )); + } + + #[test] + fn a_malformed_row_is_a_codec_error_naming_the_row() { + match decode_neighbor_triples_json(r#"[{"src":"a","node":"b"},{"src":"a"}]"#) { + Err(crate::Error::Codec { detail }) => { + assert!(detail.contains("row 1"), "{detail}"); + } + other => panic!("expected a codec error, got {other:?}"), + } + } + + #[test] + fn an_empty_node_id_is_a_codec_error() { + assert!(matches!( + decode_neighbor_triples_json(r#"[{"src":"","node":"b"}]"#), + Err(crate::Error::Codec { .. }) + )); + } + + #[test] + fn an_empty_payload_is_no_neighbors() { + assert!(decode_neighbor_triples_bytes(&[]).unwrap().is_empty()); + } } diff --git a/nodedb/src/control/server/graph_dispatch/keyed_read.rs b/nodedb/src/control/server/graph_dispatch/keyed_read.rs new file mode 100644 index 000000000..d75ba2a56 --- /dev/null +++ b/nodedb/src/control/server/graph_dispatch/keyed_read.rs @@ -0,0 +1,158 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A one-hop graph read of one node, served where that node's edges live. +//! +//! A node's edges, forward and reverse, live on its key vShard +//! (`types/record_home.rs`). The read runs on that vShard's leader: here when +//! this node leads it, through the gateway otherwise. The leader also holds a +//! transaction's staged edges (`shared/session/leader_forward.rs`). + +use crate::bridge::envelope::{Payload, PhysicalPlan}; +use crate::control::gateway::dispatcher::{DispatchRouteParams, dispatch_route}; +use crate::control::gateway::live_leaders::resolve_live_decision; +use crate::control::gateway::version_set::GatewayVersionSet; +use crate::control::gateway::{RouteDecision, TaskRoute}; +use crate::control::server::payload_merge::merge_msgpack_arrays; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, TraceId, TxnId, VShardId}; + +use super::cluster_resolve::gateway_shared; +use super::shard_reads::ShardReadLog; + +/// Run `plan`, a one-hop read of `node_key`, on the leader of the node's key +/// vShard, and return its merged payload. +pub async fn read_on_key_owner( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + node_key: &str, + plan: PhysicalPlan, + txn_id: Option, + linearizable: bool, +) -> crate::Result { + let vshard_id = VShardId::from_key(node_key.as_bytes()).as_u32(); + read_on_vshard( + state, + VShardRead { + tenant_id, + database_id, + vshard_id, + txn_id, + linearizable, + }, + plan, + ) + .await +} + +/// Where a [`read_on_vshard`] runs, and as what. +#[derive(Debug, Clone, Copy)] +pub struct VShardRead { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + pub vshard_id: u32, + pub txn_id: Option, + pub linearizable: bool, +} + +/// Run `plan`, a read of one vShard's state, on the leader of that vShard, +/// and return its merged payload. +pub async fn read_on_vshard( + state: &SharedState, + at: VShardRead, + plan: PhysicalPlan, +) -> crate::Result { + let VShardRead { + tenant_id, + database_id, + vshard_id, + txn_id, + linearizable, + } = at; + let collection = plan_collection(&plan); + let mut reads = ShardReadLog::new(); + let payload = match resolve_live_decision(state, vshard_id) { + RouteDecision::Local => { + if linearizable { + super::read_groups::confirm_graph_read(state, database_id, &plan).await?; + } + let response = crate::control::server::broadcast::broadcast_to_all_cores_txn( + state, + tenant_id, + database_id, + plan, + TraceId::ZERO, + txn_id, + ) + .await?; + reads.note([vshard_id], response.watermark_lsn, state.node_id); + response.payload + } + RouteDecision::Remote { + node_id, + vshard_id: route_vshard, + } => { + let shared_arc = gateway_shared(state)?; + let version_set = GatewayVersionSet::from_pairs(Vec::new()); + let outcome = dispatch_route(DispatchRouteParams { + route: TaskRoute { + plan, + decision: RouteDecision::Remote { + node_id, + vshard_id: route_vshard, + }, + vshard_id, + }, + shared: &shared_arc, + tenant_id, + database_id, + trace_id: TraceId::ZERO, + deadline_ms: crate::control::gateway::dispatcher::statement_deadline_ms(state), + version_set: &version_set, + txn_id, + linearizable, + }) + .await?; + reads.note_leg([vshard_id], &outcome.shard_watermarks, node_id); + let payloads = outcome.payloads; + Payload::from_vec(match payloads.len() { + 1 => payloads.into_iter().next().unwrap_or_default(), + _ => merge_msgpack_arrays(&payloads), + }) + } + RouteDecision::LeaderUnknown { vshard_id } => { + return Err(crate::Error::NotLeader { + vshard_id: VShardId::new((vshard_id % VShardId::COUNT as u64) as u32), + leader_node: 0, + leader_addr: String::new(), + leader_term: 0, + }); + } + RouteDecision::Broadcast { .. } => { + return Err(crate::Error::Internal { + detail: "graph keyed read: a single vShard resolved to a broadcast".into(), + }); + } + }; + // The node's key vShard joins the transaction read-set. + reads.publish(state, tenant_id, database_id, collection); + Ok(payload) +} + +/// The database-qualified collection a one-hop graph plan scopes, or `None` +/// when it reads every collection. +fn plan_collection(plan: &PhysicalPlan) -> Option { + use nodedb_physical::physical_plan::GraphOp; + match plan { + PhysicalPlan::Graph( + GraphOp::Neighbors { collection, .. } + | GraphOp::NeighborsMulti { collection, .. } + | GraphOp::Hop { collection, .. }, + ) => collection.as_ref().map(|c| c.as_str().to_owned()), + PhysicalPlan::Graph( + GraphOp::TemporalNeighbors { collection, .. } + | GraphOp::NodePresenceRead { collection, .. }, + ) => Some(collection.as_str().to_owned()), + _ => None, + } +} diff --git a/nodedb/src/control/server/graph_dispatch/match_broadcast.rs b/nodedb/src/control/server/graph_dispatch/match_broadcast.rs index 86a225ac1..08d7d9ce0 100644 --- a/nodedb/src/control/server/graph_dispatch/match_broadcast.rs +++ b/nodedb/src/control/server/graph_dispatch/match_broadcast.rs @@ -12,7 +12,7 @@ //! ``` //! //! The generic `gather_all_cores` / `broadcast_to_all_cores` primitives treat -//! the whole payload as a BARE msgpack array of row elements, which would +//! the whole payload as a BARE msgpack array of row elements, which will //! mis-merge this map. This module mirrors `gather_all_cores`'s per-core SPSC //! fan-out (eager dispatch → `join_all`, NotFound-tolerant) but, for each core, //! it DECODES the envelope and: @@ -31,7 +31,8 @@ use futures::future::join_all; -use crate::bridge::envelope::{Payload, PhysicalPlan, Response, Status}; +use crate::bridge::envelope::{Payload, PhysicalPlan, Response}; +use crate::control::server::exchange::core_outcome::{classify_core_response, require_every_core}; use crate::control::server::exchange::gather::eager_dispatch_to_all_cores; use crate::control::server::payload_merge::{encode_msgpack_array, extract_msgpack_elements}; use crate::data::executor::handlers::graph_match::{ @@ -41,6 +42,17 @@ use crate::engine::graph::pattern::executor::{UnresolvedExpansion, VarLenResume} use crate::types::{DatabaseId, TenantId, TraceId, TxnId}; use nodedb_query::msgpack_scan::reader::{map_header, read_str_advance, skip_value}; +/// How a graph read runs. +#[derive(Debug, Clone, Copy)] +pub struct GraphRead { + /// The session transaction whose staged edge overlay the read merges. + /// `None` for autocommit. + pub txn_id: Option, + /// The read observes every write committed before it began: each node + /// that serves part of it confirms its groups first. + pub linearizable: bool, +} + /// Result of a MATCH cross-core broadcast after envelope unwrapping. pub struct MatchBroadcastOutcome { /// Merged binding rows as a single BARE msgpack array — the exact shape @@ -55,10 +67,12 @@ pub struct MatchBroadcastOutcome { /// this is a `Vec` — a single node fanned across N cores can truncate on /// several cores at once and ALL their cursors must survive (the round loop /// re-dispatches each independently). Carried onto the cross-node wire by - /// `encode_match_envelope_raw` so remote truncation is no longer dropped. + /// `encode_match_envelope_raw` so remote truncation survives. pub resume: Vec, /// `true` if any core returned a partial (truncated) result. pub partial: bool, + /// The highest watermark any core served the MATCH at. + pub watermark_lsn: crate::types::Lsn, } /// Locate a top-level map value by key in a msgpack map payload. @@ -183,18 +197,32 @@ pub fn unwrap_match_envelope(payload: &Payload) -> crate::Result, + read: GraphRead, ) -> crate::Result { + let GraphRead { + txn_id, + linearizable, + } = read; + // A linearizable read confirms the groups the plan reads first. A MATCH + // walks the local CSR, so that is every group this node replicates. A + // no-op without a cluster. + if linearizable { + super::read_groups::confirm_graph_read(state, database_id, &plan).await?; + } + // Shared broadcast call counter (parity with the generic gather path). crate::control::server::broadcast::broadcast_call_count_increment(); @@ -217,7 +245,7 @@ pub async fn broadcast_match_to_all_cores( })?; // Await all cores in parallel, draining the full bounded response per core - // (a core's result may stream as several Partial frames before its terminal + // (a core's result can stream as several Partial frames before its terminal // frame). let max_result_bytes = state.tuning.network.max_query_result_bytes as usize; let response_futures = receivers @@ -237,40 +265,21 @@ pub async fn broadcast_match_to_all_cores( }); let results: Vec> = join_all(response_futures).await; + // `NotFound` is an empty CSR slice on that core. Any other core error fails + // the MATCH: rows from the surviving cores are not the full answer. + let answered = require_every_core(results.into_iter().map(classify_core_response))?; let mut all_row_elements: Vec> = Vec::new(); let mut frontier: Vec = Vec::new(); let mut resume: Vec = Vec::new(); let mut partial = false; - // First error seen across cores, kept as a TYPED error: a core cut short by - // the statement's deadline reports the deadline rather than collapsing into - // a generic dispatch failure. - let mut first_error: Option = None; - - for result in results { - let resp = match result { - Ok(r) => r, - Err(error) => { - if first_error.is_none() { - first_error = Some(error); - } - continue; - } - }; - - if resp.status == Status::Error { - // `NotFound` is an empty CSR slice on this core, not an error. - if let Err(error) = crate::control::local_dispatch::reject_data_plane_error(&resp) - && first_error.is_none() - { - first_error = Some(error); - } - continue; - } + let mut watermark_lsn = crate::types::Lsn::ZERO; + for resp in answered { if resp.partial { partial = true; } + watermark_lsn = watermark_lsn.max(resp.watermark_lsn); if resp.payload.is_empty() { continue; @@ -285,12 +294,6 @@ pub async fn broadcast_match_to_all_cores( resume.append(&mut decoded.resume); } - if all_row_elements.is_empty() - && let Some(error) = first_error - { - return Err(error); - } - let merged_rows = encode_msgpack_array(&all_row_elements); Ok(MatchBroadcastOutcome { @@ -298,6 +301,7 @@ pub async fn broadcast_match_to_all_cores( frontier, resume, partial, + watermark_lsn, }) } @@ -348,7 +352,7 @@ mod tests { // Rows preserved: 2 elements. Merging them back into a bare array // reproduces the exact `rows` map values embedded in the envelope — // compare against the SAME bytes the envelope carries (a second - // independent `rows_to_msgpack` call could differ only in HashMap key + // independent `rows_to_msgpack` call can differ only in HashMap key // order, so we reconstruct the expected bare array from the envelope's // own `rows` field rather than re-serializing). assert_eq!(decoded.row_elements.len(), 2); diff --git a/nodedb/src/control/server/graph_dispatch/match_scatter/coord.rs b/nodedb/src/control/server/graph_dispatch/match_scatter/coord.rs index 1098a582a..01765db8d 100644 --- a/nodedb/src/control/server/graph_dispatch/match_scatter/coord.rs +++ b/nodedb/src/control/server/graph_dispatch/match_scatter/coord.rs @@ -7,14 +7,16 @@ use std::collections::{BTreeMap, HashMap, HashSet}; use crate::bridge::envelope::Payload; use crate::control::gateway::RouteDecision; -use crate::control::server::graph_dispatch::cluster_resolve::resolve_for_vshard; +use crate::control::gateway::live_leaders::resolve_live_decision; use crate::control::state::SharedState; use crate::engine::graph::pattern::executor::{UnresolvedExpansion, VarLenResume, rows_to_msgpack}; -use crate::types::{DatabaseId, TenantId, TxnId, VShardId}; +use crate::types::{DatabaseId, TenantId, VShardId}; use nodedb_cluster::distributed_graph::{ DistributedMatchCoordinator, PatternContinuation, ResolvedContinuationArgs, ShardMatchResult, }; +use crate::control::server::graph_dispatch::match_broadcast::GraphRead; + use super::resume_queue::{PendingResume, resume_seed_key, resume_to_pending}; use super::round_loop::{dispatch_continuations, dispatch_resumes}; use super::round_zero::scatter_round_zero; @@ -72,16 +74,18 @@ pub(super) struct TaggedShardResult { /// Orchestrate a cross-shard MATCH. Caller guarantees cluster mode /// (`cluster_routing.is_some()`); single-node never enters here. /// -/// `txn_id` is threaded onto every LOCAL scatter/resume leg so this node's cores -/// merge the transaction's staged edge overlay for read-your-own-writes; remote -/// legs read committed CSR (multi-node overlay forwarding is a separate unit). +/// `read.txn_id` is threaded onto every LOCAL scatter/resume leg so this node's +/// cores merge the transaction's staged edge overlay for read-your-own-writes; +/// remote legs read committed CSR (multi-node overlay forwarding is a separate +/// unit). `read.linearizable` makes every leg confirm its groups where it is +/// served. pub async fn scatter_match( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, query_bytes: Vec, deadline_ms: u64, - txn_id: Option, + read: GraphRead, ) -> crate::Result { // Round budget = the maximum number of hops the pattern can take (summed // per-triple `max_hops`), since each round advances every frontier by one @@ -108,17 +112,17 @@ pub async fn scatter_match( let mut resume_rounds: u32 = 0; // Latches once the coordinator's hop budget (`max_rounds`) is exhausted with // continuations still pending, so the loop stops re-attempting `advance()` - // (which would spin) but keeps draining any remaining resume cursors. + // (which will spin) but keeps draining any remaining resume cursors. let mut continuations_exhausted = false; // ---- Round 0: scatter the Match plan to local + every remote owner. ---- - let round0 = scatter_round_zero( + let (round0, reads) = scatter_round_zero( state, tenant_id, database_id, &query_bytes, deadline_ms, - txn_id, + read, ) .await?; for tagged in round0 { @@ -138,7 +142,7 @@ pub async fn scatter_match( if !coordinator.advance() { // Exhausted max_rounds with hops still pending: surface as // partial rather than silently dropping the continuations. - // Latch so the loop stops re-attempting `advance()` (it would + // Latch so the loop stops re-attempting `advance()` (it will // spin) but still drains any pending resume cursors below. if coordinator.has_pending() { partial = true; @@ -152,7 +156,7 @@ pub async fn scatter_match( database_id, &query_bytes, deadline_ms, - txn_id, + read, pending, ) .await?; @@ -190,7 +194,7 @@ pub async fn scatter_match( database_id, &query_bytes, deadline_ms, - txn_id, + read, batch, ) .await?; @@ -208,6 +212,13 @@ pub async fn scatter_match( // ---- Dedup + encode. ---- let rows_payload = dedup_and_encode(&coordinator.completed)?; + // Every vShard round 0 read joins the transaction read-set. + reads.publish( + state, + tenant_id, + database_id, + pattern_collection(database_id, &query_bytes), + ); Ok(MatchScatterOutcome { rows_payload, partial, @@ -221,8 +232,8 @@ pub async fn scatter_match( /// re-dispatch. /// /// A truncation that yields a resume cursor is RECOVERABLE (drained across -/// resume rounds) and is enqueued rather than surfaced — so this no longer -/// reports a partial. The only surfaced partials come from the coordinator's +/// resume rounds) and is enqueued rather than surfaced — so this +/// reports no partial. The only surfaced partials come from the coordinator's /// hop budget and the resume-round budget, both handled in `scatter_match`. pub(super) fn feed_result( state: &SharedState, @@ -277,7 +288,7 @@ fn frontier_to_continuations( let mut out = Vec::new(); for entry in frontier { let target_vshard = VShardId::from_key(entry.node_name.as_bytes()).as_u32(); - let decision = resolve_for_vshard(state, target_vshard); + let decision = resolve_live_decision(state, target_vshard); let owner_node = match decision { RouteDecision::Local => state.node_id, RouteDecision::Remote { node_id, .. } => node_id, @@ -286,6 +297,7 @@ fn frontier_to_continuations( vshard_id: VShardId::new((vshard_id % VShardId::COUNT as u64) as u32), leader_node: 0, leader_addr: String::new(), + leader_term: 0, }); } RouteDecision::Broadcast { .. } => { @@ -353,6 +365,18 @@ fn dedup_and_encode(rows: &[HashMap]) -> crate::Result Ok(Payload::from_vec(bytes)) } +/// The database-qualified collection a serialized `MatchQuery` is scoped to +/// with `IN ''`, or `None` for a pattern over every collection. +fn pattern_collection(database_id: DatabaseId, query_bytes: &[u8]) -> Option { + use crate::engine::graph::pattern::ast::MatchQuery; + let query: MatchQuery = zerompk::from_msgpack(query_bytes).ok()?; + query.collection.map(|bare| { + nodedb_types::QualifiedCollection::new(database_id, &bare) + .as_str() + .to_owned() + }) +} + /// Count the total pattern triples across every clause/chain in the serialized /// `MatchQuery`. Used to bound the continuation rounds. A malformed query (or a /// query with no triples) yields 0 — the caller floors `max_rounds` at 1. diff --git a/nodedb/src/control/server/graph_dispatch/match_scatter/resume_queue.rs b/nodedb/src/control/server/graph_dispatch/match_scatter/resume_queue.rs index f362de348..3a7fc14f3 100644 --- a/nodedb/src/control/server/graph_dispatch/match_scatter/resume_queue.rs +++ b/nodedb/src/control/server/graph_dispatch/match_scatter/resume_queue.rs @@ -15,7 +15,7 @@ use std::collections::hash_map::DefaultHasher; use std::hash::{Hash, Hasher}; use crate::control::gateway::RouteDecision; -use crate::control::server::graph_dispatch::cluster_resolve::resolve_for_vshard; +use crate::control::gateway::live_leaders::resolve_live_decision; use crate::control::state::SharedState; use crate::engine::graph::pattern::executor::VarLenResume; use crate::types::VShardId; @@ -25,7 +25,7 @@ use crate::types::VShardId; /// Keys on the anchor bindings (`source_row`), the frontier node identities, the /// triple index, and the hop depth — deliberately EXCLUDING each frontier /// entry's accumulating `path_so_far`, which grows one node longer every round -/// and would otherwise make every re-emission look unique and defeat dedup. +/// and will otherwise make every re-emission look unique and defeat dedup. /// /// Two resumes sharing a key re-expand the same frontier at the same depth from /// the same anchor and therefore reach the same onward bindings, so the @@ -86,7 +86,7 @@ pub(super) fn resume_to_pending( return Ok(None); }; let target_vshard = VShardId::from_key(node_name.as_bytes()).as_u32(); - let remote_coords = match resolve_for_vshard(state, target_vshard) { + let remote_coords = match resolve_live_decision(state, target_vshard) { RouteDecision::Local => None, RouteDecision::Remote { node_id, vshard_id } => Some((node_id, vshard_id)), RouteDecision::LeaderUnknown { vshard_id } => { @@ -94,11 +94,12 @@ pub(super) fn resume_to_pending( vshard_id: VShardId::new((vshard_id % VShardId::COUNT as u64) as u32), leader_node: 0, leader_addr: String::new(), + leader_term: 0, }); } RouteDecision::Broadcast { .. } => { return Err(crate::Error::Internal { - detail: "match scatter: resolve_for_vshard returned Broadcast for a \ + detail: "match scatter: resolve_live_decision returned Broadcast for a \ single vShard" .into(), }); diff --git a/nodedb/src/control/server/graph_dispatch/match_scatter/round_loop.rs b/nodedb/src/control/server/graph_dispatch/match_scatter/round_loop.rs index 0eef1118f..0d70e3a77 100644 --- a/nodedb/src/control/server/graph_dispatch/match_scatter/round_loop.rs +++ b/nodedb/src/control/server/graph_dispatch/match_scatter/round_loop.rs @@ -12,13 +12,16 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::gateway::dispatcher::{DispatchRouteParams, dispatch_route}; use crate::control::gateway::version_set::GatewayVersionSet; use crate::control::gateway::{RouteDecision, TaskRoute}; -use crate::control::server::graph_dispatch::match_broadcast::broadcast_match_to_all_cores; +use crate::control::server::graph_dispatch::match_broadcast::{ + GraphRead, broadcast_match_to_all_cores, +}; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, TraceId, TxnId, VShardId}; +use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; use nodedb_cluster::distributed_graph::PatternContinuation; use nodedb_physical::physical_plan::GraphOp; -use crate::control::server::graph_dispatch::cluster_resolve::{gateway_shared, resolve_for_vshard}; +use crate::control::gateway::live_leaders::resolve_live_decision; +use crate::control::server::graph_dispatch::cluster_resolve::gateway_shared; use super::coord::{TaggedShardResult, decode_rows}; use super::resume_queue::PendingResume; @@ -38,7 +41,7 @@ struct DispatchCtx<'f> { tenant_id: TenantId, database_id: DatabaseId, deadline_ms: u64, - txn_id: Option, + read: GraphRead, shared_arc: Arc, version_set: GatewayVersionSet, } @@ -55,12 +58,12 @@ fn push_dispatch_fut<'f>( plan: PhysicalPlan, remote_coords: Option<(u64, u64)>, ) { - let (state, tenant_id, database_id, deadline_ms, txn_id) = ( + let (state, tenant_id, database_id, deadline_ms, read) = ( ctx.state, ctx.tenant_id, ctx.database_id, ctx.deadline_ms, - ctx.txn_id, + ctx.read, ); match remote_coords { None => { @@ -76,7 +79,7 @@ fn push_dispatch_fut<'f>( database_id, plan, TraceId::ZERO, - txn_id, + read, ) .await?; Ok::<_, crate::Error>(vec![TaggedShardResult { @@ -104,7 +107,8 @@ fn push_dispatch_fut<'f>( trace_id: TraceId::ZERO, deadline_ms, version_set: &version_set, - txn_id, + txn_id: read.txn_id, + linearizable: read.linearizable, }) .await? .payloads; @@ -121,7 +125,7 @@ pub(super) async fn dispatch_continuations( database_id: DatabaseId, query_bytes: &[u8], deadline_ms: u64, - txn_id: Option, + read: GraphRead, pending: HashMap>, ) -> crate::Result> { let shared_arc = gateway_shared(state)?; @@ -130,7 +134,7 @@ pub(super) async fn dispatch_continuations( tenant_id, database_id, deadline_ms, - txn_id, + read, shared_arc, version_set: GatewayVersionSet::from_pairs(Vec::new()), }; @@ -139,9 +143,9 @@ pub(super) async fn dispatch_continuations( for (target_shard, conts) in pending { // Resolve once per target shard, not once per continuation: every // continuation targeting the same vShard gets the same routing - // decision, and `resolve_for_vshard` acquires a routing-table read - // lock on each call. - let decision = resolve_for_vshard(state, target_shard); + // decision, and `resolve_live_decision` takes a Raft snapshot and a + // routing-table read lock on each call. + let decision = resolve_live_decision(state, target_shard); // Extract the remote node coordinates (Copy-able u64 fields) so the // inner loop can reuse them without re-acquiring the routing lock. @@ -152,6 +156,7 @@ pub(super) async fn dispatch_continuations( vshard_id: VShardId::new((vshard_id % VShardId::COUNT as u64) as u32), leader_node: 0, leader_addr: String::new(), + leader_term: 0, }); } RouteDecision::Broadcast { .. } => { @@ -205,7 +210,7 @@ pub(super) async fn dispatch_resumes( database_id: DatabaseId, query_bytes: &[u8], deadline_ms: u64, - txn_id: Option, + read: GraphRead, pending_resumes: Vec, ) -> crate::Result> { let shared_arc = gateway_shared(state)?; @@ -214,7 +219,7 @@ pub(super) async fn dispatch_resumes( tenant_id, database_id, deadline_ms, - txn_id, + read, shared_arc, version_set: GatewayVersionSet::from_pairs(Vec::new()), }; diff --git a/nodedb/src/control/server/graph_dispatch/match_scatter/round_zero.rs b/nodedb/src/control/server/graph_dispatch/match_scatter/round_zero.rs index ee9b1781a..4b6aba2c8 100644 --- a/nodedb/src/control/server/graph_dispatch/match_scatter/round_zero.rs +++ b/nodedb/src/control/server/graph_dispatch/match_scatter/round_zero.rs @@ -3,46 +3,67 @@ //! Round-0 scatter: local broadcast + one remote dispatch per distinct //! non-local group leader, issued concurrently. -use std::collections::HashSet; +use std::collections::HashMap; use futures::future::join_all; use crate::bridge::envelope::{Payload, PhysicalPlan}; use crate::control::gateway::dispatcher::{DispatchRouteParams, dispatch_route}; +use crate::control::gateway::live_leaders::LiveLeaders; use crate::control::gateway::version_set::GatewayVersionSet; use crate::control::gateway::{RouteDecision, TaskRoute}; use crate::control::server::graph_dispatch::cluster_resolve::gateway_shared; use crate::control::server::graph_dispatch::match_broadcast::{ - broadcast_match_to_all_cores, unwrap_match_envelope, + GraphRead, broadcast_match_to_all_cores, unwrap_match_envelope, }; +use crate::control::server::graph_dispatch::shard_reads::ShardReadLog; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TenantId, TraceId, TxnId, VShardId}; +use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; use nodedb_physical::physical_plan::GraphOp; use super::coord::{TaggedShardResult, decode_rows}; -/// A distinct remote owner node and one vShard it owns (used as the dispatch -/// target for the round-0 remote `Match`). +/// A distinct remote owner node, one vShard it owns (the dispatch target for +/// the round-0 remote `Match`), and every vShard it leads. pub(super) struct RemoteOwner { pub(super) node_id: u64, pub(super) vshard_id: u64, + pub(super) vshards: Vec, +} + +/// Who leads each data vShard, from one routing snapshot: the vShards this +/// node leads, and one entry per remote leader. +struct RoundZeroOwners { + local_vshards: Vec, + remote: Vec, } /// Round-0 scatter: local broadcast + one remote dispatch per distinct /// non-local group leader, all issued concurrently. +/// +/// Every leg walks every vShard its node leads, so round 0 reads every vShard +/// of the graph. A MATCH result depends on every one of them: an edge written +/// to any vShard can add a match. The returned log notes each vShard under the +/// leg it was dispatched to, at the watermark that leg served, from the same +/// routing snapshot that picked the legs. Later rounds read vShards round 0 +/// already noted. pub(super) async fn scatter_round_zero( state: &SharedState, tenant_id: TenantId, database_id: DatabaseId, query_bytes: &[u8], deadline_ms: u64, - txn_id: Option, -) -> crate::Result> { + read: GraphRead, +) -> crate::Result<(Vec, ShardReadLog)> { + let GraphRead { + txn_id, + linearizable, + } = read; // Local cores: fan to all and unwrap each `{rows, frontier}` envelope. The // active `txn_id` is threaded onto this LOCAL leg so each core merges the // transaction's staged edge overlay for read-your-own-writes; with the // fixed-hop overlay merge un-gated in cluster mode, a bound zero-degree - // source still emits its cross-shard frontier. The same `txn_id` is now + // source still emits its cross-shard frontier. The same `txn_id` is // forwarded to remote owners below so their leg can resolve the transaction's // staged overlay; the staging/forwarding of that overlay to the leader is a // separate unit, so the forwarded id is inert until that lands. @@ -57,11 +78,14 @@ pub(super) async fn scatter_round_zero( database_id, local_plan, TraceId::ZERO, - txn_id, + read, ); // Remote owners: one batched dispatch per distinct non-local group leader. - let remote_owners = distinct_remote_owners(state)?; + let RoundZeroOwners { + local_vshards, + remote: remote_owners, + } = round_zero_owners(state)?; let shared_arc = gateway_shared(state)?; let version_set = GatewayVersionSet::from_pairs(Vec::new()); let remote_futs = remote_owners.into_iter().map(|owner| { @@ -80,9 +104,10 @@ pub(super) async fn scatter_round_zero( }; let version_set = version_set.clone(); let node_id = owner.node_id; + let leg_vshards = owner.vshards; let shared_arc = shared_arc.clone(); Box::pin(async move { - let payloads = dispatch_route(DispatchRouteParams { + let outcome = dispatch_route(DispatchRouteParams { route, shared: &shared_arc, tenant_id, @@ -91,10 +116,12 @@ pub(super) async fn scatter_round_zero( deadline_ms, version_set: &version_set, txn_id, + linearizable, }) - .await? - .payloads; - collect_remote_envelopes(node_id, payloads) + .await?; + let mut log = ShardReadLog::new(); + log.note_leg(leg_vshards, &outcome.shard_watermarks, node_id); + Ok::<_, crate::Error>((collect_remote_envelopes(node_id, outcome.payloads)?, log)) }) }); @@ -103,7 +130,9 @@ pub(super) async fn scatter_round_zero( futures::future::join(local_fut, join_all(remote_futs)).await; let mut out: Vec = Vec::new(); + let mut log = ShardReadLog::new(); let local_outcome = local_outcome?; + log.note(local_vshards, local_outcome.watermark_lsn, state.node_id); out.push(TaggedShardResult { emitting_node: state.node_id, rows: decode_rows(&local_outcome.rows_payload)?, @@ -111,33 +140,31 @@ pub(super) async fn scatter_round_zero( resume: local_outcome.resume, }); for res in remote_results { - out.extend(res?); + let (tagged, leg_log) = res?; + out.extend(tagged); + log.merge(leg_log); } - Ok(out) + Ok((out, log)) } /// Enumerate the distinct non-local data-group leaders, each paired with one -/// vShard the group owns. The metadata group (0) holds no vShards and is -/// skipped. Resolution uses LIVE Raft leadership where available so a stale -/// routing hint cannot misdirect the scatter. -fn distinct_remote_owners(state: &SharedState) -> crate::Result> { +/// vShard the group owns and every vShard it leads, and the vShards this node +/// leads. The metadata group (0) holds no vShards and is skipped. Resolution +/// uses LIVE Raft leadership where available so a stale routing hint cannot +/// misdirect the scatter. +fn round_zero_owners(state: &SharedState) -> crate::Result { + let mut owners = RoundZeroOwners { + local_vshards: Vec::new(), + remote: Vec::new(), + }; let Some(routing_lock) = state.cluster_routing.as_ref() else { - return Ok(Vec::new()); + return Ok(owners); }; + // Raft snapshot first, routing guard second: see `LiveLeaders`. + let live = LiveLeaders::snapshot(state); let routing = routing_lock.read().unwrap_or_else(|p| p.into_inner()); - let raft_snapshot: Vec = - state.raft_status_fn.get().map(|f| f()).unwrap_or_default(); - let live_leader = |group_id: u64| -> u64 { - raft_snapshot - .iter() - .find(|gs| gs.group_id == group_id) - .map(|gs| gs.leader_id) - .unwrap_or(0) - }; - - let mut seen: HashSet = HashSet::new(); - let mut owners = Vec::new(); + let mut remote_index: HashMap = HashMap::new(); for group_id in routing.group_ids() { // Skip the metadata group — it owns no vShards. if group_id == 0 { @@ -148,13 +175,14 @@ fn distinct_remote_owners(state: &SharedState) -> crate::Result continue; }; // Prefer live Raft leadership; fall back to the routing-table hint. - let mut leader = live_leader(group_id); + let mut leader = live.leader_of(group_id); if leader == 0 { leader = routing.group_info(group_id).map(|g| g.leader).unwrap_or(0); } if leader == state.node_id { // This group is LOCAL — already covered by the local // `broadcast_match_to_all_cores`; skip from the remote-owner set. + owners.local_vshards.extend(vshards); continue; } if leader == 0 { @@ -165,13 +193,19 @@ fn distinct_remote_owners(state: &SharedState) -> crate::Result vshard_id: VShardId::new(vshard_id), leader_node: 0, leader_addr: String::new(), + leader_term: 0, }); } - if seen.insert(leader) { - owners.push(RemoteOwner { - node_id: leader, - vshard_id: vshard_id as u64, - }); + match remote_index.get(&leader) { + Some(&index) => owners.remote[index].vshards.extend(vshards), + None => { + remote_index.insert(leader, owners.remote.len()); + owners.remote.push(RemoteOwner { + node_id: leader, + vshard_id: vshard_id as u64, + vshards, + }); + } } } Ok(owners) diff --git a/nodedb/src/control/server/graph_dispatch/mod.rs b/nodedb/src/control/server/graph_dispatch/mod.rs index 01a99d19d..20e2d6829 100644 --- a/nodedb/src/control/server/graph_dispatch/mod.rs +++ b/nodedb/src/control/server/graph_dispatch/mod.rs @@ -13,10 +13,8 @@ //! responses decode identically and are merged before the next depth level //! begins. See `hop::execute_neighbor_hop`. //! -//! `GRAPH PATH` (`shortest_path`) still uses the post-expansion -//! `control::scatter_gather::coordinate_cross_shard_hop` scatter path; the -//! per-frontier owner-targeted expansion above is specific to the BFS / -//! subgraph traversal read path. +//! `GRAPH PATH` (`shortest_path`), BFS and subgraph traversal all run each +//! hop through that owner-targeted expansion. pub mod bfs; pub mod bsp_pagerank; @@ -24,17 +22,29 @@ pub mod bsp_wcc; pub(crate) mod cluster_resolve; pub mod helpers; pub(crate) mod hop; +pub mod keyed_read; pub mod match_broadcast; pub mod match_scatter; +pub mod rag_fusion; +pub mod read_groups; +pub(crate) mod run_cut; +pub(crate) mod shard_reads; pub mod shortest_path; pub mod traverse_subgraph; +pub mod walk_reads; +pub mod whole_graph; pub use bfs::{CrossCoreBfsParams, cross_core_bfs_with_options}; pub use bsp_pagerank::run_bsp_pagerank; pub use bsp_wcc::run_bsp_wcc; +pub use keyed_read::{VShardRead, read_on_key_owner, read_on_vshard}; pub use match_broadcast::{ - MatchBroadcastOutcome, broadcast_match_to_all_cores, unwrap_match_envelope, + GraphRead, MatchBroadcastOutcome, broadcast_match_to_all_cores, unwrap_match_envelope, }; pub use match_scatter::{MatchScatterOutcome, scatter_match}; +pub use rag_fusion::serve_rag_plan; +pub use read_groups::{confirm_graph_read, graph_read_groups}; pub use shortest_path::{CrossCoreShortestPathParams, cross_core_shortest_path}; pub use traverse_subgraph::{CrossCoreTraverseSubgraphParams, cross_core_traverse_subgraph}; +pub use walk_reads::serve_walk_plan; +pub use whole_graph::{run_graph_algo, scatter_to_graph_owners}; diff --git a/nodedb/src/control/server/graph_dispatch/rag_fusion/coord.rs b/nodedb/src/control/server/graph_dispatch/rag_fusion/coord.rs new file mode 100644 index 000000000..ce963f52f --- /dev/null +++ b/nodedb/src/control/server/graph_dispatch/rag_fusion/coord.rs @@ -0,0 +1,373 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The cluster RAG fusion coordinator. +//! +//! One core runs a fusion over its own vector index, text index and graph +//! partition. In a cluster those live apart, so the coordinator runs the same +//! pipeline in stages: +//! +//! 1. The collection's owner exports the ranked vector hits, and the BM25 hits +//! of a three-source fusion (`stages::export_legs`). +//! 2. Every graph owner names the graph node each hit's surrogate is bound to +//! (`stages::bindings`). A hit keys on that name, or on its surrogate +//! identity when it names no node, as on one core. +//! 3. The coordinator walks the collection's edges from those nodes, hop by +//! hop, each frontier node expanded at its key vShard's leader +//! (`hop::execute_neighbor_hop`). Hop distances are breadth-first, and the +//! visit cap is the single core's: `max_visited`, bounded by the BFS memory +//! budget. +//! 4. Every graph owner reports which reached nodes carry a surrogate; the +//! rest count as unaddressable. +//! 5. The weighted reciprocal-rank fusion and the response body are the +//! single core's own functions (`graph_rag::rag_response_body` and the +//! ranked-list builders), so the answer has the same shape and ranks. +//! +//! Under the visit cap both walks admit each level's nodes in node-name order, +//! so a capped walk admits the same nodes here as on one core. + +use std::collections::{HashMap, HashSet}; + +use nodedb_physical::physical_plan::{GraphOp, RagStage}; +use nodedb_types::{RowIdentity, Surrogate}; + +use crate::bridge::envelope::{Payload, PhysicalPlan, Response}; +use crate::control::server::dispatch_utils::{not_found_response, ok_payload_response}; +use crate::control::state::SharedState; +use crate::data::executor::handlers::graph_rag::{ + RagResponseParams, graph_nodes_to_ranked_results, rag_response_body, vector_ranked_list, +}; +use crate::data::executor::handlers::graph_rag_triple::text_ranked_list; +use crate::engine::graph::edge_store::Direction; +use crate::engine::graph::traversal_options::GraphTraversalOptions; +use crate::query::fusion::reciprocal_rank_fusion_weighted; +use crate::types::{DatabaseId, TenantId}; + +use super::super::hop::{NeighborHopParams, execute_neighbor_hop}; +use super::super::shard_reads::ShardReadLog; +use super::stages::{BindingsRequest, ExportScope, bindings, export_legs}; + +/// Run `plan` when it is a whole RAG fusion (`RagStage::Local`) in a cluster. +/// Returns `None` on a single node and for every other plan: one core runs +/// those. +pub async fn serve_rag_plan( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + plan: &PhysicalPlan, + linearizable: bool, +) -> Option> { + state.cluster_routing.as_ref()?; + let PhysicalPlan::Graph(GraphOp::RagFusion { + stage: RagStage::Local, + .. + }) = plan + else { + return None; + }; + Some( + run( + state, + RagScope { + tenant_id, + database_id, + linearizable, + }, + plan, + ) + .await, + ) +} + +#[derive(Clone, Copy)] +struct RagScope { + tenant_id: TenantId, + database_id: DatabaseId, + linearizable: bool, +} + +async fn run(state: &SharedState, scope: RagScope, plan: &PhysicalPlan) -> crate::Result { + let PhysicalPlan::Graph(GraphOp::RagFusion { + collection, + edge_label, + direction, + expansion_depth, + final_top_k, + rrf_k, + rrf_k_triple, + options, + bm25_query, + bm25_field, + .. + }) = plan + else { + return Err(crate::Error::Internal { + detail: "rag fusion coordinator received a plan that is not a RAG fusion".into(), + }); + }; + let RagScope { + tenant_id, + database_id, + linearizable, + } = scope; + let qualified = collection.as_str().to_owned(); + let vshard = nodedb_types::CollectionKey::from_qualified(database_id, collection)? + .vshard() + .as_u32(); + let mut reads = ShardReadLog::new(); + + // 1. The owner's legs. + let Some(exported) = export_legs( + state, + ExportScope { + tenant_id, + database_id, + vshard, + linearizable, + }, + plan, + &mut reads, + ) + .await? + else { + reads.publish(state, tenant_id, database_id, Some(qualified)); + return Ok(not_found_response()); + }; + let legs = exported.legs; + + // 2. Name the hits' graph nodes. + let mut hit_surrogates: Vec = Vec::new(); + for hit in &legs.vector { + if let Some(raw) = hit.surrogate + && !hit_surrogates.contains(&raw) + { + hit_surrogates.push(raw); + } + } + let seed_bindings = if hit_surrogates.is_empty() { + Default::default() + } else { + bindings( + state, + plan, + BindingsRequest { + tenant_id, + database_id, + collection: qualified.clone(), + surrogates: hit_surrogates, + names: Vec::new(), + linearizable, + }, + ) + .await? + }; + let mut vector_scores: HashMap = HashMap::new(); + let mut seeds: Vec = Vec::new(); + for (rank, hit) in legs.vector.iter().enumerate() { + let key = match hit.surrogate { + Some(raw) => match seed_bindings.by_surrogate.get(&raw) { + Some(name) => { + seeds.push(name.clone()); + RowIdentity::from_user_key(name.clone()) + } + None => RowIdentity::for_surrogate(Surrogate::new(raw)), + }, + None => RowIdentity::from_user_key(format!("__unbound_{}", hit.entry_id)), + }; + vector_scores.insert(key, (rank, hit.distance)); + } + + // 3. Walk the collection's edges from the seeds. A collection no + // partition holds edges of reaches nothing, not even its seeds. + let walk = if seed_bindings.knows_collection { + walk_from_seeds( + state, + scope, + WalkSpec { + collection: &qualified, + edge_label: edge_label.as_deref(), + direction: *direction, + max_depth: *expansion_depth, + options, + }, + seeds, + &mut reads, + ) + .await? + } else { + Walk::default() + }; + + // 4. Which reached nodes carry a surrogate. + let unaddressable = if walk.order.is_empty() { + 0 + } else { + let reached = bindings( + state, + plan, + BindingsRequest { + tenant_id, + database_id, + collection: qualified.clone(), + surrogates: Vec::new(), + names: walk.order.clone(), + linearizable, + }, + ) + .await?; + walk.order + .iter() + .filter(|name| !reached.bound_names.contains(name.as_str())) + .count() + }; + + // 5. Fuse, exactly as one core does. + let graph_expanded_count = walk.order.len(); + let graph_list = graph_nodes_to_ranked_results(walk.order, &walk.distances); + let vector_list = vector_ranked_list(&vector_scores); + let (fused, op_name) = match (bm25_query, bm25_field, rrf_k_triple) { + (Some(_), Some(_), Some((vector_k, text_k, graph_k))) => { + let text_hits: Vec<(Surrogate, f32)> = legs + .text + .iter() + .map(|hit| (Surrogate::new(hit.surrogate), hit.score)) + .collect(); + let text_list = text_ranked_list(&text_hits); + ( + reciprocal_rank_fusion_weighted( + &[vector_list, text_list, graph_list], + &[*vector_k, *text_k, *graph_k], + *final_top_k, + ), + "graph rag fusion triple", + ) + } + _ => { + let (vector_k, graph_k) = *rrf_k; + ( + reciprocal_rank_fusion_weighted( + &[vector_list, graph_list], + &[vector_k, graph_k], + *final_top_k, + ), + "graph rag fusion", + ) + } + }; + let body = rag_response_body( + &RagResponseParams { + fused: &fused, + vector_scores: &vector_scores, + hop_distances: &walk.distances, + vector_candidate_count: legs.vector.len(), + graph_expanded_count, + bfs_truncated: walk.truncated, + graph_unaddressable: unaddressable, + op_name, + }, + exported.watermark_lsn.as_u64(), + ); + let payload = zerompk::to_msgpack_vec(&body).map_err(|e| crate::Error::Codec { + detail: format!("{op_name} encode: {e}"), + })?; + // Every vShard the fusion read joins the transaction read-set. + reads.publish(state, tenant_id, database_id, Some(qualified)); + Ok(ok_payload_response(Payload::from_vec(payload))) +} + +/// What a walk reads: the collection's edges under a label and direction. +struct WalkSpec<'a> { + collection: &'a str, + edge_label: Option<&'a str>, + direction: Direction, + max_depth: usize, + options: &'a GraphTraversalOptions, +} + +/// Every node a walk reached, in discovery order, at its hop distance. +#[derive(Default)] +struct Walk { + order: Vec, + distances: HashMap, + truncated: bool, +} + +/// Breadth-first walk from `seeds`, as one core's collection-scoped +/// expansion runs it (`CsrIndex::traverse_surrogates_in_collection`): seeds +/// at distance 0, each level's new nodes admitted in node-name order one hop +/// past the frontier, and the walk stops, marked truncated, when a new node +/// will pass the visit cap. The admitted set under the cap is therefore the +/// single core's. +async fn walk_from_seeds( + state: &SharedState, + scope: RagScope, + spec: WalkSpec<'_>, + seeds: Vec, + reads: &mut ShardReadLog, +) -> crate::Result { + let WalkSpec { + collection, + edge_label, + direction, + max_depth, + options, + } = spec; + let query = &state.tuning.query; + let budget = query.bfs_memory_budget_bytes / query.bfs_bytes_per_node.max(1); + let cap = options.max_visited.min(budget); + // Each hop returns every neighbor of the frontier; the cap is applied + // here, in name order, as one core applies it. + let whole_hop = crate::control::server::graph_dispatch::bfs::whole_hop(); + + let mut walk = Walk::default(); + let mut visited: HashSet = HashSet::new(); + let mut frontier: Vec = Vec::new(); + for seed in seeds { + if visited.insert(seed.clone()) { + walk.distances.insert(seed.clone(), 0); + walk.order.push(seed.clone()); + frontier.push(seed); + } + } + 'walk: for depth in 0..max_depth { + if frontier.is_empty() { + break; + } + let hop = execute_neighbor_hop( + state, + scope.tenant_id, + scope.database_id, + NeighborHopParams { + collection: Some(collection), + frontier: &frontier, + edge_label, + direction, + options: &whole_hop, + discovered_so_far: visited.len(), + linearizable: scope.linearizable, + }, + ) + .await?; + reads.merge(hop.reads); + let mut candidates: Vec = hop + .local_triples + .into_iter() + .map(|(_src, _label, dst)| dst) + .filter(|dst| !visited.contains(dst)) + .collect(); + candidates.sort(); + candidates.dedup(); + let mut next: Vec = Vec::new(); + for dst in candidates { + if visited.len() >= cap { + walk.truncated = true; + break 'walk; + } + visited.insert(dst.clone()); + walk.distances.insert(dst.clone(), depth + 1); + walk.order.push(dst.clone()); + next.push(dst); + } + frontier = next; + } + Ok(walk) +} diff --git a/nodedb/src/control/server/graph_dispatch/rag_fusion/mod.rs b/nodedb/src/control/server/graph_dispatch/rag_fusion/mod.rs new file mode 100644 index 000000000..a1922fb40 --- /dev/null +++ b/nodedb/src/control/server/graph_dispatch/rag_fusion/mod.rs @@ -0,0 +1,10 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! GraphRAG fusion in a cluster, run in stages from a coordinator. +//! +//! See `coord` for the stages and `stages` for the dispatches they make. + +pub mod coord; +pub mod stages; + +pub use coord::serve_rag_plan; diff --git a/nodedb/src/control/server/graph_dispatch/rag_fusion/stages.rs b/nodedb/src/control/server/graph_dispatch/rag_fusion/stages.rs new file mode 100644 index 000000000..30bc7018a --- /dev/null +++ b/nodedb/src/control/server/graph_dispatch/rag_fusion/stages.rs @@ -0,0 +1,228 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The dispatches a cluster RAG fusion makes: the owner's leg export and the +//! graph owners' bindings. + +use std::collections::{HashMap, HashSet}; + +use nodedb_physical::physical_plan::{GraphOp, RagBindingRow, RagLegs, RagStage}; + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::gateway::TaskRoute; +use crate::control::gateway::dispatcher::{ + DispatchRouteParams, dispatch_route, statement_deadline_ms, +}; +use crate::control::gateway::version_set::GatewayVersionSet; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, Lsn, TenantId, TraceId}; + +use super::super::cluster_resolve::gateway_shared; +use super::super::shard_reads::ShardReadLog; +use super::super::whole_graph::scatter_to_graph_owners; +use crate::control::gateway::live_leaders::resolve_live_decision; + +/// `plan`, a RAG fusion, set to run `stage`. +pub(super) fn with_stage(plan: &PhysicalPlan, stage: RagStage) -> PhysicalPlan { + let mut plan = plan.clone(); + if let PhysicalPlan::Graph(GraphOp::RagFusion { stage: slot, .. }) = &mut plan { + *slot = stage; + } + plan +} + +/// The owner's raw legs and the watermark it served them at. +pub(super) struct ExportedLegs { + pub legs: RagLegs, + pub watermark_lsn: Lsn, +} + +/// Where and how the export runs. +pub(super) struct ExportScope { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + /// The collection's own vShard, which holds its vector and text indexes. + pub vshard: u32, + pub linearizable: bool, +} + +/// Run the `ExportLegs` stage of `plan` on the leader of the collection's +/// vShard. `Ok(None)` when no vector index of the collection exists there, +/// the case a single core answers with `NotFound`. The vShard joins `reads`. +pub(super) async fn export_legs( + state: &SharedState, + scope: ExportScope, + plan: &PhysicalPlan, + reads: &mut ShardReadLog, +) -> crate::Result> { + let ExportScope { + tenant_id, + database_id, + vshard, + linearizable, + } = scope; + let shared = gateway_shared(state)?; + let decision = resolve_live_decision(state, vshard); + let served_by = match decision { + crate::control::gateway::RouteDecision::Remote { node_id, .. } => node_id, + _ => state.node_id, + }; + let route = TaskRoute { + plan: with_stage(plan, RagStage::ExportLegs), + decision, + vshard_id: vshard, + }; + let outcome = dispatch_route(DispatchRouteParams { + route, + shared: &shared, + tenant_id, + database_id, + trace_id: TraceId::ZERO, + deadline_ms: statement_deadline_ms(state), + version_set: &GatewayVersionSet::from_pairs(Vec::new()), + txn_id: None, + linearizable, + }) + .await?; + reads.note_leg([vshard], &outcome.shard_watermarks, served_by); + let watermark_lsn = outcome + .shard_watermarks + .iter() + .map(|(_, lsn)| *lsn) + .max() + .unwrap_or(Lsn::ZERO); + if outcome.not_found { + return Ok(None); + } + // One core holds the index and answers a one-element array. A node that + // fans the plan over its cores drops the other cores' `NotFound`. + let mut answers: Vec = Vec::new(); + for payload in &outcome.payloads { + if payload.is_empty() { + continue; + } + let mut legs: Vec = + zerompk::from_msgpack(payload).map_err(|e| crate::Error::Codec { + detail: format!("rag fusion legs decode: {e}"), + })?; + answers.append(&mut legs); + } + let mut answers = answers.into_iter(); + match (answers.next(), answers.next()) { + (None, _) => Ok(None), + (Some(legs), None) => Ok(Some(ExportedLegs { + legs, + watermark_lsn, + })), + (Some(_), Some(_)) => Err(crate::Error::Internal { + detail: format!( + "rag fusion: more than one core of vShard {vshard} answered the vector leg" + ), + }), + } +} + +/// What the graph owners answered for a `Bindings` stage. +#[derive(Debug, Default)] +pub(super) struct Bindings { + /// The graph node each requested surrogate names. + pub by_surrogate: HashMap, + /// The requested names that carry a surrogate. + pub bound_names: HashSet, + /// Some partition holds edges of the collection. + pub knows_collection: bool, +} + +impl Bindings { + fn absorb(&mut self, rows: Vec) { + for row in rows { + match row { + RagBindingRow::Bound { name, surrogate } => { + self.by_surrogate.insert(surrogate, name.clone()); + self.bound_names.insert(name); + } + RagBindingRow::KnowsCollection => self.knows_collection = true, + } + } + } +} + +/// The requests of one `Bindings` stage. +pub(super) struct BindingsRequest { + pub tenant_id: TenantId, + pub database_id: DatabaseId, + /// The database-qualified collection, for the transaction read-set. + pub collection: String, + pub surrogates: Vec, + pub names: Vec, + pub linearizable: bool, +} + +/// Run the `Bindings` stage of `plan` on every graph owner and merge their +/// answers. +pub(super) async fn bindings( + state: &SharedState, + plan: &PhysicalPlan, + request: BindingsRequest, +) -> crate::Result { + let BindingsRequest { + tenant_id, + database_id, + collection, + surrogates, + names, + linearizable, + } = request; + let plan = with_stage(plan, RagStage::Bindings { surrogates, names }); + let payloads = scatter_to_graph_owners( + state, + tenant_id, + database_id, + plan, + linearizable, + Some(collection), + ) + .await?; + let mut bindings = Bindings::default(); + for payload in payloads { + if payload.is_empty() { + continue; + } + let rows: Vec = + zerompk::from_msgpack(payload.as_ref()).map_err(|e| crate::Error::Codec { + detail: format!("rag fusion bindings decode: {e}"), + })?; + bindings.absorb(rows); + } + Ok(bindings) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn bindings_merge_every_owner_answer() { + let mut bindings = Bindings::default(); + bindings.absorb(vec![RagBindingRow::Bound { + name: "alice".into(), + surrogate: 7, + }]); + bindings.absorb(vec![ + RagBindingRow::KnowsCollection, + RagBindingRow::Bound { + name: "bob".into(), + surrogate: 9, + }, + ]); + assert_eq!( + bindings.by_surrogate.get(&7).map(String::as_str), + Some("alice") + ); + assert_eq!( + bindings.by_surrogate.get(&9).map(String::as_str), + Some("bob") + ); + assert!(bindings.bound_names.contains("alice")); + assert!(bindings.knows_collection); + } +} diff --git a/nodedb/src/control/server/graph_dispatch/read_groups.rs b/nodedb/src/control/server/graph_dispatch/read_groups.rs new file mode 100644 index 000000000..d95e6ff45 --- /dev/null +++ b/nodedb/src/control/server/graph_dispatch/read_groups.rs @@ -0,0 +1,214 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The data groups a graph plan reads on this node's cores. +//! +//! A graph edge lives on the key vShard of each endpoint, and a node document +//! lives on its collection's home vShard (`types/record_home.rs`). A plan's +//! read set follows from what it expands: +//! +//! - A one-hop lookup (`Neighbors`, `NeighborsMulti`, `TemporalNeighbors`) +//! reads the edges of the nodes it names. Its set is the key vShard of each +//! named node, plus the collection home for node documents and RLS checks. +//! - Every other graph read walks the local CSR from node to node. A MATCH +//! continuation keeps expanding locally until a node has no local edges +//! (`engine/graph/pattern/executor/overlay_expand.rs`). Hop, Path and +//! Subgraph traverse to any depth. Algorithms, BSP supersteps and stats scan +//! the whole partition. Such a plan can read any key vShard held here, so its +//! set is every group this node replicates. +//! - A gathered algorithm (`AlgoStage::Gathered`) runs over the edges its plan +//! carries and reads no group. +//! +//! Only groups this node replicates count. The local cores hold no rows of any +//! other group. + +use nodedb_physical::physical_plan::{AlgoStage, GraphOp}; +use nodedb_types::{CollectionKey, QualifiedCollection}; + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::cluster::linearizable_read::{ + confirm_linearizable_read, groups_hosted_here, hosted_groups_of_vshards, + statement_read_deadline, +}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, VShardId}; + +/// The groups `plan` reads when it runs on this node's cores. +pub fn graph_read_groups( + state: &SharedState, + database_id: DatabaseId, + plan: &PhysicalPlan, +) -> crate::Result> { + let PhysicalPlan::Graph(op) = plan else { + return Ok(groups_hosted_here(state)); + }; + match op { + GraphOp::Neighbors { + collection, + node_id, + .. + } => keyed_groups( + state, + database_id, + collection.as_ref(), + std::iter::once(node_id.as_str()), + ), + GraphOp::TemporalNeighbors { + collection, + node_id, + .. + } => keyed_groups( + state, + database_id, + Some(collection), + std::iter::once(node_id.as_str()), + ), + GraphOp::NeighborsMulti { + collection, + node_ids, + .. + } => keyed_groups( + state, + database_id, + collection.as_ref(), + node_ids.iter().map(String::as_str), + ), + // A presence read reads the documents on the vShard it names. + GraphOp::NodePresenceRead { vshard, .. } => { + hosted_groups_of_vshards(state, std::iter::once(*vshard)) + } + // A gathered algorithm runs over edges the plan carries and reads no + // group of this node. + GraphOp::Algo { + stage: AlgoStage::Gathered { .. }, + .. + } => Ok(Vec::new()), + _ => Ok(groups_hosted_here(state)), + } +} + +/// Confirm the groups `plan` reads on this node's cores, within the running +/// statement's budget. +pub async fn confirm_graph_read( + state: &SharedState, + database_id: DatabaseId, + plan: &PhysicalPlan, +) -> crate::Result<()> { + let groups = graph_read_groups(state, database_id, plan)?; + confirm_linearizable_read(state, &groups, statement_read_deadline(state)).await +} + +fn keyed_groups<'a>( + state: &SharedState, + database_id: DatabaseId, + collection: Option<&QualifiedCollection>, + node_keys: impl Iterator, +) -> crate::Result> { + let home = match collection { + Some(collection) => Some( + CollectionKey::from_qualified(database_id, collection)? + .vshard() + .as_u32(), + ), + None => None, + }; + let key_vshards = node_keys.map(|key| VShardId::from_key(key.as_bytes()).as_u32()); + hosted_groups_of_vshards(state, home.into_iter().chain(key_vshards)) +} + +#[cfg(test)] +mod tests { + use std::sync::{Arc, RwLock}; + + use nodedb_cluster::RoutingTable; + + use super::*; + use crate::bridge::dispatch::Dispatcher; + use crate::engine::graph::edge_store::Direction; + use crate::wal::WalManager; + + const THIS_NODE: u64 = 1; + const OTHER_NODE: u64 = 2; + + /// A node that replicates every group except `foreign_group`, which only + /// `OTHER_NODE` holds. + fn state_with_routing(foreign_group: u64) -> (Arc, tempfile::TempDir) { + let directory = tempfile::tempdir().expect("temporary WAL directory"); + let wal = Arc::new( + WalManager::open_for_testing(&directory.path().join("graph.wal")).expect("test WAL"), + ); + let (dispatcher, _sides) = Dispatcher::new(1, 8); + let mut state = SharedState::new(dispatcher, wal).expect("shared state"); + let mut routing = RoutingTable::uniform(4, &[THIS_NODE, OTHER_NODE], 2); + for group_id in routing.group_ids() { + let members = if group_id == foreign_group { + vec![OTHER_NODE] + } else { + vec![THIS_NODE, OTHER_NODE] + }; + routing.set_group_members(group_id, members); + } + let shared = Arc::get_mut(&mut state).expect("sole owner of fresh state"); + shared.node_id = THIS_NODE; + shared.cluster_routing = Some(Arc::new(RwLock::new(routing))); + (state, directory) + } + + fn group_of_key(state: &SharedState, key: &str) -> u64 { + let routing = state + .cluster_routing + .as_ref() + .expect("routing") + .read() + .expect("lock"); + routing + .group_for_vshard(VShardId::from_key(key.as_bytes()).as_u32()) + .expect("group of key") + } + + fn neighbors_multi(node_ids: &[&str]) -> PhysicalPlan { + PhysicalPlan::Graph(GraphOp::NeighborsMulti { + collection: None, + node_ids: node_ids.iter().map(|id| (*id).to_owned()).collect(), + edge_label: None, + direction: Direction::Out, + max_results: 0, + rls_filters: Vec::new(), + }) + } + + #[test] + fn a_one_hop_lookup_reads_only_the_key_groups_of_its_nodes() { + let (state, _dir) = state_with_routing(u64::MAX); + let key_group = group_of_key(&state, "alice"); + let groups = graph_read_groups(&state, DatabaseId::DEFAULT, &neighbors_multi(&["alice"])) + .expect("read groups"); + assert_eq!(groups, vec![key_group]); + assert!(groups_hosted_here(&state).len() > 1); + } + + #[test] + fn a_key_group_held_elsewhere_is_left_out() { + let (probe, _dir) = state_with_routing(u64::MAX); + let key_group = group_of_key(&probe, "alice"); + let (state, _dir) = state_with_routing(key_group); + let groups = graph_read_groups(&state, DatabaseId::DEFAULT, &neighbors_multi(&["alice"])) + .expect("read groups"); + assert!(groups.is_empty()); + } + + #[test] + fn a_walking_plan_reads_every_group_held_here() { + let (state, _dir) = state_with_routing(u64::MAX); + let plan = PhysicalPlan::Graph(GraphOp::Match { + query: Vec::new(), + frontier_bitmap: None, + cluster_mode: true, + }); + let mut groups = + graph_read_groups(&state, DatabaseId::DEFAULT, &plan).expect("read groups"); + let mut hosted = groups_hosted_here(&state); + groups.sort_unstable(); + hosted.sort_unstable(); + assert_eq!(groups, hosted); + } +} diff --git a/nodedb/src/control/server/graph_dispatch/run_cut.rs b/nodedb/src/control/server/graph_dispatch/run_cut.rs new file mode 100644 index 000000000..76a4595bf --- /dev/null +++ b/nodedb/src/control/server/graph_dispatch/run_cut.rs @@ -0,0 +1,62 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The read cut a distributed graph run (BSP PageRank, WCC) reads at. +//! +//! The coordinator names a Calvin cut marker by a fresh watermark and sends +//! it in the run's first dispatch. Every owner node proposes and waits for +//! the marker, then reads at the epoch instant the marker's sequencer place +//! fixes (`exchange::all_cores::read_cut`). Every node applies the same +//! sequencer log, so every node answers the same cut. The coordinator checks +//! that, and sends the cut in every later dispatch. + +use crate::control::state::SharedState; + +/// A fresh cut marker watermark for one run, from this node's HLC. +pub(super) fn new_cut_marker(state: &SharedState) -> u64 { + state.hlc_clock.now().wall_ns.max(1) +} + +/// The one read cut every owner node answered. A node that answered no cut, +/// or a different one, is an error: the run will read two graphs. +pub(super) fn agreed_cut( + answers: impl IntoIterator)>, +) -> crate::Result { + let mut agreed: Option<(u64, i64)> = None; + for (node_id, cut) in answers { + let Some(cut) = cut else { + return Err(crate::Error::Internal { + detail: format!("graph read cut: node {node_id} answered no read cut"), + }); + }; + match agreed { + None => agreed = Some((node_id, cut)), + Some((first, first_cut)) if first_cut != cut => { + return Err(crate::Error::Internal { + detail: format!( + "graph read cut: node {first} read at {first_cut} and node {node_id} at \ + {cut}; a run reads one graph. Retry the query" + ), + }); + } + Some(_) => {} + } + } + agreed + .map(|(_, cut)| cut) + .ok_or_else(|| crate::Error::Internal { + detail: "graph read cut: no owner node answered".into(), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn every_node_must_answer_the_same_cut() { + assert_eq!(agreed_cut([(1, Some(5)), (2, Some(5))]).unwrap(), 5); + assert!(agreed_cut([(1, Some(5)), (2, Some(6))]).is_err()); + assert!(agreed_cut([(1, Some(5)), (2, None)]).is_err()); + assert!(agreed_cut(Vec::new()).is_err()); + } +} diff --git a/nodedb/src/control/server/graph_dispatch/shard_reads.rs b/nodedb/src/control/server/graph_dispatch/shard_reads.rs new file mode 100644 index 000000000..e8fd86ed6 --- /dev/null +++ b/nodedb/src/control/server/graph_dispatch/shard_reads.rs @@ -0,0 +1,191 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The vShards a cross-shard graph read observed, for the transaction +//! read-set. +//! +//! A coordinator notes each vShard its legs read, at the watermark the +//! serving node reported, then publishes the log once the read finishes +//! (`session::graph_reads`). The request's protocol records the published +//! reads into the transaction read-set, so commit validation checks every +//! vShard the read depended on. A read that runs on one node only (no +//! cluster) publishes nothing: its cores' watermarks sit in this node's WAL, +//! which single-shard SI already compares against. + +use std::collections::BTreeMap; + +use crate::control::server::shared::session::graph_reads::{ + GraphShardReads, ShardObservation, note, +}; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, Lsn, TenantId, VShardId}; + +/// Every vShard a graph read observed, each at its earliest served watermark +/// and with the node that served it. +#[derive(Debug, Default)] +pub(crate) struct ShardReadLog { + served: BTreeMap, +} + +/// One vShard's observation: the earliest watermark, and the node whose WAL +/// numbers it (`0` once two nodes served the vShard). +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct Served { + pub(crate) watermark: Lsn, + pub(crate) node: u64, +} + +impl ShardReadLog { + pub(crate) fn new() -> Self { + Self::default() + } + + /// Note that `node` served `vshards` at `watermark`. A vShard read twice + /// keeps its earlier watermark: the first observation is the one the + /// result rests on. A vShard two nodes served has no one node whose + /// versions compare with the read, so its node becomes `0`. + pub(crate) fn note( + &mut self, + vshards: impl IntoIterator, + watermark: Lsn, + node: u64, + ) { + for vshard in vshards { + self.served + .entry(vshard) + .and_modify(|seen| { + seen.watermark = seen.watermark.min(watermark); + if seen.node != node { + seen.node = 0; + } + }) + .or_insert(Served { watermark, node }); + } + } + + /// Note one leg's response from `node`: `vshards` read at the highest + /// watermark the leg reported. + pub(crate) fn note_leg( + &mut self, + vshards: impl IntoIterator, + watermarks: &[(VShardId, Lsn)], + node: u64, + ) { + let watermark = watermarks + .iter() + .map(|(_, lsn)| *lsn) + .max() + .unwrap_or(Lsn::ZERO); + self.note(vshards, watermark, node); + } + + /// Fold `other` into this log. + pub(crate) fn merge(&mut self, other: ShardReadLog) { + for (vshard, served) in other.served { + self.note([vshard], served.watermark, served.node); + } + } + + /// Hand the log to the running request for its transaction read-set. + /// `collection` is the database-qualified collection the read scoped, or + /// `None` when it walked every collection. Without a cluster the log is + /// dropped (see the module doc). + pub(crate) fn publish( + self, + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collection: Option, + ) { + if state.cluster_routing.is_none() { + return; + } + note(GraphShardReads { + tenant_id, + database_id, + collection, + shards: self + .served + .into_iter() + .map(|(vshard, served)| ShardObservation { + vshard: VShardId::new(vshard), + watermark: served.watermark, + node: served.node, + }) + .collect(), + }); + } + + #[cfg(test)] + pub(crate) fn served(&self) -> &BTreeMap { + &self.served + } +} + +/// The stored, database-qualified name of `bare` in `database_id`. +pub(crate) fn qualified(database_id: DatabaseId, bare: &str) -> String { + nodedb_types::QualifiedCollection::new(database_id, bare) + .as_str() + .to_owned() +} + +/// The key vShard of each node in `nodes`. +pub(crate) fn key_vshards<'a>(nodes: impl IntoIterator) -> Vec { + nodes + .into_iter() + .map(|node| VShardId::from_key(node.as_bytes()).as_u32()) + .collect() +} + +#[cfg(test)] +mod tests { + use super::*; + + fn watermark(log: &ShardReadLog, vshard: u32) -> Option { + log.served().get(&vshard).map(|served| served.watermark) + } + + #[test] + fn a_vshard_read_twice_keeps_its_earliest_watermark() { + let mut log = ShardReadLog::new(); + log.note([3, 4], Lsn::new(20), 1); + log.note([3], Lsn::new(10), 1); + log.note([4], Lsn::new(30), 1); + assert_eq!(watermark(&log, 3), Some(Lsn::new(10))); + assert_eq!(watermark(&log, 4), Some(Lsn::new(20))); + assert_eq!(log.served().get(&3).map(|s| s.node), Some(1)); + } + + #[test] + fn a_vshard_two_nodes_served_has_no_node() { + let mut log = ShardReadLog::new(); + log.note([3], Lsn::new(20), 1); + log.note([3], Lsn::new(30), 2); + assert_eq!(log.served().get(&3).map(|s| s.node), Some(0)); + } + + #[test] + fn a_leg_is_noted_at_its_highest_reported_watermark() { + let mut log = ShardReadLog::new(); + log.note_leg( + [7], + &[ + (VShardId::new(1), Lsn::new(5)), + (VShardId::new(2), Lsn::new(9)), + ], + 2, + ); + assert_eq!(watermark(&log, 7), Some(Lsn::new(9))); + } + + #[test] + fn merging_keeps_the_earliest_watermark_per_vshard() { + let mut left = ShardReadLog::new(); + left.note([1], Lsn::new(8), 1); + let mut right = ShardReadLog::new(); + right.note([1], Lsn::new(4), 1); + right.note([2], Lsn::new(6), 1); + left.merge(right); + assert_eq!(watermark(&left, 1), Some(Lsn::new(4))); + assert_eq!(watermark(&left, 2), Some(Lsn::new(6))); + } +} diff --git a/nodedb/src/control/server/graph_dispatch/shortest_path.rs b/nodedb/src/control/server/graph_dispatch/shortest_path.rs index ae4feaa07..8527b9a63 100644 --- a/nodedb/src/control/server/graph_dispatch/shortest_path.rs +++ b/nodedb/src/control/server/graph_dispatch/shortest_path.rs @@ -1,44 +1,63 @@ // SPDX-License-Identifier: BUSL-1.1 -//! `cross_core_shortest_path` — mirrors `cross_core_bfs` but records -//! parent pointers so `GRAPH PATH FROM 'src' TO 'dst'` can reconstruct -//! an ordered path across every topology (single core, single-node -//! multi-core, clustered). +//! `cross_core_shortest_path` — the bidirectional search of one core's +//! `CsrIndex::shortest_path`, run hop by hop across every topology (single +//! core, single-node multi-core, clustered), so `GRAPH PATH FROM 'src' TO +//! 'dst'` answers the same path everywhere. +//! +//! Each step expands one forward level (outgoing edges of the forward +//! frontier) and then one backward level (incoming edges of the backward +//! frontier) through [`super::hop::execute_neighbor_hop`]: every frontier node +//! expands at the node that owns its key vShard, and every crossed edge comes +//! back as a `(frontier node, label, neighbour)` triple. A level relaxes its +//! edges in `(neighbour, frontier node)` name order: a new node's parent is +//! the smallest-named frontier node reaching it, and the search stops at the +//! first node both sides reached. The visit cap is checked before each step. +//! One core does all of this in the same order, so a capped search answers +//! the same path here as there. -use std::collections::{HashMap, HashSet}; +use std::collections::HashMap; +use std::collections::hash_map::Entry; -use sonic_rs; - -use crate::bridge::envelope::{PhysicalPlan, Response}; -use crate::control::scatter_gather; +use crate::bridge::envelope::Response; use crate::control::state::SharedState; +use crate::engine::graph::edge_store::Direction; use crate::engine::graph::traversal_options::GraphTraversalOptions; -use crate::types::{DatabaseId, TenantId, TraceId}; -use nodedb_physical::physical_plan::GraphOp; +use crate::types::{DatabaseId, TenantId}; +use super::bfs::{walk_visit_cap, whole_hop}; use super::helpers::{encode_path, ok_response}; +use super::hop::{NeighborHopParams, execute_neighbor_hop}; +use super::shard_reads::ShardReadLog; -/// Cross-core / cross-shard shortest-path orchestration. -/// -/// Walks the same hop-by-hop BFS as `cross_core_bfs` but records a -/// `parent` pointer for every newly-discovered node. When `dst` is -/// reached, the path is reconstructed by walking parents back to -/// `src`. Returns a JSON array `[src, hop_1, ..., dst]`, or an empty -/// array when `dst` is unreachable within `max_depth` hops. /// Parameters for [`cross_core_shortest_path`]. pub struct CrossCoreShortestPathParams { pub tenant_id: TenantId, pub database_id: DatabaseId, - /// Collection whose edges the path walks. Required: `GRAPH PATH` names one - /// so it can be authorized, and the cross-shard hop re-issues the walk as - /// SQL on the owning node, which needs the scope to plan the same edges. - pub collection: String, + /// Database-qualified collection whose edges the path walks, or `None` to + /// walk the edges of every collection, as a single node's Data Plane does + /// for a path plan with no collection. + pub collection: Option, pub src: String, pub dst: String, pub edge_label: Option, pub max_depth: usize, + /// The plan's traversal options. The visit cap is + /// `options.max_visited`, bounded by this node's graph tuning. + pub options: GraphTraversalOptions, + /// Each node that expands part of the walk confirms its groups first. + pub linearizable: bool, } +/// Node → the node it was reached from. An endpoint maps to itself. +type Parents = HashMap; + +/// Cross-core / cross-shard shortest-path orchestration. +/// +/// Returns a JSON array `[src, hop_1, ..., dst]`, `[src]` when `src == dst`, +/// or an empty array when no path is found or either endpoint is absent from +/// the graph. `GRAPH PATH` is directed: the +/// forward side follows outgoing edges, the backward side incoming ones. pub async fn cross_core_shortest_path( shared: &SharedState, params: CrossCoreShortestPathParams, @@ -51,159 +70,207 @@ pub async fn cross_core_shortest_path( dst, edge_label, max_depth, + options, + linearizable, } = params; - let options = GraphTraversalOptions::default(); - let cluster_mode = shared.cluster_routing.is_some(); - // Path semantics only make sense over outgoing edges — the - // docs-advertised `GRAPH PATH FROM 'a' TO 'b'` is directed. - let direction = crate::engine::graph::edge_store::Direction::Out; - - // Empty path when src == dst; matches the natural `[src]` case. + let whole = whole_hop(); + // One core finds no path when either endpoint is absent from its graph, + // before it compares them. The presence read spans every collection, so it + // joins the read-set unscoped. + let mut presence_reads = ShardReadLog::new(); + let present = graph_nodes_present( + shared, + PresenceScope { + tenant_id, + database_id, + options: &whole, + linearizable, + }, + &[src.clone(), dst.clone()], + &mut presence_reads, + ) + .await?; + presence_reads.publish(shared, tenant_id, database_id, None); + if !present.contains(&src) || !present.contains(&dst) { + return Ok(ok_response(encode_path::(&[])?)); + } if src == dst { - let payload = encode_path(&[src])?; - return Ok(ok_response(payload)); + return Ok(ok_response(encode_path(&[src])?)); } + let mut reads = ShardReadLog::new(); + let cap = walk_visit_cap(shared, &options); + let mut fwd: Parents = HashMap::from([(src.clone(), src.clone())]); + let mut bwd: Parents = HashMap::from([(dst.clone(), dst.clone())]); + let mut fwd_frontier = vec![src]; + let mut bwd_frontier = vec![dst]; + let mut path: Vec = Vec::new(); - let mut parent: HashMap = HashMap::new(); - let mut visited: HashSet = HashSet::new(); - visited.insert(src.clone()); - let mut frontier: Vec = vec![src.clone()]; - - for depth in 0..max_depth { - if frontier.is_empty() { + for _depth in 0..max_depth { + if fwd.len() + bwd.len() >= cap { break; } - - // Local hop: single batched broadcast carrying the whole - // frontier. The Data Plane returns `(src, label, node)` triples - // so we can still record parent pointers. - let mut discoveries: Vec<(String, String)> = Vec::new(); - // Cap the handler's allocation to the remaining `max_visited` - // budget so a wide hop can't blow out memory mid-BFS. - let remaining_budget = options - .max_visited - .saturating_sub(visited.len()) - .min(u32::MAX as usize) as u32; - let plan = PhysicalPlan::Graph(GraphOp::NeighborsMulti { - collection: Some(nodedb_types::QualifiedCollection::new( - database_id, - &collection, - )), - node_ids: frontier.clone(), - edge_label: edge_label.clone(), - direction, - max_results: remaining_budget, - rls_filters: Vec::new(), - }); - let resp = crate::control::server::broadcast::broadcast_to_all_cores( - shared, - tenant_id, - database_id, - plan, - TraceId::ZERO, - ) - .await?; - if !resp.payload.is_empty() { - let json_text = - crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); - if let Ok(arr) = sonic_rs::from_str::>(&json_text) { - for item in arr { - let src_node = item.get("src").and_then(|v| v.as_str()); - let nb = item.get("node").and_then(|v| v.as_str()); - if let (Some(s), Some(n)) = (src_node, nb) { - discoveries.push((s.to_string(), n.to_string())); - } - } - } - } - - // Cluster mode: merge cross-shard neighbor results. The - // cross-shard helper only returns neighbor ids (no parent), - // so we attribute them to the first frontier node on the - // same shard — sufficient for shortest-path correctness - // because any path through that shard is equally shortest. - let merged: Vec<(String, String)> = if cluster_mode { - let local_ids: Vec = discoveries.iter().map(|(_, n)| n.clone()).collect(); - let (local_nodes, envelope) = { - let routing = shared - .cluster_routing - .as_ref() - .expect("cluster_routing checked above"); - let rt = routing.read().unwrap_or_else(|p| p.into_inner()); - scatter_gather::partition_local_remote(&local_ids, shared.node_id, &rt) + for forward in [true, false] { + let (frontier, direction) = if forward { + (&fwd_frontier, Direction::Out) + } else { + (&bwd_frontier, Direction::In) }; - if envelope.is_empty() { - discoveries + let triples = if frontier.is_empty() { + Vec::new() } else { - let remaining = max_depth.saturating_sub(depth + 1); - let (remote_hits, _meta) = scatter_gather::coordinate_cross_shard_hop( + let hop = execute_neighbor_hop( shared, tenant_id, - scatter_gather::CrossShardHopParams { - local_nodes, - envelope, - options: &options, - collection: &collection, + database_id, + NeighborHopParams { + collection: collection.as_deref(), + frontier, edge_label: edge_label.as_deref(), direction, - remaining_depth: remaining, - database_id, + options: &whole, + discovered_so_far: fwd.len() + bwd.len(), + linearizable, }, ) .await?; - let mut out = discoveries; - if let Some(attrib_parent) = frontier.first().cloned() { - for n in remote_hits { - out.push((attrib_parent.clone(), n)); - } - } - out - } - } else { - discoveries - }; - - // Record parents and build next frontier. Early-exit the - // moment `dst` is seen so we don't keep expanding. - let mut next_frontier: Vec = Vec::new(); - for (from, to) in merged { - if !visited.insert(to.clone()) { - continue; + reads.merge(hop.reads); + hop.local_triples + }; + let (this, other) = if forward { + (&mut fwd, &bwd) + } else { + (&mut bwd, &fwd) + }; + let (next, meeting) = relax_level(triples, this, other); + if let Some(meeting) = meeting { + path = reconstruct(&meeting, &fwd, &bwd); + break; } - parent.insert(to.clone(), from); - if to == dst { - let path = reconstruct(&parent, &src, &dst); - let payload = encode_path(&path)?; - return Ok(ok_response(payload)); + if forward { + fwd_frontier = next; + } else { + bwd_frontier = next; } - next_frontier.push(to); } - frontier = next_frontier; - - if visited.len() >= options.max_visited { + if !path.is_empty() || (fwd_frontier.is_empty() && bwd_frontier.is_empty()) { break; } } - // Unreachable within max_depth: empty array, same shape the - // client sees for a successful empty result. - let payload = encode_path::(&[])?; - Ok(ok_response(payload)) + // Every vShard the walk expanded joins the transaction read-set. + reads.publish(shared, tenant_id, database_id, collection); + Ok(ok_response(encode_path(&path)?)) } -fn reconstruct(parent: &HashMap, src: &str, dst: &str) -> Vec { - let mut path: Vec = Vec::new(); - let mut cursor = dst.to_string(); - path.push(cursor.clone()); - while cursor != src { - match parent.get(&cursor) { - Some(p) => { - cursor = p.clone(); - path.push(cursor.clone()); - } - None => break, +/// The tenancy scope of a presence check. +struct PresenceScope<'a> { + tenant_id: TenantId, + database_id: DatabaseId, + options: &'a GraphTraversalOptions, + linearizable: bool, +} + +/// The nodes of `nodes` that exist in the graph: those with an edge of any +/// label, in either direction, in any collection. One core's graph holds a +/// node exactly while an edge names it, so this is the node set one core's +/// path search checks its endpoints against. +async fn graph_nodes_present( + shared: &SharedState, + scope: PresenceScope<'_>, + nodes: &[String], + reads: &mut ShardReadLog, +) -> crate::Result> { + let hop = execute_neighbor_hop( + shared, + scope.tenant_id, + scope.database_id, + NeighborHopParams { + collection: None, + frontier: nodes, + edge_label: None, + direction: Direction::Both, + options: scope.options, + discovered_so_far: 0, + linearizable: scope.linearizable, + }, + ) + .await?; + reads.merge(hop.reads); + Ok(hop + .local_triples + .into_iter() + .map(|(from, _label, _to)| from) + .collect()) +} + +/// Relax one level's `(frontier node, label, neighbour)` edges into `this` +/// side, in `(neighbour, frontier node)` name order. Returns the level's new +/// nodes, and the first neighbour the `other` side already reached. +fn relax_level( + mut triples: Vec<(String, String, String)>, + this: &mut Parents, + other: &Parents, +) -> (Vec, Option) { + triples.sort_by(|a, b| (&a.2, &a.0).cmp(&(&b.2, &b.0))); + let mut next = Vec::new(); + for (from, _label, to) in triples { + if let Entry::Vacant(slot) = this.entry(to.clone()) { + slot.insert(from); + next.push(to.clone()); } + if other.contains_key(&to) { + return (next, Some(to)); + } + } + (next, None) +} + +/// The path through `meeting`: forward parents back to the source, then +/// backward parents on to the destination. +fn reconstruct(meeting: &str, fwd: &Parents, bwd: &Parents) -> Vec { + let mut path = vec![meeting.to_string()]; + let mut cursor = meeting; + while let Some(parent) = fwd.get(cursor).filter(|p| p.as_str() != cursor) { + path.push(parent.clone()); + cursor = parent.as_str(); } path.reverse(); + let mut cursor = meeting; + while let Some(parent) = bwd.get(cursor).filter(|p| p.as_str() != cursor) { + path.push(parent.clone()); + cursor = parent.as_str(); + } path } + +#[cfg(test)] +mod tests { + use super::*; + + fn parents(pairs: &[(&str, &str)]) -> Parents { + pairs + .iter() + .map(|(child, parent)| (child.to_string(), parent.to_string())) + .collect() + } + + #[test] + fn a_path_joins_both_sides_at_the_meeting_node() { + let fwd = parents(&[("a", "a"), ("b", "a")]); + let bwd = parents(&[("d", "d"), ("c", "d"), ("b", "c")]); + assert_eq!(reconstruct("b", &fwd, &bwd), vec!["a", "b", "c", "d"]); + } + + #[test] + fn a_level_relaxes_in_name_order() { + let mut this = parents(&[("a", "a")]); + let other = parents(&[("d", "d"), ("b", "d"), ("z", "d")]); + let triples = vec![ + ("a".to_string(), "L".to_string(), "z".to_string()), + ("a".to_string(), "L".to_string(), "b".to_string()), + ]; + let (next, meeting) = relax_level(triples, &mut this, &other); + assert_eq!(meeting.as_deref(), Some("b")); + assert_eq!(next, vec!["b"]); + } +} diff --git a/nodedb/src/control/server/graph_dispatch/traverse_subgraph.rs b/nodedb/src/control/server/graph_dispatch/traverse_subgraph.rs index 0455790bf..5be90d264 100644 --- a/nodedb/src/control/server/graph_dispatch/traverse_subgraph.rs +++ b/nodedb/src/control/server/graph_dispatch/traverse_subgraph.rs @@ -26,8 +26,10 @@ use crate::engine::graph::edge_store::Direction; use crate::engine::graph::traversal_options::GraphTraversalOptions; use crate::types::{DatabaseId, TenantId}; +use super::bfs::{admit_by_name, walk_visit_cap, whole_hop}; use super::helpers::ok_response; use super::hop::{NeighborHopParams, execute_neighbor_hop}; +use super::shard_reads::ShardReadLog; /// Wire-shape JSON node entry. Field names mirror the client decoder in /// `nodedb-client/src/remote/parse.rs::parse_graph_traverse_json`. @@ -61,17 +63,24 @@ struct WireSubGraph<'a> { pub struct CrossCoreTraverseSubgraphParams<'a> { pub tenant_id: TenantId, pub database_id: DatabaseId, - /// Collection scope, or `None` for a label-only traversal. + /// Database-qualified collection scope, or `None` for a label-only + /// traversal. pub collection: Option, pub start: String, pub edge_label: Option, pub direction: Direction, pub max_depth: usize, pub options: &'a GraphTraversalOptions, + /// Each node that expands part of the walk confirms its groups first. + pub linearizable: bool, } /// BFS that returns a `{nodes,edges}` JSON subgraph for `GRAPH TRAVERSE`. /// +/// The walk runs level by level, as one core's subgraph does +/// (`CsrIndex::subgraph`): each hop records every edge of the frontier, and +/// the level's new nodes are admitted in node-name order until the visit cap. +/// /// Each hop expands every frontier node at the node that owns /// `from_key(node)` via the shared [`execute_neighbor_hop`] helper and /// records: @@ -83,6 +92,55 @@ pub async fn cross_core_traverse_subgraph( shared: &SharedState, params: CrossCoreTraverseSubgraphParams<'_>, ) -> crate::Result { + let SubgraphWalk { + node_order, + depth_of, + edges, + } = walk_subgraph(shared, params).await?; + + let wire_nodes: Vec> = node_order + .iter() + .map(|id| WireNode { + id: id.as_str(), + depth: *depth_of.get(id).unwrap_or(&0), + }) + .collect(); + let wire_edges: Vec> = edges + .iter() + .map(|(src, label, dst)| WireEdge { + from: src.as_str(), + to: dst.as_str(), + label: label.as_str(), + }) + .collect(); + let envelope = WireSubGraph { + nodes: wire_nodes, + edges: wire_edges, + }; + + let payload = sonic_rs::to_vec(&envelope).map_err(|e| crate::Error::Serialization { + format: "json".into(), + detail: e.to_string(), + })?; + + Ok(ok_response(payload)) +} + +/// What a subgraph walk found: every visited node in discovery order, the +/// hop at which each was first reached, and every `(src, label, dst)` edge +/// crossed. +pub(crate) struct SubgraphWalk { + pub node_order: Vec, + pub depth_of: HashMap, + pub edges: Vec<(String, String, String)>, +} + +/// Walk the subgraph around `params.start`, expanding every frontier node at +/// the node that owns `from_key(node)`. +pub(crate) async fn walk_subgraph( + shared: &SharedState, + params: CrossCoreTraverseSubgraphParams<'_>, +) -> crate::Result { let CrossCoreTraverseSubgraphParams { tenant_id, collection, @@ -92,6 +150,7 @@ pub async fn cross_core_traverse_subgraph( direction, max_depth, options, + linearizable, } = params; // Per-node depth: the start node is at depth 0; subsequent nodes // are tagged with the hop index that first surfaced them. @@ -100,13 +159,16 @@ pub async fn cross_core_traverse_subgraph( let mut node_order: Vec = Vec::new(); let mut edges: Vec<(String, String, String)> = Vec::new(); let mut frontier: Vec = vec![start.clone()]; + let mut reads = ShardReadLog::new(); + let cap = walk_visit_cap(shared, options); + let whole = whole_hop(); visited.insert(start.clone()); depth_of.insert(start.clone(), 0); node_order.push(start); for hop_idx in 0..max_depth { - if frontier.is_empty() { + if frontier.is_empty() || node_order.len() >= cap { break; } @@ -119,8 +181,9 @@ pub async fn cross_core_traverse_subgraph( frontier: &frontier, edge_label: edge_label.as_deref(), direction, - options, + options: &whole, discovered_so_far: node_order.len(), + linearizable, }, ) .await?; @@ -128,56 +191,25 @@ pub async fn cross_core_traverse_subgraph( // Edges are fully attributed for BOTH local-shard and remote-shard // expansion (each frontier node was expanded at its owner). Always // record them — even when the destination is already visited (an - // A→B→C graph with a back-edge B→A should surface that edge once). + // A→B→C graph with a back-edge B→A must surface that edge once). edges.extend(hop.local_triples); + reads.merge(hop.reads); - // Tag newly-discovered nodes with the current hop's depth and - // build the next frontier. `hop_idx=0` expands the depth-0 + // Admit the level's new nodes in name order under the cap, and tag + // them with the current hop's depth. `hop_idx=0` expands the depth-0 // start node into depth-1 neighbors. let next_depth_tag = (hop_idx + 1).min(u8::MAX as usize) as u8; - let mut next_frontier: Vec = Vec::new(); - for node in hop.merged_destinations { - if visited.insert(node.clone()) { - depth_of.insert(node.clone(), next_depth_tag); - node_order.push(node.clone()); - next_frontier.push(node); - if node_order.len() >= options.max_visited { - break; - } - } - } - - frontier = next_frontier; - - if node_order.len() >= options.max_visited { - break; + frontier = admit_by_name(hop.merged_destinations, &mut visited, &mut node_order, cap); + for node in &frontier { + depth_of.insert(node.clone(), next_depth_tag); } } - let wire_nodes: Vec> = node_order - .iter() - .map(|id| WireNode { - id: id.as_str(), - depth: *depth_of.get(id).unwrap_or(&0), - }) - .collect(); - let wire_edges: Vec> = edges - .iter() - .map(|(src, label, dst)| WireEdge { - from: src.as_str(), - to: dst.as_str(), - label: label.as_str(), - }) - .collect(); - let envelope = WireSubGraph { - nodes: wire_nodes, - edges: wire_edges, - }; - - let payload = sonic_rs::to_vec(&envelope).map_err(|e| crate::Error::Serialization { - format: "json".into(), - detail: e.to_string(), - })?; - - Ok(ok_response(payload)) + // Every vShard the walk expanded joins the transaction read-set. + reads.publish(shared, tenant_id, database_id, collection); + Ok(SubgraphWalk { + node_order, + depth_of, + edges, + }) } diff --git a/nodedb/src/control/server/graph_dispatch/walk_reads.rs b/nodedb/src/control/server/graph_dispatch/walk_reads.rs new file mode 100644 index 000000000..2ce6d6e89 --- /dev/null +++ b/nodedb/src/control/server/graph_dispatch/walk_reads.rs @@ -0,0 +1,241 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Multi-hop graph reads sent as a single plan (`Hop`, `Path`, `Subgraph`), +//! run through the cross-shard walk coordinators. +//! +//! In a cluster, one plan on one node walks only that node's partitions. Each +//! of these reads instead runs the hop-by-hop walk the SQL surface uses: every +//! frontier node expands at the node that owns its key vShard. The answer is +//! returned in the payload shape the Data Plane gives for the same plan, so a +//! protocol shapes it as before: +//! +//! - `Hop`: a msgpack array of every reached node, start nodes first. +//! - `Subgraph`: a msgpack array of `{src, label, dst}` edges. +//! - `Path`: a msgpack array `[src, …, dst]`, or a `NotFound` refusal when +//! `dst` is unreachable. + +use nodedb_physical::physical_plan::GraphOp; + +use crate::bridge::envelope::{Payload, PhysicalPlan, Response}; +use crate::control::server::dispatch_utils::{not_found_response, ok_payload_response}; +use crate::control::state::SharedState; +use crate::engine::graph::edge_store::Direction; +use crate::types::{DatabaseId, TenantId}; + +use super::bfs::{CrossCoreBfsParams, cross_core_bfs_with_options}; +use super::shortest_path::{CrossCoreShortestPathParams, cross_core_shortest_path}; +use super::traverse_subgraph::{CrossCoreTraverseSubgraphParams, SubgraphWalk, walk_subgraph}; + +/// One subgraph edge, in the Data Plane's `Subgraph` response shape. +#[derive(zerompk::ToMessagePack)] +#[msgpack(map)] +struct SubgraphEdgeWire<'a> { + src: &'a str, + label: &'a str, + dst: &'a str, +} + +/// Run `plan` through the walk coordinators when it is a `Hop`, `Path` or +/// `Subgraph`. Returns `None` for every other plan. +pub async fn serve_walk_plan( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + plan: &PhysicalPlan, + linearizable: bool, +) -> Option> { + let PhysicalPlan::Graph(op) = plan else { + return None; + }; + Some(match op { + GraphOp::Hop { + collection, + start_nodes, + edge_label, + direction, + depth, + options, + .. + } => cross_core_bfs_with_options( + state, + CrossCoreBfsParams { + tenant_id, + database_id, + collection: collection.as_ref().map(|c| c.as_str()), + start_nodes: start_nodes.clone(), + edge_label: edge_label.clone(), + direction: *direction, + max_depth: *depth, + options, + linearizable, + }, + ) + .await + .and_then(|response| rewrap_as_msgpack(&response.payload)), + GraphOp::Subgraph { + collection, + start_nodes, + edge_label, + depth, + options, + .. + } => { + subgraph( + state, + tenant_id, + database_id, + SubgraphRequest { + collection: collection.as_ref().map(|c| c.as_str().to_owned()), + start_nodes, + edge_label: edge_label.clone(), + depth: *depth, + options, + linearizable, + }, + ) + .await + } + GraphOp::Path { + collection, + src, + dst, + edge_label, + max_depth, + options, + .. + } => { + path( + state, + tenant_id, + database_id, + PathRequest { + collection: collection.as_ref(), + src, + dst, + edge_label: edge_label.clone(), + max_depth: *max_depth, + options, + linearizable, + }, + ) + .await + } + _ => return None, + }) +} + +/// The inputs of a subgraph walk sent as one plan. +struct SubgraphRequest<'a> { + collection: Option, + start_nodes: &'a [String], + edge_label: Option, + depth: usize, + options: &'a crate::engine::graph::traversal_options::GraphTraversalOptions, + linearizable: bool, +} + +async fn subgraph( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + request: SubgraphRequest<'_>, +) -> crate::Result { + let [start] = request.start_nodes else { + return Err(crate::Error::BadRequest { + detail: format!( + "a subgraph walk starts from one node, got {}", + request.start_nodes.len() + ), + }); + }; + // The Data Plane's subgraph is the out-edge closure of its start node. + let SubgraphWalk { edges, .. } = walk_subgraph( + state, + CrossCoreTraverseSubgraphParams { + tenant_id, + database_id, + collection: request.collection, + start: start.clone(), + edge_label: request.edge_label, + direction: Direction::Out, + max_depth: request.depth, + options: request.options, + linearizable: request.linearizable, + }, + ) + .await?; + let wire: Vec> = edges + .iter() + .map(|(src, label, dst)| SubgraphEdgeWire { + src: src.as_str(), + label: label.as_str(), + dst: dst.as_str(), + }) + .collect(); + let payload = zerompk::to_msgpack_vec(&wire).map_err(|e| crate::Error::Codec { + detail: format!("subgraph walk encode: {e}"), + })?; + Ok(ok_payload_response(Payload::from_vec(payload))) +} + +/// The inputs of a shortest-path walk sent as one plan. +struct PathRequest<'a> { + collection: Option<&'a nodedb_types::QualifiedCollection>, + src: &'a str, + dst: &'a str, + edge_label: Option, + max_depth: usize, + options: &'a crate::engine::graph::traversal_options::GraphTraversalOptions, + linearizable: bool, +} + +async fn path( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + request: PathRequest<'_>, +) -> crate::Result { + // A path with no collection walks every collection's edges, as the Data + // Plane's path does. + let response = cross_core_shortest_path( + state, + CrossCoreShortestPathParams { + tenant_id, + database_id, + collection: request.collection.map(|c| c.as_str().to_owned()), + src: request.src.to_owned(), + dst: request.dst.to_owned(), + edge_label: request.edge_label, + max_depth: request.max_depth, + options: request.options.clone(), + linearizable: request.linearizable, + }, + ) + .await?; + let path: Vec = + sonic_rs::from_slice(response.payload.as_ref()).map_err(|e| crate::Error::Codec { + detail: format!("graph path decode: {e}"), + })?; + // The Data Plane refuses an unreachable destination with `NotFound`. + if path.is_empty() { + return Ok(not_found_response()); + } + encode_names(&path) +} + +/// Re-encode a JSON array of node names as the msgpack array the Data Plane +/// returns. +fn rewrap_as_msgpack(payload: &Payload) -> crate::Result { + let names: Vec = + sonic_rs::from_slice(payload.as_ref()).map_err(|e| crate::Error::Codec { + detail: format!("graph walk decode: {e}"), + })?; + encode_names(&names) +} + +fn encode_names(names: &[String]) -> crate::Result { + let payload = zerompk::to_msgpack_vec(&names.to_vec()).map_err(|e| crate::Error::Codec { + detail: format!("graph walk encode: {e}"), + })?; + Ok(ok_payload_response(Payload::from_vec(payload))) +} diff --git a/nodedb/src/control/server/graph_dispatch/whole_graph.rs b/nodedb/src/control/server/graph_dispatch/whole_graph.rs new file mode 100644 index 000000000..1cd21678b --- /dev/null +++ b/nodedb/src/control/server/graph_dispatch/whole_graph.rs @@ -0,0 +1,301 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Graph reads that need every partition of a graph: `SHOW GRAPH STATS` and +//! `GRAPH ALGO`. +//! +//! A graph edge lives on the key vShard of each endpoint +//! (`types/record_home.rs`), so a collection's edges spread over every data +//! group. Each of these reads goes to one node per data group: the group's +//! leader, carrying every group it leads, exactly as the BSP coordinators +//! enumerate shards (`bsp_pagerank::enumerate`). Each node answers from every +//! one of its cores, and every node must answer, or the read fails. +//! +//! How an algorithm uses the partitions: +//! +//! - PageRank and WCC run as BSP supersteps across the owners in a cluster. +//! Both are vertex-centric and converge to the single-node answer from +//! per-shard message passing. +//! - Every other algorithm gathers the collection's edges from every owner +//! and runs once, on one core of this node, over their union. +//! Label propagation and Louvain depend on update order, so a synchronous +//! BSP round will diverge from the single-node answer. LCC and triangles +//! need two-hop neighbourhoods. Betweenness, closeness, harmonic, diameter +//! and SSSP need shortest paths over the whole graph. K-core peels the +//! whole graph. Degree needs both endpoints of each edge. The gathered edges +//! are sorted, so the CSR, and the answer, match a single node's exactly. +//! - On a single node every algorithm gathers, so an algorithm sees the edges +//! of every core, not one core's share. +//! - A historical run (`AS OF SYSTEM TIME`) gathers for every algorithm: each +//! owner exports the edges live at that time. +//! - A gathered run needs the whole graph in one CSR on one core, as a single +//! node's run does. It is bounded by `GraphTuning::max_gathered_algo_edges`: +//! past the cap it is refused with the edge count, never answered from part +//! of the graph. + +use std::collections::BTreeMap; + +use futures::future::join_all; + +use crate::bridge::envelope::{Payload, PhysicalPlan}; +use crate::control::gateway::dispatcher::statement_deadline_ms; +use crate::control::gateway::version_set::GatewayVersionSet; +use crate::control::server::exchange::execute_plan_all_local_cores; +use crate::control::state::SharedState; +use crate::types::{DatabaseId, TenantId, TraceId, VShardId}; +use nodedb_graph::{AlgoParams, GraphAlgorithm}; +use nodedb_physical::physical_plan::{AlgoEdge, AlgoStage, GraphOp}; + +use super::bsp_pagerank::enumerate::enumerate_shards; +use super::cluster_resolve::{DispatchSuperstepParams, dispatch_superstep_to_node, gateway_shared}; +use super::shard_reads::{ShardReadLog, qualified}; + +/// Run `plan` on one node per data group and return each node's payload. +/// +/// On a single node the plan fans across this node's cores and yields one +/// payload. A linearizable read is confirmed on each node that serves it. +/// Every vShard read joins the transaction read-set, scoped to `collection` +/// (database-qualified), or to every collection when `None`. +pub async fn scatter_to_graph_owners( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + plan: PhysicalPlan, + linearizable: bool, + collection: Option, +) -> crate::Result> { + if state.cluster_routing.is_none() { + let node = + execute_plan_all_local_cores(state, tenant_id, database_id, plan, TraceId::ZERO, None) + .await?; + return Ok(vec![Payload::from_vec(node.payload)]); + } + let targets = enumerate_shards(state)?.targets; + let shared_arc = gateway_shared(state)?; + let version_set = GatewayVersionSet::from_pairs(Vec::new()); + let deadline_ms = statement_deadline_ms(state); + let legs = targets.into_iter().map(|target| { + let plan = plan.clone(); + let shared_arc = shared_arc.clone(); + let version_set = version_set.clone(); + async move { + let read = dispatch_superstep_to_node( + &shared_arc, + DispatchSuperstepParams { + tenant_id, + database_id, + deadline_ms, + node_id: target.node_id, + is_local: target.is_local, + route_vshard: target.route_vshard(), + plan, + version_set: &version_set, + linearizable, + }, + ) + .await?; + Ok::<_, crate::Error>((target.owned_vshards, target.node_id, read)) + } + }); + let mut reads = ShardReadLog::new(); + let mut payloads = Vec::new(); + for leg in join_all(legs).await { + let (owned_vshards, node_id, read) = leg?; + reads.note(owned_vshards, read.watermark_lsn, node_id); + payloads.push(read.payload); + } + reads.publish(state, tenant_id, database_id, collection); + Ok(payloads) +} + +/// Run `algorithm` over the whole graph of `params.collection` and return its +/// `AlgoResultBatch` payload. `system_as_of_ms` runs it over the edges live at +/// that system time; `None` runs it over the current edges. +pub async fn run_graph_algo( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + algorithm: GraphAlgorithm, + params: AlgoParams, + system_as_of_ms: Option, + linearizable: bool, +) -> crate::Result { + let deadline_ms = statement_deadline_ms(state); + // A current-state BSP run pins its own read cut from a Calvin cut marker + // (`run_cut`). A historical run gathers the edges live at its system time. + if state.cluster_routing.is_some() && system_as_of_ms.is_none() { + match algorithm { + GraphAlgorithm::PageRank => { + return super::run_bsp_pagerank( + state, + tenant_id, + database_id, + params, + deadline_ms, + linearizable, + ) + .await; + } + GraphAlgorithm::Wcc => { + return super::run_bsp_wcc( + state, + tenant_id, + database_id, + params, + deadline_ms, + linearizable, + ) + .await; + } + _ => {} + } + } + let edges = gather_algo_edges( + state, + tenant_id, + database_id, + algorithm, + ¶ms, + system_as_of_ms, + linearizable, + ) + .await?; + let plan = PhysicalPlan::Graph(GraphOp::Algo { + algorithm, + params, + stage: AlgoStage::Gathered { edges }, + }); + let response = crate::control::server::dispatch_utils::dispatch_to_data_plane( + state, + tenant_id, + database_id, + VShardId::new(0), + plan, + TraceId::ZERO, + ) + .await?; + crate::control::local_dispatch::reject_data_plane_error(&response)?; + Ok(response.payload) +} + +/// The union of the collection's edges on every owner, sorted by +/// `(src, label, dst)`. An edge held by more than one node (both endpoint +/// homes, or a replica) appears once. +async fn gather_algo_edges( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + algorithm: GraphAlgorithm, + params: &AlgoParams, + system_as_of_ms: Option, + linearizable: bool, +) -> crate::Result> { + let plan = PhysicalPlan::Graph(GraphOp::Algo { + algorithm, + params: params.clone(), + stage: AlgoStage::ExportEdges { system_as_of_ms }, + }); + let payloads = scatter_to_graph_owners( + state, + tenant_id, + database_id, + plan, + linearizable, + Some(qualified(database_id, ¶ms.collection)), + ) + .await?; + merge_edge_parts(payloads, state.tuning.graph.max_gathered_algo_edges) +} + +/// Union each owner's exported edges, sorted by `(src, label, dst)`. An edge +/// two owners hold (both endpoint homes, or a replica) appears once. The run +/// needs every edge in one CSR on one core, so past `cap` distinct edges it is +/// refused with the count gathered so far, never answered from part of the +/// graph. +fn merge_edge_parts(payloads: Vec, cap: usize) -> crate::Result> { + let mut union: BTreeMap<(String, String, String), f64> = BTreeMap::new(); + for payload in payloads { + if payload.is_empty() { + continue; + } + let part: Vec = + zerompk::from_msgpack(payload.as_ref()).map_err(|e| crate::Error::Codec { + detail: format!("graph algorithm edge export decode: {e}"), + })?; + for edge in part { + union.insert((edge.src, edge.label, edge.dst), edge.weight); + } + if union.len() > cap { + return Err(crate::Error::LimitExceeded { + limit_name: "graph.max_gathered_algo_edges", + value: union.len() as u64, + max: cap as u64, + }); + } + } + Ok(union + .into_iter() + .map(|((src, label, dst), weight)| AlgoEdge { + src, + label, + dst, + weight, + }) + .collect()) +} + +#[cfg(test)] +mod tests { + use super::*; + + fn edge(src: &str, dst: &str) -> AlgoEdge { + AlgoEdge { + src: src.into(), + label: "l".into(), + dst: dst.into(), + weight: 1.0, + } + } + + fn part(edges: &[AlgoEdge]) -> Payload { + Payload::from_vec(zerompk::to_msgpack_vec(&edges.to_vec()).expect("encode part")) + } + + #[test] + fn owners_union_into_one_sorted_edge_set() { + let merged = merge_edge_parts( + vec![ + part(&[edge("b", "c"), edge("a", "b")]), + Payload::from_vec(Vec::new()), + part(&[edge("a", "b"), edge("c", "a")]), + ], + 10, + ) + .expect("merge"); + let keys: Vec<(String, String)> = merged.into_iter().map(|e| (e.src, e.dst)).collect(); + assert_eq!( + keys, + vec![ + ("a".to_string(), "b".to_string()), + ("b".to_string(), "c".to_string()), + ("c".to_string(), "a".to_string()), + ] + ); + } + + #[test] + fn a_graph_past_the_cap_is_refused_with_its_edge_count() { + let error = merge_edge_parts( + vec![part(&[edge("a", "b"), edge("b", "c"), edge("c", "d")])], + 2, + ) + .expect_err("three edges exceed a cap of two"); + assert!(matches!( + error, + crate::Error::LimitExceeded { + limit_name: "graph.max_gathered_algo_edges", + value: 3, + max: 2, + } + )); + } +} diff --git a/nodedb/src/control/server/http/auth.rs b/nodedb/src/control/server/http/auth.rs index d45ace268..719fad66f 100644 --- a/nodedb/src/control/server/http/auth.rs +++ b/nodedb/src/control/server/http/auth.rs @@ -45,7 +45,7 @@ pub struct AppState { /// /// Validation is awaited, never blocked on: a JWKS cache miss fetches the /// provider's key set over the network, and the whole HTTP request path runs on -/// Tokio worker threads, so blocking here would stall a worker at best and +/// Tokio worker threads, so blocking here will stall a worker at best and /// abort the request at worst. async fn try_validate_jwt( state: &AppState, @@ -330,6 +330,10 @@ impl IntoResponse for ApiError { fn into_response(self) -> Response { use super::types::HttpError; + let plain = |status: StatusCode, message: String| { + (status, axum::Json(HttpError::new(message))).into_response() + }; + match self { ApiError::RateLimited { message, @@ -355,22 +359,14 @@ impl IntoResponse for ApiError { }; (status, axum::Json(body)).into_response() } - other => { - let (status, message) = match other { - ApiError::Unauthorized(msg) => (StatusCode::UNAUTHORIZED, msg), - ApiError::Forbidden(msg) => (StatusCode::FORBIDDEN, msg), - ApiError::BadRequest(msg) => (StatusCode::BAD_REQUEST, msg), - ApiError::Internal(msg) => (StatusCode::INTERNAL_SERVER_ERROR, msg), - ApiError::RateLimited { .. } => unreachable!(), - ApiError::Coded { .. } => unreachable!(), - ApiError::HttpStatus(code, msg) => ( - StatusCode::from_u16(code).unwrap_or(StatusCode::INTERNAL_SERVER_ERROR), - msg, - ), - }; - let body = HttpError::new(message); - (status, axum::Json(body)).into_response() - } + ApiError::Unauthorized(msg) => plain(StatusCode::UNAUTHORIZED, msg), + ApiError::Forbidden(msg) => plain(StatusCode::FORBIDDEN, msg), + ApiError::BadRequest(msg) => plain(StatusCode::BAD_REQUEST, msg), + ApiError::Internal(msg) => plain(StatusCode::INTERNAL_SERVER_ERROR, msg), + ApiError::HttpStatus(code, msg) => plain( + StatusCode::from_u16(code).unwrap_or(StatusCode::INTERNAL_SERVER_ERROR), + msg, + ), } } } @@ -378,7 +374,7 @@ impl IntoResponse for ApiError { /// Axum extractor that resolves and enforces HTTP auth before a handler runs. /// /// Add this as the first parameter to any handler that performs tenant-scoped -/// or admin work. Handlers that should remain public (health probes, etc.) must +/// or admin work. Handlers that remain public (health probes, etc.) must /// NOT include this extractor. /// /// Produces a 401/403 response and short-circuits the handler if auth fails. @@ -568,6 +564,7 @@ mod tests { vshard_id: VShardId::new(1), leader_node: 2, leader_addr: "10.0.0.1:9000".into(), + leader_term: 3, }, StatusCode::SERVICE_UNAVAILABLE, ), diff --git a/nodedb/src/control/server/http/routes/cdc.rs b/nodedb/src/control/server/http/routes/cdc.rs index dda094363..bb26d78a2 100644 --- a/nodedb/src/control/server/http/routes/cdc.rs +++ b/nodedb/src/control/server/http/routes/cdc.rs @@ -18,7 +18,9 @@ use super::super::admission::{admit, admit_without_rate_limit}; use super::super::auth::{ApiError, AppState, ResolvedIdentity}; use super::super::peer::PeerAddr; use super::query::{DatabaseQueryParam, resolve_database_id}; -use crate::control::change_stream::{ChangeCursor, ReplayError, ReplayStart, SequencedChangeEvent}; +use crate::control::change_stream::{ + ChangeCursor, CursorStep, ReplayError, ReplayStart, SequencedChangeEvent, +}; use crate::control::security::audit::ArcAuditEmitter; use crate::control::security::identity::Permission; use crate::control::server::shared::authorization::{authorize_collection, authorize_database}; @@ -81,23 +83,27 @@ pub async fn sse_stream( Some(tenant_id), database_id, ); + // The ring is bounded, so the replay returns every event it holds. Live + // events the replay already covered are skipped by the running cursor. let snapshot = shared .change_stream - .query_changes_in_database(tenant_id, database_id, Some(&collection), start, 10_000) + .query_changes_in_database(tenant_id, database_id, Some(&collection), start, usize::MAX) .map_err(reset_error)?; - let snapshot_cursor = snapshot.snapshot_cursor; + let mut cursor = snapshot.cursor; let stream = async_stream::stream! { - for event in snapshot.events { yield Ok(format_sse_event(&event)); } + for replayed in snapshot.events { yield Ok(format_sse_event(&replayed.event, &replayed.cursor)); } loop { match subscription.recv_sequenced().await { - Ok(event) => { - if !event.cursor().same_epoch(snapshot_cursor) { - yield Ok(Event::default().event("reset_required").data("change stream epoch changed; reconnect with a fresh snapshot")); + Ok(event) => match cursor.accept(&event) { + CursorStep::Deliver => { + yield Ok(format_sse_event(&event, &cursor)); + } + CursorStep::Skip => {} + CursorStep::Reset => { + yield Ok(Event::default().event("reset_required").data("this node's change feed has a gap above the stream position; reconnect with a fresh snapshot")); break; } - if !event.cursor().is_after_in_same_epoch(snapshot_cursor) { continue; } - yield Ok(format_sse_event(&event)); - } + }, Err(tokio::sync::broadcast::error::RecvError::Lagged(_)) => { yield Ok(Event::default().event("reset_required").data("change stream lagged; reconnect with a fresh snapshot")); break; @@ -146,23 +152,22 @@ pub async fn poll_changes( .map(ReplayStart::Cursor) .unwrap_or(ReplayStart::Timestamp(params.since_ms.unwrap_or(0))); let limit = params.limit.unwrap_or(100).clamp(1, 10_000); - let mut snapshot = state + let snapshot = state .shared .change_stream - .query_changes_in_database(tenant_id, database_id, Some(&collection), start, limit + 1) + .query_changes_in_database(tenant_id, database_id, Some(&collection), start, limit) .map_err(reset_error)?; - let has_more = snapshot.events.len() > limit; - if has_more { - snapshot.events.truncate(limit); - } - let changes: Vec<_> = snapshot.events.iter().map(change_json).collect(); - let next_cursor = snapshot + let changes: Vec<_> = snapshot .events - .last() - .map(|event| serde_json::json!({"cursor": event.cursor().to_string()})); + .iter() + .map(|replayed| change_json(&replayed.event, &replayed.cursor)) + .collect(); + // The cursor resumes past every held event the poll passed over, even + // when it returned none. + let next_cursor = serde_json::json!({"cursor": snapshot.cursor.to_string()}); Ok(( rate_limit_headers, - Json(serde_json::json!({ "changes": changes, "next_cursor": next_cursor, "has_more": has_more, "count": snapshot.events.len() })), + Json(serde_json::json!({ "changes": changes, "next_cursor": next_cursor, "has_more": snapshot.has_more, "count": snapshot.events.len() })), ) .into_response()) } @@ -221,18 +226,18 @@ fn parse_last_event_id(headers: &HeaderMap) -> Result, ApiE fn reset_error(_: ReplayError) -> ApiError { ApiError::HttpStatus( 410, - "reset_required: cursor is expired, from a different stream epoch, or ahead of the stream" - .into(), + "reset_required: this node no longer holds every change past the cursor".into(), ) } -fn change_json(event: &SequencedChangeEvent) -> serde_json::Value { - serde_json::json!({ "operation": event.operation.as_str(), "document_id": event.document_id.as_str(), "timestamp_ms": event.timestamp_ms, "lsn": event.lsn.as_u64(), "collection": event.collection, "cursor": event.cursor().to_string() }) +/// One change, and the cursor that resumes right after it. +fn change_json(event: &SequencedChangeEvent, cursor: &ChangeCursor) -> serde_json::Value { + serde_json::json!({ "operation": event.operation.as_str(), "document_id": event.document_id.as_str(), "timestamp_ms": event.timestamp_ms, "lsn": event.lsn.as_u64(), "collection": event.collection, "cursor": cursor.to_string() }) } -fn format_sse_event(event: &SequencedChangeEvent) -> Event { +fn format_sse_event(event: &SequencedChangeEvent, cursor: &ChangeCursor) -> Event { Event::default() - .id(event.cursor().to_string()) + .id(cursor.to_string()) .event(event.operation.as_str().to_lowercase()) - .data(change_json(event).to_string()) + .data(change_json(event, cursor).to_string()) } diff --git a/nodedb/src/control/server/http/routes/cluster.rs b/nodedb/src/control/server/http/routes/cluster.rs index f02515bbb..f2cee3119 100644 --- a/nodedb/src/control/server/http/routes/cluster.rs +++ b/nodedb/src/control/server/http/routes/cluster.rs @@ -7,9 +7,9 @@ //! peer, every Raft group hosted on this node — sourced from the //! `ClusterObserver` published by `control::cluster::start_raft`. //! -//! In single-node mode (no `[cluster]` config) the endpoint returns -//! `503 Service Unavailable` with a short JSON error body so clients -//! can distinguish "cluster mode disabled" from "cluster mode broken". +//! Every node runs a cluster, a node with no `[cluster]` config included: +//! it runs a one-node cluster. `start_raft` publishes the observer before +//! the startup gate admits this route. use axum::extract::State; use axum::http::{StatusCode, header}; @@ -42,25 +42,19 @@ pub async fn cluster_status( return error.into_response(); } - match state.shared.cluster_observer.get() { - Some(observer) => { - let snap = observer.snapshot(); - match sonic_rs::to_string(&snap) { - Ok(body) => json_response(StatusCode::OK, body), - Err(e) => { - tracing::warn!(error = %e, "cluster snapshot serialization failed"); - json_response( - StatusCode::INTERNAL_SERVER_ERROR, - r#"{"error":"snapshot serialization failed"}"#.to_string(), - ) - } - } + let Some(observer) = state.shared.cluster_observer.get() else { + return super::cluster_debug::guard::cluster_not_started(); + }; + let snap = observer.snapshot(); + match sonic_rs::to_string(&snap) { + Ok(body) => json_response(StatusCode::OK, body), + Err(e) => { + tracing::warn!(error = %e, "cluster snapshot serialization failed"); + json_response( + StatusCode::INTERNAL_SERVER_ERROR, + r#"{"error":"snapshot serialization failed"}"#.to_string(), + ) } - None => json_response( - StatusCode::SERVICE_UNAVAILABLE, - r#"{"error":"cluster mode not enabled","detail":"this node is running in single-node mode; /v1/cluster/status requires a [cluster] config section"}"# - .to_string(), - ), } } diff --git a/nodedb/src/control/server/http/routes/cluster_debug/guard.rs b/nodedb/src/control/server/http/routes/cluster_debug/guard.rs index 6cad7d911..6d11b8160 100644 --- a/nodedb/src/control/server/http/routes/cluster_debug/guard.rs +++ b/nodedb/src/control/server/http/routes/cluster_debug/guard.rs @@ -66,7 +66,7 @@ pub fn json_response(status: StatusCode, body: String) -> Response { /// Serialise `value` with `sonic_rs` and wrap in a 200 response. /// On serialisation failure returns a 500 with a short JSON error /// body — the only realistic failure mode for in-memory snapshots is -/// a non-UTF8 key, which would indicate corrupted memory, not a +/// a non-UTF8 key, which indicates corrupted memory, not a /// legitimate caller error. pub fn ok_json(value: &T) -> Response { match sonic_rs::to_string(value) { @@ -81,12 +81,14 @@ pub fn ok_json(value: &T) -> Response { } } -/// 503 response used when the cluster subsystem required by a handler -/// is absent (single-node mode). Kept in one place so every endpoint -/// returns the same shape for "feature not wired on this node". -pub fn cluster_disabled() -> Response { +/// 500 response for a cluster handle a handler reads before `start_raft` +/// installed it. Every node runs a cluster, a one-node cluster included, +/// and `start_raft` runs before the startup gate admits a cluster route. +/// Kept in one place so every endpoint returns the same shape. +pub fn cluster_not_started() -> Response { json_response( - StatusCode::SERVICE_UNAVAILABLE, - r#"{"error":"cluster mode not enabled"}"#.to_string(), + StatusCode::INTERNAL_SERVER_ERROR, + r#"{"error":"cluster not started","detail":"start_raft has not run on this node"}"# + .to_string(), ) } diff --git a/nodedb/src/control/server/http/routes/cluster_debug/leases.rs b/nodedb/src/control/server/http/routes/cluster_debug/leases.rs index 2b8097e01..887b2756b 100644 --- a/nodedb/src/control/server/http/routes/cluster_debug/leases.rs +++ b/nodedb/src/control/server/http/routes/cluster_debug/leases.rs @@ -21,6 +21,8 @@ struct LeaseRow { #[derive(serde::Serialize)] struct DrainRow { descriptor_id: String, + /// The operation that owns the drain. + owner: String, up_to_version: u64, expires_at: String, } @@ -71,13 +73,18 @@ pub async fn leases_debug( .lease_drain .snapshot() .into_iter() - .map(|(descriptor_id, entry)| DrainRow { + .map(|(descriptor_id, owner, entry)| DrainRow { descriptor_id: format!("{descriptor_id:?}"), + owner: format!("{owner:?}"), up_to_version: entry.up_to_version, expires_at: format!("{:?}", entry.expires_at), }) .collect(); - rows.sort_by(|a, b| a.descriptor_id.cmp(&b.descriptor_id)); + rows.sort_by(|a, b| { + a.descriptor_id + .cmp(&b.descriptor_id) + .then_with(|| a.owner.cmp(&b.owner)) + }); rows }; diff --git a/nodedb/src/control/server/http/routes/cluster_debug/raft.rs b/nodedb/src/control/server/http/routes/cluster_debug/raft.rs index 7e8b232b1..45500cce6 100644 --- a/nodedb/src/control/server/http/routes/cluster_debug/raft.rs +++ b/nodedb/src/control/server/http/routes/cluster_debug/raft.rs @@ -14,7 +14,7 @@ use axum::response::Response; use super::super::super::auth::{AppState, ResolvedIdentity}; use super::super::super::peer::PeerAddr; -use super::guard::{cluster_disabled, ensure_debug_access, json_response, ok_json}; +use super::guard::{cluster_not_started, ensure_debug_access, json_response, ok_json}; #[derive(serde::Serialize)] struct RaftDebugResponse { @@ -33,7 +33,7 @@ pub async fn raft_debug( return resp; } let Some(status_fn) = state.shared.raft_status_fn.get() else { - return cluster_disabled(); + return cluster_not_started(); }; let statuses = status_fn(); match statuses.into_iter().find(|s| s.group_id == group_id) { diff --git a/nodedb/src/control/server/http/routes/cluster_debug/transport.rs b/nodedb/src/control/server/http/routes/cluster_debug/transport.rs index ae452833d..55f485687 100644 --- a/nodedb/src/control/server/http/routes/cluster_debug/transport.rs +++ b/nodedb/src/control/server/http/routes/cluster_debug/transport.rs @@ -10,7 +10,7 @@ use nodedb_cluster::{BreakerSnapshot, TransportPeerSnapshot}; use super::super::super::auth::{AppState, ResolvedIdentity}; use super::super::super::peer::PeerAddr; -use super::guard::{cluster_disabled, ensure_debug_access, ok_json}; +use super::guard::{cluster_not_started, ensure_debug_access, ok_json}; #[derive(serde::Serialize)] struct TransportDebugResponse { @@ -28,7 +28,7 @@ pub async fn transport_debug( return resp; } let Some(transport) = state.shared.cluster_transport.as_ref() else { - return cluster_disabled(); + return cluster_not_started(); }; let response = TransportDebugResponse { node_id: state.shared.node_id, diff --git a/nodedb/src/control/server/http/routes/crdt.rs b/nodedb/src/control/server/http/routes/crdt.rs index f45ed92bf..607eb2703 100644 --- a/nodedb/src/control/server/http/routes/crdt.rs +++ b/nodedb/src/control/server/http/routes/crdt.rs @@ -96,15 +96,15 @@ pub async fn crdt_apply( let _trace_id = extract_request_id(&headers); - let surrogate = state - .shared - .surrogate_assigner - .assign( - nodedb_types::CollectionKey::from_bare(crate::types::DatabaseId::DEFAULT, &collection), - identity.tenant_id, - body.doc_id.as_bytes(), - ) - .map_err(ApiError::from)?; + let surrogate = crate::control::server::surrogate_exchange::assign_surrogate_routed( + &state.shared, + nodedb_types::CollectionKey::from_bare(crate::types::DatabaseId::DEFAULT, &collection), + identity.tenant_id, + body.doc_id.as_bytes(), + crate::types::TraceId::ZERO, + ) + .await + .map_err(ApiError::from)?; let plan = PhysicalPlan::Crdt(CrdtOp::Apply { collection: nodedb_types::QualifiedCollection::new( @@ -149,7 +149,7 @@ pub async fn crdt_apply( .ok_or_else(|| ApiError::Internal("authorization returned no capability".into()))?; // Route through the Raft proposer gate so the delta is quorum-durable under - // replication. A local-only dispatch would land it on the receiving node only + // replication. A local-only dispatch will land it on the receiving node only // — lost to followers and entirely on leader failover. This handler is scoped // to the default database (matching its surrogate assignment above). let _request = state.shared.tenant_request_guard(identity.tenant_id); @@ -277,7 +277,7 @@ mod tests { let permissions = PermissionStore::new(); permissions .grant( - "collection:9:_system.audit_log", + "collection:0:9:_system.audit_log", "user:writer", Permission::Write, "admin", @@ -302,7 +302,7 @@ mod tests { let permissions = PermissionStore::new(); permissions .grant( - "collection:10:orders", + "collection:0:10:orders", "user:writer", Permission::Write, "admin", diff --git a/nodedb/src/control/server/http/routes/health.rs b/nodedb/src/control/server/http/routes/health.rs index a348bbaa7..8fdfb9c35 100644 --- a/nodedb/src/control/server/http/routes/health.rs +++ b/nodedb/src/control/server/http/routes/health.rs @@ -25,7 +25,7 @@ use super::super::peer::PeerAddr; /// GET /health/live — unconditional liveness probe. /// /// Always returns 200. If this endpoint fails to respond, the -/// process is dead and should be restarted. No internal state is +/// process is dead and must be restarted. No internal state is /// checked — the mere ability to respond proves the event loop and /// HTTP listener are alive. pub async fn live() -> impl IntoResponse { @@ -193,7 +193,8 @@ pub async fn healthz(State(state): State) -> impl IntoResponse { } /// The lease state `/healthz` reports: `valid`, `invalid`, `sole_voter` for -/// a pinned lease, or `not_required` on a single node without a cluster. +/// a pinned lease, or `not_required` before `start_raft` installed the lease +/// timing. fn lease_label(state: &AppState) -> &'static str { use crate::control::security::auth_lease::{LeaseStatus, lease_status}; match lease_status(&state.shared, std::time::Instant::now()) { @@ -262,8 +263,8 @@ fn outcome_floor_bound(state: &AppState) -> std::time::Duration { statement.max(crate::control::catalog_entry::post_apply::vector_install_longest_core_wait()) } -/// Why a cross-shard Calvin write would be refused on this node right now, -/// or `None` when one would be accepted. +/// Why a cross-shard Calvin write will be refused on this node right now, +/// or `None` when one will be accepted. /// /// Mirrors the two refusals `control::planner::calvin::submit` raises before a /// transaction ever reaches the inbox, so a client that waits for `/healthz` @@ -272,10 +273,6 @@ fn sequencer_not_servable( state: &AppState, cluster: Option<&ClusterInfoSnapshot>, ) -> Option<&'static str> { - // No Calvin stack on this node (embedded / local boot with the sequencer - // never started): nothing to wait for, readiness is unchanged. - state.shared.sequencer_inbox.get()?; - // Same resolution `submit_calvin_routed` performs: a missing group entry is // leader 0, which is exactly the state that refuses a submit. let leader = cluster? diff --git a/nodedb/src/control/server/http/routes/metrics.rs b/nodedb/src/control/server/http/routes/metrics.rs index 3afb56ad1..79c3f48a7 100644 --- a/nodedb/src/control/server/http/routes/metrics.rs +++ b/nodedb/src/control/server/http/routes/metrics.rs @@ -24,7 +24,7 @@ pub async fn metrics( // Blacklist + account status, no rate limit: a Prometheus scrape runs on a // fixed interval and is not the per-query traffic the rate limiter's cost - // table models, so metering it would only risk starving monitoring. A + // table models, so metering it will only risk starving monitoring. A // blacklisted IP or suspended/banned monitor account must still be refused // before any internal counter is read. admit_without_rate_limit( @@ -246,7 +246,7 @@ pub async fn metrics( output.push_str("# TYPE nodedb_tenant_qps_total counter\n"); for (tid, usage, _quota) in tenants.iter_usage() { let t = tid.as_u64(); - // Tenant label only — database label requires tenant→DB lookup which may + // Tenant label only — database label requires tenant→DB lookup which can // be expensive; include tenant_id as a proxy for now. Full database+tenant // labeling is done in the expanded tenant loop below. let _ = std::fmt::write( @@ -389,8 +389,9 @@ fn render_loop_specific_gauges(state: &AppState, out: &mut String) { // gateway_plan_cache_hit_ratio — derived from the plan cache's // hit+miss counters. Returns 0.0 when the cache has never been - // consulted so the series never reports NaN. - if let Some(gateway) = state.shared.gateway.get() { + // consulted so the series never reports NaN. A scrape never fails, so a + // node whose gateway is not yet installed omits the series. + if let Ok(gateway) = state.shared.installed_gateway() { let hits = gateway.plan_cache.cache_hit_count(); let misses = gateway.plan_cache.cache_miss_count(); let ratio = gateway.plan_cache.hit_ratio(); diff --git a/nodedb/src/control/server/http/routes/promql/remote.rs b/nodedb/src/control/server/http/routes/promql/remote.rs index 8f613cda5..4a734addf 100644 --- a/nodedb/src/control/server/http/routes/promql/remote.rs +++ b/nodedb/src/control/server/http/routes/promql/remote.rs @@ -80,7 +80,7 @@ pub async fn remote_write( // This endpoint builds its physical tasks itself instead of going through // the SQL planner, so it has to run the planner's row-level-security pass - // over each one explicitly — otherwise remote write would be a way to + // over each one explicitly — otherwise remote write will be a way to // ingest rows a write policy forbids, with the same identity and the same // collection an `INSERT` refuses. let scope = crate::control::security::request_scope::RequestAuthScope::for_database( @@ -116,7 +116,7 @@ pub async fn remote_write( // Prometheus remote-write answers with an HTTP status, never rows, // for the same reason the line-protocol listener does. `inject_rls` // still runs over this task, so the read filter it fills in is - // simply never consulted. + // never consulted. returning: None, rls_filters: Vec::new(), }); @@ -171,25 +171,16 @@ pub async fn remote_write( } }; - // Route through gateway when available (cluster-aware dispatch); - // fall back to capability-bearing local dispatch on single-node boot. - let dispatch_result = match state.shared.gateway.get() { - Some(gw) => { - let gw_ctx = QueryContext { - tenant_id, - trace_id: TraceId::generate(), - database_id: nodedb_types::id::DatabaseId::DEFAULT, - txn_id: None, - }; - gw.execute(&gw_ctx, checked).await - } - None => crate::control::server::dispatch_utils::dispatch_authorized_durable_write( - &state.shared, - checked, - TraceId::generate(), - ) - .await - .map(|_| vec![]), + let gw_ctx = QueryContext { + tenant_id, + trace_id: TraceId::generate(), + database_id: nodedb_types::id::DatabaseId::DEFAULT, + txn_id: None, + linearizable: true, + }; + let dispatch_result = match state.shared.installed_gateway() { + Ok(gateway) => gateway.execute(&gw_ctx, checked).await, + Err(error) => Err(error), }; match dispatch_result { diff --git a/nodedb/src/control/server/http/routes/query/materialized/append.rs b/nodedb/src/control/server/http/routes/query/materialized/append.rs new file mode 100644 index 000000000..a3e6b55a8 --- /dev/null +++ b/nodedb/src/control/server/http/routes/query/materialized/append.rs @@ -0,0 +1,126 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Per-task helpers the dispatch loop and the orchestrated plans share: +//! authorization without the clone-write check, metering, and shaping one +//! task's answer into the statement's JSON rows. + +use std::sync::Arc; + +use nodedb_physical::physical_task::PhysicalTask; + +use crate::bridge::envelope::Status; +use crate::control::security::audit::ArcAuditEmitter; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::security::request_scope::RequestAuthScope; +use crate::control::server::response_shape::redaction::QueryRedaction; +use crate::control::server::response_shape::request::MaterializedShapeRequest; +use crate::control::server::response_shape::schema::OutputSchema; +use crate::control::server::response_shape::types::PlanKind; +use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; +use crate::types::{DatabaseId, TenantId}; + +use super::super::super::super::auth::{ApiError, AppState}; +use super::super::super::result_shape::{ + HttpShaped, passthrough_json_row, shape_error_to_api, shape_http_payload, +}; +use super::encode::response_error; + +/// Authorize one task with no clone-write check. Used only for the +/// Control-Plane orchestrated plans (see `orchestrated`), which are never +/// clone-write shapes. +pub(super) fn authorize_materialized_task( + shared: &crate::control::state::SharedState, + identity: &AuthenticatedIdentity, + task: &PhysicalTask, +) -> crate::Result { + let emitter = ArcAuditEmitter(Arc::clone(&shared.audit)); + crate::control::server::shared::authorization::authorize_task_set( + identity, + std::slice::from_ref(task), + &shared.permissions, + &shared.roles, + &emitter, + ) + .map_err(crate::Error::from)? + .into_tasks() + .into_iter() + .next() + .ok_or_else(|| crate::Error::Internal { + detail: "authorization returned an empty capability set".into(), + }) +} + +/// Meter one task's dispatch after its rows are appended to `result_rows` — +/// the row count is the delta since `rows_before`. +pub(super) fn meter_task_dispatch( + state: &crate::control::state::SharedState, + scope: &RequestAuthScope<'_>, + info: &Option, + rows_before: usize, + result_rows: &[serde_json::Value], +) { + if let Some(info) = info { + let task_rows = (result_rows.len() - rows_before) as u64; + meter_dispatch(state, scope, info, Some(task_rows)); + } +} + +/// Everything one task's answer needs to be shaped and appended. Resolved +/// once per task and reused for every payload the task produced. +pub(super) struct ShapedAppend<'a> { + pub(super) plan: &'a crate::bridge::envelope::PhysicalPlan, + pub(super) plan_kind: PlanKind, + pub(super) output_schema: &'a OutputSchema, + pub(super) state: &'a AppState, + pub(super) database_id: DatabaseId, + pub(super) tenant_id: TenantId, + pub(super) redaction: &'a QueryRedaction, +} + +/// Shape a response. A refusal becomes the HTTP error its code maps to. +pub(super) fn append_response( + result_rows: &mut Vec, + response: crate::bridge::envelope::Response, + append: &ShapedAppend<'_>, +) -> Result<(), ApiError> { + if response.status != Status::Ok { + return Err(response_error(&response)); + } + append_payload(result_rows, &response.payload.to_vec(), append) +} + +/// Shape one payload into rows, or pass a write's or tag's payload through +/// as one JSON row. An empty payload adds nothing. +pub(super) fn append_payload( + result_rows: &mut Vec, + payload: &[u8], + append: &ShapedAppend<'_>, +) -> Result<(), ApiError> { + if payload.is_empty() { + return Ok(()); + } + // HTTP carries no session: `nextval` advances the registry, `currval` + // reports "not yet called in this session". + let sequences = crate::control::sequence::SessionSequenceAccess::for_session( + &append.state.shared, + None, + append.database_id, + append.tenant_id, + ); + match shape_http_payload(MaterializedShapeRequest { + payload, + plan: append.plan, + plan_kind: append.plan_kind, + projection: Some(append.output_schema), + state: &append.state.shared, + database_id: append.database_id, + tenant_id: append.tenant_id, + redaction: Some(append.redaction.ctx(&append.state.shared.redaction)), + sequences: Some(&sequences), + }) { + Ok(HttpShaped::Rows(rows)) => result_rows.extend(rows), + Ok(HttpShaped::Passthrough) => result_rows.push(passthrough_json_row(payload)), + Err(e) => return Err(shape_error_to_api(e)), + } + Ok(()) +} diff --git a/nodedb/src/control/server/http/routes/query/materialized/mod.rs b/nodedb/src/control/server/http/routes/query/materialized/mod.rs index a277777ae..b16412010 100644 --- a/nodedb/src/control/server/http/routes/query/materialized/mod.rs +++ b/nodedb/src/control/server/http/routes/query/materialized/mod.rs @@ -2,11 +2,17 @@ //! `/v1/query`: materialized (buffer-then-respond) SQL execution. //! -//! `request.rs` holds the entry point through admission; `shape.rs` runs -//! the per-task dispatch loop and shapes each task's response; `encode.rs` -//! maps errors onto the HTTP surface. +//! - `request.rs`: the entry point through admission. +//! - `shape.rs`: the per-task dispatch loop. +//! - `orchestrated.rs`: the plans that run on the Control Plane instead of +//! the gateway route, cluster array ops among them. +//! - `append.rs`: shaping each task's answer into JSON rows, and the +//! per-task authorization and metering helpers. +//! - `encode.rs`: maps errors onto the HTTP surface. +mod append; mod encode; +mod orchestrated; mod request; mod shape; diff --git a/nodedb/src/control/server/http/routes/query/materialized/orchestrated.rs b/nodedb/src/control/server/http/routes/query/materialized/orchestrated.rs new file mode 100644 index 000000000..3cb1ef0c8 --- /dev/null +++ b/nodedb/src/control/server/http/routes/query/materialized/orchestrated.rs @@ -0,0 +1,122 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Plans that run on the Control Plane instead of the gateway route: +//! +//! - array DDL, which proposes a replicated catalog entry; +//! - a cluster array op, which this node's array coordinator routes to the +//! shards that own its cells; +//! - `INSERT ... SELECT`, an unresolved `MERGE` or `UPDATE ... FROM`, and a +//! governed columnar predicate write, which issue their own writes. +//! +//! None of these is a clone-write shape, so each is authorized without the +//! clone-write check. + +use std::sync::Arc; + +use nodedb_physical::physical_plan::DocumentOp; +use nodedb_physical::physical_task::PhysicalTask; + +use crate::bridge::envelope::{PhysicalPlan, Response, Status}; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::shared::cluster_array_dispatch::{is_cluster_array, run_cluster_array}; +use crate::control::state::SharedState; + +use super::super::super::super::auth::ApiError; +use super::append::authorize_materialized_task; +use super::encode::{gateway_error, response_error}; + +/// The payload an orchestrated plan answered with. +pub(super) struct Orchestrated { + pub(super) payload: Vec, + /// Whether the task is metered. Array DDL is not. + pub(super) metered: bool, +} + +/// Run `task` when it is a Control-Plane orchestrated plan. `None` means the +/// task takes the clone-write gate and the gateway route. +pub(super) async fn run_orchestrated( + shared: &Arc, + identity: &AuthenticatedIdentity, + task: &PhysicalTask, +) -> Result, ApiError> { + let authorize = || authorize_materialized_task(shared, identity, task).map_err(gateway_error); + + if crate::control::array_catalog::ddl::is_array_ddl(&task.plan) { + let response = + crate::control::array_catalog::ddl::run_authorized_array_ddl(shared, authorize()?) + .await + .map_err(gateway_error)?; + return answered(response, false); + } + + if is_cluster_array(&task.plan) { + let payload = run_cluster_array(shared, authorize()?) + .await + .map_err(gateway_error)?; + return Ok(Some(Orchestrated { + payload, + metered: true, + })); + } + + let response = if let PhysicalPlan::Document(DocumentOp::InsertSelect { .. }) = &task.plan { + crate::control::insert_select::run_authorized_insert_select(shared, authorize()?).await + } else if let PhysicalPlan::Document(DocumentOp::Merge { + target_collection: _, + source_collection: _, + source_alias: _, + target_join_col: _, + source_join_col: _, + clauses: _, + returning: _, + resolved_inserts: None, + resolved_insert_identities: _, + source_rows: _, + rls_filters: _, + rls_write_check: _, + resolved_sum_targets: _, + declared_primary_key: _, + }) = &task.plan + { + crate::control::merge_orchestrator::run_authorized_merge(shared, authorize()?).await + } else if let PhysicalPlan::Document(DocumentOp::UpdateFromJoin { + target_collection: _, + source_collection: _, + source_alias: _, + target_join_col: _, + source_join_col: _, + updates: _, + target_filters: _, + returning: _, + source_rows: None, + rls_filters: _, + rls_write_check: _, + resolved_sum_targets: _, + declared_primary_key: _, + }) = &task.plan + { + crate::control::update_from_join_orchestrator::run_authorized_update_from_join( + shared, + authorize()?, + ) + .await + } else if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) { + crate::control::write_resolve::run_authorized_write_resolve(shared, authorize()?, resolver) + .await + } else { + return Ok(None); + }; + answered(response.map_err(gateway_error)?, true) +} + +/// The payload of an orchestrated response. A refusal becomes the HTTP error +/// its code maps to. +fn answered(response: Response, metered: bool) -> Result, ApiError> { + if response.status != Status::Ok { + return Err(response_error(&response)); + } + Ok(Some(Orchestrated { + payload: response.payload.to_vec(), + metered, + })) +} diff --git a/nodedb/src/control/server/http/routes/query/materialized/request.rs b/nodedb/src/control/server/http/routes/query/materialized/request.rs index 71d0bf71b..99ff58e79 100644 --- a/nodedb/src/control/server/http/routes/query/materialized/request.rs +++ b/nodedb/src/control/server/http/routes/query/materialized/request.rs @@ -121,7 +121,7 @@ pub async fn query( .map_err(ApiError::from)?; let tasks = admission.tasks; let output_schema = admission.output_schema; - let _lease_scope = admission.lease_scope; + let lease_scope = admission.lease_scope; if tasks.is_empty() { return Ok(( @@ -133,7 +133,13 @@ pub async fn query( // Track active request for quota accounting. let _request = state.shared.tenant_request_guard(tenant_id); - let result_rows = run_task_loop( + // A statement admitted under a lease this node then loses ends with a + // retryable error: a read mid-flight, a write only before dispatch. + lease_scope.check_not_revoked().map_err(ApiError::from)?; + let read_only = tasks + .iter() + .all(|task| !crate::control::server::shared::write_admission::plan_is_write(&task.plan)); + let run = run_task_loop( tasks, TaskLoopParams { state: &state, @@ -144,8 +150,12 @@ pub async fn query( tenant_id, trace_id, }, - ) - .await?; + ); + let result_rows = if read_only { + lease_scope.guard(run).await.map_err(ApiError::from)?? + } else { + run.await? + }; Ok(( rate_limit_headers, diff --git a/nodedb/src/control/server/http/routes/query/materialized/shape.rs b/nodedb/src/control/server/http/routes/query/materialized/shape.rs index 073777685..d7be9c1f5 100644 --- a/nodedb/src/control/server/http/routes/query/materialized/shape.rs +++ b/nodedb/src/control/server/http/routes/query/materialized/shape.rs @@ -1,33 +1,30 @@ // SPDX-License-Identifier: BUSL-1.1 -//! The per-task dispatch loop: runs each admitted task, orchestrating the -//! Control-Plane-side plan shapes (`InsertSelect`, `Merge`, -//! `UpdateFromJoin`, a governed predicate resolution) directly and routing -//! everything else through the general clone-write / gateway / local-SPSC -//! dispatch path, then shapes each response into the JSON row set. +//! The per-task dispatch loop: runs each admitted task, then shapes each +//! response into the JSON row set. +//! +//! A Control-Plane orchestrated plan runs in `orchestrated`. Every other +//! task takes the clone-write gate, then the gateway route to its owner. use std::sync::Arc; use nodedb_physical::physical_task::PhysicalTask; -use crate::bridge::envelope::Status; use crate::control::gateway::core::QueryContext; use crate::control::security::audit::ArcAuditEmitter; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::request_scope::RequestAuthScope; use crate::control::server::response_shape::redaction::QueryRedaction; -use crate::control::server::response_shape::request::MaterializedShapeRequest; use crate::control::server::response_shape::schema::OutputSchema; -use crate::control::server::response_shape::types::{PlanKind, describe_plan}; -use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; +use crate::control::server::response_shape::types::describe_plan; +use crate::control::server::shared::metering::PlanMeteringInfo; use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; use crate::types::{DatabaseId, TenantId, TraceId}; use super::super::super::super::auth::{ApiError, AppState}; -use super::super::super::result_shape::{ - HttpShaped, passthrough_json_row, shape_error_to_api, shape_http_payload, -}; -use super::encode::{gateway_error, response_error}; +use super::append::{ShapedAppend, append_payload, append_response, meter_task_dispatch}; +use super::encode::gateway_error; +use super::orchestrated::run_orchestrated; /// Everything the per-task loop needs, beyond the tasks themselves. pub(super) struct TaskLoopParams<'a> { @@ -69,199 +66,35 @@ pub(super) async fn run_task_loop( admit_quota_for_dispatch(&state.shared, &scope, info).map_err(gateway_error)?; } let rows_before = result_rows.len(); - // `INSERT ... SELECT` orchestrates on the Control Plane and issues its own - // WAL-backed writes, so the outer per-task WAL append is skipped for it. - // Never a clone-write shape. - if let crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. }, - ) = &task.plan - { - let plan_kind = describe_plan(&task.plan); - let plan_for_shape = task.plan.clone(); - let authorized_task = authorize_materialized_task(&state.shared, identity, &task) - .map_err(gateway_error)?; - let resp = crate::control::insert_select::run_authorized_insert_select( - &state.shared, - authorized_task, - ) - .await - .map_err(gateway_error)?; - append_response( - &mut result_rows, - resp, - ShapedAppend { - plan: &plan_for_shape, - plan_kind, - output_schema: &output_schema, - state, - database_id, - tenant_id, - redaction: &QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape), - }, - )?; - meter_task_dispatch( - &state.shared, - &scope, - &plan_metering_info, - rows_before, - &result_rows, - ); - continue; - } - // Autocommit `MERGE` orchestrates on the Control Plane and issues its own - // writes, so the per-task WAL append below is skipped for it. - if let crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::Merge { - target_collection: _, - source_collection: _, - source_alias: _, - target_join_col: _, - source_join_col: _, - clauses: _, - returning: _, - resolved_inserts: None, - resolved_insert_identities: _, - source_rows: _, - rls_filters: _, - rls_write_check: _, - resolved_sum_targets: _, - declared_primary_key: _, - }, - ) = &task.plan - { - let plan_kind = describe_plan(&task.plan); - let plan_for_shape = task.plan.clone(); - let authorized_task = authorize_materialized_task(&state.shared, identity, &task) - .map_err(gateway_error)?; - let resp = crate::control::merge_orchestrator::run_authorized_merge( - &state.shared, - authorized_task, - ) - .await - .map_err(gateway_error)?; - append_response( - &mut result_rows, - resp, - ShapedAppend { - plan: &plan_for_shape, - plan_kind, - output_schema: &output_schema, - state, - database_id, - tenant_id, - redaction: &QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape), - }, - )?; - meter_task_dispatch( - &state.shared, - &scope, - &plan_metering_info, - rows_before, - &result_rows, - ); - continue; - } + // Captured before dispatch moves `task.plan`. Resolved once per task, + // reused for every payload it produced. + let plan_for_shape = task.plan.clone(); + let redaction = QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape); + let append = ShapedAppend { + plan: &plan_for_shape, + plan_kind: describe_plan(&plan_for_shape), + output_schema: &output_schema, + state, + database_id, + tenant_id, + redaction: &redaction, + }; - // Autocommit `UPDATE ... FROM ` scans the source on its own core and - // ships it into the plan; the orchestrator's own write skips the WAL append below. - if let crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { - target_collection: _, - source_collection: _, - source_alias: _, - target_join_col: _, - source_join_col: _, - updates: _, - target_filters: _, - returning: _, - source_rows: None, - rls_filters: _, - rls_write_check: _, - resolved_sum_targets: _, - declared_primary_key: _, - }, - ) = &task.plan - { - let plan_kind = describe_plan(&task.plan); - let plan_for_shape = task.plan.clone(); - let authorized_task = authorize_materialized_task(&state.shared, identity, &task) - .map_err(gateway_error)?; - let resp = - crate::control::update_from_join_orchestrator::run_authorized_update_from_join( + if let Some(orchestrated) = run_orchestrated(&state.shared, identity, &task).await? { + append_payload(&mut result_rows, &orchestrated.payload, &append)?; + if orchestrated.metered { + meter_task_dispatch( &state.shared, - authorized_task, - ) - .await - .map_err(gateway_error)?; - append_response( - &mut result_rows, - resp, - ShapedAppend { - plan: &plan_for_shape, - plan_kind, - output_schema: &output_schema, - state, - database_id, - tenant_id, - redaction: &QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape), - }, - )?; - meter_task_dispatch( - &state.shared, - &scope, - &plan_metering_info, - rows_before, - &result_rows, - ); - continue; - } - - // A governed columnar predicate UPDATE/DELETE resolves to a concrete row set - // before proposing, skipping the WAL append below; local (non-Raft) path skips this. - if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) - && state.shared.async_raft_proposer().is_some() - { - let plan_kind = describe_plan(&task.plan); - let plan_for_shape = task.plan.clone(); - let authorized_task = authorize_materialized_task(&state.shared, identity, &task) - .map_err(gateway_error)?; - let resp = crate::control::write_resolve::run_authorized_write_resolve( - &state.shared, - authorized_task, - resolver, - ) - .await - .map_err(gateway_error)?; - append_response( - &mut result_rows, - resp, - ShapedAppend { - plan: &plan_for_shape, - plan_kind, - output_schema: &output_schema, - state, - database_id, - tenant_id, - redaction: &QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape), - }, - )?; - meter_task_dispatch( - &state.shared, - &scope, - &plan_metering_info, - rows_before, - &result_rows, - ); + &scope, + &plan_metering_info, + rows_before, + &result_rows, + ); + } continue; } - // Captured before dispatch moves `task.plan` — needed by shaping below. - let plan_kind = describe_plan(&task.plan); - let plan_for_shape = task.plan.clone(); - // Resolved once per task, reused for every payload it produced. - let redaction = QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape); - // Clone CoW write-path interception, then authorization, run once // per task before dispatch — same protocol-neutral gate every // transport runs. @@ -281,19 +114,7 @@ pub(super) async fn run_task_loop( .map_err(gateway_error)? { crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(resp) => { - append_response( - &mut result_rows, - resp, - ShapedAppend { - plan: &plan_for_shape, - plan_kind, - output_schema: &output_schema, - state, - database_id, - tenant_id, - redaction: &redaction, - }, - )?; + append_response(&mut result_rows, resp, &append)?; meter_task_dispatch( &state.shared, &scope, @@ -308,63 +129,21 @@ pub(super) async fn run_task_loop( } }; - // Prefer gateway (cluster-aware, owns WAL durability), else fall back to - // local SPSC dispatch, where WAL append precedes enqueue so LSN order matches. - let payloads = match state.shared.gateway.get() { - Some(gw) => { - let gw_ctx = QueryContext { - tenant_id: checked.tenant_id(), - trace_id, - database_id, - txn_id: None, - }; - gw.execute(&gw_ctx, checked).await.map_err(gateway_error)? - } - None => { - // Single-node boot: gateway not yet initialised — dispatch locally. - let response = - crate::control::server::dispatch_utils::dispatch_authorized_durable_write( - &state.shared, - checked, - trace_id, - ) - .await - .map_err(gateway_error)?; - if response.status != Status::Ok { - return Err(response_error(&response)); - } - vec![response.payload.to_vec()] - } - }; - - // HTTP carries no session, so `currval` reports "not yet called - // in this session" while `nextval` still advances the registry — - // the same answer the plan-time adapter gives this transport. - let sequences = crate::control::sequence::SessionSequenceAccess::for_session( - &state.shared, - None, + // The gateway routes the task to its owner and owns WAL durability there. + let gateway = state.shared.installed_gateway().map_err(gateway_error)?; + let gw_ctx = QueryContext { + tenant_id: checked.tenant_id(), + trace_id, database_id, - tenant_id, - ); + txn_id: None, + linearizable: true, + }; + let payloads = gateway + .execute(&gw_ctx, checked) + .await + .map_err(gateway_error)?; for payload in &payloads { - if payload.is_empty() { - continue; - } - match shape_http_payload(MaterializedShapeRequest { - payload, - plan: &plan_for_shape, - plan_kind, - projection: Some(&output_schema), - state: &state.shared, - database_id, - tenant_id, - redaction: Some(redaction.ctx(&state.shared.redaction)), - sequences: Some(&sequences), - }) { - Ok(HttpShaped::Rows(rows)) => result_rows.extend(rows), - Ok(HttpShaped::Passthrough) => result_rows.push(passthrough_json_row(payload)), - Err(e) => return Err(shape_error_to_api(e)), - } + append_payload(&mut result_rows, payload, &append)?; } meter_task_dispatch( &state.shared, @@ -377,95 +156,3 @@ pub(super) async fn run_task_loop( Ok(result_rows) } - -/// Authorize one task with no clone-write check — used only by the -/// Control-Plane orchestrator branches ahead of the general dispatch tail, -/// whose plan shapes (`InsertSelect`, `Merge`, `UpdateFromJoin`, a governed -/// predicate resolution) are never clone-write shapes. -fn authorize_materialized_task( - shared: &crate::control::state::SharedState, - identity: &AuthenticatedIdentity, - task: &PhysicalTask, -) -> crate::Result { - let emitter = ArcAuditEmitter(Arc::clone(&shared.audit)); - crate::control::server::shared::authorization::authorize_task_set( - identity, - std::slice::from_ref(task), - &shared.permissions, - &shared.roles, - &emitter, - ) - .map_err(crate::Error::from)? - .into_tasks() - .into_iter() - .next() - .ok_or_else(|| crate::Error::Internal { - detail: "authorization returned an empty capability set".into(), - }) -} - -/// Meter one task's dispatch after its rows are appended to `result_rows` — -/// the row count is the delta since `rows_before`. -fn meter_task_dispatch( - state: &crate::control::state::SharedState, - scope: &RequestAuthScope<'_>, - info: &Option, - rows_before: usize, - result_rows: &[serde_json::Value], -) { - if let Some(info) = info { - let task_rows = (result_rows.len() - rows_before) as u64; - meter_dispatch(state, scope, info, Some(task_rows)); - } -} - -/// Everything one orchestrated task's response needs to be shaped and -/// appended. Grouped so the append helper stays within the argument budget as -/// it gained the per-statement redaction resolution. -struct ShapedAppend<'a> { - plan: &'a crate::bridge::envelope::PhysicalPlan, - plan_kind: PlanKind, - output_schema: &'a OutputSchema, - state: &'a AppState, - database_id: DatabaseId, - tenant_id: TenantId, - redaction: &'a QueryRedaction, -} - -fn append_response( - result_rows: &mut Vec, - response: crate::bridge::envelope::Response, - append: ShapedAppend<'_>, -) -> Result<(), ApiError> { - if response.status != Status::Ok { - return Err(response_error(&response)); - } - let payload = response.payload.to_vec(); - if payload.is_empty() { - return Ok(()); - } - // HTTP carries no session: `nextval` advances the registry, `currval` - // reports "not yet called in this session". - let sequences = crate::control::sequence::SessionSequenceAccess::for_session( - &append.state.shared, - None, - append.database_id, - append.tenant_id, - ); - match shape_http_payload(MaterializedShapeRequest { - payload: &payload, - plan: append.plan, - plan_kind: append.plan_kind, - projection: Some(append.output_schema), - state: &append.state.shared, - database_id: append.database_id, - tenant_id: append.tenant_id, - redaction: Some(append.redaction.ctx(&append.state.shared.redaction)), - sequences: Some(&sequences), - }) { - Ok(HttpShaped::Rows(rows)) => result_rows.extend(rows), - Ok(HttpShaped::Passthrough) => result_rows.push(passthrough_json_row(&payload)), - Err(e) => return Err(shape_error_to_api(e)), - } - Ok(()) -} diff --git a/nodedb/src/control/server/http/routes/query/ndjson.rs b/nodedb/src/control/server/http/routes/query/ndjson.rs index 9d99551af..4480375ca 100644 --- a/nodedb/src/control/server/http/routes/query/ndjson.rs +++ b/nodedb/src/control/server/http/routes/query/ndjson.rs @@ -13,6 +13,7 @@ use crate::control::server::response_shape::redaction::QueryRedaction; use crate::control::server::response_shape::request::MaterializedShapeRequest; use crate::control::server::response_shape::types::describe_plan; use crate::control::server::shared::authorization::authorize_database; +use crate::control::server::shared::cluster_array_dispatch::{is_cluster_array, run_cluster_array}; use crate::control::server::shared::metering::{ DetachedMeterGuard, PlanMeteringInfo, meter_dispatch, }; @@ -181,12 +182,25 @@ pub async fn query_ndjson( Err(error) => return ApiError::from(error).into_response(), } - let _lease_scope = lease_scope; + // Held until the materialized body is built. + let Some(lease_scope) = lease_scope.take() else { + return ApiError::from(crate::Error::Internal { + detail: "query lease scope missing before NDJSON dispatch".into(), + }) + .into_response(); + }; let mut ndjson = String::new(); // Checked once, not per task: keeps per-task extraction a no-op when metering is // disabled. This fallback fully materializes the body, so it meters like `/v1/query`. let metering_enabled = state.shared.metering_config.enabled; for task in tasks { + // A lease this node lost ends the body with a retryable error line + // before the next task dispatches. + if let Err(revoked) = lease_scope.check_not_revoked() { + ndjson.push_str(&serde_json::json!({"error": revoked.to_string()}).to_string()); + ndjson.push('\n'); + break; + } // A spent hard quota refuses the task before it runs; reported as an error // line and the task skipped, matching this stream's error reporting. let plan_metering_info = metering_enabled.then(|| PlanMeteringInfo::extract(&task.plan)); @@ -204,122 +218,154 @@ pub async fn query_ndjson( // Resolved once per task, reused for every payload it produced. let redaction = QueryRedaction::for_plan(tenant_id, scope.auth(), &plan_for_shape); - let dispatch_result: crate::Result>> = if matches!( - &task.plan, - crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. } - ) - ) { - match authorize_ndjson_task(&state.shared, &identity, &task) { - Ok(authorized_task) => crate::control::insert_select::run_authorized_insert_select( - &state.shared, - authorized_task, - ) - .await - .map(|response| vec![response.payload.to_vec()]), - Err(e) => Err(e), - } - } else if matches!( - &task.plan, - crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::Merge { - resolved_inserts: None, - .. - } - ) - ) { - match authorize_ndjson_task(&state.shared, &identity, &task) { - Ok(authorized_task) => crate::control::merge_orchestrator::run_authorized_merge( - &state.shared, - authorized_task, - ) - .await - .map(|response| vec![response.payload.to_vec()]), - Err(e) => Err(e), - } - } else if matches!( - &task.plan, - crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { - source_rows: None, - .. - } - ) - ) { - match authorize_ndjson_task(&state.shared, &identity, &task) { - Ok(authorized_task) => { - crate::control::update_from_join_orchestrator::run_authorized_update_from_join( - &state.shared, - authorized_task, + // A read is cancelled mid-flight once this node loses a lease it + // holds. A write is not: its outcome will be unknown to the client. + let task_is_write = + crate::control::server::shared::write_admission::plan_is_write(&task.plan); + let dispatch = async { + let result: crate::Result>> = + if crate::control::array_catalog::ddl::is_array_ddl(&task.plan) { + match authorize_ndjson_task(&state.shared, &identity, &task) { + Ok(authorized_task) => { + crate::control::array_catalog::ddl::run_authorized_array_ddl( + &state.shared, + authorized_task, + ) + .await + .map(|response| vec![response.payload.to_vec()]) + } + Err(e) => Err(e), + } + } else if is_cluster_array(&task.plan) { + // This node's array coordinator routes the op to the + // shards that own its cells. + match authorize_ndjson_task(&state.shared, &identity, &task) { + Ok(authorized_task) => run_cluster_array(&state.shared, authorized_task) + .await + .map(|payload| vec![payload]), + Err(e) => Err(e), + } + } else if matches!( + &task.plan, + crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. } ) - .await - .map(|response| vec![response.payload.to_vec()]) - } - Err(e) => Err(e), - } - } else if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) - && state.shared.async_raft_proposer().is_some() - { - // A governed columnar predicate UPDATE/DELETE resolves to a concrete row set - // before proposing; local (non-Raft) path skips this branch. - match authorize_ndjson_task(&state.shared, &identity, &task) { - Ok(authorized_task) => crate::control::write_resolve::run_authorized_write_resolve( - &state.shared, - authorized_task, - resolver, - ) - .await - .map(|response| vec![response.payload.to_vec()]), - Err(e) => Err(e), - } - } else { - // Clone CoW write-path interception, then authorization, run once - // per task before dispatch — same protocol-neutral gate every - // transport runs. - let emitter = crate::control::security::audit::ArcAuditEmitter(std::sync::Arc::clone( - &state.shared.audit, - )); - match crate::control::server::shared::clone_write::intercept_and_authorize( - crate::control::server::shared::clone_write::InterceptAndAuthorizeParams { - state: &state.shared, - task, - identity: &identity, - tenant_id, - permissions: &state.shared.permissions, - roles: &state.shared.roles, - emitter: &emitter, - }, - ) - .await - { - Ok(crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled( - resp, - )) => Ok(vec![resp.payload.to_vec()]), - Ok(crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed( - checked, - )) => match state.shared.gateway.get() { - Some(gw) => { - let gw_ctx = QueryContext { - tenant_id: checked.tenant_id(), - trace_id, - database_id, - txn_id: None, - }; - gw.execute(&gw_ctx, checked).await + ) { + match authorize_ndjson_task(&state.shared, &identity, &task) { + Ok(authorized_task) => { + crate::control::insert_select::run_authorized_insert_select( + &state.shared, + authorized_task, + ) + .await + .map(|response| vec![response.payload.to_vec()]) + } + Err(e) => Err(e), + } + } else if matches!( + &task.plan, + crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::Merge { + resolved_inserts: None, + .. + } + ) + ) { + match authorize_ndjson_task(&state.shared, &identity, &task) { + Ok(authorized_task) => { + crate::control::merge_orchestrator::run_authorized_merge( + &state.shared, + authorized_task, + ) + .await + .map(|response| vec![response.payload.to_vec()]) + } + Err(e) => Err(e), } - // A write takes the durable route, a read the read route. - None => { - crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( + } else if matches!( + &task.plan, + crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { + source_rows: None, + .. + } + ) + ) { + match authorize_ndjson_task(&state.shared, &identity, &task) { + Ok(authorized_task) => { + crate::control::update_from_join_orchestrator::run_authorized_update_from_join( &state.shared, - checked, - trace_id, + authorized_task, ) .await .map(|response| vec![response.payload.to_vec()]) } - }, - Err(e) => Err(e), - } + Err(e) => Err(e), + } + } else if let Some(resolver) = + crate::control::write_resolve::resolver_for_plan(&task.plan) + { + // A governed columnar predicate UPDATE/DELETE resolves to a concrete row set + // before proposing. + match authorize_ndjson_task(&state.shared, &identity, &task) { + Ok(authorized_task) => { + crate::control::write_resolve::run_authorized_write_resolve( + &state.shared, + authorized_task, + resolver, + ) + .await + .map(|response| vec![response.payload.to_vec()]) + } + Err(e) => Err(e), + } + } else { + // Clone CoW write-path interception, then authorization, run once + // per task before dispatch — same protocol-neutral gate every + // transport runs. + let emitter = crate::control::security::audit::ArcAuditEmitter( + std::sync::Arc::clone(&state.shared.audit), + ); + match crate::control::server::shared::clone_write::intercept_and_authorize( + crate::control::server::shared::clone_write::InterceptAndAuthorizeParams { + state: &state.shared, + task, + identity: &identity, + tenant_id, + permissions: &state.shared.permissions, + roles: &state.shared.roles, + emitter: &emitter, + }, + ) + .await + { + Ok(crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled( + resp, + )) => Ok(vec![resp.payload.to_vec()]), + Ok(crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed( + checked, + )) => match state.shared.installed_gateway() { + Ok(gateway) => { + let gw_ctx = QueryContext { + tenant_id: checked.tenant_id(), + trace_id, + database_id, + txn_id: None, + linearizable: true, + }; + gateway.execute(&gw_ctx, checked).await + } + Err(error) => Err(error), + }, + Err(e) => Err(e), + } + }; + result + }; + let dispatch_result = if task_is_write { + dispatch.await + } else { + lease_scope.guard(dispatch).await.and_then(|result| result) }; match dispatch_result { diff --git a/nodedb/src/control/server/http/routes/query_stream.rs b/nodedb/src/control/server/http/routes/query_stream.rs index 9ca180aab..27cd25374 100644 --- a/nodedb/src/control/server/http/routes/query_stream.rs +++ b/nodedb/src/control/server/http/routes/query_stream.rs @@ -24,7 +24,6 @@ use crate::control::gateway::GatewayErrorMap; use crate::control::gateway::core::QueryContext; use crate::control::security::audit::ArcAuditEmitter; use crate::control::security::identity::AuthenticatedIdentity; -use crate::control::server::exchange::gather::gather_all_cores_stream_authorized; use crate::control::server::exchange::streamable::streamable_gather_child; use crate::control::server::response_shape::cell::row_to_wire_json; use crate::control::server::response_shape::compose::shape_decoded_rows; @@ -48,8 +47,8 @@ use super::super::auth::AppState; /// shaper does not carry, so the materialized path answers instead. /// /// Returns `Ok(Some((stream, limit)))` when eligible, `Ok(None)` when the -/// caller should fall back to the materialized path, or `Err` when the -/// stream could not be opened. +/// caller must fall back to the materialized path, or `Err` when the +/// stream cannot be opened. pub(super) async fn try_open_stream( state: &AppState, tasks: &[PhysicalTask], @@ -98,17 +97,15 @@ pub(super) async fn try_open_stream( } }; - let stream = if let Some(gw) = state.shared.gateway.get() { - let ctx = QueryContext { - tenant_id: task.tenant_id, - trace_id, - database_id, - txn_id: None, - }; - gw.execute_stream(&ctx, checked_child).await - } else { - gather_all_cores_stream_authorized(&state.shared, checked_child.into_authorized(), trace_id) - }?; + let gateway = state.shared.installed_gateway()?; + let ctx = QueryContext { + tenant_id: task.tenant_id, + trace_id, + database_id, + txn_id: None, + linearizable: true, + }; + let stream = gateway.execute_stream(&ctx, checked_child).await?; Ok(Some((stream, limit))) } @@ -120,7 +117,7 @@ pub(super) struct NdjsonBody { pub limit: usize, pub projection: Option, /// The statement's redaction inputs, resolved ONCE before the first batch - /// is pulled. Re-resolving per batch would risk the first NDJSON lines + /// is pulled. Re-resolving per batch will risk the first NDJSON lines /// going out unredacted. pub redaction: Option, /// Owned so the body, which outlives the handler frame, can reach the @@ -155,9 +152,9 @@ pub(super) fn ndjson_body_stream( // The body owns this scope for its complete polling lifetime. Dropping // the body on completion or client disconnect releases descriptors only // after the ResultStream is no longer reachable. - let _lease_scope = lease_scope; + let lease_scope = lease_scope; // Owned by this generator for its whole polling lifetime, exactly - // like `_lease_scope` above — whether the stream runs to completion, + // like `lease_scope` above — whether the stream runs to completion, // ends on a mid-stream error, or is dropped early by a disconnected // client, this guard's `Drop` fires and bills exactly the rows // accumulated into it via `add_rows` below, never more. @@ -165,7 +162,13 @@ pub(super) fn ndjson_body_stream( let mut emitted: usize = 0; let mut batches = stream; while emitted < limit { - let batch = match batches.next().await { + // A lease this node loses mid-stream ends the body with a + // retryable error line rather than rows from a stale descriptor. + let next = lease_scope + .guard(batches.next()) + .await + .unwrap_or_else(|revoked| Some(Err(revoked))); + let batch = match next { None => break, Some(Ok(b)) => b, Some(Err(e)) => { @@ -348,7 +351,7 @@ mod tests { /// The streaming metering contract: a client that disconnects mid-stream /// must be billed for exactly the rows it received, never for the rows a - /// full scan would have produced. Polls only 3 of 2000 available rows, + /// full scan will produce. Polls only 3 of 2000 available rows, /// then drops the stream without reaching the end of the generator — /// exactly what happens when axum drops a response body because the /// connected client went away. @@ -408,7 +411,7 @@ mod tests { // The stream (and the guard it owns) must be dropped before draining, // which is what a client disconnecting mid-response does. `pin_mut!` // shadows the binding with a `Pin<&mut _>`, so `drop`ping that name - // would only release the borrow and leave the stream — and its + // will only release the borrow and leave the stream — and its // pending row count — alive until end of scope. An inner block drops // the real value. { diff --git a/nodedb/src/control/server/http/routes/stream_poll.rs b/nodedb/src/control/server/http/routes/stream_poll.rs index 54de19cfe..58e46a479 100644 --- a/nodedb/src/control/server/http/routes/stream_poll.rs +++ b/nodedb/src/control/server/http/routes/stream_poll.rs @@ -49,7 +49,7 @@ pub struct PollParams { pub struct PollResponse { /// Events in this batch. pub events: Vec, - /// Per-partition latest canonical `:` offset in this batch. + /// Per-partition latest canonical `::` offset in this batch. pub partition_offsets: std::collections::BTreeMap, /// Total events returned. pub count: usize, @@ -168,7 +168,7 @@ pub async fn poll_stream( limit, }; - let mut result = match consume_stream(&state.shared, &consume_params) { + let mut result = match consume_stream(&state.shared, &consume_params).await { Ok(r) => r, Err(ConsumeError::RemotePartition { leader_node, .. }) => { // Forward to remote node. @@ -180,6 +180,10 @@ pub async fn poll_stream( .await { Ok(r) => r, + Err(ConsumeError::OffsetOutOfRange { + partition_id, + available_from, + }) => return reset_required(partition_id, available_from), Err(e) => { return ( StatusCode::BAD_GATEWAY, @@ -189,6 +193,10 @@ pub async fn poll_stream( } } } + Err(ConsumeError::OffsetOutOfRange { + partition_id, + available_from, + }) => return reset_required(partition_id, available_from), Err(ConsumeError::BufferEmpty(_)) => ConsumeResult { events: Vec::new(), partition_offsets: Vec::new(), @@ -239,3 +247,20 @@ pub async fn poll_stream( ) .into_response() } + +/// `409 Conflict`: the consumer group's offset lies below the events any +/// reachable replica holds. The consumer commits `available_from` or later. +fn reset_required( + partition_id: u32, + available_from: crate::event::cdc::CdcOffset, +) -> axum::response::Response { + ( + StatusCode::CONFLICT, + Json(serde_json::json!({ + "error": "reset_required", + "partition": partition_id, + "available_from": available_from.token(), + })), + ) + .into_response() +} diff --git a/nodedb/src/control/server/http/routes/stream_sse.rs b/nodedb/src/control/server/http/routes/stream_sse.rs index 80c4b0a63..c8ec8e96d 100644 --- a/nodedb/src/control/server/http/routes/stream_sse.rs +++ b/nodedb/src/control/server/http/routes/stream_sse.rs @@ -6,7 +6,7 @@ //! //! Pushes events as Server-Sent Events in real-time. On each poll cycle, //! reads new events from the buffer since the consumer group's committed -//! offset. The consumer should COMMIT OFFSET via SQL to advance the cursor. +//! offset. The consumer must COMMIT OFFSET via SQL to advance the cursor. use std::convert::Infallible; use std::sync::Arc; @@ -193,7 +193,7 @@ pub async fn stream_events( limit: 100, }; - let mut result = match consume_stream(&state.shared, &consume_params) { + let mut result = match consume_stream(&state.shared, &consume_params).await { Ok(r) => r, Err(ConsumeError::RemotePartition { leader_node, .. }) => { match crate::event::cdc::consume::consume_remote( @@ -204,6 +204,12 @@ pub async fn stream_events( .await { Ok(r) => r, + Err(e @ ConsumeError::OffsetOutOfRange { .. }) => { + yield Ok(Event::default() + .event("reset_required") + .data(e.to_string())); + return; + } Err(e) => { yield Ok(Event::default() .event("error") @@ -212,6 +218,12 @@ pub async fn stream_events( } } } + Err(e @ ConsumeError::OffsetOutOfRange { .. }) => { + yield Ok(Event::default() + .event("reset_required") + .data(e.to_string())); + return; + } Err(ConsumeError::BufferEmpty(_)) => { // No events yet — wait and retry. tokio::time::sleep(Duration::from_millis(100)).await; diff --git a/nodedb/src/control/server/http/routes/wasm_upload.rs b/nodedb/src/control/server/http/routes/wasm_upload.rs index ea6d7afc7..a0db0f971 100644 --- a/nodedb/src/control/server/http/routes/wasm_upload.rs +++ b/nodedb/src/control/server/http/routes/wasm_upload.rs @@ -97,22 +97,15 @@ pub async fn upload_wasm( // `func` was fetched using `database_id`; retaining that descriptor in // the proposal keeps the update within the authenticated database scope. let entry = crate::control::catalog_entry::CatalogEntry::PutFunction(Box::new(func)); - let outcome = - match crate::control::metadata_proposer::propose_catalog_entry(&state.shared, &entry) { - Ok(outcome) => outcome, - Err(e) => { - return Ok(( - StatusCode::INTERNAL_SERVER_ERROR, - format!("metadata propose error: {e}"), - ) - .into_response()); - } - }; - crate::control::catalog_entry::apply::local::apply_locally_if_needed( - &state.shared, - &entry, - outcome, - ); + if let Err(e) = + crate::control::metadata_proposer::propose_catalog_entry_async(&state.shared, &entry).await + { + return Ok(( + StatusCode::INTERNAL_SERVER_ERROR, + format!("metadata propose error: {e}"), + ) + .into_response()); + } state.shared.audit_record_with_db( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs b/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs index 4fd83ccea..57611f31c 100644 --- a/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs +++ b/nodedb/src/control/server/http/routes/ws_rpc/execute_sql.rs @@ -10,6 +10,7 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::request_scope::RequestAuthScope; use crate::control::server::response_shape::redaction::{QueryRedaction, redact_decoded_value}; use crate::control::server::shared::authorization::authorize_database; +use crate::control::server::shared::cluster_array_dispatch::{is_cluster_array, run_cluster_array}; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::server::shared::plan_admission::{ PlanAdmissionRequest, plan_authorize_and_admit, @@ -20,8 +21,9 @@ use crate::types::{DatabaseId, TraceId}; /// Execute SQL and return result as JSON. /// -/// Routes through the gateway when available; falls back to local SPSC -/// dispatch on single-node boot before the gateway is initialised. +/// A Control-Plane orchestrated task (array DDL, a cluster array op, +/// `INSERT ... SELECT`) runs here. Every other task routes through the +/// gateway. pub async fn execute_sql( shared: &Arc, query_ctx: &crate::control::planner::context::QueryContext, @@ -63,276 +65,354 @@ pub async fn execute_sql( }) .await?; let tasks = admission.tasks; - let _lease_scope = admission.lease_scope; - - let _request = shared.tenant_request_guard(tenant_id); - - // Resolved once and reused at every decode site — orchestrated rows and plain - // dispatch rows must be redacted by the same policy snapshot. - let redaction = - QueryRedaction::for_plans(tenant_id, scope.auth(), tasks.iter().map(|t| &t.plan)); - - let mut results = Vec::new(); - // Checked once, not per task: keeps per-task extraction a no-op when metering - // is disabled (the default). - let metering_enabled = shared.metering_config.enabled; - for task in tasks { - // Extracted before `task.plan` is cloned/moved; `results.len()` gives this - // task's row-count baseline for the delta metered below. - let plan_metering_info = metering_enabled.then(|| PlanMeteringInfo::extract(&task.plan)); - // A spent hard quota refuses the task before it runs; charging below is - // success-path only and never refuses. - if let Some(info) = &plan_metering_info { - admit_quota_for_dispatch(shared, &scope, info)?; - } - let rows_before = results.len(); - - // `INSERT ... SELECT` orchestrates on the Control Plane, never dispatched - // to the Data Plane as a single op, and is never a clone-write shape. - if let crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. }, - ) = &task.plan - { - let authorized_task = authorize_ws_rpc_task(shared, identity, &task)?; - match crate::control::insert_select::run_authorized_insert_select( - shared, - authorized_task, - ) - .await - { - Ok(resp) => { - let payload = resp.payload.to_vec(); - if !payload.is_empty() { - let json = - crate::data::executor::response_codec::decode_payload_to_json(&payload); - match sonic_rs::from_str::(&json) { - Ok(mut v) => { - redact_decoded_value(Some(&redaction), &shared.redaction, &mut v); - results.push(v); - } - Err(_) => results.push(serde_json::Value::String(json)), + let lease_scope = admission.lease_scope; + + // A statement admitted under a lease this node then loses ends with a + // retryable error: a read mid-flight, a write only before dispatch. + lease_scope.check_not_revoked()?; + let read_only = tasks + .iter() + .all(|task| !crate::control::server::shared::write_admission::plan_is_write(&task.plan)); + let body = async { + let _request = shared.tenant_request_guard(tenant_id); + + // Resolved once and reused at every decode site — orchestrated rows and plain + // dispatch rows must be redacted by the same policy snapshot. + let redaction = + QueryRedaction::for_plans(tenant_id, scope.auth(), tasks.iter().map(|t| &t.plan)); + + let mut results = Vec::new(); + // Checked once, not per task: keeps per-task extraction a no-op when metering + // is disabled (the default). + let metering_enabled = shared.metering_config.enabled; + for task in tasks { + // Extracted before `task.plan` is cloned/moved; `results.len()` gives this + // task's row-count baseline for the delta metered below. + let plan_metering_info = + metering_enabled.then(|| PlanMeteringInfo::extract(&task.plan)); + // A spent hard quota refuses the task before it runs; charging below is + // success-path only and never refuses. + if let Some(info) = &plan_metering_info { + admit_quota_for_dispatch(shared, &scope, info)?; + } + let rows_before = results.len(); + + // Array DDL proposes a replicated catalog entry, never a core task. + if crate::control::array_catalog::ddl::is_array_ddl(&task.plan) { + let authorized_task = authorize_ws_rpc_task(shared, identity, &task)?; + let resp = crate::control::array_catalog::ddl::run_authorized_array_ddl( + shared, + authorized_task, + ) + .await?; + let json = + crate::data::executor::response_codec::decode_payload_to_json(&resp.payload); + match sonic_rs::from_str::(&json) { + Ok(value) => results.push(value), + Err(_) => results.push(serde_json::Value::String(json)), + } + continue; + } + + // A cluster array op runs through this node's array coordinator, + // which routes it to the shards that own its cells. + if is_cluster_array(&task.plan) { + let authorized_task = authorize_ws_rpc_task(shared, identity, &task)?; + let payload = run_cluster_array(shared, authorized_task).await?; + if !payload.is_empty() { + let json = + crate::data::executor::response_codec::decode_payload_to_json(&payload); + match sonic_rs::from_str::(&json) { + Ok(mut v) => { + redact_decoded_value(Some(&redaction), &shared.redaction, &mut v); + results.push(v); } + Err(_) => results.push(serde_json::Value::String(json)), } } - Err(e) => return Err(e), + meter_task(shared, &scope, &plan_metering_info, rows_before, &results); + continue; } - meter_task(shared, &scope, &plan_metering_info, rows_before, &results); - continue; - } - // Autocommit `MERGE` orchestrates on the Control Plane, never dispatched - // to the Data Plane as a single op. - if let crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::Merge { - target_collection: _, - source_collection: _, - source_alias: _, - target_join_col: _, - source_join_col: _, - clauses: _, - returning: _, - resolved_inserts: None, - resolved_insert_identities: _, - source_rows: _, - rls_filters: _, - rls_write_check: _, - resolved_sum_targets: _, - declared_primary_key: _, - }, - ) = &task.plan - { - let authorized_task = authorize_ws_rpc_task(shared, identity, &task)?; - match crate::control::merge_orchestrator::run_authorized_merge(shared, authorized_task) - .await + // `INSERT ... SELECT` orchestrates on the Control Plane, never dispatched + // to the Data Plane as a single op, and is never a clone-write shape. + if let crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. }, + ) = &task.plan { - Ok(resp) => { - let payload = resp.payload.to_vec(); - if !payload.is_empty() { - let json = - crate::data::executor::response_codec::decode_payload_to_json(&payload); - match sonic_rs::from_str::(&json) { - Ok(mut v) => { - redact_decoded_value(Some(&redaction), &shared.redaction, &mut v); - results.push(v); + let authorized_task = authorize_ws_rpc_task(shared, identity, &task)?; + match crate::control::insert_select::run_authorized_insert_select( + shared, + authorized_task, + ) + .await + { + Ok(resp) => { + let payload = resp.payload.to_vec(); + if !payload.is_empty() { + let json = + crate::data::executor::response_codec::decode_payload_to_json( + &payload, + ); + match sonic_rs::from_str::(&json) { + Ok(mut v) => { + redact_decoded_value( + Some(&redaction), + &shared.redaction, + &mut v, + ); + results.push(v); + } + Err(_) => results.push(serde_json::Value::String(json)), } - Err(_) => results.push(serde_json::Value::String(json)), } } + Err(e) => return Err(e), } - Err(e) => return Err(e), + meter_task(shared, &scope, &plan_metering_info, rows_before, &results); + continue; } - meter_task(shared, &scope, &plan_metering_info, rows_before, &results); - continue; - } - // Autocommit `UPDATE ... FROM ` scans the source on its own core and - // ships it into the plan, never dispatched as a single op. - if let crate::bridge::envelope::PhysicalPlan::Document( - nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { - target_collection: _, - source_collection: _, - source_alias: _, - target_join_col: _, - source_join_col: _, - updates: _, - target_filters: _, - returning: _, - source_rows: None, - rls_filters: _, - rls_write_check: _, - resolved_sum_targets: _, - declared_primary_key: _, - }, - ) = &task.plan - { - let authorized_task = authorize_ws_rpc_task(shared, identity, &task)?; - match crate::control::update_from_join_orchestrator::run_authorized_update_from_join( - shared, - authorized_task, - ) - .await + // Autocommit `MERGE` orchestrates on the Control Plane, never dispatched + // to the Data Plane as a single op. + if let crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::Merge { + target_collection: _, + source_collection: _, + source_alias: _, + target_join_col: _, + source_join_col: _, + clauses: _, + returning: _, + resolved_inserts: None, + resolved_insert_identities: _, + source_rows: _, + rls_filters: _, + rls_write_check: _, + resolved_sum_targets: _, + declared_primary_key: _, + }, + ) = &task.plan { - Ok(resp) => { - let payload = resp.payload.to_vec(); - if !payload.is_empty() { - let json = - crate::data::executor::response_codec::decode_payload_to_json(&payload); - match sonic_rs::from_str::(&json) { - Ok(mut v) => { - redact_decoded_value(Some(&redaction), &shared.redaction, &mut v); - results.push(v); + let authorized_task = authorize_ws_rpc_task(shared, identity, &task)?; + match crate::control::merge_orchestrator::run_authorized_merge( + shared, + authorized_task, + ) + .await + { + Ok(resp) => { + let payload = resp.payload.to_vec(); + if !payload.is_empty() { + let json = + crate::data::executor::response_codec::decode_payload_to_json( + &payload, + ); + match sonic_rs::from_str::(&json) { + Ok(mut v) => { + redact_decoded_value( + Some(&redaction), + &shared.redaction, + &mut v, + ); + results.push(v); + } + Err(_) => results.push(serde_json::Value::String(json)), } - Err(_) => results.push(serde_json::Value::String(json)), } } + Err(e) => return Err(e), } - Err(e) => return Err(e), + meter_task(shared, &scope, &plan_metering_info, rows_before, &results); + continue; } - meter_task(shared, &scope, &plan_metering_info, rows_before, &results); - continue; - } - // A governed columnar predicate UPDATE/DELETE resolves to a concrete row set - // before proposing; local (non-Raft) path skips this. - if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) - && shared.async_raft_proposer().is_some() - { - let authorized_task = authorize_ws_rpc_task(shared, identity, &task)?; - match crate::control::write_resolve::run_authorized_write_resolve( - shared, - authorized_task, - resolver, - ) - .await + // Autocommit `UPDATE ... FROM ` scans the source on its own core and + // ships it into the plan, never dispatched as a single op. + if let crate::bridge::envelope::PhysicalPlan::Document( + nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { + target_collection: _, + source_collection: _, + source_alias: _, + target_join_col: _, + source_join_col: _, + updates: _, + target_filters: _, + returning: _, + source_rows: None, + rls_filters: _, + rls_write_check: _, + resolved_sum_targets: _, + declared_primary_key: _, + }, + ) = &task.plan { - Ok(resp) => { - let payload = resp.payload.to_vec(); - if !payload.is_empty() { - let json = - crate::data::executor::response_codec::decode_payload_to_json(&payload); - match sonic_rs::from_str::(&json) { - Ok(mut v) => { - redact_decoded_value(Some(&redaction), &shared.redaction, &mut v); - results.push(v); + let authorized_task = authorize_ws_rpc_task(shared, identity, &task)?; + match crate::control::update_from_join_orchestrator::run_authorized_update_from_join( + shared, + authorized_task, + ) + .await + { + Ok(resp) => { + let payload = resp.payload.to_vec(); + if !payload.is_empty() { + let json = + crate::data::executor::response_codec::decode_payload_to_json(&payload); + match sonic_rs::from_str::(&json) { + Ok(mut v) => { + redact_decoded_value(Some(&redaction), &shared.redaction, &mut v); + results.push(v); + } + Err(_) => results.push(serde_json::Value::String(json)), } - Err(_) => results.push(serde_json::Value::String(json)), } } + Err(e) => return Err(e), } - Err(e) => return Err(e), + meter_task(shared, &scope, &plan_metering_info, rows_before, &results); + continue; } - meter_task(shared, &scope, &plan_metering_info, rows_before, &results); - continue; - } - // Clone CoW write-path interception, then authorization, run once per - // task before dispatch — same protocol-neutral gate every transport runs. - let emitter = crate::control::security::audit::ArcAuditEmitter(Arc::clone(&shared.audit)); - let checked = match crate::control::server::shared::clone_write::intercept_and_authorize( - crate::control::server::shared::clone_write::InterceptAndAuthorizeParams { - state: shared, - task, - identity, - tenant_id, - permissions: &shared.permissions, - roles: &shared.roles, - emitter: &emitter, - }, - ) - .await? - { - crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(resp) => { - let payload = resp.payload.to_vec(); - if !payload.is_empty() { - let json = - crate::data::executor::response_codec::decode_payload_to_json(&payload); - match sonic_rs::from_str::(&json) { - Ok(mut v) => { - redact_decoded_value(Some(&redaction), &shared.redaction, &mut v); - results.push(v); + // A governed columnar predicate UPDATE/DELETE resolves to a concrete row set + // before proposing. + if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) { + let authorized_task = authorize_ws_rpc_task(shared, identity, &task)?; + match crate::control::write_resolve::run_authorized_write_resolve( + shared, + authorized_task, + resolver, + ) + .await + { + Ok(resp) => { + let payload = resp.payload.to_vec(); + if !payload.is_empty() { + let json = + crate::data::executor::response_codec::decode_payload_to_json( + &payload, + ); + match sonic_rs::from_str::(&json) { + Ok(mut v) => { + redact_decoded_value( + Some(&redaction), + &shared.redaction, + &mut v, + ); + results.push(v); + } + Err(_) => results.push(serde_json::Value::String(json)), + } } - Err(_) => results.push(serde_json::Value::String(json)), } + Err(e) => return Err(e), } meter_task(shared, &scope, &plan_metering_info, rows_before, &results); continue; } - crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed(checked) => { - checked - } - }; - // WebSocket RPC has no session transaction (no BEGIN / COMMIT), so - // every task is autocommit and forwards with no transaction id. - let payloads: crate::Result>> = match shared.gateway.get() { - Some(gw) => { - let gw_ctx = QueryContext { - tenant_id: checked.tenant_id(), - trace_id, - database_id: checked.database_id(), - txn_id: None, - }; - gw.execute(&gw_ctx, checked).await - } - None => { - // Single-node boot: gateway not yet initialised — dispatch - // locally. A write takes the durable route, a read the read route. - crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( - shared, checked, trace_id, + // Clone CoW write-path interception, then authorization, run once per + // task before dispatch — same protocol-neutral gate every transport runs. + let emitter = + crate::control::security::audit::ArcAuditEmitter(Arc::clone(&shared.audit)); + let checked = + match crate::control::server::shared::clone_write::intercept_and_authorize( + crate::control::server::shared::clone_write::InterceptAndAuthorizeParams { + state: shared, + task, + identity, + tenant_id, + permissions: &shared.permissions, + roles: &shared.roles, + emitter: &emitter, + }, ) - .await - .map(|r| vec![r.payload.to_vec()]) - } - }; + .await? + { + crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled( + resp, + ) => { + let payload = resp.payload.to_vec(); + if !payload.is_empty() { + let json = + crate::data::executor::response_codec::decode_payload_to_json( + &payload, + ); + match sonic_rs::from_str::(&json) { + Ok(mut v) => { + redact_decoded_value( + Some(&redaction), + &shared.redaction, + &mut v, + ); + results.push(v); + } + Err(_) => results.push(serde_json::Value::String(json)), + } + } + meter_task(shared, &scope, &plan_metering_info, rows_before, &results); + continue; + } + crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed( + checked, + ) => checked, + }; - match payloads { - Ok(vecs) => { - for payload in vecs { - if !payload.is_empty() { - let json = - crate::data::executor::response_codec::decode_payload_to_json(&payload); - match sonic_rs::from_str::(&json) { - Ok(mut v) => { - redact_decoded_value(Some(&redaction), &shared.redaction, &mut v); - results.push(v); + // WebSocket RPC has no session transaction (no BEGIN / COMMIT), so + // every task is autocommit and forwards with no transaction id. + let gw_ctx = QueryContext { + tenant_id: checked.tenant_id(), + trace_id, + database_id: checked.database_id(), + txn_id: None, + linearizable: true, + }; + let payloads: crate::Result>> = match shared.installed_gateway() { + Ok(gateway) => gateway.execute(&gw_ctx, checked).await, + Err(error) => Err(error), + }; + + match payloads { + Ok(vecs) => { + for payload in vecs { + if !payload.is_empty() { + let json = + crate::data::executor::response_codec::decode_payload_to_json( + &payload, + ); + match sonic_rs::from_str::(&json) { + Ok(mut v) => { + redact_decoded_value( + Some(&redaction), + &shared.redaction, + &mut v, + ); + results.push(v); + } + Err(_) => results.push(serde_json::Value::String(json)), } - Err(_) => results.push(serde_json::Value::String(json)), } } } + Err(e) => return Err(e), } - Err(e) => return Err(e), + meter_task(shared, &scope, &plan_metering_info, rows_before, &results); } - meter_task(shared, &scope, &plan_metering_info, rows_before, &results); - } - match results.len() { - 0 => Ok(serde_json::Value::Null), - 1 => Ok(results - .into_iter() - .next() - .unwrap_or(serde_json::Value::Null)), - _ => Ok(serde_json::Value::Array(results)), + let outcome: crate::Result = match results.len() { + 0 => Ok(serde_json::Value::Null), + 1 => Ok(results + .into_iter() + .next() + .unwrap_or(serde_json::Value::Null)), + _ => Ok(serde_json::Value::Array(results)), + }; + outcome + }; + if read_only { + lease_scope.guard(body).await? + } else { + body.await } } @@ -416,7 +496,7 @@ mod tests { /// Guards `peer_addr` against regressing to a hardcoded placeholder: a /// non-IP value can't be parsed by `normalize_peer_ip`, so `check_ip` - /// would silently match nothing and let a blacklisted client through. + /// will silently match nothing and let a blacklisted client through. #[tokio::test] async fn blacklisted_peer_ip_is_rejected() { let (state, _dir) = test_state().await; @@ -426,7 +506,7 @@ mod tests { .blacklist_ip("203.0.113.0/24", "test ban", "admin", 0) .expect("blacklist CIDR range"); - let query_ctx = PlannerQueryContext::new(); + let query_ctx = PlannerQueryContext::for_state(&state); let trace_id = TraceId::generate(); let result = execute_sql( @@ -460,7 +540,7 @@ mod tests { .blacklist_ip("203.0.113.0/24", "test ban", "admin", 0) .expect("blacklist CIDR range"); - let query_ctx = PlannerQueryContext::new(); + let query_ctx = PlannerQueryContext::for_state(&state); let trace_id = TraceId::generate(); let result = execute_sql( @@ -509,7 +589,7 @@ mod tests { let identity = regular_identity(9201); let scope = scope_for(&identity, &state); // `Some(...)` regardless of config, to prove `meter_dispatch`'s own - // enabled-check protects this call, not just the caller's gate. + // enabled-check protects this call, not only the caller's gate. let info = Some(PlanMeteringInfo::extract(&kv_get_plan())); let results = vec![serde_json::Value::Null; 3]; diff --git a/nodedb/src/control/server/http/routes/ws_rpc/format.rs b/nodedb/src/control/server/http/routes/ws_rpc/format.rs index 0b66f3974..ea9e138f4 100644 --- a/nodedb/src/control/server/http/routes/ws_rpc/format.rs +++ b/nodedb/src/control/server/http/routes/ws_rpc/format.rs @@ -4,16 +4,21 @@ use nodedb_sql::parser::preprocess::lex::find_ascii_case_insensitive; -use crate::control::change_stream::SequencedChangeEvent; +use crate::control::change_stream::{ChangeCursor, SequencedChangeEvent}; use crate::control::gateway::GatewayErrorMap; -/// Format a cursor-aware LIVE SELECT notification. -pub fn format_sequenced_live_notification(sub_id: u64, event: &SequencedChangeEvent) -> String { +/// Format a cursor-aware LIVE SELECT notification. `cursor` resumes right +/// after the event. +pub fn format_sequenced_live_notification( + sub_id: u64, + event: &SequencedChangeEvent, + cursor: &ChangeCursor, +) -> String { serde_json::json!({ "method": "live", "params": { "subscription_id": sub_id, - "cursor": event.cursor().to_string(), + "cursor": cursor.to_string(), "wal_lsn": event.lsn.as_u64(), "database_id": event.database_id().as_u64(), "collection": event.collection, @@ -25,10 +30,11 @@ pub fn format_sequenced_live_notification(sub_id: u64, event: &SequencedChangeEv .to_string() } -/// Format a cursor-aware connection resume notification. -pub fn format_resume_notification(event: &SequencedChangeEvent) -> String { +/// Format a cursor-aware connection resume notification. `cursor` resumes +/// right after the event. +pub fn format_resume_notification(event: &SequencedChangeEvent, cursor: &ChangeCursor) -> String { serde_json::json!({"method": "change", "params": { - "cursor": event.cursor().to_string(), "wal_lsn": event.lsn.as_u64(), + "cursor": cursor.to_string(), "wal_lsn": event.lsn.as_u64(), "database_id": event.database_id().as_u64(), "collection": event.collection, "operation": event.operation.as_str(), "document_id": event.document_id.as_str(), "timestamp_ms": event.timestamp_ms, }}) diff --git a/nodedb/src/control/server/http/routes/ws_rpc/handler.rs b/nodedb/src/control/server/http/routes/ws_rpc/handler.rs index a4ce8f7ba..70d2c6ccc 100644 --- a/nodedb/src/control/server/http/routes/ws_rpc/handler.rs +++ b/nodedb/src/control/server/http/routes/ws_rpc/handler.rs @@ -28,6 +28,7 @@ use axum::extract::ws::{Message, WebSocket}; use axum::extract::{ConnectInfo, State, WebSocketUpgrade}; use axum::http::HeaderMap; use axum::response::IntoResponse; +use futures::future::BoxFuture; use futures::{SinkExt, StreamExt}; use tracing::{debug, warn}; @@ -145,7 +146,29 @@ impl Drop for AbortOnDropJoinHandle { } /// Handle a single WebSocket connection. -async fn handle_ws_connection( +/// +/// The connection future is boxed, once per connection. Every request path +/// nests inside it, and unboxed it can overflow the compiler's layout depth +/// limit in the upgrade task. +fn handle_ws_connection( + socket: WebSocket, + state: AppState, + identity: crate::control::security::identity::AuthenticatedIdentity, + database_id: DatabaseId, + trace_id: nodedb_types::TraceId, + peer_addr: String, +) -> BoxFuture<'static, ()> { + Box::pin(serve_ws_connection( + socket, + state, + identity, + database_id, + trace_id, + peer_addr, + )) +} + +async fn serve_ws_connection( socket: WebSocket, state: AppState, identity: crate::control::security::identity::AuthenticatedIdentity, diff --git a/nodedb/src/control/server/http/routes/ws_rpc/process_message.rs b/nodedb/src/control/server/http/routes/ws_rpc/process_message.rs index 4965b2089..c5d660295 100644 --- a/nodedb/src/control/server/http/routes/ws_rpc/process_message.rs +++ b/nodedb/src/control/server/http/routes/ws_rpc/process_message.rs @@ -5,7 +5,7 @@ use std::str::FromStr; use std::sync::Arc; -use crate::control::change_stream::{ChangeCursor, LiveSubscriptionSet, ReplayStart}; +use crate::control::change_stream::{ChangeCursor, CursorStep, LiveSubscriptionSet, ReplayStart}; use crate::control::security::audit::ArcAuditEmitter; use crate::control::security::identity::{AuthenticatedIdentity, Permission}; use crate::control::server::shared::authorization::authorize_collection; @@ -158,23 +158,29 @@ pub(super) async fn process_message( ); let sub_id = sub.id; let live_tx = live_tx.clone(); + // The subscription receives exactly the events past its start. + let mut cursor = sub.start_cursor().clone(); live_set.spawn_task(async move { - let mut last_cursor = None; loop { match sub.recv_sequenced().await { - Ok(event) => { - if !advance_live_cursor(&mut last_cursor, event.cursor()) { - let _ = live_tx.send(serde_json::json!({"method":"reset_required","params":{"subscription_id":sub_id,"reason":"change stream epoch changed"}}).to_string()).await; - break; + Ok(event) => match cursor.accept(&event) { + CursorStep::Deliver => { + if live_tx + .send(format_sequenced_live_notification( + sub_id, &event, &cursor, + )) + .await + .is_err() + { + break; + } } - if live_tx - .send(format_sequenced_live_notification(sub_id, &event)) - .await - .is_err() - { + CursorStep::Skip => {} + CursorStep::Reset => { + let _ = live_tx.send(serde_json::json!({"method":"reset_required","params":{"subscription_id":sub_id,"reason":"change feed gap"}}).to_string()).await; break; } - } + }, Err(tokio::sync::broadcast::error::RecvError::Lagged(_)) => { let _ = live_tx.send(serde_json::json!({"method":"reset_required","params":{"subscription_id":sub_id,"reason":"change stream lagged"}}).to_string()).await; break; @@ -192,16 +198,6 @@ pub(super) async fn process_message( } } -fn advance_live_cursor(last_cursor: &mut Option, cursor: ChangeCursor) -> bool { - match *last_cursor { - Some(previous) if !cursor.is_after_in_same_epoch(previous) => false, - _ => { - *last_cursor = Some(cursor); - true - } - } -} - async fn resume_auth( context: ResumeContext<'_>, id: &serde_json::Value, @@ -262,12 +258,13 @@ async fn resume_auth( let start = cursor .map(ReplayStart::Cursor) .unwrap_or(ReplayStart::Timestamp(0)); + // The ring is bounded, so the replay returns every event it holds. let snapshot = match shared.change_stream.query_changes_in_database( identity.tenant_id, database_id, None, start, - 10_000, + usize::MAX, ) { Ok(snapshot) => snapshot, Err(_) => { @@ -282,7 +279,8 @@ async fn resume_auth( }; let emitter = ArcAuditEmitter(Arc::clone(&shared.audit)); let mut replayed = 0usize; - for event in &snapshot.events { + for change in &snapshot.events { + let event = &change.event; if authorize_collection( identity, database_id, @@ -295,7 +293,7 @@ async fn resume_auth( .is_ok() { if live_tx - .send(format_resume_notification(event)) + .send(format_resume_notification(event, &change.cursor)) .await .is_err() { @@ -304,23 +302,26 @@ async fn resume_auth( replayed += 1; } } - let snapshot_cursor = snapshot.snapshot_cursor; + let snapshot_cursor = snapshot.cursor.to_string(); + let mut cursor = snapshot.cursor; let live_tx = live_tx.clone(); let shared = Arc::clone(&shared); let identity = identity.clone(); resume_set.spawn_task(async move { loop { match subscription.recv_sequenced().await { - Ok(event) => { - if !event.cursor().same_epoch(snapshot_cursor) { - let _ = live_tx.send(serde_json::json!({"method":"reset_required","params":{"reason":"change stream epoch changed"}}).to_string()).await; + Ok(event) => match cursor.accept(&event) { + CursorStep::Deliver => { + let emitter = ArcAuditEmitter(Arc::clone(&shared.audit)); + if authorize_collection(&identity, database_id, &event.collection, Permission::Read, &shared.permissions, &shared.roles, &emitter).is_ok() + && live_tx.send(format_resume_notification(&event, &cursor)).await.is_err() { break; } + } + CursorStep::Skip => {} + CursorStep::Reset => { + let _ = live_tx.send(serde_json::json!({"method":"reset_required","params":{"reason":"change feed gap"}}).to_string()).await; break; } - if !event.cursor().is_after_in_same_epoch(snapshot_cursor) { continue; } - let emitter = ArcAuditEmitter(Arc::clone(&shared.audit)); - if authorize_collection(&identity, database_id, &event.collection, Permission::Read, &shared.permissions, &shared.roles, &emitter).is_ok() - && live_tx.send(format_resume_notification(&event)).await.is_err() { break; } - } + }, Err(tokio::sync::broadcast::error::RecvError::Lagged(_)) => { let _ = live_tx.send(serde_json::json!({"method":"reset_required","params":{"reason":"change stream lagged"}}).to_string()).await; break; @@ -329,21 +330,6 @@ async fn resume_auth( } } }); - let response = serde_json::json!({"id": id, "result": {"session_id": session_id, "replayed": replayed, "snapshot_cursor": snapshot_cursor.to_string()}}).to_string(); + let response = serde_json::json!({"id": id, "result": {"session_id": session_id, "replayed": replayed, "snapshot_cursor": snapshot_cursor}}).to_string(); (response, true) } - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn live_cursor_requires_reset_after_epoch_rotation() { - let mut last = None; - assert!(advance_live_cursor( - &mut last, - ChangeCursor::new(u128::MAX, u64::MAX) - )); - assert!(!advance_live_cursor(&mut last, ChangeCursor::new(1, 1))); - } -} diff --git a/nodedb/src/control/server/ilp_batch/dispatch.rs b/nodedb/src/control/server/ilp_batch/dispatch.rs index 6e18c5e6e..875eafa36 100644 --- a/nodedb/src/control/server/ilp_batch/dispatch.rs +++ b/nodedb/src/control/server/ilp_batch/dispatch.rs @@ -9,15 +9,19 @@ use tokio::sync::Semaphore; use tracing::{debug, warn}; use crate::bridge::envelope::PhysicalPlan; +use crate::control::lease::QueryLeaseScope; +use crate::control::metadata_proposer::DEFAULT_DRAIN_TIMEOUT; use crate::control::planner::calvin::{ - TxnDispatchPosition, dispatch_authorized_strict_atomic_tasks_to_calvin, + TxnDispatchPosition, TxnProvenance, dispatch_strict_atomic_tasks_to_calvin, }; use crate::control::security::audit::ArcAuditEmitter; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::request_scope::ClientRequestScope; use crate::control::server::ilp_auth::AuthenticatedIlpContext; use crate::control::server::shared::authorization::authorize_task_set; +use crate::control::server::shared::clone_write::write_lease; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; +use crate::control::server::shared::retry::retry_through_drain; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId}; use nodedb_physical::physical_plan::TimeseriesOp; @@ -43,7 +47,8 @@ pub(crate) async fn flush_ilp_batch( } /// Strictly parse, authorize, and atomically ingest canonical ILP produced by -/// another authenticated external transport such as OTLP. +/// another authenticated external transport such as OTLP. Returns the number +/// of rows stored: the lines received less the lines the resolve rejected. pub(crate) async fn flush_authenticated_ilp_batch( state: &Arc, identity: &AuthenticatedIdentity, @@ -53,7 +58,7 @@ pub(crate) async fn flush_authenticated_ilp_batch( ) -> crate::Result { // Blacklist + account status, no rate limit: ILP/OTLP ingest is not the // per-query traffic the rate-limiter's cost table models, so charging it - // against a query rate limit would throttle a legitimate high-volume + // against a query rate limit will throttle a legitimate high-volume // ingest client. A blacklisted or suspended/banned account must not be // able to keep ingesting, though — `check_blacklist_and_status` runs // that half of `check_request_admission`'s gate (plus the @@ -92,25 +97,24 @@ pub(crate) async fn flush_authenticated_ilp_batch( /// /// The merge is a replicated catalog DDL, and every catalog DDL already /// serializes on `SharedState::metadata_ddl_lock`. A second concurrent merge -/// could therefore only park a second blocking-pool thread on a lock that -/// admits one holder, so one permit is both the useful and the safe bound — -/// ingest can never grow the blocking pool no matter how many ILP connections -/// or OTLP requests are live. +/// can therefore only park a second task on a lock that admits one holder, +/// so one permit is both the useful and the safe bound — ingest can never +/// queue merges no matter how many ILP connections or OTLP requests are live. static SCHEMA_PROJECTION_SLOT: LazyLock = LazyLock::new(|| Semaphore::new(1)); /// Merge the ingest-inferred schema projection for `groups` off the caller's /// task. /// -/// `merge_collection_fields_replicated` is a fully synchronous replicated DDL: -/// it takes the metadata DDL preparation lock, acquires the distributed -/// preparation lease, drains prior-version descriptor leases and waits for the -/// local apply — a chain whose bounds are tens of seconds. Running it inline on -/// the ingest task is what starved ILP: `handle_ilp_connection` awaits the -/// flush inside its `select!`, so for the whole of that chain the connection -/// polls neither the socket-read branch nor the coalescing timer, and no -/// subsequent batch is dispatched at all. +/// `merge_collection_fields_replicated` is a replicated DDL: it takes the +/// metadata DDL preparation lock, acquires the distributed preparation lease, +/// drains prior-version descriptor leases and waits for the local apply and +/// its Data Plane register — a chain whose bounds are tens of seconds. Running +/// it inline on the ingest task is what starved ILP: `handle_ilp_connection` +/// awaits the flush inside its `select!`, so for the whole of that chain the +/// connection polls neither the socket-read branch nor the coalescing timer, +/// and no subsequent batch is dispatched at all. /// -/// It is therefore run on the blocking pool and deliberately NOT awaited. That +/// It therefore runs on its own task and is deliberately NOT awaited. That /// costs no durability: the Calvin write above is already committed, and the /// projection is rebuildable and self-healing — every ILP batch re-supplies its /// measurement's full field set, so a merge skipped because the slot was busy @@ -122,7 +126,7 @@ fn spawn_schema_projection_merge( tenant_id: TenantId, groups: Vec, ) { - // Bound the permit to `'static` explicitly: it is moved into the blocking + // Bound the permit to `'static` explicitly: it is moved into the merge // task and must outlive this frame. let slot: &'static Semaphore = &SCHEMA_PROJECTION_SLOT; let Ok(permit) = slot.try_acquire() else { @@ -133,9 +137,9 @@ fn spawn_schema_projection_merge( return; }; let state = Arc::clone(state); - tokio::task::spawn_blocking(move || { + tokio::spawn(async move { // Held for the whole merge so the next batch's `try_acquire` observes - // a busy slot rather than queueing another blocking thread. + // a busy slot rather than queueing another merge. let _permit = permit; for group in groups { match crate::control::catalog_entry::merge_collection_fields_replicated( @@ -143,12 +147,15 @@ fn spawn_schema_projection_merge( database_id, tenant_id.as_u64(), &group.measurement, + group.catalog_time_column.as_ref(), &group.catalog_fields, - ) { + ) + .await + { Ok(_) => {} // This is a rebuildable control-plane projection. The data // commit is already durable, so logging is required but - // retrying the client request would risk a duplicate write. + // retrying the client request will risk a duplicate write. Err(error) => warn!( collection = %group.measurement, error = %error, @@ -173,7 +180,7 @@ async fn flush_ilp_batch_inner( // This transport builds its physical tasks itself instead of going through // the SQL planner, so it has to run the planner's row-level-security pass - // over them explicitly — without this the line-protocol listener would be a + // over them explicitly — without this the line-protocol listener will be a // way to write rows a write policy forbids, with the same identity and the // same collection that `INSERT` refuses. The resolved scope is the same one // the metering pass below uses, so the policy is evaluated for exactly the @@ -182,7 +189,7 @@ async fn flush_ilp_batch_inner( // It runs BEFORE `authorize_task_set` and before dispatch: the pass mutates // the tasks (it compiles the write predicate onto each `Ingest`), and an // authorized task set is what gets dispatched, so injecting afterwards - // would dispatch the un-injected copies. + // will dispatch the un-injected copies. // // Resolved against the sender's real address like the admission scope // above, so a `WHEN`/`REQUIRE IP` scope grant contributes to `$auth.*` @@ -211,19 +218,37 @@ async fn flush_ilp_batch_inner( let authorized = authorize_task_set(identity, &tasks, &state.permissions, &state.roles, &emitter) .map_err(crate::Error::from)?; - + // The leases gate the batch on a drained collection and live until the + // batch's Calvin write commits. + let _leases = acquire_batch_write_leases(state, &tasks).await?; + + // Each measurement resolves to the rows it stores before the batch is + // sequenced. The lines a resolve rejected never land, so the batch + // reports the rows it stored, not the lines it received. + let submitted: Vec = authorized + .into_tasks() + .into_iter() + .map(|task| task.into_physical_task()) + .collect(); + let submitted = + match crate::control::write_resolve::resolve_tasks_for_log(state, &submitted).await? { + Some(resolved) => resolved, + None => submitted, + }; // One Calvin submit stages every measurement and makes the TransactionRedo - // the sole durability record; no per-measurement WAL or direct dispatch may + // the sole durability record; no per-measurement WAL or direct dispatch can // race ahead of a later measurement failure. - let _ = dispatch_authorized_strict_atomic_tasks_to_calvin( + let applied = dispatch_strict_atomic_tasks_to_calvin( state, - authorized, + &submitted, tenant_id, TxnDispatchPosition::Autocommit, &[], None, + TxnProvenance::client(), ) .await?; + let rejected = rejected_lines_of(&submitted, applied.as_ref())?; // Metered here, once the whole batch's atomic Calvin write has already // committed: one usage event per measurement (= one dispatched @@ -250,12 +275,76 @@ async fn flush_ilp_batch_inner( // // The merge goes through the replicated metadata path, never a local // catalog write: the projection lives inside the replicated collection - // descriptor, and mutating that record in place would leave this node's + // descriptor, and mutating that record in place will leave this node's // copy no longer byte-equal to the replicated entry at the same descriptor // version — which wedges the metadata applier on the next replay. That path // is synchronous and slow, so it runs off this task entirely. spawn_schema_projection_merge(state, database_id, tenant_id, groups); - Ok(total_rows) + Ok(total_rows.saturating_sub(rejected)) +} + +/// Take the descriptor write lease of every task in the batch. +/// +/// A collection under drain refuses its lease as `RetryableSchemaChanged`. +/// The refusal comes before anything is sequenced, so the whole acquisition +/// is retried until the drain ends. An ILP client gets no ack and cannot +/// retry, so the flush waits as long as a DDL drain can last. Each attempt +/// reads the descriptor version afresh and reuses the same tasks, so a retry +/// never stages a row twice. A failed attempt drops every lease it took. The +/// Calvin submission stays outside this unit, because its outcome can be +/// ambiguous. +async fn acquire_batch_write_leases( + state: &SharedState, + tasks: &[PhysicalTask], +) -> crate::Result> { + retry_through_drain( + &state.lease_drain, + DEFAULT_DRAIN_TIMEOUT, + move || async move { + let mut leases = Vec::with_capacity(tasks.len()); + for task in tasks { + leases + .push(write_lease(state, task.tenant_id, task.database_id, &task.plan).await?); + } + Ok::<_, crate::Error>(leases) + }, + ) + .await +} + +/// The lines the batch did not store. Each measurement's resolve rejected +/// some lines, and its install rejected those plus the rows that conflict +/// with the schema at its log position. The install's count wins for each +/// collection the apply reports. +fn rejected_lines_of( + submitted: &[PhysicalTask], + applied: Option<&crate::bridge::envelope::Response>, +) -> crate::Result { + let mut by_collection: std::collections::BTreeMap = + std::collections::BTreeMap::new(); + for task in submitted { + if let PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection, .. }) = &task.plan { + let rejected = crate::control::write_resolve::rejected_lines(&task.plan)?; + let held = by_collection + .entry(collection.as_str().to_owned()) + .or_default(); + *held = held.saturating_add(rejected); + } + } + let installed = applied.and_then(|response| { + crate::engine::timeseries::install_counts::TsInstallCounts::from_payload( + response.payload.as_bytes(), + ) + }); + if let Some(installed) = installed { + for (collection, (_, rejected)) in installed.by_collection() { + let held = by_collection.entry(collection).or_default(); + *held = (*held).max(rejected); + } + } + Ok(by_collection + .values() + .fold(0u64, |total, rejected| total.saturating_add(*rejected))) } fn preflighted_row_count(groups: &[IlpMeasurementBatch]) -> crate::Result { @@ -357,7 +446,7 @@ mod tests { } fn grant_write(permissions: &PermissionStore, collection: &str) { - let target = format!("collection:9:{collection}"); + let target = format!("collection:7:9:{collection}"); permissions .grant( &target, @@ -407,16 +496,15 @@ mod tests { } /// The line-protocol listener builds its physical tasks itself instead of - /// going through the SQL planner, so it has to run the row-level-security - /// injection pass explicitly. Before that call existed, this transport - /// reached the Data Plane without the pass running at all — a write policy - /// that refuses an `INSERT` into a collection did nothing to an ILP batch - /// into the same collection under the same identity. + /// going through the SQL planner, so it runs the row-level-security + /// injection pass explicitly. A write policy that refuses an `INSERT` + /// into a collection also refuses an ILP batch into the same collection + /// under the same identity. /// /// The policy here names an `$auth` field the identity does not carry, so /// the pass fails closed and refuses before dispatch — an outcome only the - /// injection pass can produce, which is what makes this a regression test - /// for the pass being called rather than for anything downstream. + /// injection pass can produce, which makes this a test + /// of the pass being called rather than of anything downstream. #[tokio::test] async fn ilp_ingest_runs_the_row_level_security_injection_pass() { use crate::control::security::predicate::{CompareOp, PredicateValue, RlsPredicate}; @@ -436,7 +524,7 @@ mod tests { .create_policy(RlsPolicy { name: "cpu_owner".into(), // Seeded straight into the store, so it must carry the key the - // DDL would have written: qualified for a non-default database. + // DDL writes: qualified for a non-default database. collection: nodedb_types::QualifiedCollection::new(database_id, "cpu").to_string(), display_collection: "cpu".into(), tenant_id: 9, @@ -500,9 +588,8 @@ mod tests { /// batch must pass the admission gate and fail only on its own merits — /// here, the missing write grant that preflight refuses. /// - /// Before the sender's address reached this scope there was no score to - /// enforce, so the gate failed closed: turning on `[auth.risk]` took ILP, - /// OTLP and Prometheus remote write offline for every client. + /// The sender's address reaches this scope, so the gate has a score to + /// enforce and does not fail closed when `[auth.risk]` is on. #[tokio::test] async fn scored_ingest_passes_the_admission_gate_instead_of_failing_closed() { let (state, _dir) = risk_state(crate::control::security::risk::RiskConfig { @@ -556,10 +643,102 @@ mod tests { assert_eq!(rejection_reason(&error), "denied by risk policy"); } + /// A flush that meets a descriptor drain waits it out. It takes its lease + /// at the version the draining DDL commits, so the batch still reaches + /// Calvin and the connection survives. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn batch_leases_retry_through_a_descriptor_drain() { + use crate::control::security::catalog::StoredCollection; + use nodedb_cluster::{DescriptorId, DescriptorKind, DrainOwner}; + + let cluster = crate::control::cluster::test_one_node::boot().await; + let state = Arc::clone(&cluster.state); + let database_id = DatabaseId::DEFAULT; + let mut collection = StoredCollection::stamped_for_test(9, "cpu", "admin"); + collection.database_id = database_id; + collection.collection_type = nodedb_types::CollectionType::timeseries("ts", "1h"); + collection.fields = vec![ + ("ts".to_owned(), "BIGINT TIME_KEY".to_owned()), + ("value".to_owned(), "BIGINT".to_owned()), + ]; + crate::control::catalog_entry::apply::collection::put( + &collection, + state.credentials.catalog(), + ) + .expect("store the collection at version 1"); + let descriptor = + DescriptorId::new(database_id.as_u64(), 9, DescriptorKind::Collection, "cpu"); + { + let _gate = state + .lease_admission_gate + .lock() + .unwrap_or_else(|poison| poison.into_inner()); + state.lease_drain.install_start( + descriptor.clone(), + DrainOwner::Ddl, + 1, + nodedb_types::Hlc::new(u64::MAX, 0), + state.node_id, + ); + } + let groups = vec![super::IlpMeasurementBatch { + measurement: "cpu".into(), + raw_lines: vec!["cpu value=1i 1000".into()], + catalog_time_column: None, + catalog_fields: Vec::new(), + }]; + let tasks = + super::build_ilp_calvin_tasks(TenantId::new(9), database_id, &groups).expect("tasks"); + + let refused = crate::control::server::shared::clone_write::write_lease( + &state, + tasks[0].tenant_id, + tasks[0].database_id, + &tasks[0].plan, + ) + .await; + assert!( + matches!(refused, Err(crate::Error::RetryableSchemaChanged { .. })), + "one attempt is refused while the drain holds" + ); + + // The DDL commits version 2 and ends its drain while the flush waits. + let mut altered = collection.clone(); + altered.descriptor_version = 2; + let ddl = { + let state = Arc::clone(&state); + let descriptor = descriptor.clone(); + tokio::spawn(async move { + tokio::time::sleep(std::time::Duration::from_millis(80)).await; + crate::control::catalog_entry::apply::collection::put( + &altered, + state.credentials.catalog(), + ) + .expect("store the collection at version 2"); + state.lease_drain.install_end(&descriptor, &DrainOwner::Ddl); + state.lease_drain.settle(); + }) + }; + + let leases = super::acquire_batch_write_leases(&state, &tasks) + .await + .expect("the batch takes its leases once the drain ends"); + ddl.await.expect("the DDL task completes"); + assert_eq!(leases.len(), tasks.len()); + let held = state + .lookup_lease_for_self(&descriptor) + .expect("the batch holds the collection's lease"); + assert_eq!(held.version, 2, "the lease pins the committed version"); + + drop(leases); + drop(state); + cluster.shutdown().await; + } + #[test] fn schema_projection_slot_admits_exactly_one_merge_at_a_time() { - // The bound that keeps ingest from queueing blocking-pool threads - // behind a metadata DDL lock that admits a single holder. + // The bound that keeps ingest from queueing merge tasks behind a + // metadata DDL lock that admits a single holder. let first = super::SCHEMA_PROJECTION_SLOT .try_acquire() .expect("the first merge takes the only slot"); @@ -580,11 +759,13 @@ mod tests { super::IlpMeasurementBatch { measurement: "cpu".into(), raw_lines: vec!["cpu value=1i".into(), "cpu value=2i".into()], + catalog_time_column: None, catalog_fields: Vec::new(), }, super::IlpMeasurementBatch { measurement: "mem".into(), raw_lines: vec!["mem value=3i".into()], + catalog_time_column: None, catalog_fields: Vec::new(), }, ]; diff --git a/nodedb/src/control/server/ilp_batch/preflight.rs b/nodedb/src/control/server/ilp_batch/preflight.rs index a8822c760..aecfd147f 100644 --- a/nodedb/src/control/server/ilp_batch/preflight.rs +++ b/nodedb/src/control/server/ilp_batch/preflight.rs @@ -16,12 +16,17 @@ use crate::types::DatabaseId; /// Preflighted raw ILP lines for one canonical measurement. /// /// `raw_lines` preserve physical source order; map iteration canonicalizes -/// measurement order. `catalog_fields` is a rebuildable control-plane projection -/// of the authoritative timeseries-engine schema. +/// measurement order. `catalog_time_column` and `catalog_fields` are a +/// rebuildable control-plane projection of the authoritative timeseries-engine +/// schema. #[derive(Debug, PartialEq, Eq)] pub(super) struct IlpMeasurementBatch { pub(super) measurement: String, pub(super) raw_lines: Vec, + /// The column inference names for the line timestamp. The catalog merge + /// drops it for a collection whose declared time key carries that time. + pub(super) catalog_time_column: Option<(String, String)>, + /// The inferred tag and field columns, in inference order. pub(super) catalog_fields: Vec<(String, String)>, } @@ -75,14 +80,20 @@ pub(super) fn preflight_ilp_batch( let parsed_group = crate::engine::timeseries::ilp::parse_batch(&grouped_source) .map_err(|_| IlpPreflightFailure::Parse)?; let schema = crate::engine::timeseries::ilp_ingest::infer_schema(parsed_group.lines()); - let catalog_fields = schema - .columns - .iter() - .map(|(name, ty)| (name.clone(), ty.ddl_type_name().to_owned())) - .collect(); + let mut catalog_time_column = None; + let mut catalog_fields = Vec::with_capacity(schema.columns.len()); + for (index, (name, ty)) in schema.columns.iter().enumerate() { + let projected = (name.clone(), ty.ddl_type_name().to_owned()); + if index == schema.timestamp_idx { + catalog_time_column = Some(projected); + } else { + catalog_fields.push(projected); + } + } groups.push(IlpMeasurementBatch { measurement, raw_lines, + catalog_time_column, catalog_fields, }); } @@ -168,7 +179,7 @@ mod tests { } fn grant_write(permissions: &PermissionStore, collection: &str) { - let target = format!("collection:9:{collection}"); + let target = format!("collection:7:9:{collection}"); permissions .grant(&target, "user:ingester", Permission::Write, "admin", None) .expect("in-memory grant succeeds"); @@ -204,6 +215,30 @@ mod tests { assert_eq!(groups[1].raw_lines, vec!["mem value=1i", "mem value=3i"]); } + /// The line timestamp is projected apart from the tag and field columns. + /// A tag or field that happens to be called `timestamp` stays a field. + #[test] + fn projection_holds_the_inferred_time_column_apart_from_fields() { + let permissions = PermissionStore::new(); + grant_write(&permissions, "cpu"); + + let groups = preflight("cpu,host=a value=1i,timestamp=2i 1000\n", &permissions) + .expect("the measurement is writable"); + + assert_eq!( + groups[0].catalog_time_column, + Some(("timestamp".to_owned(), "TIMESTAMP".to_owned())) + ); + assert_eq!( + groups[0].catalog_fields, + vec![ + ("host".to_owned(), "VARCHAR".to_owned()), + ("value".to_owned(), "BIGINT".to_owned()), + ("timestamp".to_owned(), "BIGINT".to_owned()), + ] + ); + } + #[test] fn comments_blanks_and_escaped_measurements_use_canonical_grouping() { let permissions = PermissionStore::new(); @@ -286,7 +321,7 @@ mod tests { let permissions = PermissionStore::new(); permissions .grant( - "collection:9:cpu", + "collection:7:9:cpu", "user:ingester", Permission::Read, "admin", @@ -331,7 +366,7 @@ mod tests { let permissions = PermissionStore::new(); permissions .grant( - "collection:10:cpu", + "collection:7:10:cpu", "user:ingester", Permission::Write, "admin", diff --git a/nodedb/src/control/server/ilp_connection.rs b/nodedb/src/control/server/ilp_connection.rs new file mode 100644 index 000000000..9d7b406a4 --- /dev/null +++ b/nodedb/src/control/server/ilp_connection.rs @@ -0,0 +1,346 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! One ILP TCP connection: the Hello/Auth prelude, admission, and the +//! adaptive batch-coalescing ingest loop. The accept loop lives in +//! `ilp_listener`. + +use std::net::SocketAddr; +use std::sync::Arc; + +use futures::future::BoxFuture; +use tokio::io::BufReader; +use tokio::sync::OwnedSemaphorePermit; +use tracing::{debug, warn}; + +use crate::config::auth::AuthMode; +use crate::control::server::conn_stream::ConnStream; +use crate::control::server::ilp_auth::AuthenticatedIlpContext; +use crate::control::state::SharedState; +use crate::types::TenantId; + +use super::ilp_batch::{IlpRateEstimator, flush_ilp_batch}; +use super::ilp_drop::{IlpDropCause, terminate_with_buffered_flush}; +use super::ilp_line_read::read_bounded_ilp_line; + +/// Maximum byte length of a single ILP line. Lines exceeding this are +/// rejected and the connection is dropped to prevent memory exhaustion. +const MAX_ILP_LINE_BYTES: usize = 10 * 1024 * 1024; // 10 MiB +/// Per-connection aggregate cap. A batch must not turn many individually +/// valid lines into an unbounded allocation before its timer/line flush. +const MAX_ILP_BATCH_BYTES: usize = MAX_ILP_LINE_BYTES; + +/// Handle a single ILP TCP connection. +/// +/// The connection future is boxed, once per connection. Every ingest path +/// nests inside it, and unboxed it can overflow the compiler's layout depth +/// limit in the listener's connection task. +pub(super) fn handle_ilp_connection<'a>( + stream: ConnStream, + peer: SocketAddr, + state: &'a Arc, + auth_mode: &'a AuthMode, +) -> BoxFuture<'a, crate::Result<()>> { + Box::pin(serve_ilp_connection(stream, peer, state, auth_mode)) +} + +/// Serve one ILP connection with adaptive batch coalescing. +/// +/// Batch size adapts to ingest rate: +/// - High rate (>100K lines/s): batch up to 10K lines or 10ms window +/// - Medium rate (1K-100K/s): batch up to 1K lines or 50ms window +/// - Low rate (<1K/s): batch per 100 lines or 100ms window +/// +/// Larger batches amortize per-batch overhead (WAL append, memtable lock, +/// partition lookup). +async fn serve_ilp_connection( + mut stream: ConnStream, + peer: SocketAddr, + state: &Arc, + auth_mode: &AuthMode, +) -> crate::Result<()> { + // Captured before the Hello/Auth prelude borrows the stream and the + // ingest loop moves it into a `BufReader`, after which the TLS session is + // no longer reachable. + let transport = stream.transport_security(); + + // The native Hello/Auth prelude must finish before line parsing, tenant + // accounting, or any ingest side effect. Authentication failures consume + // no ILP bytes and the dropped stream cannot enter the ingest loop. + let authenticated_context = crate::control::server::ilp_auth::authenticate_ilp_connection( + &mut stream, + state, + auth_mode, + &peer.to_string(), + ) + .await + .map_err(|_| crate::Error::BadRequest { + detail: "ILP authentication failed".into(), + })?; + // The TLS policy is evaluated before any ingest capacity is acquired: the + // identity (and with it the superuser flag the cleartext carve-out needs) + // exists only now, and a refused connection must not hold a slot. + if crate::control::server::session_auth::check_transport_security( + state, + authenticated_context.identity(), + transport, + authenticated_context.peer_addr(), + ) + .is_err() + { + crate::control::server::ilp_auth::write_ilp_auth_failure( + &mut stream, + &authenticated_context, + ) + .await; + return Err(crate::Error::BadRequest { + detail: "ILP authentication failed".into(), + }); + } + + let _admission = match IlpConnectionAdmission::acquire(state, &authenticated_context) { + Ok(admission) => admission, + Err(_) => { + crate::control::server::ilp_auth::write_ilp_auth_failure( + &mut stream, + &authenticated_context, + ) + .await; + return Err(crate::Error::BadRequest { + detail: "ILP authentication failed".into(), + }); + } + }; + crate::control::server::ilp_auth::write_ilp_auth_success(&mut stream, &authenticated_context) + .await + .map_err(|_| crate::Error::BadRequest { + detail: "ILP authentication failed".into(), + })?; + + debug!(%peer, "authenticated ILP connection accepted"); + + let mut reader = BufReader::new(stream); + let mut line_buf: Vec = Vec::with_capacity(4096); + let mut batch = String::new(); + let mut line_count = 0u64; + let mut total_ingested = 0u64; + + // Adaptive batch coalescing state. + let mut rate_estimator = IlpRateEstimator::new(); + let mut batch_target = 1000u64; + let mut window = tokio::time::interval(std::time::Duration::from_millis(50)); + window.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); + + loop { + tokio::select! { + // Read next line with an enforced byte-length cap. + result = read_bounded_ilp_line(&mut reader, &mut line_buf, MAX_ILP_LINE_BYTES) => { + match result { + Ok(false) => break, // Connection closed (EOF). + Ok(true) => { + // Strip trailing newline / CRLF. + let line_bytes = line_buf + .strip_suffix(b"\r\n") + .or_else(|| line_buf.strip_suffix(b"\n")) + .unwrap_or(&line_buf); + + let line = match std::str::from_utf8(line_bytes) { + Ok(s) => s, + // The framing is broken and cannot be resynchronized, so the + // connection still ends here — but the lines already accepted + // into `batch` are dispatched first and the termination is + // recorded, because ILP acks nothing and the client will + // otherwise have no way to learn they were discarded. + Err(_) => { + return Err(terminate_with_buffered_flush( + state, + &authenticated_context, + peer, + IlpDropCause::InvalidUtf8, + &batch, + line_count, + ) + .await); + } + }; + + if line.trim().is_empty() || line.trim_start().starts_with('#') { + line_buf.clear(); + continue; + } + + let separator_bytes = if batch.is_empty() { 0 } else { 1 }; + if batch + .len() + .saturating_add(separator_bytes) + .saturating_add(line.len()) + > MAX_ILP_BATCH_BYTES + { + let flushed = line_count; + total_ingested += + flush_ilp_batch(state, &authenticated_context, &batch).await?; + batch.clear(); + line_count = 0; + + rate_estimator.record(flushed); + let (new_target, new_window_ms) = rate_estimator.suggest_batch_params(); + batch_target = new_target; + window = tokio::time::interval( + std::time::Duration::from_millis(new_window_ms), + ); + window.set_missed_tick_behavior( + tokio::time::MissedTickBehavior::Delay, + ); + } + + if !batch.is_empty() { + batch.push('\n'); + } + batch.push_str(line); + line_count += 1; + line_buf.clear(); + + // Flush when batch reaches adaptive target. + if line_count >= batch_target { + let flushed = line_count; + total_ingested += + flush_ilp_batch(state, &authenticated_context, &batch).await?; + batch.clear(); + line_count = 0; + + // Update rate estimator and recalculate batch target. + rate_estimator.record(flushed); + let (new_target, new_window_ms) = rate_estimator.suggest_batch_params(); + batch_target = new_target; + window = tokio::time::interval( + std::time::Duration::from_millis(new_window_ms), + ); + window.set_missed_tick_behavior( + tokio::time::MissedTickBehavior::Delay, + ); + } + } + Err(error) => { + warn!( + %peer, + error = %error, + limit = MAX_ILP_LINE_BYTES, + "ILP line read failed — rejecting connection" + ); + // Same contract as the decode failure above: the connection + // ends, but not before the accepted lines are dispatched and + // the loss surface is recorded. + return Err(terminate_with_buffered_flush( + state, + &authenticated_context, + peer, + IlpDropCause::LineReadFailed, + &batch, + line_count, + ) + .await); + } + } + } + // Timer-based flush (for low-rate connections). + _ = window.tick() => { + if !batch.is_empty() { + let flushed = line_count; + total_ingested += + flush_ilp_batch(state, &authenticated_context, &batch).await?; + batch.clear(); + line_count = 0; + + rate_estimator.record(flushed); + let (new_target, new_window_ms) = rate_estimator.suggest_batch_params(); + batch_target = new_target; + window = tokio::time::interval( + std::time::Duration::from_millis(new_window_ms), + ); + window.set_missed_tick_behavior( + tokio::time::MissedTickBehavior::Delay, + ); + } + } + } + } + + // Flush remaining. + if !batch.is_empty() { + total_ingested += flush_ilp_batch(state, &authenticated_context, &batch).await?; + } + + debug!( + %peer, + total_ingested, + database_id = ?authenticated_context.database_id(), + "ILP connection closed" + ); + Ok(()) +} + +/// Connection-scoped tenant accounting and quota permits. +/// +/// The permit fields release configured database/tenant limits on every return +/// path, while `Drop` balances the legacy tenant activity accounting. +struct IlpConnectionAdmission<'a> { + state: &'a SharedState, + tenant_id: TenantId, + _database_permit: Option, + _tenant_permit: Option, +} + +impl<'a> IlpConnectionAdmission<'a> { + fn acquire(state: &'a SharedState, context: &AuthenticatedIlpContext) -> crate::Result { + let tenant_id = context.identity().tenant_id; + let database_id = context.database_id(); + + let database_permit = state + .admission_registry + .try_acquire_database(database_id) + .map_err(|_| crate::Error::BadRequest { + detail: "ILP admission denied".into(), + })?; + let tenant_permit = state + .admission_registry + .try_acquire_tenant(database_id, tenant_id) + .map_err(|_| crate::Error::BadRequest { + detail: "ILP admission denied".into(), + })?; + + start_tenant_connection(state, tenant_id)?; + Ok(Self { + state, + tenant_id, + _database_permit: database_permit, + _tenant_permit: tenant_permit, + }) + } +} + +impl Drop for IlpConnectionAdmission<'_> { + fn drop(&mut self) { + self.state.tenant_connection_end(self.tenant_id); + } +} + +/// Atomically check and account for the legacy per-tenant connection cap. +/// +/// The database/tenant semaphore permits above cover configured catalog +/// quotas; this lock also protects the legacy tenant-isolation counter from a +/// check-then-increment race when it has its own max-connections setting. +fn start_tenant_connection(state: &SharedState, tenant_id: TenantId) -> crate::Result<()> { + let mut tenants = state + .tenants + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()); + if !matches!( + tenants.check_connection(tenant_id), + crate::control::security::tenant::QuotaCheck::Allowed + ) { + return Err(crate::Error::BadRequest { + detail: "ILP admission denied".into(), + }); + } + tenants.connection_start(tenant_id); + Ok(()) +} diff --git a/nodedb/src/control/server/ilp_listener.rs b/nodedb/src/control/server/ilp_listener.rs index 79eb1ad59..49ef28f23 100644 --- a/nodedb/src/control/server/ilp_listener.rs +++ b/nodedb/src/control/server/ilp_listener.rs @@ -8,39 +8,32 @@ //! //! Protocol: native Hello/Auth prelude followed by one ILP line per newline. //! The prelude is mandatory; direct unauthenticated ILP clients are rejected. +//! +//! This module holds the accept loop. One connection's prelude, admission +//! and ingest loop live in `ilp_connection`. use std::net::SocketAddr; use std::sync::Arc; -use tokio::io::BufReader; - -/// Maximum byte length of a single ILP line. Lines exceeding this are -/// rejected and the connection is dropped to prevent memory exhaustion. -const MAX_ILP_LINE_BYTES: usize = 10 * 1024 * 1024; // 10 MiB -/// Per-connection aggregate cap. A batch must not turn many individually -/// valid lines into an unbounded allocation before its timer/line flush. -const MAX_ILP_BATCH_BYTES: usize = MAX_ILP_LINE_BYTES; use tokio::net::TcpListener; -use tokio::sync::{OwnedSemaphorePermit, Semaphore}; +use tokio::sync::Semaphore; use tracing::{debug, info, warn}; use crate::config::auth::AuthMode; use crate::control::server::conn_stream::ConnStream; -use crate::control::server::ilp_auth::AuthenticatedIlpContext; use crate::control::server::shared::{ConnectionFutureOutcome, isolate_connection_future}; use crate::control::state::SharedState; -use crate::types::TenantId; #[path = "ilp_batch/mod.rs"] mod ilp_batch; +#[path = "ilp_connection.rs"] +mod ilp_connection; #[path = "ilp_drop.rs"] mod ilp_drop; #[path = "ilp_line_read.rs"] mod ilp_line_read; pub(crate) use ilp_batch::flush_authenticated_ilp_batch; -use ilp_batch::{IlpRateEstimator, flush_ilp_batch}; -use ilp_drop::{IlpDropCause, terminate_with_buffered_flush}; -use ilp_line_read::read_bounded_ilp_line; +use ilp_connection::handle_ilp_connection; /// ILP TCP listener. pub struct IlpListener { @@ -198,305 +191,3 @@ impl IlpListener { Ok(()) } } - -/// Handle a single ILP TCP connection with adaptive batch coalescing. -/// -/// Batch size adapts to ingest rate: -/// - High rate (>100K lines/s): batch up to 10K lines or 10ms window -/// - Medium rate (1K-100K/s): batch up to 1K lines or 50ms window -/// - Low rate (<1K/s): batch per 100 lines or 100ms window -/// -/// Larger batches amortize per-batch overhead (WAL append, memtable lock, -/// partition lookup). -async fn handle_ilp_connection( - mut stream: ConnStream, - peer: SocketAddr, - state: &Arc, - auth_mode: &AuthMode, -) -> crate::Result<()> { - // Captured before the Hello/Auth prelude borrows the stream and the - // ingest loop moves it into a `BufReader`, after which the TLS session is - // no longer reachable. - let transport = stream.transport_security(); - - // The native Hello/Auth prelude must finish before line parsing, tenant - // accounting, or any ingest side effect. Authentication failures consume - // no ILP bytes and the dropped stream cannot enter the ingest loop. - let authenticated_context = crate::control::server::ilp_auth::authenticate_ilp_connection( - &mut stream, - state, - auth_mode, - &peer.to_string(), - ) - .await - .map_err(|_| crate::Error::BadRequest { - detail: "ILP authentication failed".into(), - })?; - // The TLS policy is evaluated before any ingest capacity is acquired: the - // identity (and with it the superuser flag the cleartext carve-out needs) - // exists only now, and a refused connection must not hold a slot. - if crate::control::server::session_auth::check_transport_security( - state, - authenticated_context.identity(), - transport, - authenticated_context.peer_addr(), - ) - .is_err() - { - crate::control::server::ilp_auth::write_ilp_auth_failure( - &mut stream, - &authenticated_context, - ) - .await; - return Err(crate::Error::BadRequest { - detail: "ILP authentication failed".into(), - }); - } - - let _admission = match IlpConnectionAdmission::acquire(state, &authenticated_context) { - Ok(admission) => admission, - Err(_) => { - crate::control::server::ilp_auth::write_ilp_auth_failure( - &mut stream, - &authenticated_context, - ) - .await; - return Err(crate::Error::BadRequest { - detail: "ILP authentication failed".into(), - }); - } - }; - crate::control::server::ilp_auth::write_ilp_auth_success(&mut stream, &authenticated_context) - .await - .map_err(|_| crate::Error::BadRequest { - detail: "ILP authentication failed".into(), - })?; - - debug!(%peer, "authenticated ILP connection accepted"); - - let mut reader = BufReader::new(stream); - let mut line_buf: Vec = Vec::with_capacity(4096); - let mut batch = String::new(); - let mut line_count = 0u64; - let mut total_ingested = 0u64; - - // Adaptive batch coalescing state. - let mut rate_estimator = IlpRateEstimator::new(); - let mut batch_target = 1000u64; - let mut window = tokio::time::interval(std::time::Duration::from_millis(50)); - window.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); - - loop { - tokio::select! { - // Read next line with an enforced byte-length cap. - result = read_bounded_ilp_line(&mut reader, &mut line_buf, MAX_ILP_LINE_BYTES) => { - match result { - Ok(false) => break, // Connection closed (EOF). - Ok(true) => { - // Strip trailing newline / CRLF. - let line_bytes = line_buf - .strip_suffix(b"\r\n") - .or_else(|| line_buf.strip_suffix(b"\n")) - .unwrap_or(&line_buf); - - let line = match std::str::from_utf8(line_bytes) { - Ok(s) => s, - // The framing is broken and cannot be resynchronized, so the - // connection still ends here — but the lines already accepted - // into `batch` are dispatched first and the termination is - // recorded, because ILP acks nothing and the client would - // otherwise have no way to learn they were discarded. - Err(_) => { - return Err(terminate_with_buffered_flush( - state, - &authenticated_context, - peer, - IlpDropCause::InvalidUtf8, - &batch, - line_count, - ) - .await); - } - }; - - if line.trim().is_empty() || line.trim_start().starts_with('#') { - line_buf.clear(); - continue; - } - - let separator_bytes = if batch.is_empty() { 0 } else { 1 }; - if batch - .len() - .saturating_add(separator_bytes) - .saturating_add(line.len()) - > MAX_ILP_BATCH_BYTES - { - let flushed = line_count; - total_ingested += - flush_ilp_batch(state, &authenticated_context, &batch).await?; - batch.clear(); - line_count = 0; - - rate_estimator.record(flushed); - let (new_target, new_window_ms) = rate_estimator.suggest_batch_params(); - batch_target = new_target; - window = tokio::time::interval( - std::time::Duration::from_millis(new_window_ms), - ); - window.set_missed_tick_behavior( - tokio::time::MissedTickBehavior::Delay, - ); - } - - if !batch.is_empty() { - batch.push('\n'); - } - batch.push_str(line); - line_count += 1; - line_buf.clear(); - - // Flush when batch reaches adaptive target. - if line_count >= batch_target { - let flushed = line_count; - total_ingested += - flush_ilp_batch(state, &authenticated_context, &batch).await?; - batch.clear(); - line_count = 0; - - // Update rate estimator and recalculate batch target. - rate_estimator.record(flushed); - let (new_target, new_window_ms) = rate_estimator.suggest_batch_params(); - batch_target = new_target; - window = tokio::time::interval( - std::time::Duration::from_millis(new_window_ms), - ); - window.set_missed_tick_behavior( - tokio::time::MissedTickBehavior::Delay, - ); - } - } - Err(error) => { - warn!( - %peer, - error = %error, - limit = MAX_ILP_LINE_BYTES, - "ILP line read failed — rejecting connection" - ); - // Same contract as the decode failure above: the connection - // ends, but not before the accepted lines are dispatched and - // the loss surface is recorded. - return Err(terminate_with_buffered_flush( - state, - &authenticated_context, - peer, - IlpDropCause::LineReadFailed, - &batch, - line_count, - ) - .await); - } - } - } - // Timer-based flush (for low-rate connections). - _ = window.tick() => { - if !batch.is_empty() { - let flushed = line_count; - total_ingested += - flush_ilp_batch(state, &authenticated_context, &batch).await?; - batch.clear(); - line_count = 0; - - rate_estimator.record(flushed); - let (new_target, new_window_ms) = rate_estimator.suggest_batch_params(); - batch_target = new_target; - window = tokio::time::interval( - std::time::Duration::from_millis(new_window_ms), - ); - window.set_missed_tick_behavior( - tokio::time::MissedTickBehavior::Delay, - ); - } - } - } - } - - // Flush remaining. - if !batch.is_empty() { - total_ingested += flush_ilp_batch(state, &authenticated_context, &batch).await?; - } - - debug!( - %peer, - total_ingested, - database_id = ?authenticated_context.database_id(), - "ILP connection closed" - ); - Ok(()) -} - -/// Connection-scoped tenant accounting and quota permits. -/// -/// The permit fields release configured database/tenant limits on every return -/// path, while `Drop` balances the legacy tenant activity accounting. -struct IlpConnectionAdmission<'a> { - state: &'a SharedState, - tenant_id: TenantId, - _database_permit: Option, - _tenant_permit: Option, -} - -impl<'a> IlpConnectionAdmission<'a> { - fn acquire(state: &'a SharedState, context: &AuthenticatedIlpContext) -> crate::Result { - let tenant_id = context.identity().tenant_id; - let database_id = context.database_id(); - - let database_permit = state - .admission_registry - .try_acquire_database(database_id) - .map_err(|_| crate::Error::BadRequest { - detail: "ILP admission denied".into(), - })?; - let tenant_permit = state - .admission_registry - .try_acquire_tenant(database_id, tenant_id) - .map_err(|_| crate::Error::BadRequest { - detail: "ILP admission denied".into(), - })?; - - start_tenant_connection(state, tenant_id)?; - Ok(Self { - state, - tenant_id, - _database_permit: database_permit, - _tenant_permit: tenant_permit, - }) - } -} - -impl Drop for IlpConnectionAdmission<'_> { - fn drop(&mut self) { - self.state.tenant_connection_end(self.tenant_id); - } -} - -/// Atomically check and account for the legacy per-tenant connection cap. -/// -/// The database/tenant semaphore permits above cover configured catalog -/// quotas; this lock also protects the legacy tenant-isolation counter from a -/// check-then-increment race when it has its own max-connections setting. -fn start_tenant_connection(state: &SharedState, tenant_id: TenantId) -> crate::Result<()> { - let mut tenants = state - .tenants - .lock() - .unwrap_or_else(|poisoned| poisoned.into_inner()); - if !matches!( - tenants.check_connection(tenant_id), - crate::control::security::tenant::QuotaCheck::Allowed - ) { - return Err(crate::Error::BadRequest { - detail: "ILP admission denied".into(), - }); - } - tenants.connection_start(tenant_id); - Ok(()) -} diff --git a/nodedb/src/control/server/native/dispatch/ctx.rs b/nodedb/src/control/server/native/dispatch/ctx.rs index 9b6245dd3..04aadc7cf 100644 --- a/nodedb/src/control/server/native/dispatch/ctx.rs +++ b/nodedb/src/control/server/native/dispatch/ctx.rs @@ -1,7 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 //! Dispatch context: holds references needed by all per-opcode handlers. -//! Split out of `mod.rs` to keep that file declarations/re-exports only. use std::sync::Arc; @@ -69,10 +68,19 @@ impl DispatchCtx<'_> { collection: &str, ) -> VShardId { if let PhysicalPlan::Graph(op) = plan { + // A presence guard, its read and a TRUNCATE's edge share name + // their vShard. + if let GraphOp::NodePresenceGuard { vshard, .. } + | GraphOp::NodePresenceRead { vshard, .. } + | GraphOp::TruncateEdges { vshard, .. } = op + { + return VShardId::new(*vshard); + } let node_key = match op { GraphOp::EdgePut { src_id, .. } | GraphOp::EdgeDelete { src_id, .. } => { src_id.as_str() } + GraphOp::NodeEdgeGuard { node_id, .. } => node_id.as_str(), GraphOp::EdgePutBatch { .. } | GraphOp::ResolveEdgeDelete(_) | GraphOp::EdgeDeleteBatch { .. } @@ -92,7 +100,10 @@ impl DispatchCtx<'_> { | GraphOp::RemoveNodeLabels { .. } | GraphOp::TemporalNeighbors { .. } | GraphOp::TemporalAlgorithm { .. } - | GraphOp::Stats { .. } => document_id.unwrap_or(collection), + | GraphOp::Stats { .. } + | GraphOp::NodePresenceGuard { .. } + | GraphOp::NodePresenceRead { .. } + | GraphOp::TruncateEdges { .. } => document_id.unwrap_or(collection), }; return VShardId::from_key(node_key.as_bytes()); } diff --git a/nodedb/src/control/server/native/dispatch/direct_ops.rs b/nodedb/src/control/server/native/dispatch/direct_ops.rs index 5eece82c3..b357719e4 100644 --- a/nodedb/src/control/server/native/dispatch/direct_ops.rs +++ b/nodedb/src/control/server/native/dispatch/direct_ops.rs @@ -12,11 +12,11 @@ use crate::control::planner::calvin::{ use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; use crate::control::server::shared::session::TransactionState; +use crate::control::server::shared::txn_route::statement_needs_implicit_txn; use crate::control::server::shared::write_admission::all_writes_bufferable; use crate::types::TraceId; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; -use super::response::data_plane_response_to_native; use super::single_task::dispatch_single_task; use super::{DispatchCtx, error_to_native, error_to_native_with_sqlstate}; use crate::control::server::native::sqlstate_code::sqlstate_error; @@ -65,9 +65,34 @@ pub(crate) async fn handle_direct_op( return error_to_native(seq, &e); } - let mut plan = match super::plan_builder::build_plan(ctx, op, fields, &collection) { + // A point delete's plan depends on whether the collection is edge-bearing. + // The collection's lease is held from before that read until the op's + // outcome, so the flag cannot change between the plan and the write: the + // flag's catalog write drains every lease on the version the plan read. + let _plan_lease = if op == OpCode::PointDelete { + match crate::control::server::shared::clone_write::collections_write_lease( + ctx.state, + tenant_id, + ctx.database_id(), + std::iter::once(collection.clone()), + ) + .await + { + Ok(lease) => Some(lease), + Err(e) => return error_to_native(seq, &e), + } + } else { + None + }; + + // A malformed request is a syntax error. Any other error comes from + // resolving surrogates at the collection home and keeps its own code. + let mut plan = match super::plan_builder::build_plan(ctx, op, fields, &collection).await { Ok(p) => p, - Err(e) => return error_to_native_with_sqlstate(seq, "42601", &e), + Err(e @ crate::Error::BadRequest { .. }) => { + return error_to_native_with_sqlstate(seq, "42601", &e); + } + Err(e) => return error_to_native(seq, &e), }; let vshard_id = ctx.task_vshard(&plan, fields.document_id.as_deref(), &collection); @@ -109,129 +134,25 @@ pub(crate) async fn handle_direct_op( } // False for `dispatch_single_task`, which meters itself — re-metering here - // would double-bill a `Staged` dispatch and wrongly bill a `Buffered` one. + // will double-bill a `Staged` dispatch and wrongly bill a `Buffered` one. let mut needs_top_level_metering = true; // Wrapped in an async block so `return` inside each branch exits only this // block, letting the metering call below run exactly once regardless of branch. let response: NativeResponse = async { - // `INSERT ... SELECT` orchestrates on the Control Plane; never reaches the - // Data Plane as a single op. - if matches!( - &plan, - PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. }) - ) { - let task = PhysicalTask { - tenant_id, - vshard_id, - database_id: ctx.database_id(), - plan: plan.clone(), - post_set_op: PostSetOp::None, - txn_id: None, - }; - let authorized = match super::sql_gateway::authorize_native_task(ctx, &task) { - Ok(authorized) => authorized, - Err(error) => return error_to_native(seq, &error), - }; - let _request = ctx.state.tenant_request_guard(tenant_id); - let result = - crate::control::insert_select::run_authorized_insert_select(ctx.state, authorized) - .await; - return match result { - Ok(resp) => data_plane_response_to_native(ctx, seq, &plan, &resp), - Err(e) => error_to_native(seq, &e), - }; - } - - // Autocommit `MERGE` orchestrates on the Control Plane; never reaches the - // Data Plane as a single op. - if matches!( - &plan, - PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Merge { - resolved_inserts: None, - .. - }) - ) { - let task = PhysicalTask { - tenant_id, - vshard_id, - database_id: ctx.database_id(), - plan: plan.clone(), - post_set_op: PostSetOp::None, - txn_id: None, - }; - let authorized = match super::sql_gateway::authorize_native_task(ctx, &task) { - Ok(authorized) => authorized, - Err(error) => return error_to_native(seq, &error), - }; - let _request = ctx.state.tenant_request_guard(tenant_id); - let result = - crate::control::merge_orchestrator::run_authorized_merge(ctx.state, authorized) - .await; - return match result { - Ok(resp) => data_plane_response_to_native(ctx, seq, &plan, &resp), - Err(e) => error_to_native(seq, &e), - }; - } - - // Autocommit `UPDATE ... FROM ` scans the source on its own core and - // ships it into the plan; never reaches the Data Plane as a single op. - if matches!( - &plan, - PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { - source_rows: None, - .. - }) - ) { - let task = PhysicalTask { - tenant_id, - vshard_id, - database_id: ctx.database_id(), - plan: plan.clone(), - post_set_op: PostSetOp::None, - txn_id: None, - }; - let authorized = match super::sql_gateway::authorize_native_task(ctx, &task) { - Ok(authorized) => authorized, - Err(error) => return error_to_native(seq, &error), - }; - let _request = ctx.state.tenant_request_guard(tenant_id); - let result = - crate::control::update_from_join_orchestrator::run_authorized_update_from_join( - ctx.state, authorized, - ) - .await; - return match result { - Ok(resp) => data_plane_response_to_native(ctx, seq, &plan, &resp), - Err(e) => error_to_native(seq, &e), - }; + // Plans that orchestrate on the Control Plane never reach the Data + // Plane as a single op. + if let Some(response) = + super::direct_orchestrated::dispatch_orchestrated(ctx, seq, &plan, vshard_id).await + { + return response; } - // A governed predicate resolves to a concrete row set before proposing — see - // `control::write_resolve`. Local (non-Raft) path skips this. - if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&plan) - && ctx.state.async_raft_proposer().is_some() + // A graph algorithm or neighbors read runs where the graph's partitions + // live, not on the one node the collection routes to. + if let Some(response) = + super::graph_owner::dispatch_graph_owner_read(ctx, seq, &plan, vshard_id).await { - let task = PhysicalTask { - tenant_id, - vshard_id, - database_id: ctx.database_id(), - plan: plan.clone(), - post_set_op: PostSetOp::None, - txn_id: None, - }; - let authorized = match super::sql_gateway::authorize_native_task(ctx, &task) { - Ok(authorized) => authorized, - Err(error) => return error_to_native(seq, &error), - }; - let _request = ctx.state.tenant_request_guard(tenant_id); - let result = crate::control::write_resolve::run_authorized_write_resolve( - ctx.state, authorized, resolver, - ) - .await; - return match result { - Ok(resp) => data_plane_response_to_native(ctx, seq, &plan, &resp), - Err(e) => error_to_native(seq, &e), - }; + return response; } // Stamp the connection's active txn id so the Data Plane resolves the staging @@ -262,6 +183,19 @@ pub(crate) async fn handle_direct_op( ) { return error_to_native(seq, &crate::Error::from(error)); } + // The lease gates the direct op on a drained collection and lives + // until the op's outcome. + let lease = match crate::control::server::shared::clone_write::write_lease( + ctx.state, + tenant_id, + ctx.database_id(), + &tasks[0].plan, + ) + .await + { + Ok(lease) => std::sync::Arc::new(lease), + Err(e) => return error_to_native(seq, &e), + }; if let Err(e) = crate::control::planner::implicit_edges::append_implicit_edge_tasks( ctx.state, @@ -302,7 +236,7 @@ pub(crate) async fn handle_direct_op( } // Period-lock reference rows, resolved into the same plan slot as the - // materialized-sum targets just above. + // materialized-sum targets right above. if let Err(e) = crate::control::planner::period_lock::resolve_period_lock_targets( ctx.state, &mut tasks, @@ -328,6 +262,34 @@ pub(crate) async fn handle_direct_op( Err(error) => return error_to_native(seq, &crate::Error::from(error)), }; + // A write that fires a BEFORE, INSTEAD OF or SYNC AFTER body, or a + // MERGE into an edge-bearing collection, runs in one transaction, as + // the SQL path's does. The loop there clone-checks, authorizes and + // meters each task itself. + if statement_needs_implicit_txn(ctx.state, &tasks) { + let _request = ctx.state.tenant_request_guard(tenant_id); + needs_top_level_metering = false; + return super::direct_txn::dispatch_direct_in_txn( + ctx, + seq, + super::direct_txn::DirectWrite { + tasks, + sum_target_reads, + lease: std::sync::Arc::clone(&lease), + }, + ) + .await; + } + + // A delete or update on an edge-bearing collection commits its edge + // cleanup in the same Calvin transaction, as the SQL path's does. + use super::edge_recon_gate::{EdgeReconResult, try_edge_recon_dispatch}; + let (tasks, authorized_tasks) = + match try_edge_recon_dispatch(ctx, seq, tasks, authorized_tasks).await { + EdgeReconResult::Outcome(outcome) => return outcome.into_response(), + EdgeReconResult::NotFired(tasks, authorized) => (tasks, authorized), + }; + if tasks.len() == 1 { // No-edge fast path. Local-path WAL append lives inside `dispatch_single_task`, // shared with the single-shard edge loop. diff --git a/nodedb/src/control/server/native/dispatch/direct_orchestrated.rs b/nodedb/src/control/server/native/dispatch/direct_orchestrated.rs new file mode 100644 index 000000000..377d2e361 --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/direct_orchestrated.rs @@ -0,0 +1,135 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Direct operations that orchestrate on the Control Plane and never reach the +//! Data Plane as a single op: `INSERT ... SELECT`, autocommit `MERGE`, +//! autocommit `UPDATE ... FROM `, and governed-predicate writes. + +use nodedb_types::protocol::NativeResponse; + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::shared::txn_route::statement_needs_implicit_txn; +use crate::types::VShardId; +use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; + +use super::response::data_plane_response_to_native; +use super::{DispatchCtx, error_to_native}; + +/// A task carrying `plan` with no post-set op and no transaction. +fn bare_task(ctx: &DispatchCtx<'_>, plan: &PhysicalPlan, vshard_id: VShardId) -> PhysicalTask { + PhysicalTask { + tenant_id: ctx.tenant_id(), + vshard_id, + database_id: ctx.database_id(), + plan: plan.clone(), + post_set_op: PostSetOp::None, + txn_id: None, + } +} + +/// Run `plan` on the Control Plane when it needs orchestration there. +/// +/// Returns `None` when the plan dispatches as a single Data Plane op. +pub(super) async fn dispatch_orchestrated( + ctx: &DispatchCtx<'_>, + seq: u64, + plan: &PhysicalPlan, + vshard_id: VShardId, +) -> Option { + let tenant_id = ctx.tenant_id(); + + // `INSERT ... SELECT` orchestrates on the Control Plane; never reaches the + // Data Plane as a single op. + if matches!( + plan, + PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::InsertSelect { .. }) + ) { + let task = bare_task(ctx, plan, vshard_id); + let authorized = match super::sql_gateway::authorize_native_task(ctx, &task) { + Ok(authorized) => authorized, + Err(error) => return Some(error_to_native(seq, &error)), + }; + let _request = ctx.state.tenant_request_guard(tenant_id); + let result = + crate::control::insert_select::run_authorized_insert_select(ctx.state, authorized) + .await; + return Some(match result { + Ok(resp) => data_plane_response_to_native(ctx, seq, plan, &resp), + Err(e) => error_to_native(seq, &e), + }); + } + + // Autocommit `MERGE` orchestrates on the Control Plane; never reaches the + // Data Plane as a single op. + if matches!( + plan, + PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Merge { + resolved_inserts: None, + .. + }) + ) { + let task = bare_task(ctx, plan, vshard_id); + // A MERGE into an edge-bearing collection runs in the implicit + // transaction, which stages each removed row's edge tasks. + if !statement_needs_implicit_txn(ctx.state, std::slice::from_ref(&task)) { + let authorized = match super::sql_gateway::authorize_native_task(ctx, &task) { + Ok(authorized) => authorized, + Err(error) => return Some(error_to_native(seq, &error)), + }; + let _request = ctx.state.tenant_request_guard(tenant_id); + let result = + crate::control::merge_orchestrator::run_authorized_merge(ctx.state, authorized) + .await; + return Some(match result { + Ok(resp) => data_plane_response_to_native(ctx, seq, plan, &resp), + Err(e) => error_to_native(seq, &e), + }); + } + } + + // Autocommit `UPDATE ... FROM ` scans the source on its own core and + // ships it into the plan; never reaches the Data Plane as a single op. + if matches!( + plan, + PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::UpdateFromJoin { + source_rows: None, + .. + }) + ) { + let task = bare_task(ctx, plan, vshard_id); + let authorized = match super::sql_gateway::authorize_native_task(ctx, &task) { + Ok(authorized) => authorized, + Err(error) => return Some(error_to_native(seq, &error)), + }; + let _request = ctx.state.tenant_request_guard(tenant_id); + let result = + crate::control::update_from_join_orchestrator::run_authorized_update_from_join( + ctx.state, authorized, + ) + .await; + return Some(match result { + Ok(resp) => data_plane_response_to_native(ctx, seq, plan, &resp), + Err(e) => error_to_native(seq, &e), + }); + } + + // A governed predicate resolves to a concrete row set before proposing — see + // `control::write_resolve`. + if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(plan) { + let task = bare_task(ctx, plan, vshard_id); + let authorized = match super::sql_gateway::authorize_native_task(ctx, &task) { + Ok(authorized) => authorized, + Err(error) => return Some(error_to_native(seq, &error)), + }; + let _request = ctx.state.tenant_request_guard(tenant_id); + let result = crate::control::write_resolve::run_authorized_write_resolve( + ctx.state, authorized, resolver, + ) + .await; + return Some(match result { + Ok(resp) => data_plane_response_to_native(ctx, seq, plan, &resp), + Err(e) => error_to_native(seq, &e), + }); + } + + None +} diff --git a/nodedb/src/control/server/native/dispatch/direct_txn.rs b/nodedb/src/control/server/native/dispatch/direct_txn.rs new file mode 100644 index 000000000..7ecbcd69f --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/direct_txn.rs @@ -0,0 +1,70 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! A direct write whose collection fires a BEFORE, INSTEAD OF or SYNC AFTER +//! body, in the statement's transaction. +//! +//! The write's tasks run through the SQL path's per-task loop, which routes +//! each one through the shared `txn_route`: the bodies join the transaction, +//! a shadowed clone takes its copy-on-write steps, and the writes stage. +//! Inside a transaction block that is the client's transaction. Outside one +//! it is an implicit transaction that commits when every task succeeds and +//! rolls back when one fails. + +use std::sync::Arc; + +use nodedb_types::protocol::NativeResponse; + +use crate::control::lease::QueryLeaseScope; +use crate::control::server::shared::session::read_set::ReadSetEntry; +use crate::control::server::shared::session::{DmlTxnCtx, TransactionState}; +use crate::control::server::shared::txn_route::unbufferable_joined_statement; +use crate::control::server::shared::write_admission::all_writes_bufferable; +use nodedb_physical::physical_task::PhysicalTask; + +use super::sql_loop::{PlannedStatement, run_dispatch_loop, run_implicit_statement}; +use super::{DispatchCtx, error_to_native}; + +/// A direct write's tasks and what they carry into the transaction. +pub(super) struct DirectWrite { + pub tasks: Vec, + /// The row images the write's cross-shard balances were settled from. + pub sum_target_reads: Vec, + /// The write's descriptor leases, kept on every task it buffers. + pub lease: Arc, +} + +/// Run a direct write's tasks in the statement's transaction. The answer is +/// the statement fold the SQL path gives: the write's count and verb. +pub(super) async fn dispatch_direct_in_txn( + ctx: &DispatchCtx<'_>, + seq: u64, + write: DirectWrite, +) -> NativeResponse { + let DirectWrite { + tasks, + sum_target_reads, + lease, + } = write; + let statement = PlannedStatement { + tasks, + output_schema: None, + database_id: ctx.database_id(), + plan_lease_scope: lease, + sum_target_reads, + }; + if ctx.sessions.transaction_state(ctx.peer_addr) == TransactionState::InBlock { + let client = DmlTxnCtx { + sessions: ctx.sessions, + session_id: ctx.peer_addr.into(), + }; + return run_dispatch_loop(ctx, seq, statement, Some(&client)) + .await + .into_response(); + } + if !all_writes_bufferable(&statement.tasks) { + return error_to_native(seq, &unbufferable_joined_statement()); + } + run_implicit_statement(ctx, seq, statement) + .await + .into_response() +} diff --git a/nodedb/src/control/server/native/dispatch/edge_recon_gate.rs b/nodedb/src/control/server/native/dispatch/edge_recon_gate.rs index 72284e501..e82f61c0a 100644 --- a/nodedb/src/control/server/native/dispatch/edge_recon_gate.rs +++ b/nodedb/src/control/server/native/dispatch/edge_recon_gate.rs @@ -10,7 +10,7 @@ //! the Data Plane — leaving mirrored CSR edges dangling. //! //! This module exposes a single entry point: [`try_edge_recon_dispatch`], which -//! implements the same three-guard check as the pgwire `execute.rs` gate and +//! implements the same two-guard check as the pgwire `execute.rs` gate and //! returns the native protocol outcome when the gate fires. use nodedb_types::protocol::NativeResponse; @@ -28,14 +28,14 @@ use super::{DispatchCtx, SqlOutcome, error_to_native}; /// /// Returns `Some(outcome)` when the gate fires (the task set contains a /// `BulkDelete`/`BulkUpdate` on an edge-bearing collection that is not inside -/// an explicit transaction block and the Calvin sequencer registry is up). The +/// an explicit transaction block). The /// caller MUST return this outcome immediately — the tasks have been consumed. /// /// Returns `None` when the gate does not fire; the caller proceeds with the /// normal classify/dispatch path. /// /// A genuine catalog I/O error propagates as `Some(Err-shaped SqlOutcome)` so -/// the caller surfaces it correctly — misrouting on a real I/O fault would +/// the caller surfaces it correctly — misrouting on a real I/O fault will /// silently skip edge cleanup (dangling edges). pub(super) async fn try_edge_recon_dispatch( ctx: &DispatchCtx<'_>, @@ -49,19 +49,14 @@ pub(super) async fn try_edge_recon_dispatch( // is identical across both protocol paths. Edge-bearing predicate writes // inside an explicit native transaction block are NOT recon-routed (same // limitation as pgwire — buffering a multi-step OLLP inside an explicit txn - // would require full two-phase commit across the outer txn boundary). + // will require full two-phase commit across the outer txn boundary). if ctx.sessions.transaction_state(ctx.peer_addr) == TransactionState::InBlock { return EdgeReconResult::NotFired(tasks, authorized); } - // Guard 2: Calvin completion registry available (sequencer is up). - if ctx.state.calvin_completion_registry.get().is_none() { - return EdgeReconResult::NotFired(tasks, authorized); - } - - // Guard 3: at least one BulkDelete/BulkUpdate targets an edge-bearing + // Guard 2: at least one BulkDelete/BulkUpdate targets an edge-bearing // collection. A genuine catalog I/O error propagates rather than falling - // through — misrouting on a real fault would skip edge cleanup. + // through — misrouting on a real fault will skip edge cleanup. let (_coll, database_id) = match plan_needs_implicit_edge_recon(ctx.state, &tasks, ctx.tenant_id()) { Err(e) => return EdgeReconResult::Outcome(resp(error_to_native(seq, &e))), @@ -75,17 +70,13 @@ pub(super) async fn try_edge_recon_dispatch( // drains. let plans: Vec<_> = tasks.iter().map(|t| t.plan.clone()).collect(); - // All three guards passed — run the OLLP/Calvin coordinator. This is the - // normal multi-shard OLLP path (NOT the contended single-shard route from - // `route_write_to_calvin`), so it stays on the strict multi-vshard - // dependent `TxClass` builder (`allow_single_vshard: false`). + // Both guards passed — run the OLLP/Calvin coordinator. let outcome = dispatch_authorized_dependent_edge_recon( ctx.state, authorized, ctx.identity, ctx.tenant_id(), database_id, - false, ) .await; diff --git a/nodedb/src/control/server/native/dispatch/graph_match.rs b/nodedb/src/control/server/native/dispatch/graph_match.rs index f659ee765..f6f3749a0 100644 --- a/nodedb/src/control/server/native/dispatch/graph_match.rs +++ b/nodedb/src/control/server/native/dispatch/graph_match.rs @@ -41,7 +41,7 @@ pub(crate) async fn handle_graph_match( } let mut plan = - match super::plan_builder::build_plan(ctx, OpCode::GraphMatch, fields, &collection) { + match super::plan_builder::build_plan(ctx, OpCode::GraphMatch, fields, &collection).await { Ok(plan) => plan, Err(error) => return error_to_native_with_sqlstate(seq, "42601", &error), }; @@ -88,6 +88,20 @@ pub(crate) async fn handle_graph_match( return error_to_native_with_sqlstate(seq, "53400", &e); } let _request = ctx.state.tenant_request_guard(tenant_id); + + // In a cluster the pattern crosses every node's partitions: it runs + // through the cross-shard MATCH scatter, as the SQL surface's MATCH does. + if ctx.state.cluster_routing.is_some() { + let native = match_across_shards(ctx, seq, vshard_id, &plan, txn_id).await; + if native.status != nodedb_types::protocol::ResponseStatus::Error + && let Some(info) = &plan_metering_info + { + let rows = native.rows.as_ref().map(|rows| rows.len() as u64); + meter_dispatch(ctx.state, &ctx.scope, info, rows); + } + return native; + } + let raw = dispatch_authorized_single_task(ctx, tenant_id, vshard_id, plan, txn_id).await; let response = match raw { @@ -146,3 +160,70 @@ pub(crate) async fn handle_graph_match( } native } + +/// Run a native MATCH across every node's partitions through +/// `graph_dispatch::scatter_match`. The rows come back as the bare array a +/// local MATCH yields once its envelope is unwrapped. A result the scatter +/// cannot finish is refused with `54001`, never returned partial. The +/// scatter notes every vShard it read, and the session loop records them into +/// the transaction read-set (`session::graph_reads`). +async fn match_across_shards( + ctx: &DispatchCtx<'_>, + seq: u64, + vshard_id: crate::types::VShardId, + plan: &crate::bridge::envelope::PhysicalPlan, + txn_id: Option, +) -> NativeResponse { + let crate::bridge::envelope::PhysicalPlan::Graph( + nodedb_physical::physical_plan::GraphOp::Match { query, .. }, + ) = plan + else { + return error_to_native( + seq, + &crate::Error::Internal { + detail: "a native MATCH built a plan that is not a MATCH".into(), + }, + ); + }; + let task = nodedb_physical::physical_task::PhysicalTask { + tenant_id: ctx.tenant_id(), + vshard_id, + database_id: ctx.database_id(), + plan: plan.clone(), + post_set_op: nodedb_physical::physical_task::PostSetOp::None, + txn_id, + }; + if let Err(error) = super::sql_gateway::authorize_native_task(ctx, &task) { + return error_to_native(seq, &error); + } + let outcome = crate::control::server::graph_dispatch::scatter_match( + ctx.state, + ctx.tenant_id(), + ctx.database_id(), + query.clone(), + crate::control::gateway::dispatcher::statement_deadline_ms(ctx.state), + crate::control::server::graph_dispatch::GraphRead { + txn_id, + // The native protocol has no read-consistency setting. + linearizable: true, + }, + ) + .await; + match outcome { + Ok(outcome) if outcome.partial => error_to_native_with_sqlstate( + seq, + "54001", + &crate::Error::BadRequest { + detail: crate::control::server::shared::ddl::neutral::match_ops::MATCH_INCOMPLETE_MESSAGE + .into(), + }, + ), + Ok(outcome) => data_plane_response_to_native( + ctx, + seq, + plan, + &crate::control::server::dispatch_utils::ok_payload_response(outcome.rows_payload), + ), + Err(error) => error_to_native(seq, &error), + } +} diff --git a/nodedb/src/control/server/native/dispatch/graph_owner.rs b/nodedb/src/control/server/native/dispatch/graph_owner.rs new file mode 100644 index 000000000..04a13b386 --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/graph_owner.rs @@ -0,0 +1,155 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Native graph reads served where the graph's partitions live. +//! +//! A graph algorithm reads every partition of the collection's graph, a +//! one-hop neighbors read reads one node's edges, a hop, path or subgraph +//! walk crosses partitions hop by hop, and a RAG fusion reads the collection's +//! indexes on its owner and its edges everywhere. Routed by collection, each +//! will read a single node's share. Each runs through the same coordinators +//! as the SQL surface (`graph_dispatch::whole_graph`, +//! `graph_dispatch::keyed_read`, `graph_dispatch::walk_reads`, +//! `graph_dispatch::rag_fusion`). Every running server has cluster routing, +//! the synthesized one-node cluster included, so a walk and a fusion run +//! through these coordinators on every server. A walk answers the payload +//! shape the Data Plane gives, and `response_shape::walk` turns its node +//! names into rows. The native protocol has no read-consistency setting, so +//! every one of these reads is strong. + +use nodedb_physical::physical_plan::GraphOp; +use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; +use nodedb_types::protocol::NativeResponse; + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::server::dispatch_utils::ok_payload_response; +use crate::types::VShardId; + +use super::response::data_plane_response_to_native; +use super::{DispatchCtx, error_to_native}; + +/// A graph read this module serves. +enum OwnerRead { + /// An algorithm, over the edges live at `Option` system time when set. + /// The parameters are boxed: they dwarf the other variants. + Algo( + nodedb_graph::GraphAlgorithm, + Box, + Option, + ), + Neighbors(String), + /// A `Hop`, `Path` or `Subgraph` in a cluster. + Walk, + /// A RAG fusion in a cluster. + Rag, +} + +/// Serve `plan` when it is a graph algorithm, a one-hop neighbors read, or a +/// walk or RAG fusion in a cluster. Returns `None` for every other plan. +pub(super) async fn dispatch_graph_owner_read( + ctx: &DispatchCtx<'_>, + seq: u64, + plan: &PhysicalPlan, + vshard_id: VShardId, +) -> Option { + let read = match plan { + PhysicalPlan::Graph(GraphOp::Algo { + algorithm, params, .. + }) => OwnerRead::Algo(*algorithm, Box::new(params.clone()), None), + PhysicalPlan::Graph(GraphOp::TemporalAlgorithm { + algorithm, + params, + system_time, + }) => match system_time { + nodedb_types::SystemTimeScope::Current => { + OwnerRead::Algo(*algorithm, Box::new(params.clone()), None) + } + nodedb_types::SystemTimeScope::AsOf(ms) => { + OwnerRead::Algo(*algorithm, Box::new(params.clone()), Some(*ms)) + } + nodedb_types::SystemTimeScope::AllVersions => { + return Some(error_to_native( + seq, + &crate::Error::BadRequest { + detail: "a graph algorithm runs over one system time; all-versions is not supported" + .into(), + }, + )); + } + }, + PhysicalPlan::Graph(GraphOp::Neighbors { node_id, .. }) => { + OwnerRead::Neighbors(node_id.clone()) + } + PhysicalPlan::Graph( + GraphOp::Hop { .. } | GraphOp::Path { .. } | GraphOp::Subgraph { .. }, + ) if ctx.state.cluster_routing.is_some() => OwnerRead::Walk, + PhysicalPlan::Graph(GraphOp::RagFusion { .. }) if ctx.state.cluster_routing.is_some() => { + OwnerRead::Rag + } + _ => return None, + }; + let tenant_id = ctx.tenant_id(); + let database_id = ctx.database_id(); + let txn_id = ctx.sessions.tx_id(ctx.peer_addr); + let task = PhysicalTask { + tenant_id, + vshard_id, + database_id, + plan: plan.clone(), + post_set_op: PostSetOp::None, + txn_id, + }; + if let Err(error) = super::sql_gateway::authorize_native_task(ctx, &task) { + return Some(error_to_native(seq, &error)); + } + let _request = ctx.state.tenant_request_guard(tenant_id); + let served = match read { + OwnerRead::Algo(algorithm, params, system_as_of_ms) => { + crate::control::server::graph_dispatch::run_graph_algo( + ctx.state, + tenant_id, + database_id, + algorithm, + *params, + system_as_of_ms, + true, + ) + .await + .map(ok_payload_response) + } + OwnerRead::Neighbors(node_id) => crate::control::server::graph_dispatch::read_on_key_owner( + ctx.state, + tenant_id, + database_id, + &node_id, + plan.clone(), + txn_id, + true, + ) + .await + .map(ok_payload_response), + OwnerRead::Walk => { + crate::control::server::graph_dispatch::serve_walk_plan( + ctx.state, + tenant_id, + database_id, + plan, + true, + ) + .await? + } + OwnerRead::Rag => { + crate::control::server::graph_dispatch::serve_rag_plan( + ctx.state, + tenant_id, + database_id, + plan, + true, + ) + .await? + } + }; + Some(match served { + Ok(response) => data_plane_response_to_native(ctx, seq, plan, &response), + Err(error) => error_to_native(seq, &error), + }) +} diff --git a/nodedb/src/control/server/native/dispatch/mod.rs b/nodedb/src/control/server/native/dispatch/mod.rs index 14cd8c68b..863edf2f9 100644 --- a/nodedb/src/control/server/native/dispatch/mod.rs +++ b/nodedb/src/control/server/native/dispatch/mod.rs @@ -8,8 +8,11 @@ mod cluster_array; mod conversion; mod ctx; mod direct_ops; +mod direct_orchestrated; +mod direct_txn; mod edge_recon_gate; mod graph_match; +mod graph_owner; mod index_ddl_op; mod limits; mod plan_builder; @@ -21,8 +24,10 @@ mod sorted_read_op; mod sql; mod sql_admin; mod sql_dispatch_task; +mod sql_fold; mod sql_gateway; mod sql_loop; +mod sql_planned; mod streaming; mod transaction; mod transaction_savepoint; diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/columnar.rs b/nodedb/src/control/server/native/dispatch/plan_builder/columnar.rs index 5f1d19d5b..27d146746 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/columnar.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/columnar.rs @@ -11,7 +11,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::server::native::dispatch::DispatchCtx; use nodedb_physical::physical_plan::{ColumnarInsertIntent, ColumnarOp}; -pub(crate) fn build_scan( +pub(crate) async fn build_scan( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -33,7 +33,7 @@ pub(crate) fn build_scan( })) } -pub(crate) fn build_insert( +pub(crate) async fn build_insert( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -46,13 +46,21 @@ pub(crate) fn build_insert( detail: "missing 'payload' or 'data'".to_string(), })? .clone(); - let format = fields.format.as_deref().unwrap_or("json").to_string(); + let format = fields.format.as_deref().unwrap_or("json"); // Decode the payload at the CP boundary so each row's primary key // can be resolved into a stable cross-engine identity before the // batch lands on the Data Plane. Row order matches the payload's // wire order one-to-one. - let surrogates = derive_surrogates(ctx, collection, &payload, &format)?; + let surrogates = derive_surrogates(ctx, collection, &payload, format).await?; + + // The Data Plane decodes a columnar payload as MessagePack only. A JSON + // payload converts here, at the API boundary, with its row order kept. + let (payload, format) = match format { + "json" => (json_payload_to_msgpack(collection, &payload)?, "msgpack"), + other => (payload, other), + }; + let format = format.to_string(); Ok(PhysicalPlan::Columnar(ColumnarOp::Insert { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -75,9 +83,9 @@ pub(crate) fn build_insert( /// Decode a bulk insert payload (JSON array of objects, or MessagePack /// array of maps), extract the conventional primary-key column from /// each row (`id` / `document_id` / `key`), and resolve a per-row -/// surrogate via the assigner. Rows with no PK column receive -/// `Surrogate::ZERO` (matching the SQL VALUES path's empty-PK fallback). -fn derive_surrogates( +/// surrogate through the async routed exchange. A row with no PK column gets +/// a fresh anonymous surrogate. +async fn derive_surrogates( ctx: &DispatchCtx<'_>, collection: &str, payload: &[u8], @@ -91,22 +99,52 @@ fn derive_surrogates( // and let the engine integration re-derive identity. _ => return Ok(Vec::new()), }; - let assigner = &ctx.state.surrogate_assigner; + // Every keyed row's identity in one batch at the collection's home. A + // row with no key gets a fresh anonymous surrogate. + let keyed: Vec<&[u8]> = pks + .iter() + .filter(|pk| !pk.is_empty()) + .map(Vec::as_slice) + .collect(); + let mut bound = super::helpers::assign_surrogates(ctx, collection, &keyed) + .await? + .into_iter(); let mut out = Vec::with_capacity(pks.len()); - for pk in pks { + for pk in &pks { if pk.is_empty() { - out.push(Surrogate::ZERO); + out.push( + ctx.state + .surrogate_assigner + .assign_anonymous( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + ) + .await?, + ); } else { - out.push(assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - &pk, - )?); + out.push(bound.next().ok_or_else(|| crate::Error::Internal { + detail: format!( + "columnar ingest into '{collection}': a keyed row got no surrogate" + ), + })?); } } Ok(out) } +/// The MessagePack form of a JSON columnar payload. +fn json_payload_to_msgpack(collection: &str, bytes: &[u8]) -> crate::Result> { + let value: serde_json::Value = + crate::util::bounded_json::from_slice(bytes).map_err(|e| crate::Error::Serialization { + format: "json".into(), + detail: format!("columnar insert into '{collection}': decode the JSON payload: {e}"), + })?; + nodedb_types::json_to_msgpack(&value).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("columnar insert into '{collection}': encode the payload: {e}"), + }) +} + fn extract_pks_json(bytes: &[u8]) -> crate::Result>> { let value: sonic_rs::Value = crate::util::bounded_json::from_slice(bytes).map_err(|e| crate::Error::Serialization { diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/crdt.rs b/nodedb/src/control/server/native/dispatch/plan_builder/crdt.rs index da2c204c9..b94022929 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/crdt.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/crdt.rs @@ -12,7 +12,7 @@ use nodedb_physical::physical_plan::CrdtOp; use super::super::DispatchCtx; use super::require_doc_id; -pub(crate) fn build_read( +pub(crate) async fn build_read( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -24,7 +24,7 @@ pub(crate) fn build_read( })) } -pub(crate) fn build_apply( +pub(crate) async fn build_apply( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -41,11 +41,8 @@ pub(crate) fn build_apply( crate::util::fnv1a_hash(&combined) }); - let surrogate = ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - document_id.as_bytes(), - )?; + let surrogate = + super::helpers::assign_surrogate(ctx, collection, document_id.as_bytes()).await?; Ok(PhysicalPlan::Crdt(CrdtOp::Apply { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -80,7 +77,7 @@ fn bounded_delta(fields: &TextFields) -> crate::Result> { Ok(delta.clone()) } -pub(crate) fn build_alter_policy( +pub(crate) async fn build_alter_policy( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -125,7 +122,7 @@ fn require_list_index(value: Option, field_name: &str) -> crate::Result, fields: &TextFields, collection: &str, @@ -142,11 +139,8 @@ pub(crate) fn build_list_insert( detail: "missing 'list_fields_json'".to_string(), })?; - let surrogate = ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - document_id.as_bytes(), - )?; + let surrogate = + super::helpers::assign_surrogate(ctx, collection, document_id.as_bytes()).await?; Ok(PhysicalPlan::Crdt(CrdtOp::ListInsert { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -158,7 +152,7 @@ pub(crate) fn build_list_insert( })) } -pub(crate) fn build_list_delete( +pub(crate) async fn build_list_delete( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -167,11 +161,8 @@ pub(crate) fn build_list_delete( let list_path = require_list_path(fields)?; let index = require_list_index(fields.list_index, "list_index")?; - let surrogate = ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - document_id.as_bytes(), - )?; + let surrogate = + super::helpers::assign_surrogate(ctx, collection, document_id.as_bytes()).await?; Ok(PhysicalPlan::Crdt(CrdtOp::ListDelete { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -182,7 +173,7 @@ pub(crate) fn build_list_delete( })) } -pub(crate) fn build_list_move( +pub(crate) async fn build_list_move( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -192,11 +183,8 @@ pub(crate) fn build_list_move( let from_index = require_list_index(fields.list_from_index, "list_from_index")?; let to_index = require_list_index(fields.list_to_index, "list_to_index")?; - let surrogate = ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - document_id.as_bytes(), - )?; + let surrogate = + super::helpers::assign_surrogate(ctx, collection, document_id.as_bytes()).await?; Ok(PhysicalPlan::Crdt(CrdtOp::ListMove { collection: QualifiedCollection::new(ctx.database_id(), collection), diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs b/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs index f86701665..a2625fd37 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/dispatch.rs @@ -14,7 +14,7 @@ use super::{ }; /// Build a PhysicalPlan from an opcode and request fields. -pub(crate) fn build_plan( +pub(crate) async fn build_plan( ctx: &DispatchCtx<'_>, op: OpCode, fields: &TextFields, @@ -22,75 +22,81 @@ pub(crate) fn build_plan( ) -> crate::Result { match op { // Point operations (collection-type-aware). - OpCode::PointGet => document::build_point_get(ctx, fields, collection), - OpCode::PointPut => document::build_point_put(ctx, fields, collection), - OpCode::PointDelete => document::build_point_delete(ctx, fields, collection), - OpCode::RangeScan => document::build_range_scan(ctx, fields, collection), - OpCode::DocumentBatchInsert => document::build_batch_insert(ctx, fields, collection), - OpCode::DocumentUpdate => document::build_update(ctx, fields, collection), - OpCode::DocumentScan => document::build_scan(ctx, fields, collection), - OpCode::DocumentUpsert => document::build_upsert(ctx, fields, collection), - OpCode::DocumentBulkUpdate => document_bulk::build_bulk_update(ctx, fields, collection), - OpCode::DocumentBulkDelete => document_bulk::build_bulk_delete(ctx, fields, collection), + OpCode::PointGet => document::build_point_get(ctx, fields, collection).await, + OpCode::PointPut => document::build_point_put(ctx, fields, collection).await, + OpCode::PointDelete => document::build_point_delete(ctx, fields, collection).await, + OpCode::RangeScan => document::build_range_scan(ctx, fields, collection).await, + OpCode::DocumentBatchInsert => document::build_batch_insert(ctx, fields, collection).await, + OpCode::DocumentUpdate => document::build_update(ctx, fields, collection).await, + OpCode::DocumentScan => document::build_scan(ctx, fields, collection).await, + OpCode::DocumentUpsert => document::build_upsert(ctx, fields, collection).await, + OpCode::DocumentBulkUpdate => { + document_bulk::build_bulk_update(ctx, fields, collection).await + } + OpCode::DocumentBulkDelete => { + document_bulk::build_bulk_delete(ctx, fields, collection).await + } // Vector. - OpCode::VectorSearch => vector::build_search(ctx, fields, collection), - OpCode::VectorBatchInsert => vector::build_batch_insert(ctx, fields, collection), - OpCode::VectorInsert => vector::build_insert(ctx, fields, collection), - OpCode::VectorMultiSearch => vector::build_multi_search(ctx, fields, collection), - OpCode::VectorDelete => vector::build_delete(ctx, fields, collection), + OpCode::VectorSearch => vector::build_search(ctx, fields, collection).await, + OpCode::VectorBatchInsert => vector::build_batch_insert(ctx, fields, collection).await, + OpCode::VectorInsert => vector::build_insert(ctx, fields, collection).await, + OpCode::VectorMultiSearch => vector::build_multi_search(ctx, fields, collection).await, + OpCode::VectorDelete => vector::build_delete(ctx, fields, collection).await, // Graph. - OpCode::GraphRagFusion => graph::build_rag_fusion(ctx, fields, collection), - OpCode::GraphHop => graph::build_hop(ctx, fields), - OpCode::GraphNeighbors => graph::build_neighbors(ctx, fields), - OpCode::GraphPath => graph::build_path(ctx, fields), - OpCode::GraphSubgraph => graph::build_subgraph(ctx, fields), - OpCode::EdgePut => graph::build_edge_put(ctx, fields, collection), - OpCode::EdgeDelete => graph::build_edge_delete(ctx, fields, collection), + OpCode::GraphRagFusion => graph::build_rag_fusion(ctx, fields, collection).await, + OpCode::GraphHop => graph::build_hop(ctx, fields).await, + OpCode::GraphNeighbors => graph::build_neighbors(ctx, fields).await, + OpCode::GraphPath => graph::build_path(ctx, fields).await, + OpCode::GraphSubgraph => graph::build_subgraph(ctx, fields).await, + OpCode::EdgePut => graph::build_edge_put(ctx, fields, collection).await, + OpCode::EdgeDelete => graph::build_edge_delete(ctx, fields, collection).await, // KV. - OpCode::KvScan => kv::build_scan(ctx, fields, collection), - OpCode::KvExpire => kv::build_expire(ctx, fields, collection), - OpCode::KvPersist => kv::build_persist(ctx, fields, collection), - OpCode::KvGetTtl => kv::build_get_ttl(ctx, fields, collection), - OpCode::KvBatchGet => kv::build_batch_get(ctx, fields, collection), - OpCode::KvBatchPut => kv::build_batch_put(ctx, fields, collection), - OpCode::KvFieldGet => kv::build_field_get(ctx, fields, collection), - OpCode::KvFieldSet => kv::build_field_set(ctx, fields, collection), + OpCode::KvScan => kv::build_scan(ctx, fields, collection).await, + OpCode::KvExpire => kv::build_expire(ctx, fields, collection).await, + OpCode::KvPersist => kv::build_persist(ctx, fields, collection).await, + OpCode::KvGetTtl => kv::build_get_ttl(ctx, fields, collection).await, + OpCode::KvBatchGet => kv::build_batch_get(ctx, fields, collection).await, + OpCode::KvBatchPut => kv::build_batch_put(ctx, fields, collection).await, + OpCode::KvFieldGet => kv::build_field_get(ctx, fields, collection).await, + OpCode::KvFieldSet => kv::build_field_set(ctx, fields, collection).await, // CRDT. - OpCode::CrdtRead => crdt::build_read(ctx, fields, collection), - OpCode::CrdtApply => crdt::build_apply(ctx, fields, collection), - OpCode::AlterCollectionPolicy => crdt::build_alter_policy(ctx, fields, collection), - OpCode::CrdtListInsert => crdt::build_list_insert(ctx, fields, collection), - OpCode::CrdtListDelete => crdt::build_list_delete(ctx, fields, collection), - OpCode::CrdtListMove => crdt::build_list_move(ctx, fields, collection), + OpCode::CrdtRead => crdt::build_read(ctx, fields, collection).await, + OpCode::CrdtApply => crdt::build_apply(ctx, fields, collection).await, + OpCode::AlterCollectionPolicy => crdt::build_alter_policy(ctx, fields, collection).await, + OpCode::CrdtListInsert => crdt::build_list_insert(ctx, fields, collection).await, + OpCode::CrdtListDelete => crdt::build_list_delete(ctx, fields, collection).await, + OpCode::CrdtListMove => crdt::build_list_move(ctx, fields, collection).await, // Text/Search. - OpCode::TextSearch => text::build_search(ctx, fields, collection), - OpCode::HybridSearch => text::build_hybrid_search(ctx, fields, collection), + OpCode::TextSearch => text::build_search(ctx, fields, collection).await, + OpCode::HybridSearch => text::build_hybrid_search(ctx, fields, collection).await, // Spatial. - OpCode::SpatialScan => spatial::build_scan(ctx, fields, collection), + OpCode::SpatialScan => spatial::build_scan(ctx, fields, collection).await, // Timeseries. - OpCode::TimeseriesScan => timeseries::build_scan(ctx, fields, collection), - OpCode::TimeseriesIngest => timeseries::build_ingest(ctx, fields, collection), + OpCode::TimeseriesScan => timeseries::build_scan(ctx, fields, collection).await, + OpCode::TimeseriesIngest => timeseries::build_ingest(ctx, fields, collection).await, // Columnar. - OpCode::ColumnarScan => columnar::build_scan(ctx, fields, collection), - OpCode::ColumnarInsert => columnar::build_insert(ctx, fields, collection), + OpCode::ColumnarScan => columnar::build_scan(ctx, fields, collection).await, + OpCode::ColumnarInsert => columnar::build_insert(ctx, fields, collection).await, // Graph DDL. - OpCode::GraphAlgo => graph::build_algo(fields, collection), - OpCode::GraphMatch => graph::build_match(fields, collection), + OpCode::GraphAlgo => graph::build_algo(fields, collection).await, + OpCode::GraphMatch => graph::build_match(fields, collection).await, // Document DDL. - OpCode::DocumentTruncate => document_bulk::build_truncate(ctx, collection), + OpCode::DocumentTruncate => document_bulk::build_truncate(ctx, collection).await, OpCode::DocumentEstimateCount => { - document_bulk::build_estimate_count(ctx, fields, collection) + document_bulk::build_estimate_count(ctx, fields, collection).await + } + OpCode::DocumentInsertSelect => { + document_bulk::build_insert_select(ctx, fields, collection).await } - OpCode::DocumentInsertSelect => document_bulk::build_insert_select(ctx, fields, collection), // KV truncate. - OpCode::KvTruncate => kv::build_truncate(ctx, collection), + OpCode::KvTruncate => kv::build_truncate(ctx, collection).await, // KV atomic operations. - OpCode::KvIncr => kv_counter::build_incr(ctx, collection, fields), - OpCode::KvIncrFloat => kv_counter::build_incr_float(ctx, collection, fields), - OpCode::KvCas => kv::build_cas(ctx, collection, fields), - OpCode::KvGetSet => kv::build_getset(ctx, collection, fields), + OpCode::KvIncr => kv_counter::build_incr(ctx, collection, fields).await, + OpCode::KvIncrFloat => kv_counter::build_incr_float(ctx, collection, fields).await, + OpCode::KvCas => kv::build_cas(ctx, collection, fields).await, + OpCode::KvGetSet => kv::build_getset(ctx, collection, fields).await, // Query. - OpCode::RecursiveScan => query::build_recursive_scan(ctx, fields, collection), + OpCode::RecursiveScan => query::build_recursive_scan(ctx, fields, collection).await, _ => Err(crate::Error::BadRequest { detail: format!("operation {op:?} not supported as direct dispatch"), }), diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs index 4916459b3..545327475 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/document.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/document.rs @@ -14,7 +14,7 @@ use nodedb_physical::physical_plan::{DocumentOp, KvOp, TimeseriesOp}; use super::super::DispatchCtx; use super::{collection_type, declared_primary_key, require_doc_id}; -pub(crate) fn build_point_get( +pub(crate) async fn build_point_get( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -41,15 +41,7 @@ pub(crate) fn build_point_get( }), Some(CollectionType::Document(_)) | None => { let pk_bytes = doc_id.as_bytes().to_vec(); - let surrogate = ctx - .state - .surrogate_assigner - .lookup( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - &pk_bytes, - )? - .unwrap_or(nodedb_types::Surrogate::ZERO); + let surrogate = super::helpers::existing_surrogate(ctx, collection, &pk_bytes).await?; Ok(PhysicalPlan::Document(DocumentOp::PointGet { collection: QualifiedCollection::new(ctx.database_id(), collection), document_id: doc_id, @@ -63,7 +55,7 @@ pub(crate) fn build_point_get( } } -pub(crate) fn build_point_put( +pub(crate) async fn build_point_put( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -73,11 +65,7 @@ pub(crate) fn build_point_put( match collection_type(ctx, collection)? { Some(CollectionType::KeyValue(_)) => { let key = doc_id.into_bytes(); - let surrogate = ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - &key, - )?; + let surrogate = super::helpers::assign_surrogate(ctx, collection, &key).await?; Ok(PhysicalPlan::Kv(KvOp::Put { collection: QualifiedCollection::new(ctx.database_id(), collection), key, @@ -94,10 +82,14 @@ pub(crate) fn build_point_put( let ilp_line = format!("{collection} value={json_str}\n"); // The line's own surrogate keys its staged row, so a read later in // the same transaction observes it. - let (surrogate, _identity) = ctx.state.surrogate_assigner.assign_fresh( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - )?; + let (surrogate, _identity) = ctx + .state + .surrogate_assigner + .assign_fresh( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + ) + .await?; Ok(PhysicalPlan::Timeseries(TimeseriesOp::Ingest { collection: QualifiedCollection::new(ctx.database_id(), collection), payload: ilp_line.into_bytes(), @@ -117,11 +109,7 @@ pub(crate) fn build_point_put( }), Some(CollectionType::Document(_)) | None => { let pk_bytes = doc_id.as_bytes().to_vec(); - let surrogate = ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - &pk_bytes, - )?; + let surrogate = super::helpers::assign_surrogate(ctx, collection, &pk_bytes).await?; Ok(PhysicalPlan::Document(DocumentOp::PointPut { collection: QualifiedCollection::new(ctx.database_id(), collection), document_id: doc_id, @@ -138,7 +126,7 @@ pub(crate) fn build_point_put( } } -pub(crate) fn build_point_delete( +pub(crate) async fn build_point_delete( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -168,16 +156,14 @@ pub(crate) fn build_point_delete( .to_string(), }), Some(CollectionType::Document(_)) | None => { + // A row of an edge-bearing collection is also a graph node. Its + // delete is a `BulkDelete` on its key, so the edge reconnaissance + // gate commits the node's edge tombstones with it. + if super::helpers::collection_is_edge_bearing(ctx, collection)? { + return edge_bearing_key_delete(ctx, collection, doc_id); + } let pk_bytes = doc_id.as_bytes().to_vec(); - let surrogate = ctx - .state - .surrogate_assigner - .lookup( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - &pk_bytes, - )? - .unwrap_or(nodedb_types::Surrogate::ZERO); + let surrogate = super::helpers::existing_surrogate(ctx, collection, &pk_bytes).await?; Ok(PhysicalPlan::Document(DocumentOp::PointDelete { collection: QualifiedCollection::new(ctx.database_id(), collection), document_id: doc_id, @@ -192,7 +178,45 @@ pub(crate) fn build_point_delete( } } -pub(crate) fn build_range_scan( +/// A `BulkDelete` of the one row whose identity column holds `doc_id`: the +/// declared primary key column, else `id`. +fn edge_bearing_key_delete( + ctx: &DispatchCtx<'_>, + collection: &str, + doc_id: String, +) -> crate::Result { + use crate::bridge::scan_filter::{FilterOp, ScanFilter}; + let declared_primary_key = declared_primary_key(ctx, collection)?; + let filter = ScanFilter { + field: declared_primary_key + .clone() + .unwrap_or_else(|| nodedb_types::DEFAULT_IDENTITY_COLUMN.to_string()), + op: FilterOp::Eq, + value: nodedb_types::Value::String(doc_id), + clauses: Vec::new(), + expr: None, + }; + let key_filters: Vec = std::iter::once(filter).collect(); + let filters = + zerompk::to_msgpack_vec(&key_filters).map_err(|e| crate::Error::Serialization { + format: "msgpack".into(), + detail: format!("key delete filter encode: {e}"), + })?; + Ok(PhysicalPlan::Document(DocumentOp::BulkDelete { + collection: QualifiedCollection::new(ctx.database_id(), collection), + filters, + returning: None, + ollp_predicted_surrogates: None, + ollp_predicted_edges: None, + rls_filters: Vec::new(), + rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), + // Filled in by the materialized-sum resolution pass. + resolved_sum_targets: Vec::new(), + declared_primary_key, + })) +} + +pub(crate) async fn build_range_scan( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -215,7 +239,7 @@ pub(crate) fn build_range_scan( })) } -pub(crate) fn build_batch_insert( +pub(crate) async fn build_batch_insert( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -231,20 +255,16 @@ pub(crate) fn build_batch_insert( detail: "documents array is empty".to_string(), }); } + // Every row's identity in one batch at the collection's home. + let pks: Vec<&[u8]> = batch_docs.iter().map(|d| d.id.as_bytes()).collect(); + let surrogates = super::helpers::assign_surrogates(ctx, collection, &pks).await?; let mut documents: Vec<(String, Vec)> = Vec::with_capacity(batch_docs.len()); - let mut surrogates: Vec = Vec::with_capacity(batch_docs.len()); for d in batch_docs { let value_bytes = sonic_rs::to_vec(&d.fields).map_err(|e| crate::Error::Serialization { format: "json".into(), detail: format!("failed to serialize document '{}': {e}", d.id), })?; - let surrogate = ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - d.id.as_bytes(), - )?; documents.push((d.id.clone(), value_bytes)); - surrogates.push(surrogate); } Ok(PhysicalPlan::Document(DocumentOp::BatchInsert { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -258,7 +278,7 @@ pub(crate) fn build_batch_insert( })) } -pub(crate) fn build_update( +pub(crate) async fn build_update( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -279,15 +299,7 @@ pub(crate) fn build_update( }) .collect(); let pk_bytes = doc_id.as_bytes().to_vec(); - let surrogate = ctx - .state - .surrogate_assigner - .lookup( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - &pk_bytes, - )? - .unwrap_or(nodedb_types::Surrogate::ZERO); + let surrogate = super::helpers::existing_surrogate(ctx, collection, &pk_bytes).await?; Ok(PhysicalPlan::Document(DocumentOp::PointUpdate { collection: QualifiedCollection::new(ctx.database_id(), collection), document_id: doc_id, @@ -308,19 +320,21 @@ pub(crate) fn build_update( /// above: a document scan reads the sparse store only, so every other engine /// takes its own scan builder. A spatial collection's plain scan is the /// columnar scan; the geometry query is `SpatialScan`. -pub(crate) fn build_scan( +pub(crate) async fn build_scan( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, ) -> crate::Result { match collection_type(ctx, collection)? { - Some(CollectionType::KeyValue(_)) => return super::kv::build_scan(ctx, fields, collection), + Some(CollectionType::KeyValue(_)) => { + return super::kv::build_scan(ctx, fields, collection).await; + } Some(CollectionType::Columnar(ColumnarProfile::Timeseries { .. })) => { - return super::timeseries::build_scan(ctx, fields, collection); + return super::timeseries::build_scan(ctx, fields, collection).await; } Some(CollectionType::Columnar(ColumnarProfile::Plain)) | Some(CollectionType::Columnar(ColumnarProfile::Spatial { .. })) => { - return super::columnar::build_scan(ctx, fields, collection); + return super::columnar::build_scan(ctx, fields, collection).await; } Some(CollectionType::Document(_)) | None => {} } @@ -342,18 +356,14 @@ pub(crate) fn build_scan( })) } -pub(crate) fn build_upsert( +pub(crate) async fn build_upsert( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, ) -> crate::Result { let doc_id = require_doc_id(fields)?; let value = fields.data.clone().unwrap_or_default(); - let surrogate = ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - doc_id.as_bytes(), - )?; + let surrogate = super::helpers::assign_surrogate(ctx, collection, doc_id.as_bytes()).await?; Ok(PhysicalPlan::Document(DocumentOp::Upsert { collection: QualifiedCollection::new(ctx.database_id(), collection), document_id: doc_id, diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/document_bulk.rs b/nodedb/src/control/server/native/dispatch/plan_builder/document_bulk.rs index 45681ae25..97cb16530 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/document_bulk.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/document_bulk.rs @@ -12,7 +12,7 @@ use nodedb_physical::physical_plan::DocumentOp; use super::super::DispatchCtx; use super::declared_primary_key; -pub(crate) fn build_bulk_update( +pub(crate) async fn build_bulk_update( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -54,7 +54,7 @@ pub(crate) fn build_bulk_update( })) } -pub(crate) fn build_bulk_delete( +pub(crate) async fn build_bulk_delete( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -81,7 +81,7 @@ pub(crate) fn build_bulk_delete( })) } -pub(crate) fn build_truncate( +pub(crate) async fn build_truncate( ctx: &DispatchCtx<'_>, collection: &str, ) -> crate::Result { @@ -95,7 +95,7 @@ pub(crate) fn build_truncate( })) } -pub(crate) fn build_estimate_count( +pub(crate) async fn build_estimate_count( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -108,7 +108,7 @@ pub(crate) fn build_estimate_count( })) } -pub(crate) fn build_insert_select( +pub(crate) async fn build_insert_select( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/graph.rs b/nodedb/src/control/server/native/dispatch/plan_builder/graph.rs index a69c0c785..6f6edd6d9 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/graph.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/graph.rs @@ -29,7 +29,7 @@ fn clamped_depth(value: Option, default: usize, field: &str) -> crate::Resu Ok(v) } -pub(crate) fn build_rag_fusion( +pub(crate) async fn build_rag_fusion( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -57,10 +57,14 @@ pub(crate) fn build_rag_fusion( options: Default::default(), bm25_query: None, bm25_field: None, + stage: nodedb_physical::physical_plan::RagStage::Local, })) } -pub(crate) fn build_hop(ctx: &DispatchCtx<'_>, fields: &TextFields) -> crate::Result { +pub(crate) async fn build_hop( + ctx: &DispatchCtx<'_>, + fields: &TextFields, +) -> crate::Result { let start = fields .start_node .as_ref() @@ -82,7 +86,7 @@ pub(crate) fn build_hop(ctx: &DispatchCtx<'_>, fields: &TextFields) -> crate::Re })) } -pub(crate) fn build_neighbors( +pub(crate) async fn build_neighbors( ctx: &DispatchCtx<'_>, fields: &TextFields, ) -> crate::Result { @@ -104,7 +108,7 @@ pub(crate) fn build_neighbors( })) } -pub(crate) fn build_path( +pub(crate) async fn build_path( ctx: &DispatchCtx<'_>, fields: &TextFields, ) -> crate::Result { @@ -135,7 +139,7 @@ pub(crate) fn build_path( })) } -pub(crate) fn build_subgraph( +pub(crate) async fn build_subgraph( ctx: &DispatchCtx<'_>, fields: &TextFields, ) -> crate::Result { @@ -158,7 +162,7 @@ pub(crate) fn build_subgraph( })) } -pub(crate) fn build_edge_put( +pub(crate) async fn build_edge_put( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -192,16 +196,9 @@ pub(crate) fn build_edge_put( })?, None => String::new(), }; - let src_surrogate = ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - src.as_bytes(), - )?; - let dst_surrogate = ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - dst.as_bytes(), - )?; + // The endpoint surrogates come from the collection home, the one place a + // key's surrogate is minted (`surrogate_exchange::authority`). + let [src_surrogate, dst_surrogate] = endpoint_surrogates(ctx, collection, src, dst).await?; Ok(PhysicalPlan::Graph(GraphOp::EdgePut { collection: QualifiedCollection::new(ctx.database_id(), collection), src_id: src.clone(), @@ -213,7 +210,7 @@ pub(crate) fn build_edge_put( })) } -pub(crate) fn build_edge_delete( +pub(crate) async fn build_edge_delete( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -241,19 +238,10 @@ pub(crate) fn build_edge_delete( .ok_or_else(|| crate::Error::BadRequest { detail: "missing 'edge_type'".to_string(), })?; - // Resolve endpoint surrogates exactly as `build_edge_put` does (get-or-assign - // returns the existing node identities) so a cross-shard delete dual-homes - // and locks against a concurrent insert of the same edge. - let src_surrogate = ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - src.as_bytes(), - )?; - let dst_surrogate = ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - dst.as_bytes(), - )?; + // The endpoint surrogates come from the collection home, as for + // `build_edge_put`, so a cross-shard delete dual-homes and locks against a + // concurrent insert of the same edge. + let [src_surrogate, dst_surrogate] = endpoint_surrogates(ctx, collection, src, dst).await?; Ok(PhysicalPlan::Graph(GraphOp::EdgeDelete { collection: QualifiedCollection::new(ctx.database_id(), collection), src_id: src.clone(), @@ -267,7 +255,10 @@ pub(crate) fn build_edge_delete( })) } -pub(crate) fn build_algo(fields: &TextFields, collection: &str) -> crate::Result { +pub(crate) async fn build_algo( + fields: &TextFields, + collection: &str, +) -> crate::Result { let algo_name = fields .algorithm .as_deref() @@ -312,7 +303,13 @@ pub(crate) fn build_algo(fields: &TextFields, collection: &str) -> crate::Result personalization_vector, }; - Ok(PhysicalPlan::Graph(GraphOp::Algo { algorithm, params })) + // The native dispatch runs the algorithm over every partition + // (`graph_owner`); the stage it builds from is chosen there. + Ok(PhysicalPlan::Graph(GraphOp::Algo { + algorithm, + params, + stage: nodedb_physical::physical_plan::AlgoStage::Local, + })) } /// Extract the Personalized PageRank seed map from the raw-protocol @@ -347,7 +344,10 @@ fn parse_algo_personalization( Ok(Some(map)) } -pub(crate) fn build_match(fields: &TextFields, _collection: &str) -> crate::Result { +pub(crate) async fn build_match( + fields: &TextFields, + _collection: &str, +) -> crate::Result { let query_str = fields .match_query .as_ref() @@ -370,6 +370,27 @@ pub(crate) fn build_match(fields: &TextFields, _collection: &str) -> crate::Resu })) } +/// Both endpoints' surrogates, in one batch at the collection's home. +async fn endpoint_surrogates( + ctx: &DispatchCtx<'_>, + collection: &str, + src: &str, + dst: &str, +) -> crate::Result<[nodedb_types::Surrogate; 2]> { + let bound = + super::helpers::assign_surrogates(ctx, collection, &[src.as_bytes(), dst.as_bytes()]) + .await?; + match bound.as_slice() { + [src_surrogate, dst_surrogate] => Ok([*src_surrogate, *dst_surrogate]), + _ => Err(crate::Error::Internal { + detail: format!( + "edge write in '{collection}': the home answered {} endpoint surrogates", + bound.len() + ), + }), + } +} + #[cfg(test)] mod tests { use super::*; @@ -390,45 +411,45 @@ mod tests { params } - #[test] - fn build_algo_parses_personalization_from_algo_params() { + #[tokio::test] + async fn build_algo_parses_personalization_from_algo_params() { let fields = algo_fields(Some(json!({ "personalization_vector": { "alice": 1.0, "bob": 0.5 } }))); - let pv = params_of(build_algo(&fields, "social").unwrap()) + let pv = params_of(build_algo(&fields, "social").await.unwrap()) .personalization_vector .expect("personalization present"); assert_eq!(pv.get("alice"), Some(&1.0)); assert_eq!(pv.get("bob"), Some(&0.5)); } - #[test] - fn build_algo_without_personalization_is_none() { + #[tokio::test] + async fn build_algo_without_personalization_is_none() { assert!( - params_of(build_algo(&algo_fields(None), "social").unwrap()) + params_of(build_algo(&algo_fields(None), "social").await.unwrap()) .personalization_vector .is_none() ); // An algo_params object that omits the key is also None. let fields = algo_fields(Some(json!({ "other": 1 }))); assert!( - params_of(build_algo(&fields, "social").unwrap()) + params_of(build_algo(&fields, "social").await.unwrap()) .personalization_vector .is_none() ); } - #[test] - fn build_algo_rejects_non_numeric_weight() { + #[tokio::test] + async fn build_algo_rejects_non_numeric_weight() { let fields = algo_fields(Some( json!({ "personalization_vector": { "alice": "high" } }), )); - assert!(build_algo(&fields, "social").is_err()); + assert!(build_algo(&fields, "social").await.is_err()); } - #[test] - fn build_algo_rejects_non_object_personalization() { + #[tokio::test] + async fn build_algo_rejects_non_object_personalization() { let fields = algo_fields(Some(json!({ "personalization_vector": [1, 2, 3] }))); - assert!(build_algo(&fields, "social").is_err()); + assert!(build_algo(&fields, "social").await.is_err()); } } diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/helpers.rs b/nodedb/src/control/server/native/dispatch/plan_builder/helpers.rs index 78a5976cf..9bea7a47f 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/helpers.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/helpers.rs @@ -24,6 +24,22 @@ pub(in crate::control::server::native::dispatch) fn collection_type( .map(|coll| coll.collection_type)) } +/// Whether an edge was ever written into `collection`. An absent collection +/// row is not edge-bearing. +pub(in crate::control::server::native::dispatch) fn collection_is_edge_bearing( + ctx: &DispatchCtx<'_>, + collection: &str, +) -> crate::Result { + let catalog = ctx.state.credentials.catalog(); + Ok(catalog + .get_collection( + ctx.database_id(), + ctx.identity.tenant_id.as_u64(), + collection, + )? + .is_some_and(|coll| coll.has_implicit_edges)) +} + /// `collection`'s DDL-declared `PRIMARY KEY` column name, for the apply-time /// NOT NULL guard on `PointUpdate` / `BulkUpdate`. `None` means no `PRIMARY /// KEY` was declared, so the guard has nothing to enforce. @@ -61,3 +77,56 @@ pub(in crate::control::server::native::dispatch) fn parse_direction( _ => crate::engine::graph::edge_store::Direction::Out, } } + +/// The surrogates of `pks` in `collection`, bound at the collection's home +/// when a key has none, in `pks` order. The async routed exchange answers +/// them in one batch, and a binding this node's catalog holds answers without +/// a request. +pub(super) async fn assign_surrogates( + ctx: &DispatchCtx<'_>, + collection: &str, + pks: &[&[u8]], +) -> crate::Result> { + crate::control::server::surrogate_exchange::assign_surrogates_routed( + ctx.state, + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + pks, + crate::types::TraceId::ZERO, + ) + .await +} + +/// [`assign_surrogates`] for one key. +pub(super) async fn assign_surrogate( + ctx: &DispatchCtx<'_>, + collection: &str, + pk: &[u8], +) -> crate::Result { + crate::control::server::surrogate_exchange::assign_surrogate_routed( + ctx.state, + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + pk, + crate::types::TraceId::ZERO, + ) + .await +} + +/// The surrogate `pk` is bound to in `collection`, or `None` when the key +/// names no row: the home's binding, through the async routed exchange. +/// Never binds. +pub(super) async fn existing_surrogate( + ctx: &DispatchCtx<'_>, + collection: &str, + pk: &[u8], +) -> crate::Result> { + crate::control::server::surrogate_exchange::lookup_surrogate_routed( + ctx.state, + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + pk, + crate::types::TraceId::ZERO, + ) + .await +} diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs index 14232fafe..3d78ac297 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv.rs @@ -9,7 +9,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::server::native::dispatch::DispatchCtx; use nodedb_physical::physical_plan::KvOp; -pub(crate) fn build_scan( +pub(crate) async fn build_scan( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -32,7 +32,7 @@ pub(crate) fn build_scan( })) } -pub(crate) fn build_expire( +pub(crate) async fn build_expire( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -52,7 +52,7 @@ pub(crate) fn build_expire( })) } -pub(crate) fn build_persist( +pub(crate) async fn build_persist( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -66,7 +66,7 @@ pub(crate) fn build_persist( })) } -pub(crate) fn build_get_ttl( +pub(crate) async fn build_get_ttl( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -79,7 +79,7 @@ pub(crate) fn build_get_ttl( })) } -pub(crate) fn build_batch_get( +pub(crate) async fn build_batch_get( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -104,7 +104,7 @@ pub(crate) fn build_batch_get( })) } -pub(crate) fn build_batch_put( +pub(crate) async fn build_batch_put( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -123,15 +123,12 @@ pub(crate) fn build_batch_put( } let ttl_ms = fields.ttl_ms.unwrap_or(0); - // Assign each entry's stable cross-engine surrogate the SAME way a - // single-key `Put` does (`assign_kv_surrogate` below): an existing key - // resolves to its already-bound surrogate, a new key mints a fresh one. - // Without this every batch-put row would land with `Surrogate::ZERO`, - // making it invisible to any surrogate-keyed cross-engine read/join. - let surrogates = entries - .iter() - .map(|(key, _value)| assign_kv_surrogate(ctx, collection, key)) - .collect::>>()?; + // Every entry's stable cross-engine surrogate, in one batch at the + // collection's home: an existing key resolves to its bound surrogate, a + // new key mints one. A row left at `Surrogate::ZERO` is invisible to any + // surrogate-keyed cross-engine read or join. + let keys: Vec<&[u8]> = entries.iter().map(|(key, _value)| key.as_slice()).collect(); + let surrogates = super::helpers::assign_surrogates(ctx, collection, &keys).await?; Ok(PhysicalPlan::Kv(KvOp::BatchPut { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -143,7 +140,7 @@ pub(crate) fn build_batch_put( })) } -pub(crate) fn build_field_get( +pub(crate) async fn build_field_get( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -165,7 +162,7 @@ pub(crate) fn build_field_get( })) } -pub(crate) fn build_field_set( +pub(crate) async fn build_field_set( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -178,7 +175,7 @@ pub(crate) fn build_field_set( detail: "missing 'updates'".to_string(), })? .clone(); - let surrogate = assign_kv_surrogate(ctx, collection, &key)?; + let surrogate = super::helpers::assign_surrogate(ctx, collection, &key).await?; Ok(PhysicalPlan::Kv(KvOp::FieldSet { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -207,7 +204,7 @@ fn require_key_bytes(fields: &TextFields) -> crate::Result> { }) } -pub(crate) fn build_truncate( +pub(crate) async fn build_truncate( ctx: &DispatchCtx<'_>, collection: &str, ) -> crate::Result { @@ -217,22 +214,7 @@ pub(crate) fn build_truncate( })) } -/// Resolve the stable cross-engine surrogate for a KV atomic op, content- -/// addressed on `(collection, key)` — the same binding a normal insert of that -/// key allocated, so an atomic op on an existing key keeps its identity. -pub(super) fn assign_kv_surrogate( - ctx: &DispatchCtx<'_>, - collection: &str, - key: &[u8], -) -> crate::Result { - ctx.state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - key, - ) -} - -pub(crate) fn build_cas( +pub(crate) async fn build_cas( ctx: &DispatchCtx<'_>, collection: &str, fields: &TextFields, @@ -250,7 +232,7 @@ pub(crate) fn build_cas( .ok_or_else(|| crate::Error::BadRequest { detail: "missing 'new_value'".to_string(), })?; - let surrogate = assign_kv_surrogate(ctx, collection, key.as_bytes())?; + let surrogate = super::helpers::assign_surrogate(ctx, collection, key.as_bytes()).await?; Ok(PhysicalPlan::Kv(KvOp::Cas { collection: QualifiedCollection::new(ctx.database_id(), collection), @@ -262,7 +244,7 @@ pub(crate) fn build_cas( })) } -pub(crate) fn build_getset( +pub(crate) async fn build_getset( ctx: &DispatchCtx<'_>, collection: &str, fields: &TextFields, @@ -279,7 +261,7 @@ pub(crate) fn build_getset( .ok_or_else(|| crate::Error::BadRequest { detail: "missing 'new_value'".to_string(), })?; - let surrogate = assign_kv_surrogate(ctx, collection, key.as_bytes())?; + let surrogate = super::helpers::assign_surrogate(ctx, collection, key.as_bytes()).await?; Ok(PhysicalPlan::Kv(KvOp::GetSet { collection: QualifiedCollection::new(ctx.database_id(), collection), diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs b/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs index 017b9cc1a..eaeaf6329 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/kv_counter.rs @@ -7,7 +7,6 @@ use nodedb_sql::planner::dml_helpers::KvCounterKind; use nodedb_types::QualifiedCollection; use nodedb_types::protocol::TextFields; -use super::kv::assign_kv_surrogate; use crate::bridge::envelope::PhysicalPlan; use crate::control::planner::sql_plan_convert::kv_counter_shape::kv_counter_shape; use crate::control::server::native::dispatch::DispatchCtx; @@ -21,7 +20,7 @@ fn required_key(fields: &TextFields) -> crate::Result<&str> { }) } -pub(crate) fn build_incr( +pub(crate) async fn build_incr( ctx: &DispatchCtx<'_>, collection: &str, fields: &TextFields, @@ -29,7 +28,7 @@ pub(crate) fn build_incr( let key = required_key(fields)?; let delta = fields.incr_delta.unwrap_or(1); let ttl_ms = fields.ttl_ms.unwrap_or(0); - let surrogate = assign_kv_surrogate(ctx, collection, key.as_bytes())?; + let surrogate = super::helpers::assign_surrogate(ctx, collection, key.as_bytes()).await?; // An absent key takes the collection's shape, as a SQL `KV_INCR` does. let shape = kv_counter_shape( ctx.state, @@ -51,7 +50,7 @@ pub(crate) fn build_incr( })) } -pub(crate) fn build_incr_float( +pub(crate) async fn build_incr_float( ctx: &DispatchCtx<'_>, collection: &str, fields: &TextFields, @@ -65,7 +64,7 @@ pub(crate) fn build_incr_float( detail: format!("KvIncrFloat: delta must be a decimal number, got '{delta}'"), }); } - let surrogate = assign_kv_surrogate(ctx, collection, key.as_bytes())?; + let surrogate = super::helpers::assign_surrogate(ctx, collection, key.as_bytes()).await?; let shape = kv_counter_shape( ctx.state, ctx.tenant_id(), diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/query.rs b/nodedb/src/control/server/native/dispatch/plan_builder/query.rs index 2c62ad1f4..ea679c401 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/query.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/query.rs @@ -9,7 +9,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::server::native::dispatch::DispatchCtx; use nodedb_physical::physical_plan::QueryOp; -pub(crate) fn build_recursive_scan( +pub(crate) async fn build_recursive_scan( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/spatial.rs b/nodedb/src/control/server/native/dispatch/plan_builder/spatial.rs index 1ed5a7678..1e8c3530f 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/spatial.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/spatial.rs @@ -9,7 +9,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::server::native::dispatch::DispatchCtx; use nodedb_physical::physical_plan::{SpatialOp, SpatialPredicate}; -pub(crate) fn build_scan( +pub(crate) async fn build_scan( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/text.rs b/nodedb/src/control/server/native/dispatch/plan_builder/text.rs index 2cb1c02ac..3f95c7255 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/text.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/text.rs @@ -9,7 +9,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::server::native::dispatch::DispatchCtx; use nodedb_physical::physical_plan::TextOp; -pub(crate) fn build_search( +pub(crate) async fn build_search( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -33,7 +33,7 @@ pub(crate) fn build_search( })) } -pub(crate) fn build_hybrid_search( +pub(crate) async fn build_hybrid_search( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/timeseries.rs b/nodedb/src/control/server/native/dispatch/plan_builder/timeseries.rs index 319ae1dc1..941e16625 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/timeseries.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/timeseries.rs @@ -9,7 +9,7 @@ use crate::bridge::envelope::PhysicalPlan; use crate::control::server::native::dispatch::DispatchCtx; use nodedb_physical::physical_plan::TimeseriesOp; -pub(crate) fn build_scan( +pub(crate) async fn build_scan( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -45,7 +45,7 @@ pub(crate) fn build_scan( })) } -pub(crate) fn build_ingest( +pub(crate) async fn build_ingest( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, diff --git a/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs b/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs index 37a682aac..f1ccb99be 100644 --- a/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs +++ b/nodedb/src/control/server/native/dispatch/plan_builder/vector.rs @@ -10,7 +10,7 @@ use super::super::DispatchCtx; use crate::bridge::envelope::PhysicalPlan; use nodedb_physical::physical_plan::VectorOp; -pub(crate) fn build_search( +pub(crate) async fn build_search( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -49,7 +49,7 @@ pub(crate) fn build_search( })) } -pub(crate) fn build_batch_insert( +pub(crate) async fn build_batch_insert( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -74,10 +74,14 @@ pub(crate) fn build_batch_insert( let assigner = &ctx.state.surrogate_assigner; let mut surrogates = Vec::with_capacity(vectors.len()); for _ in &vectors { - surrogates.push(assigner.assign_anonymous( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - )?); + surrogates.push( + assigner + .assign_anonymous( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + ) + .await?, + ); } Ok(PhysicalPlan::Vector(VectorOp::BatchInsert { @@ -88,7 +92,7 @@ pub(crate) fn build_batch_insert( })) } -pub(crate) fn build_insert( +pub(crate) async fn build_insert( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -107,18 +111,16 @@ pub(crate) fn build_insert( let assigner = &ctx.state.surrogate_assigner; let (surrogate, pk_bytes) = match fields.document_id.as_deref() { Some(pk) if !pk.is_empty() => ( - assigner.assign( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - pk.as_bytes(), - )?, + super::helpers::assign_surrogate(ctx, collection, pk.as_bytes()).await?, Some(pk.as_bytes().to_vec()), ), _ => ( - assigner.assign_anonymous( - nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), - ctx.tenant_id(), - )?, + assigner + .assign_anonymous( + nodedb_types::CollectionKey::from_bare(ctx.database_id(), collection), + ctx.tenant_id(), + ) + .await?, None, ), }; @@ -134,7 +136,7 @@ pub(crate) fn build_insert( })) } -pub(crate) fn build_multi_search( +pub(crate) async fn build_multi_search( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, @@ -158,7 +160,7 @@ pub(crate) fn build_multi_search( })) } -pub(crate) fn build_delete( +pub(crate) async fn build_delete( ctx: &DispatchCtx<'_>, fields: &TextFields, collection: &str, diff --git a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs index 665ee98f9..298667222 100644 --- a/nodedb/src/control/server/native/dispatch/raw_dispatch.rs +++ b/nodedb/src/control/server/native/dispatch/raw_dispatch.rs @@ -62,37 +62,35 @@ pub(super) async fn dispatch_authorized_single_task( // resolves to a concrete row set here, while the identity is live, the // way the planned native and pgwire writes resolve. if checked.txn_id().is_none() - && ctx.state.async_raft_proposer().is_some() && let Some(resolver) = crate::control::write_resolve::resolver_for_plan(checked.plan()) { + let (authorized, _lease) = checked.into_parts(); return crate::control::write_resolve::run_authorized_write_resolve( - ctx.state, - checked.into_authorized(), - resolver, + ctx.state, authorized, resolver, ) .await; } + let gateway = ctx.state.installed_gateway()?; // A staged write and the other transaction meta-ops run on the core of - // the task's own vShard. The gateway would route them to vShard 0. - let gateway = ctx - .state - .gateway - .get() - .filter(|_| !is_task_vshard_scoped(checked.plan())); - match gateway { - Some(gateway) => { - let query = GatewayQueryContext { - tenant_id, - trace_id: TraceId::generate(), - database_id: ctx.database_id(), - txn_id, - }; - // The typed error passes through unchanged. The native frame - // renders its SQLSTATE and numeric code from it. - gateway.execute_response(&query, checked).await - } - None => dispatch_without_gateway(ctx, checked).await, + // the task's own vShard. The gateway will route them to vShard 0. + if is_task_vshard_scoped(checked.plan()) { + return dispatch_utils::dispatch_authorized_durable_write( + ctx.state, + checked, + TraceId::ZERO, + ) + .await; } + let query = GatewayQueryContext { + tenant_id, + trace_id: TraceId::generate(), + database_id: ctx.database_id(), + txn_id, + linearizable: true, + }; + // The typed error passes through unchanged. The native frame renders its + // SQLSTATE and numeric code from it. + gateway.execute_response(&query, checked).await } async fn dispatch_external_crdt_apply( @@ -150,7 +148,7 @@ async fn dispatch_external_crdt_apply( // misses a policy on a non-default database. `collection` is already // qualified here (the plan builder qualifies it against `ctx.database_id()` // before this dispatch runs), so no re-qualification happens — doing so - // would double-prefix the name. + // will double-prefix the name. let qualified_collection = collection.as_str(); let policy = crate::control::crdt_post_image_policy::ExternalCrdtPostImagePolicy::from_identity( tenant_id, @@ -185,27 +183,3 @@ async fn dispatch_external_crdt_apply( write_set: Vec::new(), }) } - -pub(super) async fn dispatch_without_gateway( - ctx: &DispatchCtx<'_>, - checked: crate::control::server::shared::clone_write::CloneCheckedTask, -) -> crate::Result { - let vshard_id = checked.vshard_id(); - let frontier_mutation = checked.txn_id().is_none() - && matches!( - checked.plan(), - PhysicalPlan::Crdt(op) - if crate::control::crdt_admission::changes_crdt_frontier(op) - ); - let write = || async move { - dispatch_utils::dispatch_authorized_durable_write(ctx.state, checked, TraceId::ZERO).await - }; - if frontier_mutation { - ctx.state - .vshard_admission_sequencer - .run(vshard_id, write) - .await - } else { - write().await - } -} diff --git a/nodedb/src/control/server/native/dispatch/sql.rs b/nodedb/src/control/server/native/dispatch/sql.rs index d1f177a79..7afee5132 100644 --- a/nodedb/src/control/server/native/dispatch/sql.rs +++ b/nodedb/src/control/server/native/dispatch/sql.rs @@ -3,32 +3,24 @@ //! SQL dispatch: DataFusion planning + Data Plane execution. use nodedb_types::protocol::NativeResponse; +use nodedb_types::strip_prefix_ascii_case_insensitive; use nodedb_types::value::Value; -use nodedb_types::{TraceId, strip_prefix_ascii_case_insensitive}; use std::sync::Arc; -use crate::control::planner::calvin::{ - CrossShardTxnMode, DispatchClass, TxnDispatchPosition, classify_dispatch, - dispatch_authorized_tasks_to_calvin, -}; use crate::control::security::audit::ArcAuditEmitter; +use crate::control::server::native::sqlstate_code::sqlstate_error; use crate::control::server::shared::authorization::authorize_database; -use crate::control::server::shared::plan_admission::{ - PlanAdmissionRequest, plan_authorize_and_admit, -}; use crate::control::server::shared::session::TransactionState; -use crate::control::server::shared::write_admission::all_writes_bufferable; use super::sql_admin::{handle_explain, handle_set_sql, handle_show_sql, is_session_show}; -use super::sql_loop::run_dispatch_loop; -use super::streaming::{SqlOutcome, try_open_sql_stream}; +use super::sql_planned::execute_planned; +use super::streaming::SqlOutcome; use super::transaction::{handle_begin, handle_commit, handle_rollback}; use super::transaction_savepoint::{ handle_release_savepoint, handle_rollback_to_savepoint, handle_savepoint, }; use super::{DispatchCtx, error_to_native, handle_reset}; -use crate::control::server::native::sqlstate_code::sqlstate_error; /// Handle a SQL statement: transaction control, SET/SHOW, DDL, or DataFusion. /// @@ -184,7 +176,7 @@ async fn handle_sql_inner( } // DataFusion planning + dispatch. The streaming fast path (when - // `allow_stream`) may return a `SqlStream`; otherwise this collapses to a + // `allow_stream`) can return a `SqlStream`; otherwise this collapses to a // single materialized `NativeResponse`. let _request = ctx.state.tenant_request_guard(ctx.tenant_id()); let outcome = execute_planned(ctx, seq, sql_trimmed, database_id, allow_stream).await; @@ -200,229 +192,10 @@ async fn handle_sql_inner( /// Wrap a materialized response as a non-streaming [`SqlOutcome`]. #[inline] -fn resp(r: NativeResponse) -> SqlOutcome { +pub(super) fn resp(r: NativeResponse) -> SqlOutcome { SqlOutcome::Response(Box::new(r)) } -/// Plan SQL via DataFusion and dispatch tasks to the Data Plane. -/// -/// When `allow_stream` is set and the planned statement is an eligible -/// autocommit, single-task, unordered multi-row SELECT, returns -/// [`SqlOutcome::Stream`] for lazy frame emission. Every other case — writes, -/// in-block buffering, multi-task, set-ops, errors — collapses to a single -/// [`SqlOutcome::Response`]. -async fn execute_planned( - ctx: &DispatchCtx<'_>, - seq: u64, - sql: &str, - database_id: crate::types::DatabaseId, - allow_stream: bool, -) -> SqlOutcome { - // `ctx.scope` is the single request-scoped auth contract, built once per - // request in `session::request::handle_request` — it already carries - // `database_id` (agreeing with the `database_id` passed into this - // function) and a scope-grant-enriched `AuthContext`. A per-query - // `ON DENY` override (e.g. `SELECT ... ON DENY ERROR 'CODE' MESSAGE - // '...'`) rebuilds the scope rather than mutating it in place - // (`RequestAuthScope` has no `&mut` path to `on_deny_override` by - // design), so a clone is taken here: `ctx.scope` stays the canonical, - // unmodified scope for any other consumer of `ctx` during this request, - // while `scope` below is the (possibly overridden) one this statement - // dispatches and admits under. - let (clean_sql, scope) = - crate::control::server::session_auth::apply_per_query_on_deny(sql, ctx.scope.clone()); - - // Forward every per-session planning GUC (vector-dim quota, force-shuffle - // join/agg overrides + partition counts, broadcast / shuffle-aggregate cost - // thresholds) into the shared query context before planning — the same - // protocol-neutral resolution pgwire performs, so the canonical native - // transport honors these overrides identically. Native plans without a plan - // cache, so the returned bypass flags are not needed here. - crate::control::server::shared::planning_overrides::apply_planning_session_overrides( - ctx.query_ctx, - ctx.sessions, - ctx.state, - ctx.peer_addr, - ctx.tenant_id(), - ); - - // Planning, authorization, implicit-edge extraction (pgwire parity: a - // schemaless document carrying `_from`/`_to` is mirrored as a - // `GraphOp::EdgePut` task so the classify/Calvin/single-shard logic below - // routes it like an explicit edge) and lease admission run as ONE retried - // unit, so a descriptor drain starting between the planner's catalog read - // and the lease acquisition is absorbed rather than surfaced. Admission - // still follows authorization inside the unit, so denied requests consume - // no descriptor lease. The scope stays alive through all dispatch and - // response shaping below. - let admission = match plan_authorize_and_admit(PlanAdmissionRequest { - state: ctx.state, - query_ctx: ctx.query_ctx, - scope: &scope, - sql: &clean_sql, - trace_id: TraceId::ZERO, - }) - .await - { - Ok(admission) => admission, - Err(error) => return resp(error_to_native(seq, &error)), - }; - - let mut tasks = admission.tasks; - let output_schema = admission.output_schema; - // Re-derived here rather than carried from `admission`: Calvin dispatch - // and implicit-edge reconciliation are trusted internal mechanisms that - // never reach the clone-checked dispatch boundary (they don't call - // `dispatch_authorized_to_data_plane` / `Gateway::execute`), so a plain - // batch authorize is correct for them. `run_dispatch_loop` below clone-checks - // and authorizes each task itself, immediately before its own dispatch. - let mut authorized_tasks = - match crate::control::server::shared::authorization::authorize_task_set( - ctx.identity, - &tasks, - &ctx.state.permissions, - &ctx.state.roles, - &ArcAuditEmitter(Arc::clone(&ctx.state.audit)), - ) { - Ok(authorized) => authorized, - Err(error) => return resp(error_to_native(seq, &crate::Error::from(error))), - }; - let mut lease_scope = Some(admission.lease_scope); - // Covers the images every cross-shard materialized-sum balance in `tasks` - // was settled from, so Calvin's OCC check aborts rather than committing a - // total folded from an image that has since moved. - let sum_target_reads = admission.sum_target_reads; - - if tasks.is_empty() { - return resp(NativeResponse::status_row(seq, "OK")); - } - - // Implicit-edge DELETE/UPDATE routing gate (native-protocol parity with - // pgwire). See `edge_recon_gate` for the full invariant and guard - // documentation. Returns early when the gate fires, consuming `tasks`. - { - use super::edge_recon_gate::{EdgeReconResult, try_edge_recon_dispatch}; - match try_edge_recon_dispatch(ctx, seq, tasks, authorized_tasks).await { - EdgeReconResult::Outcome(outcome) => return outcome, - EdgeReconResult::NotFired(returned_tasks, returned_authorized) => { - tasks = returned_tasks; - authorized_tasks = returned_authorized; - } - } - } - - // Cross-shard write parity with pgwire: classify the planned task set and, - // for a strict multi-shard write, route the whole batch through the Calvin - // sequencer so it commits atomically. Single-shard (and best-effort) keep - // the existing per-task gateway/SPSC dispatch loop below unchanged. - // Autocommit single-statement dispatch: no session read-set to widen with. - let sum_read_vshards = match crate::control::planner::calvin::read_vshards_of(&sum_target_reads) - { - Ok(vshards) => vshards, - Err(error) => return resp(error_to_native(seq, &error)), - }; - match classify_dispatch(&tasks, &sum_read_vshards) { - DispatchClass::SingleShard { .. } => {} - DispatchClass::MultiShard { .. } => { - // Dispatching to Calvin here applies the statement durably at - // statement time, escaping the transaction buffer. Inside a block, - // fall through to the per-task staging gate when the gate can - // buffer every write — COMMIT flushes the whole buffer through - // Calvin. Anything else is refused, matching pgwire. - let in_txn_block = - ctx.sessions.transaction_state(ctx.peer_addr) == TransactionState::InBlock; - if in_txn_block && !all_writes_bufferable(&tasks) { - return resp(error_to_native( - seq, - &crate::Error::CrossShardInExplicitTransaction, - )); - } - - // Native has no per-session `cross_shard_txn` parameter wired, so it - // reads the same `SessionStore` accessor pgwire uses; an unset value - // defaults to `CrossShardTxnMode::Strict` (the documented default), - // so native multi-shard writes route through Calvin by default. - let cross_shard_mode = ctx.sessions.cross_shard_txn_mode(ctx.peer_addr); - if !in_txn_block && cross_shard_mode == CrossShardTxnMode::Strict { - return match dispatch_authorized_tasks_to_calvin( - ctx.state, - authorized_tasks, - ctx.tenant_id(), - cross_shard_mode, - TxnDispatchPosition::Autocommit, - &sum_target_reads, - None, - ) - .await - { - // Calvin committed. A RETURNING write surfaces its rows from - // the applied Response; a plain write reports the affected - // count its own mutation returned. - Ok(apply_resp) => { - let plans: Vec<_> = tasks.iter().map(|t| t.plan.clone()).collect(); - resp(super::conversion::calvin_native_response( - seq, - apply_resp, - &plans, - ctx.state, - database_id, - ctx.tenant_id(), - ctx.auth_context(), - )) - } - Err(e) => resp(error_to_native(seq, &e)), - }; - } - // An in-block statement and `BestEffortNonAtomic` both fall - // through to the per-task loop below. - } - } - - // A native lazy stream outlives this handler, so transfer its descriptor - // leases to the session-owned `SqlStream` before returning it. The session - // loop retains that owner through final emission or connection teardown. - if allow_stream { - match try_open_sql_stream(ctx, seq, &tasks, database_id, Some(&output_schema)).await { - Ok(Some(mut stream)) => { - let Some(scope) = lease_scope.take() else { - return resp(sqlstate_error( - seq, - nodedb_types::error::sqlstate::INTERNAL_ERROR, - "internal error: query lease scope missing before SQL stream dispatch", - )); - }; - if let Err(error) = stream.attach_lease_scope(scope) { - return resp(error_to_native(seq, &error)); - } - return SqlOutcome::Stream(Box::new(stream)); - } - Ok(None) => {} - Err(error) => return resp(error_to_native(seq, &error)), - } - } - - // Materialized statements retain their admitted scope in an Arc so any - // writes buffered during this statement keep the same descriptor leases - // after this local owner is dropped. Lazy streams above retain the raw - // scope directly in their stream owner. - let Some(lease_scope) = lease_scope.take() else { - return resp(sqlstate_error( - seq, - nodedb_types::error::sqlstate::INTERNAL_ERROR, - "internal error: query lease scope missing before materialized SQL dispatch", - )); - }; - run_dispatch_loop( - ctx, - seq, - tasks, - Some(&output_schema), - database_id, - Arc::new(lease_scope), - ) - .await -} - // ─── Bound parameter substitution ──────────────────────────────────── // // The native protocol carries bound parameters in `TextFields::sql_params` diff --git a/nodedb/src/control/server/native/dispatch/sql_admin.rs b/nodedb/src/control/server/native/dispatch/sql_admin.rs index 731bbf6b9..194c6cd5d 100644 --- a/nodedb/src/control/server/native/dispatch/sql_admin.rs +++ b/nodedb/src/control/server/native/dispatch/sql_admin.rs @@ -1,8 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 //! SET / SHOW / RESET (SQL form) and EXPLAIN handling for the native SQL -//! dispatch path. Split out of `sql.rs` to keep that file under the -//! file-size limit; behavior is unchanged. +//! dispatch path. use nodedb_sql::parser::preprocess::lex::find_ascii_case_insensitive; use nodedb_types::protocol::NativeResponse; @@ -101,13 +100,12 @@ pub(super) async fn handle_explain(ctx: &DispatchCtx<'_>, seq: u64, sql: &str) - }; } - let perm_cache = - match crate::control::security::auth_fence::permission_view(ctx.state, ctx.tenant_id()) + if let Err(e) = + crate::control::security::auth_fence::admit_permission_view(ctx.state, ctx.tenant_id()) .await - { - Ok(view) => view, - Err(e) => return error_to_native(seq, &e), - }; + { + return error_to_native(seq, &e); + } let sec = crate::control::planner::context::PlanSecurityContext { identity: ctx.identity, auth: ctx.auth_context(), @@ -115,7 +113,9 @@ pub(super) async fn handle_explain(ctx: &DispatchCtx<'_>, seq: u64, sql: &str) - redaction_store: &ctx.state.redaction, permissions: &ctx.state.permissions, roles: &ctx.state.roles, - permission_cache: Some(&*perm_cache), + permission_tree: crate::control::planner::context::PermissionTreeSource::Live( + &ctx.state.permission_cache, + ), }; let database_id = ctx.database_id(); match ctx @@ -129,7 +129,6 @@ pub(super) async fn handle_explain(ctx: &DispatchCtx<'_>, seq: u64, sql: &str) - .await { Ok((tasks, _output_schema)) => { - drop(perm_cache); // EXPLAIN is metadata-only. Authorize the original plan to protect // metadata, but never materialize implicit edges while describing it. let emitter = crate::control::security::audit::ArcAuditEmitter(std::sync::Arc::clone( diff --git a/nodedb/src/control/server/native/dispatch/sql_dispatch_task.rs b/nodedb/src/control/server/native/dispatch/sql_dispatch_task.rs index 9d5a8822c..78a07f960 100644 --- a/nodedb/src/control/server/native/dispatch/sql_dispatch_task.rs +++ b/nodedb/src/control/server/native/dispatch/sql_dispatch_task.rs @@ -92,10 +92,8 @@ pub(super) async fn dispatch_task( } // A governed predicate resolves to a concrete row set before proposing - // (`control::write_resolve`); local (non-Raft) path skips this. - if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) - && ctx.state.async_raft_proposer().is_some() - { + // (`control::write_resolve`). + if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) { let authorized = super::sql_gateway::authorize_native_task(ctx, &task)?; let resp = crate::control::write_resolve::run_authorized_write_resolve( ctx.state, authorized, resolver, @@ -104,38 +102,26 @@ pub(super) async fn dispatch_task( return Ok((resp, Vec::new(), Vec::new())); } - // Native DROP uses the same reversible all-core protocol as pgwire. - if matches!( - task.plan, - crate::bridge::envelope::PhysicalPlan::Array( - nodedb_physical::physical_plan::ArrayOp::DropArray { .. } - ) - ) { + // Array DDL proposes a replicated catalog entry, as on pgwire. + if crate::control::array_catalog::ddl::is_array_ddl(&task.plan) { let authorized = super::sql_gateway::authorize_native_task(ctx, &task)?; - let task = authorized.into_physical_task(); - let resp = crate::control::array_catalog::ddl::run_authorized_drop( - ctx.state, - task.tenant_id, - task.database_id, - task.plan, - TraceId::ZERO, - ) - .await?; + let resp = + crate::control::array_catalog::ddl::run_authorized_array_ddl(ctx.state, authorized) + .await?; return Ok((resp, Vec::new(), Vec::new())); } // Materialize catalog providers and resolve Exchange nodes before dispatch. - match resolve_and_materialize( - ctx.state, - ctx.identity, - task.database_id, - task.tenant_id, - task.plan, - TraceId::ZERO, - task.txn_id, - ) - .await? - { + // The native protocol has no weaker read consistency: every read is + // strong, so every leg confirms its group where it is served. + let scope = crate::control::server::exchange::ReadScope { + database_id: task.database_id, + tenant_id: task.tenant_id, + trace_id: TraceId::ZERO, + txn_id: task.txn_id, + linearizable: true, + }; + match resolve_and_materialize(ctx.state, ctx.identity, task.plan, scope).await? { Resolved::Gathered(resp, shard_watermarks, dist_reads) => { return Ok((resp, shard_watermarks, dist_reads)); } @@ -150,7 +136,7 @@ pub(super) async fn dispatch_task( } } - // Everything else routes through the gateway when available, or local SPSC otherwise. + // Everything else routes through the gateway. let resp = dispatch_task_via_gateway(ctx, task).await?; Ok((resp, Vec::new(), Vec::new())) } diff --git a/nodedb/src/control/server/native/dispatch/sql_fold.rs b/nodedb/src/control/server/native/dispatch/sql_fold.rs new file mode 100644 index 000000000..9852f8b61 --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/sql_fold.rs @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Folding each task's answer into one native SQL statement's response: its +//! columns and rows, its warnings, and its one command tag. + +use nodedb_types::protocol::NativeResponse; +use nodedb_types::value::Value; + +use crate::bridge::envelope::PhysicalPlan; +use crate::control::sequence::SessionSequenceAccess; +use crate::control::server::response_shape::compose::{ShapeOutcome, shape_response_materialized}; +use crate::control::server::response_shape::redaction::QueryRedaction; +use crate::control::server::response_shape::request::MaterializedShapeRequest; +use crate::control::server::response_shape::schema::OutputSchema; +use crate::control::server::response_shape::types::{ + DmlOutcome, FoldedTag, PlanKind, StatementTag, describe_plan, staged_dml_outcome, +}; +use crate::control::server::shared::session::staging_gate::{StagedTagKind, StagedWriteOutcome}; +use crate::types::DatabaseId; + +use super::{ + DispatchCtx, apply_dml_outcome, dml_fold_error_to_native, shape_error_to_native, + to_native_columns_rows, +}; + +/// What shaping a task's answer reads. +pub(super) struct FoldShape<'a> { + pub seq: u64, + pub output_schema: Option<&'a OutputSchema>, + pub database_id: DatabaseId, + /// The concrete session access, never `&dyn SequenceAccess`: the trait + /// object is not `Sync`, and the shape lives across the dispatch awaits + /// of a `Send` connection future. + pub sequences: &'a SessionSequenceAccess<'a>, +} + +/// One statement's response, accumulated across its tasks. +#[derive(Default)] +pub(super) struct StatementFold { + pub columns: Option>, + pub rows: Vec>, + pub warnings: Vec, + pub tag: StatementTag, + pub last_lsn: u64, +} + +impl StatementFold { + /// Fold one task's count-bearing outcome into the statement's tag. + pub(super) fn fold_dml( + &mut self, + seq: u64, + outcome: DmlOutcome, + ) -> Result<(), Box> { + self.tag + .fold(outcome) + .map_err(|e| Box::new(dml_fold_error_to_native(seq, &e))) + } + + /// Shape `payload`, the answer `plan` gave as `plan_kind`, and fold its + /// rows. Returns the row count, `None` when the payload carries no rows. + pub(super) fn fold_rows( + &mut self, + ctx: &DispatchCtx<'_>, + shape: &FoldShape<'_>, + plan: &PhysicalPlan, + plan_kind: PlanKind, + payload: &[u8], + ) -> Result, Box> { + let redaction = QueryRedaction::for_plan(ctx.tenant_id(), ctx.auth_context(), plan); + match shape_response_materialized(MaterializedShapeRequest { + payload, + plan, + plan_kind, + projection: shape.output_schema, + state: ctx.state, + database_id: shape.database_id, + tenant_id: ctx.tenant_id(), + redaction: Some(redaction.ctx(&ctx.state.redaction)), + sequences: Some(shape.sequences), + }) { + Ok(ShapeOutcome::Rows(mut shaped)) => { + if let Some(notice) = shaped.notice.take() { + self.warnings.push(notice); + } + let (cols, rows) = to_native_columns_rows(&shaped); + if !cols.is_empty() && self.columns.is_none() { + self.columns = Some(cols); + } + let count = rows.len() as u64; + self.rows.extend(rows); + Ok(Some(count)) + } + Ok(ShapeOutcome::Passthrough) => Ok(None), + Err(e) => Err(Box::new(shape_error_to_native(shape.seq, &e))), + } + } + + /// Fold a write its transaction staged. A write that answers `RETURNING` + /// folds its rows in place of a count, as an autocommit one does. A + /// computed-value payload (`RawPayload`) is shaped instead of counted. + pub(super) fn fold_staged( + &mut self, + ctx: &DispatchCtx<'_>, + shape: &FoldShape<'_>, + plan: &PhysicalPlan, + outcome: StagedWriteOutcome, + ) -> Result<(), Box> { + if !outcome.returning_rows.is_empty() { + for rows in &outcome.returning_rows { + self.fold_rows(ctx, shape, plan, PlanKind::ReturningRows, rows)?; + } + return Ok(()); + } + if !matches!(outcome.kind, StagedTagKind::RawPayload) { + return self.fold_dml( + shape.seq, + staged_dml_outcome(outcome.kind, outcome.affected), + ); + } + if outcome.payload.is_empty() { + self.tag.fold_opaque(); + return Ok(()); + } + // The value rides in the payload, never as a count. + if self + .fold_rows(ctx, shape, plan, describe_plan(plan), &outcome.payload)? + .is_none() + { + self.tag.fold_opaque(); + } + Ok(()) + } + + /// The statement's one response. + pub(super) fn finish(self, seq: u64) -> NativeResponse { + let mut r = NativeResponse::ok(seq); + r.watermark_lsn = self.last_lsn; + r.warnings = self.warnings; + if !self.rows.is_empty() { + r.columns = self.columns; + r.rows = Some(self.rows); + } + match self.tag.finish() { + Some(FoldedTag::Dml(outcome)) => apply_dml_outcome(&mut r, outcome), + Some(FoldedTag::Opaque) | None => {} + } + r + } +} diff --git a/nodedb/src/control/server/native/dispatch/sql_gateway.rs b/nodedb/src/control/server/native/dispatch/sql_gateway.rs index 7731841a8..061a31aeb 100644 --- a/nodedb/src/control/server/native/dispatch/sql_gateway.rs +++ b/nodedb/src/control/server/native/dispatch/sql_gateway.rs @@ -2,12 +2,11 @@ //! Gateway-based SQL task dispatch for the native protocol. //! -//! When `SharedState.gateway` is `Some`, tasks are routed through -//! `Gateway::execute_response` which handles cluster-aware routing, typed `NotLeader` -//! retry, and plan caching. The `None` fallback retains the original -//! `dispatch_to_data_plane` path for single-node boot before the gateway is -//! wired. This is native's SQL-TEXT opcode path — distinct from -//! `raw_dispatch.rs`, which serves only native's direct-op opcodes. +//! Tasks route through `Gateway::execute_response`, which handles +//! cluster-aware routing, typed `NotLeader` retry, and plan caching. A +//! transaction meta-op runs on its own vShard's core instead. This is native's +//! SQL-TEXT opcode path — distinct from `raw_dispatch.rs`, which serves only +//! native's direct-op opcodes. use crate::bridge::envelope::Response; use std::sync::Arc; @@ -23,7 +22,7 @@ use super::DispatchCtx; /// Authorize one task with no clone-write check — used only by the /// Control-Plane orchestrator branches ahead of this file's gateway dispatch, /// whose plan shapes (`InsertSelect`, `Merge`, `UpdateFromJoin`, a governed -/// predicate resolution, `DropArray`) are never clone-write shapes. +/// predicate resolution, array DDL) are never clone-write shapes. pub(super) fn authorize_native_task( ctx: &DispatchCtx<'_>, task: &PhysicalTask, @@ -45,11 +44,10 @@ pub(super) fn authorize_native_task( }) } -/// Dispatch a single `PhysicalTask` through the gateway when available, -/// falling back to the local SPSC path. +/// Dispatch a single `PhysicalTask` through the gateway. /// -/// Both paths return the Data-Plane `Response` shape, with a `NotFound` -/// verdict as an error status. +/// Returns the Data-Plane `Response` shape, with a `NotFound` verdict as an +/// error status. pub(super) async fn dispatch_task_via_gateway( ctx: &DispatchCtx<'_>, task: PhysicalTask, @@ -77,35 +75,28 @@ pub(super) async fn dispatch_task_via_gateway( let database_id = checked.database_id(); let txn_id = checked.txn_id(); + let gateway = ctx.state.installed_gateway()?; // A staged write and the other transaction meta-ops run on the core of - // the task's own vShard. The gateway would route them to vShard 0. - let gateway = ctx - .state - .gateway - .get() - .filter(|_| !is_task_vshard_scoped(checked.plan())); - match gateway { - Some(gw) => { - let gw_ctx = GatewayQueryContext { - tenant_id, - trace_id: TraceId::generate(), - database_id, - // Propagate the in-block transaction id so gateway local - // dispatch resolves the per-txn staging overlay. - txn_id, - }; - // The typed error passes through unchanged. The native frame - // renders its SQLSTATE and numeric code from it. - gw.execute_response(&gw_ctx, checked).await - } - // A write takes the durable route, a read the read route. - None => { - crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( - ctx.state, - checked, - TraceId::generate(), - ) - .await - } + // the task's own vShard. The gateway will route them to vShard 0. A + // write takes the durable route, a read the read route. + if is_task_vshard_scoped(checked.plan()) { + return crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( + ctx.state, + checked, + TraceId::generate(), + ) + .await; } + let gw_ctx = GatewayQueryContext { + tenant_id, + trace_id: TraceId::generate(), + database_id, + // Propagate the in-block transaction id so gateway local dispatch + // resolves the per-txn staging overlay. + txn_id, + linearizable: true, + }; + // The typed error passes through unchanged. The native frame renders its + // SQLSTATE and numeric code from it. + gateway.execute_response(&gw_ctx, checked).await } diff --git a/nodedb/src/control/server/native/dispatch/sql_loop.rs b/nodedb/src/control/server/native/dispatch/sql_loop.rs index f9c8b234e..742697c10 100644 --- a/nodedb/src/control/server/native/dispatch/sql_loop.rs +++ b/nodedb/src/control/server/native/dispatch/sql_loop.rs @@ -1,46 +1,46 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Per-task dispatch loop for the DataFusion-planned SQL path. Split out of -//! `sql.rs` to keep that file under the file-size limit, with no behavior -//! change from running inline in `execute_planned`. +//! Per-task dispatch loop for the DataFusion-planned SQL path. //! -//! The single-task dispatch primitive it calls lives in `sql_dispatch_task.rs`. +//! A statement in a transaction (the client's block, or the implicit +//! transaction a statement whose writes fire a BEFORE, INSTEAD OF or SYNC +//! AFTER body runs in) routes each task through the shared `txn_route`, the +//! route pgwire takes too: the write's trigger bodies join the transaction, +//! a shadowed clone takes its copy-on-write steps, and the write stages. The +//! single-task dispatch primitive lives in `sql_dispatch_task.rs`, and the +//! folding of each task's answer in `sql_fold.rs`. +use std::ops::ControlFlow; use std::sync::Arc; use nodedb_types::protocol::NativeResponse; -use nodedb_types::value::Value; use crate::bridge::envelope::Status; use crate::control::planner::calvin::write_class::{ plan_counts_toward_statement_tag, plans_have_user_write, }; use crate::control::sequence::SessionSequenceAccess; -use crate::control::server::response_shape::compose::{ShapeOutcome, shape_response_materialized}; -use crate::control::server::response_shape::redaction::QueryRedaction; -use crate::control::server::response_shape::request::MaterializedShapeRequest; use crate::control::server::response_shape::schema::OutputSchema; use crate::control::server::response_shape::types::{ - DmlOutcome, FoldedTag, PlanKind, StatementTag, describe_plan, payload_to_dml_outcome, - staged_dml_outcome, + PlanKind, describe_plan, payload_to_dml_outcome, replaced_write_outcome, }; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; -use crate::control::server::shared::session::expander_stage::{ - ExpanderOutcome, route_in_tx_expander, -}; -use crate::control::server::shared::session::staging_gate::{ - InTxnRoute, StagedTagKind, StagingGateError, route_in_tx_write, +use crate::control::server::shared::session::DmlTxnCtx; +use crate::control::server::shared::session::read_set::ReadSetEntry; +use crate::control::server::shared::session::staging_gate::StagingGateError; +use crate::control::server::shared::txn_route::{ + StatementEvents, TxnTaskContext, TxnTaskOutcome, route_txn_task, }; use crate::types::DatabaseId; use nodedb_physical::physical_task::PhysicalTask; use super::sql_dispatch_task::dispatch_task; +use super::sql_fold::{FoldShape, StatementFold}; use super::streaming::SqlOutcome; use super::{ - DispatchCtx, apply_dml_outcome, dml_fold_error_to_native, error_code_to_native, - error_response_to_native, error_to_native, error_to_native_with_sqlstate, - shape_error_to_native, to_native_columns_rows, + DispatchCtx, error_code_to_native, error_response_to_native, error_to_native, + error_to_native_with_sqlstate, }; use crate::control::server::native::sqlstate_code::sqlstate_error; @@ -50,44 +50,49 @@ fn resp(r: NativeResponse) -> SqlOutcome { SqlOutcome::Response(Box::new(r)) } -/// Fold one task's count-bearing outcome into the statement's tag, rendering -/// a verb mismatch as the native error frame. -fn fold_dml( - tag: &mut StatementTag, - seq: u64, - outcome: DmlOutcome, -) -> Result<(), Box> { - tag.fold(outcome) - .map_err(|e| Box::new(dml_fold_error_to_native(seq, &e))) +/// A planned statement the loop dispatches. +pub(super) struct PlannedStatement<'a> { + pub tasks: Vec, + pub output_schema: Option<&'a OutputSchema>, + pub database_id: DatabaseId, + pub plan_lease_scope: Arc, + /// The row images the statement's cross-shard balances were settled + /// from. A statement in a transaction adds them to its read set. + pub sum_target_reads: Vec, } /// Run the per-task dispatch loop for a planned, non-streamed task set, /// materializing all rows/columns/affected-count into a single -/// [`SqlOutcome::Response`]. -/// -/// Called from `execute_planned` after the streaming fast path has been -/// ruled out (or declined). Buffers writes when in an explicit transaction -/// block, exactly like the pgwire dispatch loop. +/// [`SqlOutcome::Response`]. `txn` is the statement's transaction, `None` +/// for an autocommit statement. /// /// The statement's `rows_affected` and `command` come from one -/// [`StatementTag`] folded over every task, the same fold pgwire renders as -/// its command tag: a count-bearing task contributes its reported count and -/// verb, an opaque task (buffered write, index or graph maintenance, -/// computed-value payload) contributes nothing. Neither field is ever -/// synthesised from the task count. +/// [`crate::control::server::response_shape::types::StatementTag`] folded +/// over every task, the same fold pgwire renders as its command tag: a +/// count-bearing task contributes its reported count and verb, an opaque task +/// (buffered write, index or graph maintenance, computed-value payload) +/// contributes nothing. Neither field is ever synthesised from the task +/// count. pub(super) async fn run_dispatch_loop( ctx: &DispatchCtx<'_>, seq: u64, - tasks: Vec, - output_schema: Option<&OutputSchema>, - database_id: DatabaseId, - plan_lease_scope: Arc, + statement: PlannedStatement<'_>, + txn: Option<&DmlTxnCtx<'_>>, ) -> SqlOutcome { - let mut all_columns: Option> = None; - let mut all_rows: Vec> = Vec::new(); - let mut warnings: Vec = Vec::new(); - let mut last_lsn = 0u64; - let mut statement_tag = StatementTag::default(); + let PlannedStatement { + tasks, + output_schema, + database_id, + plan_lease_scope, + sum_target_reads, + } = statement; + if let Some(txn) = txn + && !sum_target_reads.is_empty() + { + txn.sessions + .record_read_entries(txn.session_id, sum_target_reads); + } + let mut fold = StatementFold::default(); // Checked once rather than per task — metering is disabled by default, so // this keeps the per-task extraction below (which clones the collection // name) a true no-op on the hot path for every deployment that hasn't @@ -105,30 +110,35 @@ pub(super) async fn run_dispatch_loop( database_id, ctx.tenant_id(), ); + let shape = FoldShape { + seq, + output_schema, + database_id, + sequences: &sequences, + }; + let route_ctx = txn.map(|txn| TxnTaskContext { + state: ctx.state, + identity: ctx.identity, + auth: ctx.auth_context(), + txn, + lease_scope: &plan_lease_scope, + fire_triggers: true, + }); + let mut statement_events = StatementEvents::default(); for task in tasks { if task.tenant_id != ctx.tenant_id() { return resp(sqlstate_error(seq, "42501", "tenant isolation violation")); } - // Cloned before `route_in_tx_write` consumes `task`, so a staged - // write whose outcome carries a computed payload (KV `Incr` / - // `IncrFloat` / `Cas` / `GetSet` -- see `StagedTagKind::RawPayload`) - // can be shaped into the response exactly like the non-staged branch - // below shapes `task_resp.payload`. - let plan_for_staged_response = task.plan.clone(); - // Extracted from the same clone above, before `task` is moved into - // the routing call below — metering needs the collection/engine - // shape after this task's dispatch succeeds. Only covers the direct - // dispatch below (`InTxnRoute::Read` / `Autocommit`: reads, and - // writes that apply now); `Buffered`/`Staged` tasks `continue` - // before reaching the metering call and are not billed here — a - // `Buffered` task performs no dispatch yet (replayed at COMMIT), and - // a `Staged` task's dispatch happens inside `route_in_tx_write`'s - // closure, whose response is consumed before returning `Staged` and - // is not observable at this loop level. - let plan_metering_info = - metering_enabled.then(|| PlanMeteringInfo::extract(&plan_for_staged_response)); + // Cloned before routing consumes `task`: a staged write's rows and + // computed-value payload shape against it. + let plan = task.plan.clone(); + // Metering needs the collection/engine shape after this task's + // dispatch succeeds. It covers the direct dispatch below (reads, and + // writes that apply now). A staged write meters itself inside the + // staging gate, and a buffered write meters at COMMIT. + let plan_metering_info = metering_enabled.then(|| PlanMeteringInfo::extract(&plan)); // A spent hard quota refuses the task before it runs; the charging // call at the end of this loop is on the success path and so can @@ -139,159 +149,74 @@ pub(super) async fn run_dispatch_loop( return resp(error_to_native_with_sqlstate(seq, "53400", &e)); } - // In transaction: route through the protocol-neutral staging gate. - // Reads (including in-transaction reads) come back as `Read` with - // `txn_id` stamped for read-your-own-writes; non-stageable writes are - // buffered for COMMIT-time replay; stageable writes are applied to - // the per-transaction overlay immediately for a real affected count - // and statement-time constraint errors. Outside a transaction block, - // `route_in_tx_write` returns the task unchanged, as `Read` or as - // `Autocommit` for a write. - // In-transaction `MERGE` and `UPDATE ... FROM` are resolved + staged at - // STATEMENT time by the expander (read-your-own-writes for later - // statements in the same txn); every other task falls through to the - // neutral staging gate. - let buffer_start = ctx.sessions.buffered_task_count(ctx.peer_addr); - let routed = match route_in_tx_expander( - ctx.state, - ctx.sessions, - ctx.peer_addr.into(), - task, - |stage_task| async move { - dispatch_task(ctx, stage_task) - .await - .map(|(resp, _, _)| resp) - }, - ) - .await - { - Ok(ExpanderOutcome::Handled(route)) => Ok(route), - Ok(ExpanderOutcome::Passthrough(task)) => { - route_in_tx_write( - ctx.state, - ctx.sessions, - ctx.peer_addr.into(), - *task, - |stage_task| async move { - dispatch_task(ctx, stage_task) - .await - .map(|(resp, _, _)| resp) - }, - ) - .await - } - Err(e) => Err(e), - }; - if ctx.sessions.buffered_task_count(ctx.peer_addr) > buffer_start - && !ctx.sessions.attach_tx_lease_scope_since( - ctx.peer_addr, - buffer_start, - Arc::clone(&plan_lease_scope), + // The task to dispatch below: the input task, or the one routing + // hands back. Every other routing outcome answers the task here. + let task = if let Some(route) = route_ctx.as_ref() { + let routed = route_txn_task( + route, + task, + &mut statement_events, + |stage_task| async move { + dispatch_task(ctx, stage_task) + .await + .map(|(resp, _, _)| resp) + }, ) - { - return resp(sqlstate_error( - seq, - nodedb_types::error::sqlstate::INTERNAL_ERROR, - "internal error: failed to retain descriptor leases for buffered transaction tasks", - )); - } - // A buffered or staged write reports only a count: it has no stored - // row to project at statement time, and COMMIT answers with one tag for - // the whole transaction, so a statement that asked for rows would - // otherwise succeed with none. Refused here for every verb, matching - // the pgwire loop. - let returns_rows = matches!( - describe_plan(&plan_for_staged_response), - PlanKind::ReturningRows - ); - let task = match routed { - // A write here reaches the gateway, which proposes it through - // Raft or appends its redo record in the funnel. - Ok(InTxnRoute::Read(routed_task) | InTxnRoute::Autocommit(routed_task)) => *routed_task, - Ok(InTxnRoute::Buffered) => { - if returns_rows { - return resp(error_to_native( - seq, - &crate::control::server::shared::returning:: - in_transaction_returning_unsupported(), - )); + .await; + let answered = match routed { + Ok(TxnTaskOutcome::Dispatch(routed_task)) => ControlFlow::Continue(*routed_task), + Ok(TxnTaskOutcome::Buffered) => { + // Applied at COMMIT: no count and no verb yet. + fold.tag.fold_opaque(); + ControlFlow::Break(Ok(())) } - // Applied at COMMIT: no count and no verb yet. - statement_tag.fold_opaque(); - continue; - } - Ok(InTxnRoute::Staged(outcome)) => { - if returns_rows { - return resp(error_to_native( - seq, - &crate::control::server::shared::returning:: - in_transaction_returning_unsupported(), - )); + Ok(TxnTaskOutcome::Staged(outcome)) => { + ControlFlow::Break(fold.fold_staged(ctx, &shape, &plan, outcome)) } - // The staging gate decided the verb and counted the rows it - // applied to the overlay; only a computed-value payload - // (`RawPayload`) is shaped instead of folded. - if !matches!(outcome.kind, StagedTagKind::RawPayload) { - if let Err(e) = fold_dml( - &mut statement_tag, - seq, - staged_dml_outcome(outcome.kind, outcome.affected), - ) { - return SqlOutcome::Response(e); - } - } else if outcome.payload.is_empty() { - statement_tag.fold_opaque(); - } else { - let plan_kind = describe_plan(&plan_for_staged_response); - let redaction = QueryRedaction::for_plan( - ctx.tenant_id(), - ctx.auth_context(), - &plan_for_staged_response, - ); - match shape_response_materialized(MaterializedShapeRequest { - payload: &outcome.payload, - plan: &plan_for_staged_response, - plan_kind, - projection: output_schema, - state: ctx.state, - database_id, - tenant_id: ctx.tenant_id(), - redaction: Some(redaction.ctx(&ctx.state.redaction)), - sequences: Some(&sequences), - }) { - Ok(ShapeOutcome::Rows(mut shaped)) => { - if let Some(notice) = shaped.notice.take() { - warnings.push(notice); - } - let (cols, rows) = to_native_columns_rows(&shaped); - if !cols.is_empty() && all_columns.is_none() { - all_columns = Some(cols); - } - all_rows.extend(rows); + Ok(TxnTaskOutcome::CloneHandled(clone_resp)) => ControlFlow::Break( + fold_answer( + ctx, + &shape, + &mut fold, + &plan, + has_user_write, + clone_resp.payload.as_ref(), + ) + .map(drop), + ), + Ok(TxnTaskOutcome::InsteadOf) => { + ControlFlow::Break(match replaced_write_outcome(describe_plan(&plan)) { + Some(outcome) => fold.fold_dml(seq, outcome), + None => { + fold.tag.fold_opaque(); + Ok(()) } - // The value rides in the payload, never as a count. - Ok(ShapeOutcome::Passthrough) => statement_tag.fold_opaque(), - Err(e) => return resp(shape_error_to_native(seq, &e)), - } + }) } - continue; - } - Err(StagingGateError::Dispatch(e)) => return resp(error_to_native(seq, &e)), - Err(StagingGateError::Rejected { code }) => { - return resp(error_code_to_native(seq, code.as_ref())); + Err(StagingGateError::Dispatch(e)) => return resp(error_to_native(seq, &e)), + Err(StagingGateError::Rejected { code }) => { + return resp(error_code_to_native(seq, code.as_ref())); + } + }; + match answered { + ControlFlow::Continue(routed_task) => routed_task, + ControlFlow::Break(Ok(())) => continue, + ControlFlow::Break(Err(e)) => return SqlOutcome::Response(e), } + } else { + task }; // `ClusterArray` plans are handled entirely on the Control Plane by // the `ArrayCoordinator` — they must never reach the SPSC bridge or - // the trigger/DML machinery. A write reshapes into per-shard - // `ArrayOp` tasks inside the staging gate above and never reaches - // here as `ClusterArray`; only reads (`Slice`/`Agg`) and autocommit - // `Put`/`Delete` arrive as this plan by the time `routed` resolves to - // `Read`. Intercepted here, before `dispatch_task` would otherwise - // route it through the gateway toward the Data Plane. Metering is - // not applied here, matching pgwire's `ClusterArray` short-circuit, - // which also does not meter this path. + // the trigger/DML machinery. A write in a transaction reshapes into + // per-shard `ArrayOp` tasks inside the staging gate above and never + // reaches here as `ClusterArray`; only reads (`Slice`/`Agg`) and + // autocommit `Put`/`Delete` arrive as this plan. Intercepted here, + // before `dispatch_task` will otherwise route it through the gateway + // toward the Data Plane. Metering is not applied here, matching + // pgwire's `ClusterArray` short-circuit, which also does not meter + // this path. if matches!( task.plan, crate::bridge::envelope::PhysicalPlan::ClusterArray(_) @@ -309,15 +234,15 @@ pub(super) async fn run_dispatch_loop( notice, }) => { if let Some(n) = notice { - warnings.push(n); + fold.warnings.push(n); } - if !columns.is_empty() && all_columns.is_none() { - all_columns = Some(columns); + if !columns.is_empty() && fold.columns.is_none() { + fold.columns = Some(columns); } - all_rows.extend(rows); + fold.rows.extend(rows); } Ok(super::cluster_array::ClusterArrayOutcome::Affected(outcome)) => { - if let Err(e) = fold_dml(&mut statement_tag, seq, outcome) { + if let Err(e) = fold.fold_dml(seq, outcome) { return SqlOutcome::Response(e); } } @@ -326,7 +251,6 @@ pub(super) async fn run_dispatch_loop( continue; } - let plan_for_response = task.plan.clone(); let task_vshard = task.vshard_id; let task_database_id = task.database_id; let (task_resp, shard_watermarks, dist_reads) = match dispatch_task(ctx, task).await { @@ -334,20 +258,17 @@ pub(super) async fn run_dispatch_loop( Err(e) => return resp(error_to_native(seq, &e)), }; - // Track reads for snapshot-isolation / cross-shard conflict detection at - // the protocol-neutral layer — the native (canonical) transport records - // identically to pgwire. Recorded BEFORE the error short-circuit so an - // absent-key point read (a `NotFound` from the Data Plane) is captured - // too; a "not found" is a validatable phantom observation. A multi-core - // fan read records one entry per participating shard from the gather's - // per-shard watermarks; a single read falls back to its one watermark. + // Track reads for snapshot-isolation / cross-shard conflict detection + // at the protocol-neutral layer, into the statement's transaction. + // Recorded BEFORE the error short-circuit so an absent-key point read + // (a `NotFound` from the Data Plane) is captured too; a "not found" is + // a validatable phantom observation. A multi-core fan read records one + // entry per participating shard from the gather's per-shard + // watermarks; a single read falls back to its one watermark. let records_read = task_resp.status == Status::Ok || task_resp.error_code.as_deref() == Some(&crate::bridge::envelope::ErrorCode::NotFound); - if records_read - && ctx.sessions.transaction_state(ctx.peer_addr) - == crate::control::server::shared::session::TransactionState::InBlock - { + if records_read && let Some(txn) = txn { let watermarks = if shard_watermarks.is_empty() { vec![(task_vshard, task_resp.watermark_lsn)] } else { @@ -355,11 +276,11 @@ pub(super) async fn run_dispatch_loop( }; crate::control::server::shared::session::record_reads_for_response( ctx.state, - ctx.sessions, - ctx.peer_addr.into(), + txn.sessions, + txn.session_id, ctx.tenant_id(), crate::control::server::shared::session::ResponseReads { - plan: &plan_for_response, + plan: &plan, watermarks: &watermarks, read_version_lsn: task_resp.read_version_lsn, found: task_resp.status == Status::Ok, @@ -377,7 +298,7 @@ pub(super) async fn run_dispatch_loop( // --- TRUNCATE RESTART IDENTITY --- // Autocommit only: a buffered truncate restarts its sequences at // COMMIT. Same rule as the pgwire dispatch loop. - if let Some((collection, true)) = plan_for_response.truncate_target() { + if let Some((collection, true)) = plan.truncate_target() { ctx.state .sequence_registry .restart_sequences_for_collection( @@ -387,69 +308,19 @@ pub(super) async fn run_dispatch_loop( ); } - last_lsn = task_resp.watermark_lsn.as_u64(); + fold.last_lsn = task_resp.watermark_lsn.as_u64(); - // This task's own row count, for metering below — distinct from - // `statement_tag`/`all_rows`, which accumulate across every task in - // the loop. - let mut task_rows: Option = None; - let plan_kind = describe_plan(&plan_for_response); - let counts_toward_tag = - plan_counts_toward_statement_tag(&plan_for_response, has_user_write); - let count_bearing = matches!(plan_kind, PlanKind::DmlResult(_) | PlanKind::DmlResultByOp); - if task_resp.payload.is_empty() && !count_bearing { - // Not a count-bearing plan (graph / vector / index write): no - // count and no verb to report. A count-bearing plan with no - // payload falls through so the count reader refuses it. - statement_tag.fold_opaque(); - } else { - let redaction = - QueryRedaction::for_plan(ctx.tenant_id(), ctx.auth_context(), &plan_for_response); - match shape_response_materialized(MaterializedShapeRequest { - payload: &task_resp.payload, - plan: &plan_for_response, - plan_kind, - projection: output_schema, - state: ctx.state, - database_id, - tenant_id: ctx.tenant_id(), - redaction: Some(redaction.ctx(&ctx.state.redaction)), - sequences: Some(&sequences), - }) { - Ok(ShapeOutcome::Rows(mut shaped)) => { - if let Some(notice) = shaped.notice.take() { - warnings.push(notice); - } - let (cols, rows) = to_native_columns_rows(&shaped); - if !cols.is_empty() && all_columns.is_none() { - all_columns = Some(cols); - } - task_rows = Some(rows.len() as u64); - all_rows.extend(rows); - } - // A count-bearing write reports the rows it touched and its - // verb; an opaque execution reports neither. Counting one per - // dispatched task instead would report a row for a delete - // that removed nothing and for an `ON CONFLICT DO NOTHING` - // insert that skipped. - Ok(ShapeOutcome::Passthrough) if !counts_toward_tag => { - statement_tag.fold_opaque(); - } - Ok(ShapeOutcome::Passthrough) => { - match payload_to_dml_outcome(&task_resp.payload, plan_kind) { - Ok(Some(outcome)) => { - task_rows = Some(outcome.affected); - if let Err(e) = fold_dml(&mut statement_tag, seq, outcome) { - return SqlOutcome::Response(e); - } - } - Ok(None) => statement_tag.fold_opaque(), - Err(e) => return resp(error_to_native(seq, &e)), - } - } - Err(e) => return resp(shape_error_to_native(seq, &e)), - } - } + let task_rows = match fold_answer( + ctx, + &shape, + &mut fold, + &plan, + has_user_write, + task_resp.payload.as_ref(), + ) { + Ok(rows) => rows, + Err(e) => return SqlOutcome::Response(e), + }; // Metered here, once per successfully dispatched task (see the scope // note on `plan_metering_info` above for the tasks this does not @@ -460,16 +331,91 @@ pub(super) async fn run_dispatch_loop( } } - let mut r = NativeResponse::ok(seq); - r.watermark_lsn = last_lsn; - r.warnings = warnings; - if !all_rows.is_empty() { - r.columns = all_columns; - r.rows = Some(all_rows); + if let Some(route) = route_ctx.as_ref() + && let Err(e) = statement_events.fire(route).await + { + return resp(error_to_native(seq, &e)); + } + + resp(fold.finish(seq)) +} + +/// Run a statement in an implicit transaction: it commits when every task +/// succeeds, and rolls back when one fails. +pub(super) async fn run_implicit_statement( + ctx: &DispatchCtx<'_>, + seq: u64, + statement: PlannedStatement<'_>, +) -> SqlOutcome { + let client = DmlTxnCtx { + sessions: ctx.sessions, + session_id: ctx.peer_addr.into(), + }; + let outcome = crate::control::trigger::statement_txn::with_statement_txn_lifted( + ctx.state, + ctx.identity, + &client, + true, + |error| resp(error_to_native(seq, &error)), + async |txn: &DmlTxnCtx<'_>| { + let outcome = run_dispatch_loop(ctx, seq, statement, Some(txn)).await; + match &outcome { + SqlOutcome::Response(r) + if r.status == nodedb_types::protocol::ResponseStatus::Error => + { + Err(outcome) + } + _ => Ok(outcome), + } + }, + ) + .await; + match outcome { + Ok(outcome) | Err(outcome) => outcome, + } +} + +/// Fold a dispatched task's answer `payload`: its rows, or its count. Returns +/// the task's own row count for metering, `None` when it carries none. +fn fold_answer( + ctx: &DispatchCtx<'_>, + shape: &FoldShape<'_>, + fold: &mut StatementFold, + plan: &crate::bridge::envelope::PhysicalPlan, + has_user_write: bool, + payload: &[u8], +) -> Result, Box> { + let plan_kind = describe_plan(plan); + let counts_toward_tag = plan_counts_toward_statement_tag(plan, has_user_write); + let count_bearing = matches!(plan_kind, PlanKind::DmlResult(_) | PlanKind::DmlResultByOp); + if payload.is_empty() && !count_bearing { + // Not a count-bearing plan (graph / vector / index write): no count + // and no verb to report. A count-bearing plan with no payload falls + // through so the count reader refuses it. + fold.tag.fold_opaque(); + return Ok(None); + } + if let Some(rows) = fold.fold_rows(ctx, shape, plan, plan_kind, payload)? { + return Ok(Some(rows)); } - match statement_tag.finish() { - Some(FoldedTag::Dml(outcome)) => apply_dml_outcome(&mut r, outcome), - Some(FoldedTag::Opaque) | None => {} + // A count-bearing write reports the rows it touched and its verb; an + // opaque execution reports neither. Counting one per dispatched task + // instead will report a row for a delete that removed nothing and for an + // `ON CONFLICT DO NOTHING` insert that skipped. + if !counts_toward_tag { + fold.tag.fold_opaque(); + return Ok(None); + } + match payload_to_dml_outcome(payload, plan_kind) { + Ok(Some(outcome)) => { + let affected = outcome.affected; + fold.fold_dml(shape.seq, outcome)?; + Ok(Some(affected)) + } + Ok(None) => { + fold.tag.fold_opaque(); + Ok(None) + } + Err(e) => Err(Box::new(error_to_native(shape.seq, &e))), } - resp(r) } diff --git a/nodedb/src/control/server/native/dispatch/sql_planned.rs b/nodedb/src/control/server/native/dispatch/sql_planned.rs new file mode 100644 index 000000000..fa2d3eadf --- /dev/null +++ b/nodedb/src/control/server/native/dispatch/sql_planned.rs @@ -0,0 +1,321 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Planned SQL execution: DataFusion planning, admission, and dispatch of +//! the planned tasks to the Data Plane, Calvin, or a lazy stream. + +use nodedb_types::TraceId; +use nodedb_types::protocol::NativeResponse; + +use std::sync::Arc; + +use crate::control::planner::calvin::{ + CrossShardTxnMode, DispatchClass, TxnDispatchPosition, classify_dispatch, + dispatch_authorized_tasks_to_calvin, +}; +use crate::control::security::audit::ArcAuditEmitter; +use crate::control::server::native::sqlstate_code::sqlstate_error; +use crate::control::server::shared::check_constraint::{ + enforce_statement_checks, enforce_statement_enum_labels, +}; +use crate::control::server::shared::plan_admission::{ + PlanAdmissionRequest, plan_authorize_and_admit, +}; +use crate::control::server::shared::session::{DmlTxnCtx, TransactionState}; +use crate::control::server::shared::txn_route::{ + statement_needs_implicit_txn, unbufferable_joined_statement, +}; +use crate::control::server::shared::write_admission::{all_writes_bufferable, plan_is_write}; + +use super::sql::resp; +use super::sql_loop::{PlannedStatement, run_dispatch_loop, run_implicit_statement}; +use super::streaming::{SqlOutcome, try_open_sql_stream}; +use super::{DispatchCtx, ddl_result_to_native, error_to_native}; + +/// Plan SQL via DataFusion and dispatch tasks to the Data Plane. +/// +/// When `allow_stream` is set and the planned statement is an eligible +/// autocommit, single-task, unordered multi-row SELECT, returns +/// [`SqlOutcome::Stream`] for lazy frame emission. Every other case — writes, +/// in-block buffering, multi-task, set-ops, errors — collapses to a single +/// [`SqlOutcome::Response`]. +pub(super) async fn execute_planned( + ctx: &DispatchCtx<'_>, + seq: u64, + sql: &str, + database_id: crate::types::DatabaseId, + allow_stream: bool, +) -> SqlOutcome { + // `ctx.scope` is the single request-scoped auth contract, built once per + // request in `session::request::handle_request` — it already carries + // `database_id` (agreeing with the `database_id` passed into this + // function) and a scope-grant-enriched `AuthContext`. A per-query + // `ON DENY` override (e.g. `SELECT ... ON DENY ERROR 'CODE' MESSAGE + // '...'`) rebuilds the scope rather than mutating it in place + // (`RequestAuthScope` has no `&mut` path to `on_deny_override` by + // design), so a clone is taken here: `ctx.scope` stays the canonical, + // unmodified scope for any other consumer of `ctx` during this request, + // while `scope` below is the (possibly overridden) one this statement + // dispatches and admits under. + let (clean_sql, scope) = + crate::control::server::session_auth::apply_per_query_on_deny(sql, ctx.scope.clone()); + + // Forward every per-session planning GUC (vector-dim quota, force-shuffle + // join/agg overrides + partition counts, broadcast / shuffle-aggregate cost + // thresholds) into the shared query context before planning — the same + // protocol-neutral resolution pgwire performs, so the canonical native + // transport honors these overrides identically. Native plans without a plan + // cache, so the returned bypass flags are not needed here. + crate::control::server::shared::planning_overrides::apply_planning_session_overrides( + ctx.query_ctx, + ctx.sessions, + ctx.state, + ctx.peer_addr, + ctx.tenant_id(), + ); + + // General CHECK constraints and enum labels on an INSERT or UPDATE are + // enforced before planning, by the same code pgwire runs. + if let Err(error) = enforce_statement_checks( + ctx.state, + ctx.identity, + ctx.tenant_id(), + database_id, + scope.auth(), + ctx.sessions.tx_id(ctx.peer_addr), + &clean_sql, + ) + .await + .and_then(|()| { + enforce_statement_enum_labels(ctx.state, ctx.tenant_id(), database_id, &clean_sql) + }) { + return resp(ddl_result_to_native(seq, Err(error))); + } + + // Planning, authorization, implicit-edge extraction (pgwire parity: a + // schemaless document carrying `_from`/`_to` is mirrored as a + // `GraphOp::EdgePut` task so the classify/Calvin/single-shard logic below + // routes it like an explicit edge) and lease admission run as ONE retried + // unit, so a descriptor drain starting between the planner's catalog read + // and the lease acquisition is absorbed rather than surfaced. Admission + // still follows authorization inside the unit, so denied requests consume + // no descriptor lease. The scope stays alive through all dispatch and + // response shaping below. + let admission = match plan_authorize_and_admit(PlanAdmissionRequest { + state: ctx.state, + query_ctx: ctx.query_ctx, + scope: &scope, + sql: &clean_sql, + trace_id: TraceId::ZERO, + }) + .await + { + Ok(admission) => admission, + Err(error) => return resp(error_to_native(seq, &error)), + }; + + let mut tasks = admission.tasks; + let output_schema = admission.output_schema; + // Re-derived here rather than carried from `admission`: Calvin dispatch + // and implicit-edge reconciliation are trusted internal mechanisms that + // never reach the clone-checked dispatch boundary (they don't call + // `dispatch_authorized_to_data_plane` / `Gateway::execute`), so a plain + // batch authorize is correct for them. `run_dispatch_loop` below clone-checks + // and authorizes each task itself, immediately before its own dispatch. + let mut authorized_tasks = + match crate::control::server::shared::authorization::authorize_task_set( + ctx.identity, + &tasks, + &ctx.state.permissions, + &ctx.state.roles, + &ArcAuditEmitter(Arc::clone(&ctx.state.audit)), + ) { + Ok(authorized) => authorized, + Err(error) => return resp(error_to_native(seq, &crate::Error::from(error))), + }; + let mut lease_scope = Some(admission.lease_scope); + // Covers the images every cross-shard materialized-sum balance in `tasks` + // was settled from, so Calvin's OCC check aborts rather than committing a + // total folded from an image that has since moved. + let sum_target_reads = admission.sum_target_reads; + + if tasks.is_empty() { + return resp(NativeResponse::status_row(seq, "OK")); + } + + // A statement outside a transaction block whose writes fire a BEFORE, + // INSTEAD OF or SYNC AFTER body, or a MERGE into an edge-bearing + // collection, runs in an implicit transaction ahead of every autocommit + // route: its writes stage on their vShards' leaders and commit together + // at its end. + let in_txn_block = ctx.sessions.transaction_state(ctx.peer_addr) == TransactionState::InBlock; + if !in_txn_block && statement_needs_implicit_txn(ctx.state, &tasks) { + if !all_writes_bufferable(&tasks) { + return resp(error_to_native(seq, &unbufferable_joined_statement())); + } + let Some(lease_scope) = lease_scope.take() else { + return resp(sqlstate_error( + seq, + nodedb_types::error::sqlstate::INTERNAL_ERROR, + "internal error: query lease scope missing before SQL dispatch", + )); + }; + return run_implicit_statement( + ctx, + seq, + PlannedStatement { + tasks, + output_schema: Some(&output_schema), + database_id, + plan_lease_scope: Arc::new(lease_scope), + sum_target_reads, + }, + ) + .await; + } + + // Implicit-edge DELETE/UPDATE routing gate (native-protocol parity with + // pgwire). See `edge_recon_gate` for the full invariant and guard + // documentation. Returns early when the gate fires, consuming `tasks`. + { + use super::edge_recon_gate::{EdgeReconResult, try_edge_recon_dispatch}; + match try_edge_recon_dispatch(ctx, seq, tasks, authorized_tasks).await { + EdgeReconResult::Outcome(outcome) => return outcome, + EdgeReconResult::NotFired(returned_tasks, returned_authorized) => { + tasks = returned_tasks; + authorized_tasks = returned_authorized; + } + } + } + + // Cross-shard write parity with pgwire: classify the planned task set and, + // for a strict multi-shard write, route the whole batch through the Calvin + // sequencer so it commits atomically. Single-shard (and best-effort) keep + // the existing per-task gateway/SPSC dispatch loop below unchanged. + // Autocommit single-statement dispatch: no session read-set to widen with. + let sum_read_vshards = match crate::control::planner::calvin::read_vshards_of(&sum_target_reads) + { + Ok(vshards) => vshards, + Err(error) => return resp(error_to_native(seq, &error)), + }; + match classify_dispatch(&tasks, &sum_read_vshards) { + DispatchClass::SingleShard { .. } => {} + DispatchClass::MultiShard { .. } => { + // Dispatching to Calvin here applies the statement durably at + // statement time, escaping the transaction buffer. Inside a block, + // fall through to the per-task staging gate when the gate can + // buffer every write — COMMIT flushes the whole buffer through + // Calvin. Anything else is refused, matching pgwire. + if in_txn_block && !all_writes_bufferable(&tasks) { + return resp(error_to_native( + seq, + &crate::Error::CrossShardInExplicitTransaction, + )); + } + + // Native has no per-session `cross_shard_txn` parameter wired, so it + // reads the same `SessionStore` accessor pgwire uses; an unset value + // defaults to `CrossShardTxnMode::Strict` (the documented default), + // so native multi-shard writes route through Calvin by default. + let cross_shard_mode = ctx.sessions.cross_shard_txn_mode(ctx.peer_addr); + if !in_txn_block && cross_shard_mode == CrossShardTxnMode::Strict { + return match dispatch_authorized_tasks_to_calvin( + ctx.state, + authorized_tasks, + ctx.tenant_id(), + cross_shard_mode, + TxnDispatchPosition::Autocommit, + &sum_target_reads, + None, + ) + .await + { + // Calvin committed. A RETURNING write surfaces its rows from + // the applied Response; a plain write reports the affected + // count its own mutation returned. + Ok(apply_resp) => { + let plans: Vec<_> = tasks.iter().map(|t| t.plan.clone()).collect(); + resp(super::conversion::calvin_native_response( + seq, + apply_resp, + &plans, + ctx.state, + database_id, + ctx.tenant_id(), + ctx.auth_context(), + )) + } + Err(e) => resp(error_to_native(seq, &e)), + }; + } + // An in-block statement and `BestEffortNonAtomic` both fall + // through to the per-task loop below. + } + } + + // A native lazy stream outlives this handler, so transfer its descriptor + // leases to the session-owned `SqlStream` before returning it. The session + // loop retains that owner through final emission or connection teardown. + if allow_stream { + match try_open_sql_stream(ctx, seq, &tasks, database_id, Some(&output_schema)).await { + Ok(Some(mut stream)) => { + let Some(scope) = lease_scope.take() else { + return resp(sqlstate_error( + seq, + nodedb_types::error::sqlstate::INTERNAL_ERROR, + "internal error: query lease scope missing before SQL stream dispatch", + )); + }; + if let Err(error) = stream.attach_lease_scope(scope) { + return resp(error_to_native(seq, &error)); + } + return SqlOutcome::Stream(Box::new(stream)); + } + Ok(None) => {} + Err(error) => return resp(error_to_native(seq, &error)), + } + } + + // Materialized statements retain their admitted scope in an Arc so any + // writes buffered during this statement keep the same descriptor leases + // after this local owner is dropped. Lazy streams above retain the raw + // scope directly in their stream owner. + let Some(lease_scope) = lease_scope.take() else { + return resp(sqlstate_error( + seq, + nodedb_types::error::sqlstate::INTERNAL_ERROR, + "internal error: query lease scope missing before materialized SQL dispatch", + )); + }; + // A statement admitted under a lease this node then loses ends with a + // retryable error: a read mid-flight, a write only before dispatch. + let lease_scope = Arc::new(lease_scope); + if let Err(revoked) = lease_scope.check_not_revoked() { + return resp(error_to_native(seq, &revoked)); + } + let read_only = tasks.iter().all(|task| !plan_is_write(&task.plan)); + let guard_scope = Arc::clone(&lease_scope); + let client_txn = DmlTxnCtx { + sessions: ctx.sessions, + session_id: ctx.peer_addr.into(), + }; + let run = run_dispatch_loop( + ctx, + seq, + PlannedStatement { + tasks, + output_schema: Some(&output_schema), + database_id, + plan_lease_scope: lease_scope, + sum_target_reads, + }, + in_txn_block.then_some(&client_txn), + ); + if read_only { + match guard_scope.guard(run).await { + Ok(outcome) => outcome, + Err(revoked) => resp(error_to_native(seq, &revoked)), + } + } else { + run.await + } +} diff --git a/nodedb/src/control/server/native/dispatch/streaming.rs b/nodedb/src/control/server/native/dispatch/streaming.rs index d63bd3b1e..fa37d43fe 100644 --- a/nodedb/src/control/server/native/dispatch/streaming.rs +++ b/nodedb/src/control/server/native/dispatch/streaming.rs @@ -18,7 +18,6 @@ use nodedb_physical::physical_task::PostSetOp; use nodedb_types::protocol::NativeResponse; use crate::control::gateway::core::QueryContext; -use crate::control::server::exchange::gather::gather_all_cores_stream_authorized; use crate::control::server::exchange::streamable::streamable_gather_child; use crate::control::server::response_shape::redaction::QueryRedaction; use crate::control::server::response_shape::schema::OutputSchema; @@ -71,8 +70,8 @@ pub(crate) struct SqlStream { /// projection; no `apply_kv_wrap` / `translate_search_response` applies here. pub projection: Option, /// The statement's column-level redaction inputs, resolved ONCE here. - /// Re-resolving them per batch would risk an early batch shipping rows a - /// later one would have redacted. + /// Re-resolving them per batch will risk an early batch shipping rows a + /// later one will redact. pub redaction: Option, /// Descriptor leases acquired after planning. The session loop owns this /// stream, so retaining the scope here holds leases until final emission @@ -109,7 +108,7 @@ impl SqlStream { /// streamable unordered scan (via [`streamable_gather_child`]). /// /// Returns `Ok(Some(stream))` when eligible, `Ok(None)` to fall back to the -/// materialized path, or `Err` if the stream could not be opened. +/// materialized path, or `Err` if the stream cannot be opened. pub(crate) async fn try_open_sql_stream( ctx: &DispatchCtx<'_>, seq: u64, @@ -162,22 +161,15 @@ pub(crate) async fn try_open_sql_stream( } }; - let gateway = ctx.state.gateway.get(); - let stream = if let Some(gw) = gateway { - let gw_ctx = QueryContext { - tenant_id: task.tenant_id, - trace_id: crate::types::TraceId::ZERO, - database_id, - txn_id: task.txn_id, - }; - gw.execute_stream(&gw_ctx, checked_child).await? - } else { - gather_all_cores_stream_authorized( - ctx.state, - checked_child.into_authorized(), - crate::types::TraceId::ZERO, - )? + let gateway = ctx.state.installed_gateway()?; + let gw_ctx = QueryContext { + tenant_id: task.tenant_id, + trace_id: crate::types::TraceId::ZERO, + database_id, + txn_id: task.txn_id, + linearizable: true, }; + let stream = gateway.execute_stream(&gw_ctx, checked_child).await?; Ok(Some(SqlStream { seq, diff --git a/nodedb/src/control/server/native/session/run.rs b/nodedb/src/control/server/native/session/run.rs index 8f6469a45..34edd61b6 100644 --- a/nodedb/src/control/server/native/session/run.rs +++ b/nodedb/src/control/server/native/session/run.rs @@ -8,6 +8,7 @@ use std::sync::atomic::{AtomicBool, Ordering}; use std::sync::{Arc, OnceLock}; use std::time::Duration; +use futures::future::BoxFuture; use tokio::sync::Notify; use tracing::{debug, instrument}; @@ -189,10 +190,16 @@ impl Drop for NativeTxnCleanupGuard { impl NativeSession { /// Run the session. The guard begins detached cleanup on normal return, /// panic unwinding, or task cancellation; normal completion waits for it. - pub async fn run(self) -> crate::Result<()> { + /// + /// The session future is boxed, once per connection. Every request path + /// nests inside it, and unboxed it overflows the compiler's layout depth + /// limit in the listener's connection task. + pub fn run(self) -> BoxFuture<'static, crate::Result<()>> { // The connection-scoped slots wrap the cleanup guard too, so its // synchronous take still reaches this connection's DDL buffer. - crate::control::server::shared::session::conn_scope::scoped(self.run_guarded()).await + Box::pin(crate::control::server::shared::session::conn_scope::scoped( + self.run_guarded(), + )) } async fn run_guarded(mut self) -> crate::Result<()> { @@ -293,9 +300,20 @@ impl NativeSession { // Crash-injection coverage verifies that a panic after a request // mutates transaction state still runs detached connection cleanup. crate::fail_point!("native_session::after_request"); + // Cross-shard graph reads the request made join the transaction's + // read-set. + crate::control::server::shared::session::graph_reads::record_pending( + &self.sessions, + self.peer_addr.into(), + ); match outcome { - dispatch::SqlOutcome::Response(response) => { + dispatch::SqlOutcome::Response(mut response) => { + // Notices raised below the response shaper during this + // request (`session::statement_notice`). + response + .warnings + .extend(crate::control::server::shared::session::statement_notice::take()); // Encode and write response — chunk if it exceeds frame limit. let resp_bytes = codec::encode_response(&response, format)?; if resp_bytes.len() <= MAX_FRAME_SIZE as usize { diff --git a/nodedb/src/control/server/native/session/session_stream.rs b/nodedb/src/control/server/native/session/session_stream.rs index 6f484ec0a..e3157af60 100644 --- a/nodedb/src/control/server/native/session/session_stream.rs +++ b/nodedb/src/control/server/native/session/session_stream.rs @@ -81,7 +81,7 @@ pub(super) async fn emit_sql_stream( stream: mut rows_stream, projection, redaction, - lease_scope: _lease_scope, + lease_scope, } = sql_stream; let mut emitted: usize = 0; @@ -89,7 +89,17 @@ pub(super) async fn emit_sql_stream( let mut last_lsn: u64 = 0; while emitted < limit { - let batch = match rows_stream.next().await { + // A lease this node loses mid-stream ends the stream with a + // retryable error rather than more rows from a stale descriptor. + let next = match &lease_scope { + Some(scope) => scope.guard(rows_stream.next()).await, + None => Ok(rows_stream.next().await), + }; + let next = match next { + Ok(next) => next, + Err(revoked) => Some(Err(revoked)), + }; + let batch = match next { None => break, Some(Ok(b)) => b, Some(Err(e)) => { @@ -157,7 +167,9 @@ pub(super) async fn emit_sql_stream( watermark_lsn: last_lsn, error: None, auth: None, - warnings: Vec::new(), + // Notices raised below the response shaper while the stream ran + // (`session::statement_notice`). + warnings: crate::control::server::shared::session::statement_notice::take(), }; let bytes = codec::encode_response(&terminal, format)?; codec::write_frame(stream, &bytes).await?; diff --git a/nodedb/src/control/server/pgwire/catalog/dropped_collections.rs b/nodedb/src/control/server/pgwire/catalog/dropped_collections.rs index 35d5afb19..2dfaf47fb 100644 --- a/nodedb/src/control/server/pgwire/catalog/dropped_collections.rs +++ b/nodedb/src/control/server/pgwire/catalog/dropped_collections.rs @@ -119,6 +119,7 @@ async fn query_collection_size( txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: crate::bridge::envelope::Admission::Exempt( crate::bridge::envelope::ExemptReason::AlreadyOrdered, ), diff --git a/nodedb/src/control/server/pgwire/connection.rs b/nodedb/src/control/server/pgwire/connection.rs index 6561d41a9..226d5425b 100644 --- a/nodedb/src/control/server/pgwire/connection.rs +++ b/nodedb/src/control/server/pgwire/connection.rs @@ -13,6 +13,7 @@ use std::panic::AssertUnwindSafe; use std::sync::Arc; use std::time::Duration; +use futures::future::BoxFuture; use futures::{FutureExt, SinkExt, StreamExt}; use pgwire::api::{ClientInfo, ErrorHandler, PgWireConnectionState}; use pgwire::error::ErrorInfo; @@ -36,7 +37,7 @@ pub(crate) enum ConnectionOutcome { Panicked, } -/// Read the negotiated transport out of the socket pgwire just produced. +/// Read the negotiated transport out of the socket pgwire produced. /// /// `MaybeTls` is `#[non_exhaustive]`, so the catch-all arm is required; every /// non-TLS arm (plain TCP, Unix socket) is cleartext as far as the policy is @@ -70,7 +71,7 @@ fn fixed_panic_response() -> PgWireBackendMessage { macro_rules! recover_from_panic { ($socket:expr, $response:expr) => {{ - // A queued response may already be partially visible to the peer. Do + // A queued response can already be partially visible to the peer. Do // not append another frame in that case; dropping the connection is // the only conservative recovery. The response was allocated before // panic-prone dispatch and is consumed at most once. @@ -96,22 +97,25 @@ macro_rules! recover_from_panic { /// escape its connection task. /// /// Panics before TLS negotiation yields a framed socket cannot be replied to, -/// so the TCP stream is simply dropped. Once a socket exists, every panic-prone +/// so the TCP stream is dropped. Once a socket exists, every panic-prone /// stage is caught locally and gets one fixed fatal response only if pgwire has -/// no buffered output that could make the wire stream ambiguous. -pub(crate) async fn run( +/// no buffered output that can make the wire stream ambiguous. +/// +/// The connection future is boxed, once per connection. Every statement path +/// nests inside it, and unboxed it can overflow the compiler's layout depth +/// limit in the listener's connection task. +pub(crate) fn run( stream: TcpStream, tls_acceptor: Option, factory: Arc, context: PgConnectionContext, -) -> ConnectionOutcome { +) -> BoxFuture<'static, ConnectionOutcome> { // Session slots are installed for the whole connection, not per statement: // a transaction's statements are polled on whichever worker tokio picks, // so the DDL buffer must follow the task across every await. - crate::control::server::shared::session::conn_scope::scoped(isolate_connection_future( - run_inner(stream, tls_acceptor, factory, context), + Box::pin(crate::control::server::shared::session::conn_scope::scoped( + isolate_connection_future(run_inner(stream, tls_acceptor, factory, context)), )) - .await } async fn isolate_connection_future(future: F) -> ConnectionOutcome @@ -153,7 +157,7 @@ async fn run_inner( // reachable here, between `negotiate_tls` returning and the framed socket // being handed to the message loop. They are stashed in the connection's // typed session-extension store — not `metadata`, which the startup - // handler fills from client-supplied startup parameters and a client could + // handler fills from client-supplied startup parameters and a client can // therefore forge — and read back at identity resolution. socket .session_extensions() diff --git a/nodedb/src/control/server/pgwire/handler/copy_handler.rs b/nodedb/src/control/server/pgwire/handler/copy_handler.rs index 6ef694a62..340c247fe 100644 --- a/nodedb/src/control/server/pgwire/handler/copy_handler.rs +++ b/nodedb/src/control/server/pgwire/handler/copy_handler.rs @@ -32,11 +32,12 @@ use crate::control::backup::state::AppendError; use crate::control::security::audit::ArcAuditEmitter; use crate::control::security::identity::{AuthenticatedIdentity, Permission}; use crate::control::security::request_scope::RequestAuthScope; -use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; +use crate::control::server::shared::backup_metering::{ + admit_backup_restore_quota, meter_backup_restore, +}; use crate::control::server::shared::session::{ConnectionId, SessionId}; use crate::control::state::SharedState; use crate::types::TenantId; -use nodedb_types::calvin::EngineTag; use super::core::NodeDbPgHandler; @@ -56,7 +57,7 @@ impl NodeDbPgHandler { // Blacklist + account status, no rate limit: backup/restore is // admin-scoped bulk data movement, not the per-query traffic the // rate-limiter's cost table models, so charging it against a query - // rate limit would throttle a legitimate restore. A blacklisted or + // rate limit will throttle a legitimate restore. A blacklisted or // suspended/banned account must not be able to run backup or // restore, though — `check_blacklist_and_status` runs that half of // `check_request_admission`'s gate (plus the internal-service @@ -174,57 +175,6 @@ impl NodeDbPgHandler { } } -/// Meter one completed whole-tenant backup or restore. -/// -/// Shared by the backup branch of `intent_to_response` and -/// `on_copy_done`'s restore completion below — both operate on a whole -/// tenant rather than a `PhysicalPlan`-shaped single-collection dispatch, so -/// this builds a [`PlanMeteringInfo`] directly via -/// [`PlanMeteringInfo::for_collection`] instead of extracting one from a -/// plan. -/// Refuse a backup/restore whose covering scope has already spent its cap. -/// -/// The sibling of [`meter_backup_restore`], and it describes the request the -/// same way: a whole-tenant operation has no `PhysicalPlan`, so the collection -/// dimension is the synthetic `tenant:` marker that function bills under. -/// A scope grant therefore only caps this if it was written against that same -/// marker — which is exactly the entitlement an operator would define to cap -/// backups. -fn admit_backup_restore_quota( - state: &SharedState, - scope: &RequestAuthScope<'_>, - tenant_id: u64, -) -> crate::Result<()> { - if !state.metering_config.enabled { - return Ok(()); - } - let info = PlanMeteringInfo::for_collection( - format!("tenant:{tenant_id}"), - EngineTag::Meta, - "sql", - Permission::Backup, - ); - crate::control::server::shared::quota_admission::admit_quota_for_dispatch(state, scope, &info) -} - -fn meter_backup_restore( - state: &SharedState, - scope: &RequestAuthScope<'_>, - tenant_id: u64, - rows: Option, -) { - if !state.metering_config.enabled { - return; - } - let info = PlanMeteringInfo::for_collection( - format!("tenant:{tenant_id}"), - EngineTag::Meta, - "sql", - Permission::Backup, - ); - meter_dispatch(state, scope, &info, rows); -} - /// CopyHandler shared by the factory's per-connection handler. Holds /// only the `Arc` it needs — the rest of the SharedState /// is reachable via the cloned handle for the dispatch on `on_copy_done`. @@ -381,9 +331,8 @@ mod tests { ) } - /// The regression this admission check exists to prevent: before it - /// existed, `intent_to_response` ran no blacklist or account-status - /// check at all, so a blacklisted client could still run backup/restore. + /// `intent_to_response` runs the blacklist and account-status check, so a + /// blacklisted client cannot run backup/restore. #[tokio::test] async fn intent_to_response_rejects_blacklisted_identity() { let (handler, _dir) = test_handler(); diff --git a/nodedb/src/control/server/pgwire/handler/core.rs b/nodedb/src/control/server/pgwire/handler/core.rs index 891c36e98..5c3a3ec05 100644 --- a/nodedb/src/control/server/pgwire/handler/core.rs +++ b/nodedb/src/control/server/pgwire/handler/core.rs @@ -124,7 +124,7 @@ impl NodeDbPgHandler { /// /// The override is installed via `SET TENANT = '' | | DEFAULT` /// or `SET nodedb.tenant_id = `; the SET handler enforces that only - /// superuser sessions may install one and that no active transaction is + /// superuser sessions can install one and that no active transaction is /// in flight. Honoring it here — at the single chokepoint every query /// path passes through immediately after authentication — keeps every /// downstream `identity.tenant_id` read correct without threading the @@ -180,13 +180,25 @@ impl ExtendedQueryHandler for NodeDbPgHandler { let result = self.execute_prepared(client, portal, max_rows).await; // Mirror the simple-query path: surface any queued NOTICE messages - // (e.g. `truncated_before_horizon`) before returning. - for message in self.sessions.drain_notices(session_id) { + // (e.g. `truncated_before_horizon`), shaped or raised below the + // shaper, before returning. + for message in self + .sessions + .drain_notices(session_id) + .into_iter() + .chain(crate::control::server::shared::session::statement_notice::take()) + { let notice = notice_warning(&message); let _ = client .send(PgWireBackendMessage::NoticeResponse(notice)) .await; } + // Cross-shard graph reads the statement made join the transaction's + // read-set. + crate::control::server::shared::session::graph_reads::record_pending( + &self.sessions, + session_id, + ); result } diff --git a/nodedb/src/control/server/pgwire/handler/core/simple_query.rs b/nodedb/src/control/server/pgwire/handler/core/simple_query.rs index f3f08106b..b3b46c31c 100644 --- a/nodedb/src/control/server/pgwire/handler/core/simple_query.rs +++ b/nodedb/src/control/server/pgwire/handler/core/simple_query.rs @@ -105,15 +105,27 @@ impl SimpleQueryHandler for NodeDbPgHandler { let result = self.execute_sql(&identity, session_id, query).await; // Drain queued NOTICE messages emitted by response shapers (e.g. - // `truncated_before_horizon` on array slices) and send them before + // `truncated_before_horizon` on array slices) and raised below them + // (`session::statement_notice`), and send them before // the query result so the client associates the warning with the // current statement. - for message in self.sessions.drain_notices(session_id) { + for message in self + .sessions + .drain_notices(session_id) + .into_iter() + .chain(crate::control::server::shared::session::statement_notice::take()) + { let notice = super::super::super::types::notice_warning(&message); let _ = client .send(PgWireBackendMessage::NoticeResponse(notice)) .await; } + // Cross-shard graph reads the query made join the transaction's + // read-set. + crate::control::server::shared::session::graph_reads::record_pending( + &self.sessions, + session_id, + ); // Drain pending LIVE SELECT notifications and send as pgwire // async NotificationResponse messages. This is the standard diff --git a/nodedb/src/control/server/pgwire/handler/cursor_query.rs b/nodedb/src/control/server/pgwire/handler/cursor_query.rs index b190690e7..dfc1a628c 100644 --- a/nodedb/src/control/server/pgwire/handler/cursor_query.rs +++ b/nodedb/src/control/server/pgwire/handler/cursor_query.rs @@ -56,41 +56,43 @@ impl NodeDbPgHandler { // Admission still follows the explicit authorization boundary, so a // rejected cursor declaration consumes no descriptor lease. The scope // remains live while every cursor-materialization task is dispatched. - let (tasks, _lease_scope) = retry_on_schema_change(move || async move { - let perm_cache = - crate::control::security::auth_fence::permission_view(&self.state, tenant_id) + let (tasks, _lease_scope) = + retry_on_schema_change(&self.state.lease_drain, move || async move { + crate::control::security::auth_fence::admit_permission_view(&self.state, tenant_id) + .await + .map_err(StatementSetupError::from)?; + let sec = crate::control::planner::context::PlanSecurityContext { + identity, + auth: auth_ctx, + rls_store: &self.state.rls, + redaction_store: &self.state.redaction, + permissions: &self.state.permissions, + roles: &self.state.roles, + permission_tree: crate::control::planner::context::PermissionTreeSource::Live( + &self.state.permission_cache, + ), + }; + let (tasks, _output_schema, versions, _) = query_ctx + .plan_sql_with_rls_and_versions(sql, tenant_id, database_id, &sec, None) .await .map_err(StatementSetupError::from)?; - let sec = crate::control::planner::context::PlanSecurityContext { - identity, - auth: auth_ctx, - rls_store: &self.state.rls, - redaction_store: &self.state.redaction, - permissions: &self.state.permissions, - roles: &self.state.roles, - permission_cache: Some(&*perm_cache), - }; - let (tasks, _output_schema, versions, _) = query_ctx - .plan_sql_with_rls_and_versions(sql, tenant_id, database_id, &sec, None) - .await - .map_err(StatementSetupError::from)?; - drop(perm_cache); - // Deliberate gate: proves the planned set is authorizable before the - // descriptor lease is acquired. Dispatch below re-derives the - // capability per task through the clone-check gate. - let _preauthorized_tasks = self - .authorize_tasks(identity, &tasks) - .map_err(StatementSetupError::from)?; + // Deliberate gate: proves the planned set is authorizable before the + // descriptor lease is acquired. Dispatch below re-derives the + // capability per task through the clone-check gate. + let _preauthorized_tasks = self + .authorize_tasks(identity, &tasks) + .map_err(StatementSetupError::from)?; - let lease_scope = self - .state - .acquire_plan_lease_scope(&versions) - .map_err(StatementSetupError::from)?; - Ok::<_, StatementSetupError>((tasks, lease_scope)) - }) - .await - .map_err(PgWireError::from)?; + let lease_scope = self + .state + .acquire_plan_lease_scope(&versions) + .await + .map_err(StatementSetupError::from)?; + Ok::<_, StatementSetupError>((tasks, lease_scope)) + }) + .await + .map_err(PgWireError::from)?; let mut rows = Vec::new(); for task in tasks { diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/authorize.rs b/nodedb/src/control/server/pgwire/handler/dispatch/authorize.rs index 594bc47b6..48599c2db 100644 --- a/nodedb/src/control/server/pgwire/handler/dispatch/authorize.rs +++ b/nodedb/src/control/server/pgwire/handler/dispatch/authorize.rs @@ -68,7 +68,7 @@ mod tests { delta: Vec::new(), peer_id: 1, mutation_id: 1, - surrogate: Surrogate::ZERO, + surrogate: Surrogate::new(1), provenance: None, constraint_version_required: 0, expected_frontier_digest: None, diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs b/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs index b8ab69fa0..cd6f60c8c 100644 --- a/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs +++ b/nodedb/src/control/server/pgwire/handler/dispatch/entry.rs @@ -21,12 +21,14 @@ impl NodeDbPgHandler { /// /// In cluster mode, writes propose to Raft first and execute only after /// quorum commit; reads bypass Raft. `identity` must be passed for every - /// externally derived task. + /// externally derived task. `linearizable` makes a read confirm each group + /// it observes where it is served. pub(in crate::control::server::pgwire::handler) async fn dispatch_authorized_task( &self, task: PhysicalTask, user_id: Option>, identity: &AuthenticatedIdentity, + linearizable: bool, ) -> crate::Result { let mut shard_watermarks = Vec::new(); let mut distributed_reads = Vec::new(); @@ -34,6 +36,7 @@ impl NodeDbPgHandler { task, user_id, identity, + linearizable, &mut shard_watermarks, &mut distributed_reads, ) @@ -48,6 +51,7 @@ impl NodeDbPgHandler { task: PhysicalTask, user_id: Option>, identity: &AuthenticatedIdentity, + linearizable: bool, ) -> crate::Result<(Response, Vec<(VShardId, Lsn)>, Vec)> { let mut shard_watermarks = Vec::new(); let mut distributed_reads = Vec::new(); @@ -56,6 +60,7 @@ impl NodeDbPgHandler { task, user_id, identity, + linearizable, &mut shard_watermarks, &mut distributed_reads, ) diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/local.rs b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs index 553ef4a97..faf34058b 100644 --- a/nodedb/src/control/server/pgwire/handler/dispatch/local.rs +++ b/nodedb/src/control/server/pgwire/handler/dispatch/local.rs @@ -16,7 +16,7 @@ use super::authorize::reject_unadmitted_crdt_apply; impl NodeDbPgHandler { /// Dispatch a task directly to the local Data Plane (single-node or reads). /// - /// WAL append happens inside the write funnel, under the admission guard just + /// WAL append happens inside the write funnel, under the admission guard right /// before enqueue, so LSN order equals apply order. Reads bypass the WAL entirely. pub(super) async fn dispatch_local( &self, @@ -30,6 +30,7 @@ impl NodeDbPgHandler { now_override: None, apply_key: 0, commit_hlc: None, + change_position: None, }, ) .await @@ -38,22 +39,11 @@ impl NodeDbPgHandler { /// Dispatch a task to the Data Plane WITHOUT individual WAL append. /// /// Used by COMMIT after the transaction is written as one `RecordType::Transaction` - /// record — per-task WAL would double-write. + /// record — per-task WAL will double-write. pub(in crate::control::server::pgwire::handler) async fn dispatch_task_no_wal( &self, task: PhysicalTask, ) -> crate::Result { - // Without this, a transaction begun before the freeze could COMMIT mid-scan and - // break the as-of contract. - use crate::control::security::identity::{Permission, required_permission}; - let perm = required_permission(&task.plan); - if matches!(perm, Permission::Write | Permission::Admin) - && self.state.materialize_freeze.is_frozen(task.database_id) - { - return Err(crate::Error::SourceFrozen { - database_id: task.database_id, - }); - } reject_unadmitted_crdt_apply(&task.plan)?; let txn_id = task.txn_id; // The caller owns the transaction's durability, so the task carries no diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/mod.rs b/nodedb/src/control/server/pgwire/handler/dispatch/mod.rs index dd3957b21..b3af6a024 100644 --- a/nodedb/src/control/server/pgwire/handler/dispatch/mod.rs +++ b/nodedb/src/control/server/pgwire/handler/dispatch/mod.rs @@ -5,7 +5,7 @@ //! Split by concern: //! - [`entry`]: the public dispatch entry points and the write-HLC //! bookkeeping wrapper around them. -//! - [`routing`]: the per-task routing decision — freeze/mirror checks, +//! - [`routing`]: the per-task routing decision — mirror checks, //! orchestrated DML, exchange resolution, and the replicated-vs-local //! choice. //! - [`replicated`]: proposing a write to Raft and shaping the response once diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/replicated.rs b/nodedb/src/control/server/pgwire/handler/dispatch/replicated.rs index 8b98e7479..72ebdd4d0 100644 --- a/nodedb/src/control/server/pgwire/handler/dispatch/replicated.rs +++ b/nodedb/src/control/server/pgwire/handler/dispatch/replicated.rs @@ -5,36 +5,27 @@ use std::sync::Arc; use crate::bridge::envelope::Response; -use crate::control::server::dispatch_utils::publish_origin_change_events; use super::super::core::NodeDbPgHandler; /// Inputs for [`NodeDbPgHandler::dispatch_replicated_write`]: the entry to -/// propose, the proposer, and the identity + plan its origin CDC publish needs. +/// propose and the proposer. pub(super) struct ReplicatedWrite<'a> { pub(super) entry: crate::control::wal_replication::ReplicatedEntry, pub(super) proposer: &'a Arc, - pub(super) authorized: crate::control::server::shared::authorization::AuthorizedTask, } impl NodeDbPgHandler { /// Dispatch a write through Raft: propose → register waiter → await apply. /// `ProposeTracker` is race-safe against an entry applying before register. /// - /// Also the origin CDC publish site; replicas publish nothing (`ChangeFeedOwner::Unowned`). + /// Every replica publishes the write's change events as it applies the + /// entry (`ChangeFeedOwner::Replicated`). pub(super) async fn dispatch_replicated_write( &self, args: ReplicatedWrite<'_>, ) -> crate::Result { - let ReplicatedWrite { - entry, - proposer, - authorized, - } = args; - let task = authorized.into_physical_task(); - let tenant_id = task.tenant_id; - let database_id = task.database_id; - let plan = task.plan; + let ReplicatedWrite { entry, proposer } = args; let request_id = self.next_request_id(); // `write_version` is the post-write `coll_write_lsn`, surfaced so the session @@ -57,9 +48,6 @@ impl NodeDbPgHandler { write_set: Vec::new(), }; - // Propose returned: entry is committed and applied. Publish once, from this plan. - publish_origin_change_events(&self.state, tenant_id, database_id, &plan, &response); - Ok(response) } } diff --git a/nodedb/src/control/server/pgwire/handler/dispatch/routing.rs b/nodedb/src/control/server/pgwire/handler/dispatch/routing.rs index 33fbb2be2..0cb929392 100644 --- a/nodedb/src/control/server/pgwire/handler/dispatch/routing.rs +++ b/nodedb/src/control/server/pgwire/handler/dispatch/routing.rs @@ -1,15 +1,20 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Per-task routing: freeze/mirror checks, orchestrated DML, exchange +//! Per-task routing: mirror checks, orchestrated DML, exchange //! resolution, and the replicated-vs-local dispatch choice. use std::sync::Arc; use crate::bridge::envelope::Response; +use crate::control::cluster::linearizable_read::{ + confirm_linearizable_read, statement_read_deadline, +}; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::exchange::ReadScope; use crate::control::server::exchange::resolve::{ DistributedReadCapture, Resolved, resolve_and_materialize, }; +use crate::control::server::shared::write_admission::plan_is_write; use crate::types::{Lsn, ReadConsistency, TraceId, VShardId}; use nodedb_physical::physical_task::PhysicalTask; @@ -23,20 +28,12 @@ impl NodeDbPgHandler { mut task: PhysicalTask, user_id: Option>, identity: &AuthenticatedIdentity, + linearizable: bool, shard_watermarks: &mut Vec<(VShardId, Lsn)>, distributed_reads: &mut Vec, ) -> crate::Result { - // Reject user writes against a database frozen by a clone materializer sweep. - // Reads/DDL pass through. use crate::control::security::identity::{Permission, required_permission}; let perm = required_permission(&task.plan); - if matches!(perm, Permission::Write | Permission::Admin) - && self.state.materialize_freeze.is_frozen(task.database_id) - { - return Err(crate::Error::SourceFrozen { - database_id: task.database_id, - }); - } // Mirror enforcement: writes reject on non-promoted mirrors; reads gate by // ReadConsistency. Catalog lookup skipped for db id=0 to stay allocation-free. @@ -131,9 +128,7 @@ impl NodeDbPgHandler { // Can't replicate bare over Raft — a follower has no writing identity to decide // `$auth.*` against. `write_resolve` resolves it while the identity is live. - if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) - && self.state.async_raft_proposer().is_some() - { + if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&task.plan) { let authorized = self.authorize_for_dispatch(identity, &task)?; return crate::control::write_resolve::run_authorized_write_resolve( &self.state, @@ -143,24 +138,13 @@ impl NodeDbPgHandler { .await; } - // `DROP ARRAY` reaches every core so each releases its store and segment dir — - // otherwise a follow-up `CREATE ARRAY` carries stale state. - if matches!( - task.plan, - crate::bridge::envelope::PhysicalPlan::Array( - nodedb_physical::physical_plan::ArrayOp::DropArray { .. } - ) - ) { - // Broadcast bypasses the write funnel, so a denied DROP must not - // delete catalog rows or surrogate bindings. + // Array DDL proposes a replicated catalog entry; every node's + // post-apply opens or drops the array on its cores. + if crate::control::array_catalog::ddl::is_array_ddl(&task.plan) { let authorized = self.authorize_for_dispatch(identity, &task)?; - let task = authorized.into_physical_task(); - return crate::control::array_catalog::ddl::run_authorized_drop( + return crate::control::array_catalog::ddl::run_authorized_array_ddl( &self.state, - task.tenant_id, - task.database_id, - task.plan, - TraceId::ZERO, + authorized, ) .await; } @@ -175,17 +159,14 @@ impl NodeDbPgHandler { } // Resolve derived Exchange plans before authorizing the dispatched task. - match resolve_and_materialize( - &self.state, - identity, - task.database_id, - task.tenant_id, - task.plan, - TraceId::ZERO, - task.txn_id, - ) - .await? - { + let scope = ReadScope { + database_id: task.database_id, + tenant_id: task.tenant_id, + trace_id: TraceId::ZERO, + txn_id: task.txn_id, + linearizable, + }; + match resolve_and_materialize(&self.state, identity, task.plan, scope).await? { Resolved::Gathered(resp, wms, caps) => { *shard_watermarks = wms; *distributed_reads = caps; @@ -210,24 +191,66 @@ impl NodeDbPgHandler { } crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed(t) => t, }; - if let Some(async_proposer) = self.state.async_raft_proposer() - && let Some(entry) = crate::control::wal_replication::to_replicated_entry( - checked.tenant_id(), - checked.database_id(), - checked.vshard_id(), - &crate::control::wal_replication::ReplicableWrite::decide_for_replication( - checked.plan(), - )?, - )? - { + // The entry carries resolved rows: a timeseries ingest resolves here, + // on the proposer, before the entry exists. + let resolved = crate::control::write_resolve::resolve_for_log( + &self.state, + crate::control::write_resolve::WriteResolveContext { + tenant_id: checked.tenant_id(), + database_id: checked.database_id(), + }, + checked.vshard_id(), + checked.plan(), + ) + .await?; + if let Some(entry) = crate::control::wal_replication::to_replicated_entry( + checked.tenant_id(), + checked.database_id(), + checked.vshard_id(), + &crate::control::wal_replication::ReplicableWrite::decide_for_replication( + resolved.as_ref().unwrap_or(checked.plan()), + )?, + )? { + let async_proposer = self.state.async_raft_proposer()?; + let (_authorized, _lease) = checked.into_parts(); return self .dispatch_replicated_write(ReplicatedWrite { entry, proposer: async_proposer, - authorized: checked.into_authorized(), }) .await; } + // A read runs here when this node replicates its group (and leads it, + // for a transaction's read: the staging overlay lives on the leader), + // confirmed first. Otherwise it runs on the group's leader. + if !plan_is_write(checked.plan()) { + use crate::control::server::dispatch_utils::{ + ReadPlacement, owner_response, read_placement, + }; + match read_placement(&self.state, checked.vshard_id(), checked.txn_id())? { + ReadPlacement::Here(groups) => { + if linearizable { + let deadline = statement_read_deadline(&self.state); + confirm_linearizable_read(&self.state, &groups, deadline).await?; + } + } + ReadPlacement::Owner => { + let gateway = self.state.installed_gateway()?; + let ctx = crate::control::gateway::core::QueryContext { + tenant_id: checked.tenant_id(), + trace_id: TraceId::ZERO, + database_id: checked.database_id(), + txn_id: checked.txn_id(), + linearizable, + }; + let outcome = gateway.execute_with_watermarks(&ctx, checked).await; + if let Ok((_, wms, _)) = &outcome { + *shard_watermarks = wms.clone(); + } + return owner_response(outcome); + } + } + } self.dispatch_local(checked, user_id).await } } diff --git a/nodedb/src/control/server/pgwire/handler/facet.rs b/nodedb/src/control/server/pgwire/handler/facet.rs index d96c806d8..2bca5d7d8 100644 --- a/nodedb/src/control/server/pgwire/handler/facet.rs +++ b/nodedb/src/control/server/pgwire/handler/facet.rs @@ -61,7 +61,7 @@ pub(super) async fn execute_facet_counts_sql( }; let resp = handler - .dispatch_authorized_task(task, None, identity) + .dispatch_authorized_task(task, None, identity, strong_read(handler, session_id)) .await .map_err(|error| { let (severity, code, message) = error_to_sqlstate(&error); @@ -124,7 +124,7 @@ pub(super) async fn execute_search_with_facets_sql( }; let facet_resp = handler - .dispatch_authorized_task(facet_task, None, identity) + .dispatch_authorized_task(facet_task, None, identity, strong_read(handler, session_id)) .await .map_err(|error| { let (severity, code, message) = error_to_sqlstate(&error); @@ -173,6 +173,14 @@ struct SearchWithFacetsArgs { } /// Parse `SELECT FACET_COUNTS(collection => 'name', filter => 'pred', fields => ['a','b'])`. +/// Whether the session's reads must be linearizable. +fn strong_read(handler: &NodeDbPgHandler, session_id: SessionId) -> bool { + handler + .sessions + .read_consistency(session_id) + .requires_leader() +} + fn parse_facet_counts_args(sql: &str) -> PgWireResult { let collection = extract_named_string_arg(sql, "collection") .ok_or_else(|| syntax_error("FACET_COUNTS requires collection => 'name' argument"))?; diff --git a/nodedb/src/control/server/pgwire/handler/plan.rs b/nodedb/src/control/server/pgwire/handler/plan.rs index db3a0e90d..cea2a6a5a 100644 --- a/nodedb/src/control/server/pgwire/handler/plan.rs +++ b/nodedb/src/control/server/pgwire/handler/plan.rs @@ -17,7 +17,7 @@ pub(super) use crate::control::server::response_shape::types::{PlanKind, describ /// Outcome of shaping a Data Plane payload into a pgwire `Response`. /// /// `notice` is set when the response shaper detected a condition the client -/// should know about (e.g. `truncated_before_horizon` on an array slice). +/// must know about (e.g. `truncated_before_horizon` on an array slice). /// Callers forward it to the per-connection notice queue. pub(super) struct ShapedResponse { pub response: Response, @@ -94,7 +94,7 @@ mod tests { key: Vec::new(), value: Vec::new(), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), returning: None, rls_filters: Vec::new(), provenance: None, diff --git a/nodedb/src/control/server/pgwire/handler/prepared/parser.rs b/nodedb/src/control/server/pgwire/handler/prepared/parser.rs index e53bb7318..aad57f217 100644 --- a/nodedb/src/control/server/pgwire/handler/prepared/parser.rs +++ b/nodedb/src/control/server/pgwire/handler/prepared/parser.rs @@ -74,9 +74,9 @@ fn ddl_col_type_to_pg(ty: &DdlColType) -> Type { /// encoding, so a lossy mapping is worse than saying nothing: an unknown /// parameter type is sent as text, which the bind layer already handles. /// `Decimal`, `Uuid`, `Vector` and `Geometry` all currently fold into -/// `DdlColType::Text`, so advertising them would tell a client holding a +/// `DdlColType::Text`, so advertising them will tell a client holding a /// `Decimal`/`Uuid` that the server wants TEXT — a client-side `WrongType` -/// failure where `Unknown` would have worked. They stay unresolved until +/// failure where `Unknown` works. They stay unresolved until /// each has a real wire type. fn inferred_param_type(inferred: &nodedb_sql::InferredParamType) -> Option { use crate::control::server::response_shape::schema::sql_data_type_to_ddl_col_type_with_width; @@ -148,7 +148,7 @@ impl NodeDbQueryParser { /// Used on both Parse paths — the schema-inferring one and the fallback /// for SQL the planner cannot plan — because inference reads only the SQL /// text: whether planning succeeded has no bearing on it, and letting the - /// advertised parameter types depend on that would be arbitrary. + /// advertised parameter types depend on that will be arbitrary. fn param_types_with_inference( sql: &str, client_types: &[Option], @@ -163,7 +163,7 @@ impl NodeDbQueryParser { /// inferred from the SQL itself. /// /// A client-declared type always wins — that is PostgreSQL semantics: the - /// Parse message's type oids are the client's contract, and the server may + /// Parse message's type oids are the client's contract, and the server can /// only resolve the positions the client left as unspecified (oid 0). /// /// Inference runs on the *unsubstituted* SQL. The schema-inference pass @@ -211,10 +211,12 @@ impl NodeDbQueryParser { ); // Parse plans against the same authorization state as every other // planning path. A refusal is an error, not "not plannable". - let permission_cache = - crate::control::security::auth_fence::permission_view(&self.state, identity.tenant_id) - .await - .map_err(|e| crate::control::server::pgwire::types::error_map::error_to_pg(&e))?; + crate::control::security::auth_fence::admit_permission_view( + &self.state, + identity.tenant_id, + ) + .await + .map_err(|e| crate::control::server::pgwire::types::error_map::error_to_pg(&e))?; let security = crate::control::planner::context::PlanSecurityContext { identity, auth: scope.auth(), @@ -222,7 +224,9 @@ impl NodeDbQueryParser { redaction_store: &self.state.redaction, permissions: &self.state.permissions, roles: &self.state.roles, - permission_cache: Some(&*permission_cache), + permission_tree: crate::control::planner::context::PermissionTreeSource::Live( + &self.state.permission_cache, + ), }; let Ok((tasks, _)) = query_ctx .plan_sql_with_rls_metadata(crate::control::planner::context::PlanSqlWithRlsParams { @@ -235,7 +239,6 @@ impl NodeDbQueryParser { else { return Ok(false); }; - drop(permission_cache); // Parse/Describe is metadata-only: authorize the original task set, // but do not materialize implicit graph edges while describing it. @@ -263,6 +266,7 @@ impl NodeDbQueryParser { ) -> crate::control::planner::catalog_adapter::OriginCatalog { crate::control::planner::catalog_adapter::OriginCatalog::new( Arc::clone(&self.state.credentials), + self.state.array_catalog.clone(), tenant_id, database_id, Some(Arc::clone(&self.state.retention_policy_registry)), @@ -298,7 +302,7 @@ impl NodeDbQueryParser { // The planner type-checks WHERE/projection expressions, which // fails on raw `$N` placeholders (no bound value to typecheck). // For schema inference we only need the collection + projection - // structure, so substitute placeholders with NULL literals just + // structure, so substitute placeholders with NULL literals // for this planning pass. Execution re-plans with real bound // values. let sql_for_inference = substitute_placeholders_with_null(&sql_stripped); @@ -402,7 +406,7 @@ impl QueryParser for NodeDbQueryParser { // One catalog per Parse, shared by parameter-type inference and schema // planning. The unplannable branch still needs it: inference resolves // `WHERE col = $1` from the catalog whether or not the statement as a - // whole could be planned. + // whole can be planned. let catalog = self.build_catalog(identity.tenant_id.as_u64(), database_id); let (param_types, result_fields) = if can_infer_schema { self.try_infer_types(sql, types, &catalog, database_id, identity.tenant_id) diff --git a/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs b/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs index 58cd7b32b..5a83c1244 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/calvin_dispatch.rs @@ -11,7 +11,7 @@ use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; use crate::control::planner::calvin::{ TxnDispatchPosition, dispatch_authorized_dependent_edge_recon, - dispatch_authorized_tasks_to_calvin, is_dependent_predicate, + dispatch_authorized_tasks_to_calvin, is_edge_recon_plan, }; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::request_scope::RequestAuthScope; @@ -33,8 +33,8 @@ use super::super::core::NodeDbPgHandler; /// /// `rows: None` — a task yields either a tag contribution or its rows, which /// the fold takes once for the statement's single result set; counting rows -/// here would mean reaching into that fold. `meter_dispatch` charges one unit -/// for `None`, correct for the write that just committed. +/// here will mean reaching into that fold. `meter_dispatch` charges one unit +/// for `None`, correct for the write that committed. fn meter_calvin_task( state: &crate::control::state::SharedState, identity: &AuthenticatedIdentity, @@ -92,37 +92,23 @@ impl NodeDbPgHandler { .get_current_database(session_id) .unwrap_or(crate::types::DatabaseId::DEFAULT); - // Presence guard preserved from the inlined implementation: BOTH the - // static and OLLP paths require the completion registry to be wired, so - // an absent registry rejects either path with `SequencerUnavailable` - // here, before any classification or scan. The OLLP body re-fetches the - // registry itself; this check keeps the static path's rejection - // behaviour byte-identical. - if self.state.calvin_completion_registry.get().is_none() { - let (severity, code, message) = error_to_sqlstate(&crate::Error::SequencerUnavailable); - return Err(PgWireError::UserError(Box::new(ErrorInfo::new( - severity.to_owned(), - code.to_owned(), - message, - )))); - } - - let dependent_task = tasks.iter().find(|t| is_dependent_predicate(&t.plan)); + // A dependent predicate, or a TRUNCATE, takes the edge recon path. + let dependent_task = tasks.iter().find(|t| is_edge_recon_plan(&t.plan)); // Static (non-OLLP) Calvin path: build the TxClass and route the // submit-and-await to the SEQUENCER-GROUP leader via - // `submit_calvin_routed`. Submitting to the LOCAL inbox here is the - // silent-loss bug this fix addresses: only the sequencer leader's service + // `submit_calvin_routed`. Only the sequencer leader's service // assigns and only its registry receives the replicated completion ack, - // so a submit on a non-leader coordinator never completes. Routing fixes - // that for cross-shard document writes from any coordinator. + // so a submit to the LOCAL inbox on a non-leader coordinator never + // completes. Routing serves cross-shard document writes from any + // coordinator. // // The OLLP (dependent-predicate) path below is COORDINATOR-OWNED: this // handler runs `run_dependent_with_retry`, which owns the // submit → await-assignment → await-completion loop and, on a post-exec // predicate-drift mismatch, runs a FRESH pre-execution reconnaissance // before resubmitting (the scheduler releases the aborted attempt's - // locks and only signals the mismatch back — it no longer re-submits a + // locks and only signals the mismatch back — it never re-submits a // stale prediction). The submit step ROUTES to the sequencer-group leader // via `submit_calvin_routed_assign` (returning the leader-assigned // assignment) while the completion is awaited on this coordinator's local @@ -207,10 +193,6 @@ impl NodeDbPgHandler { // write (if any). `tasks` is cloned into the recon call so the original // list survives to shape the per-task responses afterwards: a RETURNING // task emits its rows, every other task its command tag. - // Normal multi-shard OLLP dispatch (NOT the contended single-shard - // route from `route_write_to_calvin`), so it stays on the strict - // multi-vshard dependent `TxClass` builder (`allow_single_vshard: - // false`). let authorized = self.authorize_tasks(identity, &tasks)?; let outcome = dispatch_authorized_dependent_edge_recon( &self.state, @@ -218,7 +200,6 @@ impl NodeDbPgHandler { identity, tenant_id, database_id, - false, ) .await .map_err(|e| { diff --git a/nodedb/src/control/server/pgwire/handler/routing/check_enforcement.rs b/nodedb/src/control/server/pgwire/handler/routing/check_enforcement.rs index 02af69999..d2b2da2cb 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/check_enforcement.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/check_enforcement.rs @@ -1,53 +1,28 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Control Plane CHECK constraint enforcement for the standard SQL path. +//! CHECK constraint and enum-label enforcement for the pgwire SQL path. //! -//! Intercepts INSERT and UPDATE statements before planning to evaluate -//! general CHECK constraints. For UPDATE, fetches the current document -//! and merges SET values for cross-field CHECK evaluation. +//! The enforcement is protocol-neutral and lives in +//! `shared::check_constraint::statement`. These wrappers map its verdict to a +//! pgwire error. -use nodedb_types::{DatabaseId, strip_prefix_ascii_case_insensitive}; -use std::collections::HashMap; - -use nodedb_sql::parser::preprocess::lex::{ - find_ascii_case_insensitive, find_ascii_case_insensitive_from, -}; +use nodedb_types::DatabaseId; use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; -use crate::control::security::audit::ArcAuditEmitter; use crate::control::security::auth_context::AuthContext; -use crate::control::security::identity::{AuthenticatedIdentity, Permission}; -use crate::control::server::shared::authorization::authorize_collection; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::shared::check_constraint::{ + enforce_statement_checks, enforce_statement_enum_labels, +}; +use crate::control::server::shared::ddl::result::DdlError; use crate::types::TenantId; use super::super::core::NodeDbPgHandler; -/// Extract the collection name and operation type from an INSERT or UPDATE SQL -/// statement. Returns `None` for any other statement kind. -fn extract_collection_from_sql(sql: &str) -> Option<(String, bool)> { - if let Some(after) = strip_prefix_ascii_case_insensitive(sql, "INSERT INTO ") { - let after = after.trim_start(); - let end = after - .find(|c: char| c.is_whitespace() || c == '(') - .unwrap_or(after.len()); - Some((after[..end].to_lowercase(), true)) - } else if let Some(after) = strip_prefix_ascii_case_insensitive(sql, "UPDATE ") { - let after = after.trim_start(); - let end = after - .find(|c: char| c.is_whitespace()) - .unwrap_or(after.len()); - Some((after[..end].to_lowercase(), false)) - } else { - None - } -} - impl NodeDbPgHandler { /// Enforce general CHECK constraints before planning INSERT or UPDATE SQL. - /// - /// Extracts the collection name from the SQL text, looks up CHECK constraints - /// from the catalog, and if any exist, parses column/value pairs from the SQL - /// to evaluate each CHECK expression. + /// `txn_id` names the session's open transaction: an UPDATE's current row + /// is read as it left it. pub(super) async fn enforce_check_constraints_if_needed( &self, sql: &str, @@ -55,406 +30,36 @@ impl NodeDbPgHandler { tenant_id: TenantId, database_id: DatabaseId, auth: &AuthContext, + txn_id: Option, ) -> PgWireResult<()> { - let Some((coll_name, is_insert)) = extract_collection_from_sql(sql) else { - return Ok(()); - }; - - // CHECK evaluation is on the write path. Authorize its target before - // catalog lookup or an OLD-row read so unauthorized SQL cannot probe - // collection metadata or row existence. - let audit = ArcAuditEmitter(std::sync::Arc::clone(&self.state.audit)); - authorize_collection( - identity, - database_id, - &coll_name, - Permission::Write, - &self.state.permissions, - &self.state.roles, - &audit, - ) - .map_err(|error| pgwire_err("42501", error.resource()))?; - - // Look up collection and its CHECK constraints. - let catalog = self.state.credentials.catalog(); - let coll = match catalog.get_collection(database_id, tenant_id.as_u64(), &coll_name) { - Ok(Some(c)) => c, - _ => return Ok(()), - }; - if coll.check_constraints.is_empty() { - return Ok(()); - } - - // Extract column/value pairs from the SQL text. - let mut fields = if is_insert { - extract_insert_fields(sql).map_err(|e| pgwire_err("42601", &e))? - } else { - extract_update_fields(sql).map_err(|e| pgwire_err("42601", &e))? - }; - - if fields.is_empty() { - return Ok(()); - } - - // For UPDATE: merge SET values with current document for cross-field CHECK. - if !is_insert && let Some(doc_id) = extract_where_id(sql) { - let old = crate::control::trigger::dml_hook::fetch_old_row( - &self.state, - identity, - database_id, - auth, - &nodedb_types::QualifiedCollection::new(database_id, &coll_name), - &doc_id, - ) - .await - .map_err(|error| { - let (_, sqlstate, message) = - crate::control::server::pgwire::types::error_to_sqlstate(&error); - pgwire_err(sqlstate, &message) - })?; - let mut merged = old; - for (k, v) in &fields { - merged.insert(k.clone(), v.clone()); - } - fields = merged; - } - - crate::control::server::shared::check_constraint::enforce_check_constraints( + enforce_statement_checks( &self.state, identity, + tenant_id, database_id, - &coll.check_constraints, - &fields, + auth, + txn_id, + sql, ) .await - .map_err(|e| pgwire_err(&e.sqlstate, &e.message)) + .map_err(pgwire_err) } /// Validate enum-typed column values against the custom type registry. - /// - /// Intercepts INSERT and UPDATE for any collection whose `fields` list - /// contains a user-defined enum type. No-op if the collection has no - /// enum-typed columns or if the SQL is not an INSERT/UPDATE. - pub(super) async fn enforce_enum_labels_if_needed( + pub(super) fn enforce_enum_labels_if_needed( &self, sql: &str, - tenant_id: crate::types::TenantId, + tenant_id: TenantId, database_id: DatabaseId, ) -> PgWireResult<()> { - let Some((coll_name, is_insert)) = extract_collection_from_sql(sql) else { - return Ok(()); - }; - - let catalog = self.state.credentials.catalog(); - let coll = match catalog.get_collection(database_id, tenant_id.as_u64(), &coll_name) { - Ok(Some(c)) => c, - _ => return Ok(()), - }; - - // Quick path: no user-defined types means nothing to validate. - if coll.fields.is_empty() { - return Ok(()); - } - - let fields = if is_insert { - match extract_insert_fields(sql) { - Ok(f) => f, - Err(_) => return Ok(()), // Unparseable; let the planner handle errors. - } - } else { - match extract_update_fields(sql) { - Ok(f) => f, - Err(_) => return Ok(()), - } - }; - - for (field_name, type_name) in &coll.fields { - let Some(value) = fields.get(field_name.as_str()) else { - continue; - }; - let label = match value { - nodedb_types::Value::String(s) => s.as_str(), - _ => continue, - }; - if let Err(msg) = self.state.custom_type_registry.validate_enum_label( - database_id.as_u64(), - tenant_id.as_u64(), - type_name, - label, - ) { - return Err(pgwire_err("22P02", &msg)); - } - } - - Ok(()) - } -} - -/// Extract column/value pairs from `INSERT INTO x (col1, col2) VALUES (val1, val2)`. -fn extract_insert_fields(sql: &str) -> Result, String> { - let cols_start = sql.find('(').ok_or_else(|| { - let preview: String = sql.chars().take(60).collect(); - format!("missing '(' in INSERT: {preview}") - })?; - let cols_end = sql[cols_start + 1..] - .find(')') - .map(|p| cols_start + 1 + p) - .ok_or_else(|| "missing ')' after column list in INSERT".to_string())?; - let cols: Vec<&str> = sql[cols_start + 1..cols_end] - .split(',') - .map(|s| s.trim()) - .collect(); - - let values_pos = find_ascii_case_insensitive(sql, "VALUES") - .ok_or_else(|| "missing VALUES keyword in INSERT".to_string())? - + 6; - let vals_start = sql[values_pos..] - .find('(') - .map(|p| values_pos + p + 1) - .ok_or_else(|| "missing '(' after VALUES in INSERT".to_string())?; - - let mut depth = 1i32; - let mut vals_end = vals_start; - for (i, ch) in sql[vals_start..].char_indices() { - match ch { - '(' => depth += 1, - ')' => { - depth -= 1; - if depth == 0 { - vals_end = vals_start + i; - break; - } - } - _ => {} - } - } - if depth != 0 { - return Err("unmatched parentheses in VALUES clause".to_string()); + enforce_statement_enum_labels(&self.state, tenant_id, database_id, sql).map_err(pgwire_err) } - - let vals = split_top_level_commas(&sql[vals_start..vals_end]); - let mut fields = HashMap::new(); - for (i, col) in cols.iter().enumerate() { - if let Some(val_str) = vals.get(i) { - let col_name = col.trim_matches('"').trim_matches('`').to_lowercase(); - let val = parse_sql_literal(val_str.trim()); - fields.insert(col_name, val); - } - } - - Ok(fields) } -/// Extract column/value pairs from `UPDATE x SET col1 = val1, col2 = val2 WHERE ...`. -fn extract_update_fields(sql: &str) -> Result, String> { - let set_pos = find_ascii_case_insensitive(sql, " SET ") - .ok_or_else(|| "missing SET keyword in UPDATE".to_string())? - + 5; - - let where_pos = find_ascii_case_insensitive_from(sql, " WHERE ", set_pos).unwrap_or(sql.len()); - let assignments_str = &sql[set_pos..where_pos]; - - let mut fields = HashMap::new(); - for assignment in split_top_level_commas(assignments_str) { - let assignment = assignment.trim(); - if let Some(eq_pos) = assignment.find('=') { - let col = assignment[..eq_pos] - .trim() - .trim_matches('"') - .trim_matches('`') - .to_lowercase(); - let val_str = assignment[eq_pos + 1..].trim(); - let val = parse_sql_literal(val_str); - fields.insert(col, val); - } - } - - Ok(fields) -} - -/// Extract document ID from a `WHERE id = 'value'` clause. -/// -/// Only matches standalone `id` with word boundaries — `userid`, `order_id` etc. won't match. -fn extract_where_id(sql: &str) -> Option { - let where_pos = find_ascii_case_insensitive(sql, " WHERE ")?; - let after = &sql[where_pos + 7..]; - // Find standalone "ID" with word boundary checks. - let mut search_start = 0; - loop { - let abs_pos = find_ascii_case_insensitive_from(after, "ID", search_start)?; - - // Check word boundary before: must be start or non-alphanumeric/underscore. - if abs_pos > 0 { - let prev = after.as_bytes()[abs_pos - 1]; - if prev.is_ascii_alphanumeric() || prev == b'_' { - search_start = abs_pos + 2; - continue; - } - } - // Check word boundary after: must be end or non-alphanumeric/underscore. - let end_pos = abs_pos + 2; - if end_pos < after.len() { - let next = after.as_bytes()[end_pos]; - if next.is_ascii_alphanumeric() || next == b'_' { - search_start = end_pos; - continue; - } - } - - let after_id = after[end_pos..].trim_start(); - let Some(val_str) = after_id.strip_prefix('=') else { - search_start = end_pos; - continue; - }; - let val_str = val_str.trim_start(); - - if let Some(inner) = val_str.strip_prefix('\'') { - let end = inner.find('\'')?; - return Some(inner[..end].to_string()); - } - if let Some(inner) = val_str.strip_prefix('"') { - let end = inner.find('"')?; - return Some(inner[..end].to_string()); - } - let end = val_str - .find(|c: char| c.is_whitespace() || c == ';') - .unwrap_or(val_str.len()); - return Some(val_str[..end].to_string()); - } -} - -/// Split a string on commas, respecting parentheses and string quotes. -fn split_top_level_commas(s: &str) -> Vec<&str> { - let mut parts = Vec::new(); - let mut depth = 0i32; - let mut in_single_quote = false; - let mut in_double_quote = false; - let mut last = 0; - - for (i, ch) in s.char_indices() { - match ch { - '\'' if !in_double_quote => in_single_quote = !in_single_quote, - '"' if !in_single_quote => in_double_quote = !in_double_quote, - '(' if !in_single_quote && !in_double_quote => depth += 1, - ')' if !in_single_quote && !in_double_quote => depth -= 1, - ',' if depth == 0 && !in_single_quote && !in_double_quote => { - parts.push(&s[last..i]); - last = i + 1; - } - _ => {} - } - } - parts.push(&s[last..]); - parts -} - -/// Parse a SQL literal string into a Value (best-effort). -fn parse_sql_literal(s: &str) -> nodedb_types::Value { - let s = s.trim(); - - if s.eq_ignore_ascii_case("NULL") { - return nodedb_types::Value::Null; - } - if s.eq_ignore_ascii_case("TRUE") { - return nodedb_types::Value::Bool(true); - } - if s.eq_ignore_ascii_case("FALSE") { - return nodedb_types::Value::Bool(false); - } - if let Some(inner) = s - .strip_prefix('\'') - .and_then(|value| value.strip_suffix('\'')) - { - return nodedb_types::Value::String(inner.replace("''", "'")); - } - if let Ok(i) = s.parse::() { - return nodedb_types::Value::Integer(i); - } - if let Ok(f) = s.parse::() { - return nodedb_types::Value::Float(f); - } - nodedb_types::Value::String(s.to_string()) -} - -fn pgwire_err(code: &str, msg: &str) -> PgWireError { +fn pgwire_err(error: DdlError) -> PgWireError { PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), - code.to_owned(), - msg.to_owned(), + error.sqlstate, + error.message, ))) } - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn extract_where_id_basic() { - let sql = "UPDATE orders SET amount = 5 WHERE id = 'o1'"; - assert_eq!(extract_where_id(sql), Some("o1".to_string())); - } - - #[test] - fn extract_where_id_no_match_userid() { - // "userid" should NOT match — only standalone "id". - let sql = "UPDATE orders SET amount = 5 WHERE userid = 'u1'"; - assert_eq!(extract_where_id(sql), None); - } - - #[test] - fn extract_where_id_no_match_order_id() { - let sql = "UPDATE orders SET amount = 5 WHERE order_id = 'x'"; - assert_eq!(extract_where_id(sql), None); - } - - #[test] - fn extract_where_id_after_unicode_value_preserves_original_offsets() { - let sql = "UPDATE orders SET note = 'ǰ' WHERE id = 'o1'"; - assert_eq!(extract_where_id(sql), Some("o1".to_string())); - } - - #[test] - fn extract_insert_fields_basic() { - let fields = extract_insert_fields("INSERT INTO t (a, b) VALUES ('hello', 42)").unwrap(); - assert_eq!( - fields.get("a"), - Some(&nodedb_types::Value::String("hello".into())) - ); - assert_eq!(fields.get("b"), Some(&nodedb_types::Value::Integer(42))); - } - - #[test] - fn extract_insert_fields_error_on_bad_sql() { - let result = extract_insert_fields("INSERT INTO t no_parens"); - assert!(result.is_err()); - } - - #[test] - fn extract_insert_fields_with_unicode_before_values_preserves_original_offsets() { - let fields = extract_insert_fields("INSERT INTO tffff (a) VALUES (42)").unwrap(); - assert_eq!(fields.get("a"), Some(&nodedb_types::Value::Integer(42))); - } - - #[test] - fn malformed_insert_preview_respects_utf8_boundaries() { - let sql = format!("INSERT INTO {}é no_parens", "a".repeat(47)); - assert_eq!(sql.find('é'), Some(59)); - assert!(extract_insert_fields(&sql).is_err()); - } - - #[test] - fn extract_update_fields_basic() { - let fields = extract_update_fields("UPDATE t SET x = 10, y = 'hi' WHERE id = '1'").unwrap(); - assert_eq!(fields.get("x"), Some(&nodedb_types::Value::Integer(10))); - assert_eq!( - fields.get("y"), - Some(&nodedb_types::Value::String("hi".into())) - ); - } - - #[test] - fn extract_update_fields_with_unicode_before_set_preserves_original_offsets() { - let fields = extract_update_fields("UPDATE tǰ SET x = 10 WHERE id = '1'").unwrap(); - assert_eq!(fields.get("x"), Some(&nodedb_types::Value::Integer(10))); - } -} diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/mod.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/mod.rs index 109d57c5b..bfd9c0b5a 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/mod.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/mod.rs @@ -5,5 +5,7 @@ mod finish; mod run; mod task; +mod tracking; +mod txn_task; pub(crate) use run::DispatchTaskContext; diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs index c665af5cb..15518879b 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/run.rs @@ -1,13 +1,14 @@ // SPDX-License-Identifier: BUSL-1.1 //! The per-task dispatch loop for non-Calvin pgwire queries: tenant check, -//! in-transaction routing, streaming fast path, pre-dispatch hooks, dispatch, -//! read tracking, AFTER triggers, and metering. Shaping one task's response +//! in-transaction routing (`txn_route`: triggers, clone copy-on-write, +//! staging), streaming fast path, pre-dispatch hooks, dispatch, read tracking +//! (`tracking.rs`), auto-analyze, and metering. Shaping one task's response //! lives in `task.rs`; the statement's tail (folded RETURNING rows, the //! set-op merge, and the one folded command tag) lives in `finish.rs`. //! -//! Split out of `execute.rs`, which keeps the plan/authorize/admit entry -//! points and hands the admitted task list here. +//! `execute.rs` holds the plan/authorize/admit entry points and hands the +//! admitted task list here. use std::sync::Arc; @@ -24,12 +25,14 @@ use crate::control::server::response_shape::types::{ShapedRows, StatementTag}; use crate::control::server::shared::ddl::neutral::maintenance::auto_analyze; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; -use crate::control::server::shared::session::SessionId; +use crate::control::server::shared::session::{DmlTxnCtx, SessionId}; +use crate::control::server::shared::write_admission::plan_is_write; use crate::types::TenantId; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use super::super::super::super::types::{ - dml_fold_error_to_pg, error_to_sqlstate, response_status_to_sqlstate, sqlstate_error, + dml_fold_error_to_pg, error_to_pg, error_to_sqlstate, response_status_to_sqlstate, + sqlstate_error, }; use super::super::super::core::NodeDbPgHandler; use super::super::super::plan::{PlanKind, describe_plan}; @@ -39,6 +42,9 @@ use super::super::result_shaping::ResultShaping; use super::super::streaming::StreamSelectContext; use super::finish::StatementTail; use super::task::ShapeTaskParams; +use super::tracking::DispatchedTask; +use super::txn_task::StatementTxnOutcome; +use crate::control::server::shared::txn_route::{StatementEvents, TxnTaskContext}; pub(crate) struct DispatchTaskContext<'a> { pub(crate) plan_lease_scope: Arc, @@ -47,14 +53,23 @@ pub(crate) struct DispatchTaskContext<'a> { pub(crate) auth_ctx: &'a crate::control::security::auth_context::AuthContext, pub(crate) session_id: SessionId, pub(crate) shaping: ResultShaping<'a>, + /// The row images the statement's cross-shard balances were settled + /// from. A statement in a transaction adds them to its read set, so + /// COMMIT's conflict check covers them. + pub(crate) sum_target_reads: + Vec, } impl NodeDbPgHandler { - /// Execute the per-task dispatch loop for non-Calvin queries. - pub(crate) async fn dispatch_task_loop( + /// Execute the per-task dispatch loop for non-Calvin queries. `txn` is + /// the statement's transaction: the client's block, or the implicit + /// transaction a statement whose write fires a synchronous trigger body + /// runs in (see `txn_task`). `None` for an autocommit statement. + pub(super) async fn dispatch_task_loop_in( &self, tasks: Vec, context: DispatchTaskContext<'_>, + txn: Option<&DmlTxnCtx<'_>>, ) -> PgWireResult> { let DispatchTaskContext { plan_lease_scope, @@ -63,7 +78,14 @@ impl NodeDbPgHandler { auth_ctx, session_id, shaping, + sum_target_reads, } = context; + if let Some(txn) = txn + && !sum_target_reads.is_empty() + { + txn.sessions + .record_read_entries(txn.session_id, sum_target_reads); + } let projection = shaping.projection; let result_formats = shaping.formats; let needs_set_op = tasks.iter().any(|t| t.post_set_op != PostSetOp::None); @@ -98,6 +120,21 @@ impl NodeDbPgHandler { // A derived implicit-edge write beside the user's own never answers // the statement, exactly as Calvin's deposit rule has it. let has_user_write = plans_have_user_write(tasks.iter().map(|t| &t.plan)); + // A strong session reads linearizably, in and out of a transaction. + let strong_reads = self.sessions.read_consistency(session_id).requires_leader(); + + // A statement in a transaction routes each task through the shared + // `txn_route`, and fires its SYNC AFTER STATEMENT bodies once after + // its last task. + let route_ctx = txn.map(|txn| TxnTaskContext { + state: &self.state, + identity, + auth: auth_ctx, + txn, + lease_scope: &plan_lease_scope, + fire_triggers: true, + }); + let mut statement_events = StatementEvents::default(); for mut task in tasks { if task.tenant_id != tenant_id { @@ -113,38 +150,62 @@ impl NodeDbPgHandler { )))); } - // Whether this task would answer with rows, read BEFORE the - // routing gate consumes the task: a buffered or staged write - // reports only a command tag, and a statement that asked for rows - // must be told so rather than handed that tag. - let returns_rows = matches!(describe_plan(&task.plan), PlanKind::ReturningRows); - - // In-transaction write-routing gate: protocol-neutral decision of - // read / buffer-for-COMMIT / stage-now-and-buffer, shared with - // every other dispatch loop (native, DSL/UPSERT). Moved to - // `execute_dml_hooks.rs` to keep this file under the size limit; - // behavior is unchanged. - match self - .route_task_in_txn(session_id, identity, task, Arc::clone(&plan_lease_scope)) - .await? - { - execute_dml_hooks::TxnRouteOutcome::Proceed(routed_task) => { - task = *routed_task; - } - execute_dml_hooks::TxnRouteOutcome::Handled(handled) => { - if returns_rows { - let (severity, code, message) = error_to_sqlstate( - &crate::control::server::shared::returning:: - in_transaction_returning_unsupported(), - ); - return Err(PgWireError::UserError(Box::new(ErrorInfo::new( - severity.to_owned(), - code.to_owned(), - message, - )))); + // In a transaction: read, buffer for COMMIT, or stage now, with + // the write's synchronous trigger bodies joining the same + // transaction. A staged write that answers `RETURNING` folds its + // rows into the statement's one result set. + if let Some(route) = route_ctx.as_ref() { + let plan = task.plan.clone(); + let task_database_id = task.database_id; + let hooks = execute_dml_hooks::PreDispatchContext { + identity, + auth: auth_ctx, + tenant_id, + session_id, + plan_kind: describe_plan(&plan), + projection, + }; + match self + .route_statement_txn_task(route, task, &mut statement_events) + .await? + { + StatementTxnOutcome::Dispatch(routed_task) => task = *routed_task, + StatementTxnOutcome::Write(handled) => { + handled.fold_into(&mut statement_tag)?; + continue; + } + StatementTxnOutcome::Returning(returning) => { + for rows in returning { + self.shape_task_response( + ShapeTaskParams { + response: &staged_rows_response(rows), + plan: &plan, + plan_kind: PlanKind::ReturningRows, + counts_toward_tag: false, + projection, + result_formats, + session_id, + tenant_id, + database_id: task_database_id, + auth_ctx, + session_sequences: session_sequences.clone(), + }, + &mut responses, + &mut returning_rows, + &mut statement_tag, + )?; + } + continue; + } + StatementTxnOutcome::Clone(resp) => { + match self.clone_write_answer(hooks, &plan, &resp)? { + PreDispatchHandled::Rows(response) => responses.push(response), + PreDispatchHandled::Write(handled) => { + handled.fold_into(&mut statement_tag)?; + } + } + continue; } - handled.fold_into(&mut statement_tag)?; - continue; } } @@ -222,9 +283,7 @@ impl NodeDbPgHandler { // the request and the data plane merges the transaction's own staged // writes into the scan (read-your-own-writes); the streaming path // builds per-core requests without the transaction id. - let in_transaction = self.sessions.transaction_state(session_id) - == crate::control::server::shared::session::TransactionState::InBlock; - if !in_transaction + if txn.is_none() && let Some(stream_response) = self .maybe_stream_select( &task, @@ -246,10 +305,9 @@ impl NodeDbPgHandler { continue; } - // --- Pre-dispatch hooks: trigger interception + clone write-path - // interception (moved to execute_dml_hooks.rs to keep this file - // under the size limit; behavior is unchanged). - let (dml_info, old_row, truncate_restart_collection) = match self + // --- Pre-dispatch hooks: truncate restart-identity extraction and + // clone write-path interception (`execute_dml_hooks.rs`). + let (dml_info, truncate_restart_collection) = match self .run_pre_dispatch_hooks( execute_dml_hooks::PreDispatchContext { identity, @@ -279,19 +337,19 @@ impl NodeDbPgHandler { let execute_dml_hooks::PreDispatchProceed { task: proceeding_task, dml_info, - old_row, truncate_restart_collection, } = *proceed; task = proceeding_task; - (dml_info, old_row, truncate_restart_collection) + (dml_info, truncate_restart_collection) } }; // --- Normal dispatch --- let user_id: Option> = Some(std::sync::Arc::from(identity.username.as_str())); + let linearizable = strong_reads && !plan_is_write(&task.plan); let (resp, shard_watermarks, distributed_reads) = self - .dispatch_authorized_task_with_watermarks(task, user_id, identity) + .dispatch_authorized_task_with_watermarks(task, user_id, identity, linearizable) .await .map_err(|e| { let (severity, code, message) = error_to_sqlstate(&e); @@ -302,68 +360,20 @@ impl NodeDbPgHandler { ))) })?; - // Track reads for snapshot-isolation / cross-shard conflict detection - // at the protocol-neutral layer. Recorded BEFORE the error - // short-circuit so an absent-key point read (a `NotFound` from the - // Data Plane) is still captured — a "not found" is a validatable - // phantom observation, not a no-op. Only successful reads and - // not-found reads record; a genuine dispatch failure does not. - let records_read = resp.status == crate::bridge::envelope::Status::Ok - || resp.error_code.as_deref() - == Some(&crate::bridge::envelope::ErrorCode::NotFound); - if records_read - && self.sessions.transaction_state(session_id) - == crate::control::server::shared::session::TransactionState::InBlock - { - let watermarks = if shard_watermarks.is_empty() { - vec![(task_vshard, resp.watermark_lsn)] - } else { - shard_watermarks - }; - crate::control::server::shared::session::record_reads_for_response( - &self.state, - &self.sessions, - session_id, - identity.tenant_id, - crate::control::server::shared::session::ResponseReads { - plan: &plan_for_response, - watermarks: &watermarks, - read_version_lsn: resp.read_version_lsn, - found: resp.status == crate::bridge::envelope::Status::Ok, - distributed_reads: &distributed_reads, - read_lsn_vshard: task_vshard, - }, - ) - .await; - } - - // Record the session's OWN committed write-version so a later - // transaction's read-set capture can be floored at it - // (read-your-writes floor for cross-shard OCC). A prior autocommit - // write must still floor a later transaction's read, so this records - // regardless of transaction state — the version is the write's - // committed per-collection `coll_write_lsn`, carried on - // `read_version_lsn` by the replicated-write dispatch path. Only - // successful writes with a non-zero version are recorded. - if resp.status == crate::bridge::envelope::Status::Ok - && resp.read_version_lsn > crate::types::Lsn::ZERO - && matches!( - crate::control::security::identity::required_permission(&plan_for_response), - crate::control::security::identity::Permission::Write - ) - && let Some(collection) = - crate::control::server::shared::plan_util::extract_collection( - &plan_for_response, - ) - { - self.sessions.note_own_write( - session_id, - task_database_id, - identity.tenant_id, - collection, - resp.read_version_lsn, - ); - } + self.track_dispatched_task( + DispatchedTask { + identity, + client_session: session_id, + txn, + plan: &plan_for_response, + vshard: task_vshard, + database_id: task_database_id, + }, + &resp, + shard_watermarks, + &distributed_reads, + ) + .await; if let Some((severity, code, message)) = response_status_to_sqlstate(resp.status, resp.error_code.as_deref()) @@ -386,29 +396,9 @@ impl NodeDbPgHandler { ); } - // --- AFTER triggers --- + // An autocommit write reaching here fires no synchronous trigger + // body: one that does runs in an implicit transaction (`txn`). if let Some(ref info) = dml_info { - crate::control::trigger::dml_hook_fire::fire_post_dispatch_triggers( - crate::control::trigger::dml_hook_fire::DispatchTriggerParams { - state: &self.state, - identity, - database_id: task_database_id, - tenant_id, - info, - old_row: &old_row, - cascade_depth: 0, - }, - ) - .await - .map_err(|e| { - let (severity, code, message) = error_to_sqlstate(&e); - PgWireError::UserError(Box::new(ErrorInfo::new( - severity.to_owned(), - code.to_owned(), - message, - ))) - })?; - auto_analyze::record_and_maybe_analyze( &self.state, identity, @@ -459,6 +449,13 @@ impl NodeDbPgHandler { } } + if let Some(route) = route_ctx.as_ref() { + statement_events + .fire(route) + .await + .map_err(|error| error_to_pg(&error))?; + } + self.finish_statement( &mut responses, StatementTail { @@ -479,3 +476,19 @@ impl NodeDbPgHandler { Ok(responses) } } + +/// A staged write's `RETURNING` rows as the response the shaper reads. +fn staged_rows_response(rows: Vec) -> crate::bridge::envelope::Response { + crate::bridge::envelope::Response { + request_id: crate::types::RequestId::new(0), + status: crate::bridge::envelope::Status::Ok, + attempt: 0, + partial: false, + payload: crate::bridge::envelope::Payload::from_vec(rows), + watermark_lsn: crate::types::Lsn::ZERO, + error_code: None, + read_set_valid: None, + read_version_lsn: crate::types::Lsn::ZERO, + write_set: Vec::new(), + } +} diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/tracking.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/tracking.rs new file mode 100644 index 000000000..3a63315aa --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/tracking.rs @@ -0,0 +1,98 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! What one dispatched task leaves on its sessions: the reads a transaction +//! validates at COMMIT, and the session's own committed write version. + +use crate::bridge::envelope::{ErrorCode, PhysicalPlan, Response, Status}; +use crate::control::security::identity::{AuthenticatedIdentity, Permission, required_permission}; +use crate::control::server::exchange::resolve::DistributedReadCapture; +use crate::control::server::shared::plan_util::extract_collection; +use crate::control::server::shared::session::{ + DmlTxnCtx, ResponseReads, SessionId, record_reads_for_response, +}; +use crate::types::{DatabaseId, Lsn, VShardId}; + +use super::super::super::core::NodeDbPgHandler; + +/// One task the loop dispatched, and the sessions it ran for. +pub(super) struct DispatchedTask<'a> { + pub identity: &'a AuthenticatedIdentity, + /// The client's session, which notes its own committed writes. + pub client_session: SessionId, + /// The statement's transaction, which validates the task's reads at + /// COMMIT. `None` for an autocommit statement. + pub txn: Option<&'a DmlTxnCtx<'a>>, + pub plan: &'a PhysicalPlan, + pub vshard: VShardId, + pub database_id: DatabaseId, +} + +impl NodeDbPgHandler { + /// Record what one dispatched task observed and wrote. + /// + /// A read in a transaction joins the transaction's read set, for + /// snapshot-isolation and cross-shard conflict checks. An absent-key + /// point read (a `NotFound` from the Data Plane) joins it too: a "not + /// found" is a validatable phantom observation. A genuine dispatch + /// failure records nothing. + /// + /// A successful write records its committed per-collection version on + /// the client's session, so a later transaction's read-set capture is + /// floored at it (the read-your-writes floor for cross-shard OCC). An + /// autocommit write floors a later transaction's read too, so this + /// records outside a transaction as well. + pub(super) async fn track_dispatched_task( + &self, + task: DispatchedTask<'_>, + resp: &Response, + shard_watermarks: Vec<(VShardId, Lsn)>, + distributed_reads: &[DistributedReadCapture], + ) { + let DispatchedTask { + identity, + client_session, + txn, + plan, + vshard, + database_id, + } = task; + let records_read = + resp.status == Status::Ok || resp.error_code.as_deref() == Some(&ErrorCode::NotFound); + if records_read && let Some(txn) = txn { + let watermarks = if shard_watermarks.is_empty() { + vec![(vshard, resp.watermark_lsn)] + } else { + shard_watermarks + }; + record_reads_for_response( + &self.state, + txn.sessions, + txn.session_id, + identity.tenant_id, + ResponseReads { + plan, + watermarks: &watermarks, + read_version_lsn: resp.read_version_lsn, + found: resp.status == Status::Ok, + distributed_reads, + read_lsn_vshard: vshard, + }, + ) + .await; + } + + if resp.status == Status::Ok + && resp.read_version_lsn > Lsn::ZERO + && matches!(required_permission(plan), Permission::Write) + && let Some(collection) = extract_collection(plan) + { + self.sessions.note_own_write( + client_session, + database_id, + identity.tenant_id, + collection, + resp.read_version_lsn, + ); + } + } +} diff --git a/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/txn_task.rs b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/txn_task.rs new file mode 100644 index 000000000..247271078 --- /dev/null +++ b/nodedb/src/control/server/pgwire/handler/routing/dispatch_loop/txn_task.rs @@ -0,0 +1,149 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The transaction a pgwire statement's tasks run in, and one task's route +//! through it. +//! +//! Inside a transaction block every task routes through the client's +//! session. Outside one, a statement whose write fires a BEFORE, INSTEAD OF +//! or SYNC AFTER body runs in an implicit transaction under +//! `EventSource::ImplicitClient` ([`NodeDbPgHandler::dispatch_task_loop_implicit`]): +//! its writes and the bodies' writes stage there and commit together once the +//! statement's tasks all succeed, so its rows fire ASYNC triggers as an +//! autocommit write's do and the bodies' rows fire none. Every other +//! autocommit statement dispatches its tasks directly. +//! +//! Each task in a transaction takes the shared `txn_route`, the route the +//! native protocol takes too. + +use std::sync::Arc; + +use pgwire::api::results::Response; +use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; + +use crate::control::server::response_shape::types::{replaced_write_outcome, staged_dml_outcome}; +use crate::control::server::shared::session::staging_gate::StagingGateError; +use crate::control::server::shared::session::{DmlTxnCtx, TransactionState}; +use crate::control::server::shared::txn_route::{ + StatementEvents, TxnTaskContext, TxnTaskOutcome, route_txn_task, +}; +use crate::control::trigger::statement_txn::with_statement_txn_lifted; +use nodedb_physical::physical_task::PhysicalTask; + +use super::super::super::super::types::{error_to_pg, error_to_sqlstate}; +use super::super::super::core::NodeDbPgHandler; +use super::super::super::plan::describe_plan; +use super::super::execute_dml_hooks::HandledWrite; +use super::run::DispatchTaskContext; + +/// What one task of a statement in a transaction contributes. +pub(super) enum StatementTxnOutcome { + /// Not a write the transaction holds: dispatch it. + Dispatch(Box), + /// A write the transaction holds: its share of the command tag. + Write(HandledWrite), + /// A staged write that answers `RETURNING`: the `RowsPayload` of each + /// write it staged. Its rows answer the statement in place of a count. + Returning(Vec>), + /// A clone copy-on-write answered the write: shape this response. + Clone(crate::bridge::envelope::Response), +} + +impl NodeDbPgHandler { + /// Execute the per-task dispatch loop for non-Calvin queries: in the + /// client's transaction inside a block, directly outside one. + pub(crate) async fn dispatch_task_loop( + &self, + tasks: Vec, + context: DispatchTaskContext<'_>, + ) -> PgWireResult> { + let txn = DmlTxnCtx { + sessions: &self.sessions, + session_id: context.session_id, + }; + let in_block = + self.sessions.transaction_state(context.session_id) == TransactionState::InBlock; + self.dispatch_task_loop_in(tasks, context, in_block.then_some(&txn)) + .await + } + + /// Execute the per-task dispatch loop in an implicit transaction: the + /// statement, outside a transaction block, fires a joined trigger body. + /// The transaction commits when every task succeeds and rolls back when + /// one fails. + pub(in crate::control::server::pgwire::handler::routing) async fn dispatch_task_loop_implicit( + &self, + tasks: Vec, + context: DispatchTaskContext<'_>, + ) -> PgWireResult> { + let identity = context.identity.clone(); + let client = DmlTxnCtx { + sessions: &self.sessions, + session_id: context.session_id, + }; + with_statement_txn_lifted( + &self.state, + &identity, + &client, + true, + |error| error_to_pg(&error), + async |txn: &DmlTxnCtx<'_>| self.dispatch_task_loop_in(tasks, context, Some(txn)).await, + ) + .await + } + + /// Route one task of a statement through its transaction: its triggers, + /// its clone copy-on-write steps and its staging. + pub(super) async fn route_statement_txn_task( + &self, + route: &TxnTaskContext<'_>, + task: PhysicalTask, + events: &mut StatementEvents, + ) -> PgWireResult { + let plan_kind = describe_plan(&task.plan); + let identity = route.identity; + let user_id: Option> = Some(Arc::from(identity.username.as_str())); + let outcome = route_txn_task(route, task, events, |stage_task| { + self.dispatch_authorized_task(stage_task, user_id.clone(), identity, false) + }) + .await + .map_err(|error| staging_error_to_pg(&error))?; + Ok(match outcome { + TxnTaskOutcome::Dispatch(task) => StatementTxnOutcome::Dispatch(task), + TxnTaskOutcome::Buffered => StatementTxnOutcome::Write(HandledWrite::Opaque), + TxnTaskOutcome::Staged(staged) if !staged.returning_rows.is_empty() => { + StatementTxnOutcome::Returning(staged.returning_rows) + } + TxnTaskOutcome::Staged(staged) => StatementTxnOutcome::Write(HandledWrite::Dml( + staged_dml_outcome(staged.kind, staged.affected), + )), + TxnTaskOutcome::CloneHandled(resp) => StatementTxnOutcome::Clone(resp), + TxnTaskOutcome::InsteadOf => { + StatementTxnOutcome::Write(match replaced_write_outcome(plan_kind) { + Some(outcome) => HandledWrite::Dml(outcome), + None => HandledWrite::Opaque, + }) + } + }) + } +} + +/// The pgwire error for a task its transaction refused. +fn staging_error_to_pg(error: &StagingGateError) -> PgWireError { + let (severity, code, message) = match error { + StagingGateError::Dispatch(error) => { + let (severity, code, message) = error_to_sqlstate(error); + (severity, code.to_owned(), message) + } + StagingGateError::Rejected { code: Some(code) } => { + let (severity, sqlstate, message) = + crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(code); + (severity, sqlstate.to_owned(), message) + } + StagingGateError::Rejected { code: None } => ( + "ERROR", + nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), + "unknown data plane error".to_owned(), + ), + }; + PgWireError::UserError(Box::new(ErrorInfo::new(severity.to_owned(), code, message))) +} diff --git a/nodedb/src/control/server/pgwire/handler/routing/execute.rs b/nodedb/src/control/server/pgwire/handler/routing/execute.rs index 0d890968f..b76fe619b 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/execute.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/execute.rs @@ -14,7 +14,7 @@ use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; use crate::control::planner::calvin::{DispatchClass, classify_dispatch}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::session::{SessionId, TransactionState}; -use crate::control::server::shared::write_admission::all_writes_bufferable; +use crate::control::server::shared::write_admission::{all_writes_bufferable, plan_is_write}; use crate::types::TenantId; use super::super::super::types::error_to_sqlstate; @@ -47,7 +47,7 @@ impl NodeDbPgHandler { .get_current_database(session_id) .unwrap_or(crate::types::DatabaseId::DEFAULT); let (tasks, output_schema, auth_ctx, plan_lease_scope, sum_target_reads) = - retry_on_schema_change(move || async move { + retry_on_schema_change(&self.state.lease_drain, move || async move { let (mut tasks, output_schema, versions, auth_ctx) = self .plan_statement_to_tasks(identity, sql, tenant_id, session_id, params) .await?; @@ -104,7 +104,7 @@ impl NodeDbPgHandler { .map_err(StatementSetupError::from)?; // Period-lock reference rows are resolved here for the same - // reason as the materialized-sum targets just above, into the + // reason as the materialized-sum targets right above, into the // same plan slot. crate::control::planner::period_lock::resolve_period_lock_targets( &self.state, @@ -126,6 +126,7 @@ impl NodeDbPgHandler { let plan_lease_scope = self .state .acquire_plan_lease_scope(&versions) + .await .map_err(StatementSetupError::from)?; Ok::<_, StatementSetupError>(( @@ -140,54 +141,80 @@ impl NodeDbPgHandler { .map_err(PgWireError::from)?; let plan_lease_scope = Arc::new(plan_lease_scope); - if tasks.is_empty() { - return Ok(vec![Response::Execution(Tag::new("OK"))]); + // A statement admitted under a lease this node then loses ends with a + // retryable error. A read is cancelled mid-flight. A write is checked + // only before dispatch: cancelling a proposed write will leave its + // outcome unknown to the client. + if let Err(revoked) = plan_lease_scope.check_not_revoked() { + return Err(sql_error(&revoked)); } + let read_only = tasks.iter().all(|task| !plan_is_write(&task.plan)); + let guard_scope = Arc::clone(&plan_lease_scope); + let body = async { + if tasks.is_empty() { + return Ok(vec![Response::Execution(Tag::new("OK"))]); + } - // An externally-supplied prepared-statement schema (from the Describe - // phase) names the columns; otherwise the planner's fresh output - // schema for this statement does. The Control-Plane computed list is - // known only to this statement's plan, so it rides along under the - // Describe-phase columns: the shaper evaluates it before those - // columns project, and a computed alias never renders as NULL. - let effective_schema_owned = match shaping.projection { - Some(described) => crate::control::server::response_shape::schema::OutputSchema { - columns: described.columns.clone(), - is_star: described.is_star, - cp_computed: output_schema.cp_computed, - }, - None => output_schema, - }; - let effective_schema = Some(&effective_schema_owned); - - // Implicit-edge dependent predicates must be preempted onto the - // OLLP/Calvin path before gateway forwarding or ordinary dispatch. - if let Some(responses) = self - .maybe_dispatch_implicit_edge_recon( - &tasks, - tenant_id, - identity, - session_id, - ResultShaping { - projection: effective_schema, - formats: shaping.formats, + // An externally-supplied prepared-statement schema (from the Describe + // phase) names the columns; otherwise the planner's fresh output + // schema for this statement does. The Control-Plane computed list is + // known only to this statement's plan, so it rides along under the + // Describe-phase columns: the shaper evaluates it before those + // columns project, and a computed alias never renders as NULL. + let effective_schema_owned = match shaping.projection { + Some(described) => crate::control::server::response_shape::schema::OutputSchema { + columns: described.columns.clone(), + is_star: described.is_star, + cp_computed: output_schema.cp_computed, }, - &auth_ctx, - ) - .await? - { - return Ok(responses); - } + None => output_schema, + }; + let effective_schema = Some(&effective_schema_owned); - // Read once, ahead of the gateway gate: an in-block write must never - // forward here, or it applies durably outside the transaction. - let tx_state = self.sessions.transaction_state(session_id); - if tx_state != TransactionState::InBlock - && let Some(responses) = self - .maybe_dispatch_tasks_via_gateway( + // A statement outside a transaction block whose writes fire a + // BEFORE, INSTEAD OF or SYNC AFTER body, or a MERGE into an + // edge-bearing collection, runs in an implicit transaction, + // ahead of every autocommit route: + // its writes stage on their vShards' leaders and commit together + // at its end. + let tx_state = self.sessions.transaction_state(session_id); + if tx_state != TransactionState::InBlock + && crate::control::server::shared::txn_route::statement_needs_implicit_txn( + &self.state, + &tasks, + ) + { + if !all_writes_bufferable(&tasks) { + return Err(sql_error( + &crate::control::server::shared::txn_route::unbufferable_joined_statement(), + )); + } + return self + .dispatch_task_loop_implicit( + tasks, + DispatchTaskContext { + plan_lease_scope: Arc::clone(&plan_lease_scope), + tenant_id, + identity, + auth_ctx: &auth_ctx, + session_id, + shaping: ResultShaping { + projection: effective_schema, + formats: shaping.formats, + }, + sum_target_reads, + }, + ) + .await; + } + + // Implicit-edge dependent predicates must be preempted onto the + // OLLP/Calvin path before gateway forwarding or ordinary dispatch. + if let Some(responses) = self + .maybe_dispatch_implicit_edge_recon( &tasks, - identity, tenant_id, + identity, session_id, ResultShaping { projection: effective_schema, @@ -196,89 +223,129 @@ impl NodeDbPgHandler { &auth_ctx, ) .await? - { - return Ok(responses); - } + { + return Ok(responses); + } - // Autocommit statement routing: the only reads to widen with are the - // ones the materialized-sum settlement stamped on the source rows its - // shipped balances were folded from. - let sum_read_vshards = - match crate::control::planner::calvin::read_vshards_of(&sum_target_reads) { - Ok(vshards) => vshards, - Err(error) => { - let (severity, code, message) = error_to_sqlstate(&error); - return Err(PgWireError::UserError(Box::new(ErrorInfo::new( - severity.to_owned(), - code.to_owned(), - message, - )))); - } - }; - match classify_dispatch(&tasks, &sum_read_vshards) { - DispatchClass::SingleShard { .. } => { - // A single-shard dependent-predicate write (e.g. `DELETE ... - // WHERE `) doesn't need OLLP/Calvin: one shard is one - // Raft group, so the normal replicated-write dispatch path - // applies it deterministically. Edge-bearing dependent - // predicates are already preempted onto Calvin above; only - // genuine multi-shard bulk writes need OLLP. Fall through. + // Read once, ahead of the gateway gate: an in-block write must never + // forward here, or it applies durably outside the transaction. + if tx_state != TransactionState::InBlock + && let Some(responses) = self + .maybe_dispatch_tasks_via_gateway( + &tasks, + identity, + tenant_id, + session_id, + ResultShaping { + projection: effective_schema, + formats: shaping.formats, + }, + &auth_ctx, + ) + .await? + { + return Ok(responses); } - DispatchClass::MultiShard { .. } => { - // Dispatching to Calvin here would apply the statement durably - // at statement time, escaping the transaction buffer. Inside a - // block, fall through to the per-task staging gate when the gate - // can buffer every write — COMMIT then flushes the whole buffer - // through Calvin. Anything else is refused, never applied. - if tx_state == TransactionState::InBlock { - if !all_writes_bufferable(&tasks) { - let (severity, code, message) = - error_to_sqlstate(&crate::Error::CrossShardInExplicitTransaction); + + // Autocommit statement routing: the only reads to widen with are the + // ones the materialized-sum settlement stamped on the source rows its + // shipped balances were folded from. + let sum_read_vshards = + match crate::control::planner::calvin::read_vshards_of(&sum_target_reads) { + Ok(vshards) => vshards, + Err(error) => { + let (severity, code, message) = error_to_sqlstate(&error); return Err(PgWireError::UserError(Box::new(ErrorInfo::new( severity.to_owned(), code.to_owned(), message, )))); } - // Bufferable: fall through to the staging gate below. - } else { - let cross_shard_mode = self.sessions.cross_shard_txn_mode(session_id); - if cross_shard_mode - == crate::control::server::shared::session::cross_shard_mode::CrossShardTxnMode::Strict - { - return self - .dispatch_calvin_multishard( - tasks, - tenant_id, - super::calvin_dispatch::CalvinDispatchSession { - identity, - session_id, - result_formats: shaping.formats, - auth: &auth_ctx, - projection: effective_schema, - }, - &sum_target_reads, - ) - .await; + }; + match classify_dispatch(&tasks, &sum_read_vshards) { + DispatchClass::SingleShard { .. } => { + // A single-shard dependent-predicate write (e.g. `DELETE ... + // WHERE `) doesn't need OLLP/Calvin: one shard is one + // Raft group, so the normal replicated-write dispatch path + // applies it deterministically. Edge-bearing dependent + // predicates are already preempted onto Calvin above; only + // genuine multi-shard bulk writes need OLLP. Fall through. + } + DispatchClass::MultiShard { .. } => { + // Dispatching to Calvin here will apply the statement durably + // at statement time, escaping the transaction buffer. Inside a + // block, fall through to the per-task staging gate when the gate + // can buffer every write — COMMIT then flushes the whole buffer + // through Calvin. Anything else is refused, never applied. + if tx_state == TransactionState::InBlock { + if !all_writes_bufferable(&tasks) { + let (severity, code, message) = + error_to_sqlstate(&crate::Error::CrossShardInExplicitTransaction); + return Err(PgWireError::UserError(Box::new(ErrorInfo::new( + severity.to_owned(), + code.to_owned(), + message, + )))); + } + // Bufferable: fall through to the staging gate below. + } else { + let cross_shard_mode = self.sessions.cross_shard_txn_mode(session_id); + if cross_shard_mode + == crate::control::server::shared::session::cross_shard_mode::CrossShardTxnMode::Strict + { + return self + .dispatch_calvin_multishard( + tasks, + tenant_id, + super::calvin_dispatch::CalvinDispatchSession { + identity, + session_id, + result_formats: shaping.formats, + auth: &auth_ctx, + projection: effective_schema, + }, + &sum_target_reads, + ) + .await; + } } } } - } - self.dispatch_task_loop( - tasks, - DispatchTaskContext { - plan_lease_scope: Arc::clone(&plan_lease_scope), - tenant_id, - identity, - auth_ctx: &auth_ctx, - session_id, - shaping: ResultShaping { - projection: effective_schema, - formats: shaping.formats, + self.dispatch_task_loop( + tasks, + DispatchTaskContext { + plan_lease_scope: Arc::clone(&plan_lease_scope), + tenant_id, + identity, + auth_ctx: &auth_ctx, + session_id, + shaping: ResultShaping { + projection: effective_schema, + formats: shaping.formats, + }, + sum_target_reads, }, - }, - ) - .await + ) + .await + }; + if read_only { + match guard_scope.guard(body).await { + Ok(result) => result, + Err(revoked) => Err(sql_error(&revoked)), + } + } else { + body.await + } } } + +/// The client-facing form of an engine error. +fn sql_error(error: &crate::Error) -> PgWireError { + let (severity, code, message) = error_to_sqlstate(error); + PgWireError::UserError(Box::new(ErrorInfo::new( + severity.to_owned(), + code.to_owned(), + message, + ))) +} diff --git a/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs b/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs index a4fd1dfd8..60b5cf9df 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/execute_dml_hooks.rs @@ -1,14 +1,10 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Pre-dispatch hook interception for the `dispatch_task_loop` write path: -//! BEFORE/INSTEAD OF trigger firing (with OLD-row fetch and probe-driven -//! event reclassification), truncate `restart_identity` extraction, and -//! clone CoW write-path interception. Split out of `execute.rs` to keep -//! that file under the file-size limit, with no behavior change from -//! running inline in the per-task dispatch loop. - -use std::collections::HashMap; -use std::sync::Arc; +//! Pre-dispatch hooks for the `dispatch_task_loop` autocommit write path: +//! truncate `restart_identity` extraction and clone CoW write-path +//! interception. A statement in a transaction routes each task through the +//! shared `txn_route` instead, which fires its triggers and takes its clone +//! copy-on-write steps inside the transaction. use pgwire::api::results::Response; use pgwire::error::{ErrorInfo, PgWireError, PgWireResult}; @@ -17,7 +13,7 @@ use crate::control::security::auth_context::AuthContext; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::response_shape::schema::OutputSchema; use crate::control::server::response_shape::types::{ - DmlOutcome, StatementTag, payload_to_dml_outcome, staged_dml_outcome, + DmlOutcome, StatementTag, payload_to_dml_outcome, }; use crate::control::server::shared::session::SessionId; use crate::control::trigger::dml_hook::DmlWriteInfo; @@ -35,9 +31,9 @@ use super::super::plan::PlanKind; pub(super) enum HandledWrite { /// A count-bearing outcome: folds into the statement tag. Dml(DmlOutcome), - /// No count and no verb (a buffered write, a trigger that consumed an - /// opaque plan): the statement renders `OK` unless a DML outcome is - /// folded too. + /// No count and no verb (a buffered write, an INSTEAD OF body that + /// replaced an opaque plan): the statement renders `OK` unless a DML + /// outcome is folded too. Opaque, } @@ -56,124 +52,6 @@ impl HandledWrite { } } -/// Outcome of routing a single task through the in-transaction staging gate. -pub(super) enum TxnRouteOutcome { - /// Not staged/buffered: caller proceeds to normal dispatch with the - /// (possibly `txn_id`-stamped) task. - Proceed(Box), - /// Fully handled (buffered, or a staged write with its real count). - /// Caller folds this into the statement tag and continues the loop. - Handled(HandledWrite), -} - -impl NodeDbPgHandler { - /// Route a single task through the protocol-neutral in-transaction - /// staging gate (`shared::session::staging_gate`), translating its - /// outcome into this file's `PgWireResult`. A constraint violation on a - /// staged write surfaces here as the pgwire error, matching the - /// pre-refactor `stage_in_tx_point_write` behavior exactly. - pub(super) async fn route_task_in_txn( - &self, - session_id: SessionId, - identity: &AuthenticatedIdentity, - task: PhysicalTask, - plan_lease_scope: Arc, - ) -> PgWireResult { - use crate::control::server::shared::session::expander_stage::{ - ExpanderOutcome, route_in_tx_expander, - }; - use crate::control::server::shared::session::staging_gate::{ - InTxnRoute, StagingGateError, route_in_tx_write, - }; - - let user_id: Option> = - Some(std::sync::Arc::from(identity.username.as_str())); - - // In-transaction `MERGE` and `UPDATE ... FROM` are resolved + staged at - // STATEMENT time by the expander (read-your-own-writes for later - // statements in the same txn); every other task falls through to the - // neutral staging gate. The expander dispatches each derived point op via - // the SAME closure, so it must be `Fn` — hence `user_id.clone()` per call. - let buffer_start = self.sessions.buffered_task_count(session_id); - let routed = match route_in_tx_expander( - &self.state, - &self.sessions, - session_id, - task, - |stage_task| self.dispatch_authorized_task(stage_task, user_id.clone(), identity), - ) - .await - { - Ok(ExpanderOutcome::Handled(route)) => Ok(route), - Ok(ExpanderOutcome::Passthrough(task)) => { - route_in_tx_write( - &self.state, - &self.sessions, - session_id, - *task, - |stage_task| { - self.dispatch_authorized_task(stage_task, user_id.clone(), identity) - }, - ) - .await - } - Err(e) => Err(e), - }; - - if self.sessions.buffered_task_count(session_id) > buffer_start - && !self.sessions.attach_tx_lease_scope_since( - session_id, - buffer_start, - plan_lease_scope, - ) - { - return Err(PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - nodedb_types::error::sqlstate::INTERNAL_ERROR.to_owned(), - "internal error: failed to retain descriptor leases for buffered transaction tasks" - .to_owned(), - )))); - } - - match routed { - // The pgwire dispatch proposes a write through Raft or appends its - // redo record in the funnel. - Ok(InTxnRoute::Read(routed_task) | InTxnRoute::Autocommit(routed_task)) => { - Ok(TxnRouteOutcome::Proceed(routed_task)) - } - Ok(InTxnRoute::Buffered) => Ok(TxnRouteOutcome::Handled(HandledWrite::Opaque)), - Ok(InTxnRoute::Staged(outcome)) => Ok(TxnRouteOutcome::Handled(HandledWrite::Dml( - staged_dml_outcome(outcome.kind, outcome.affected), - ))), - Err(StagingGateError::Dispatch(e)) => { - let (severity, code, message) = error_to_sqlstate(&e); - Err(PgWireError::UserError(Box::new(ErrorInfo::new( - severity.to_owned(), - code.to_owned(), - message, - )))) - } - Err(StagingGateError::Rejected { code }) => { - let (severity, sqlstate, message) = match code { - Some(code) => { - crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(&code) - } - None => ( - "ERROR", - nodedb_types::error::sqlstate::INTERNAL_ERROR, - "unknown data plane error".to_owned(), - ), - }; - Err(PgWireError::UserError(Box::new(ErrorInfo::new( - severity.to_owned(), - sqlstate.to_owned(), - message, - )))) - } - } - } -} - /// What a pre-dispatch hook answered the task with. pub(super) enum PreDispatchHandled { /// A clone write's `RETURNING` rows, encoded. Caller pushes the response. @@ -184,14 +62,13 @@ pub(super) enum PreDispatchHandled { /// Outcome of running the pre-dispatch hooks for a single task. pub(super) enum PreDispatchOutcome { - /// The task was fully handled (trigger short-circuit, or clone write - /// interception). Caller emits the answer and continues the loop. + /// The task was fully handled by clone write interception. Caller emits + /// the answer and continues the loop. Handled(PreDispatchHandled), - /// No interception occurred (or a mutation was applied in place); - /// caller proceeds to normal dispatch with the (possibly mutated) task - /// and the trigger bookkeeping needed for the AFTER-trigger phase. + /// No interception occurred; caller proceeds to normal dispatch with the + /// (possibly retargeted) task and its write classification. /// Boxed: `PhysicalTask` makes this variant far larger than `Handled`, - /// which would otherwise bloat every `PreDispatchOutcome` on the stack. + /// which will otherwise bloat every `PreDispatchOutcome` on the stack. Proceed(Box), } @@ -199,7 +76,6 @@ pub(super) enum PreDispatchOutcome { pub(super) struct PreDispatchProceed { pub(super) task: PhysicalTask, pub(super) dml_info: Option, - pub(super) old_row: Option>, pub(super) truncate_restart_collection: Option, } @@ -223,8 +99,8 @@ pub(super) struct PreDispatchContext<'a> { } impl NodeDbPgHandler { - /// Run trigger interception and clone write-path interception for a - /// single write task, before it reaches normal dispatch. + /// Run clone write-path interception for a single write task, before it + /// reaches normal dispatch. pub(super) async fn run_pre_dispatch_hooks( &self, context: PreDispatchContext<'_>, @@ -232,124 +108,14 @@ impl NodeDbPgHandler { ) -> PgWireResult { let PreDispatchContext { identity, - auth, tenant_id, - session_id, - plan_kind, - projection, + .. } = context; - // --- Trigger interception for DML writes --- - let mut dml_info = crate::control::trigger::dml_hook::classify_dml_write(&task.plan); - - // The OLD read must retain the exact database identity of the task, - // rather than re-resolving mutable session state. - let database_id = task.database_id; - - // Fetch OLD row and fire BEFORE/INSTEAD OF triggers if applicable. - let old_row = if let Some(ref info) = dml_info - && info.document_id.is_some() - && (matches!( - info.event, - crate::control::trigger::DmlEvent::Update - | crate::control::trigger::DmlEvent::Delete - ) || info.needs_existence_probe) - { - let doc_id = info.document_id.as_deref().unwrap_or(""); - // `info.collection` is already qualified (from `DocumentOp`) — - // rebuild rather than re-qualify. - let row = crate::control::trigger::dml_hook::fetch_old_row( - &self.state, - identity, - database_id, - auth, - &nodedb_types::QualifiedCollection::from_stored(info.collection.clone()), - doc_id, - ) - .await - .map_err(|error| { - let (severity, code, message) = error_to_sqlstate(&error); - PgWireError::UserError(Box::new(ErrorInfo::new( - severity.to_owned(), - code.to_owned(), - message, - ))) - })?; - if !row.is_empty() { Some(row) } else { None } - } else { - None - }; - - // Probe-driven reclassification. - if let Some(ref mut info) = dml_info - && info.needs_existence_probe - { - info.event = if old_row.is_some() { - crate::control::trigger::DmlEvent::Update - } else { - crate::control::trigger::DmlEvent::Insert - }; - } - - if let Some(ref info) = dml_info { - use crate::control::trigger::dml_hook_fire::PreDispatchResult; - match crate::control::trigger::dml_hook_fire::fire_pre_dispatch_triggers( - crate::control::trigger::dml_hook_fire::DispatchTriggerParams { - state: &self.state, - identity, - database_id, - tenant_id, - info, - old_row: &old_row, - cascade_depth: 0, - }, - ) - .await - .map_err(|e| { - let (severity, code, message) = error_to_sqlstate(&e); - PgWireError::UserError(Box::new(ErrorInfo::new( - severity.to_owned(), - code.to_owned(), - message, - ))) - })? { - PreDispatchResult::Handled => { - // The trigger consumed the row: the statement ran and - // affected nothing. A count-bearing plan keeps its verb - // with a zero count so the fold stays on one verb; an - // opaque plan contributes no count. - let handled = match plan_kind { - PlanKind::DmlResult(verb) => { - HandledWrite::Dml(DmlOutcome { verb, affected: 0 }) - } - // The verb is resolved at apply time and no apply - // happened. The statement is an `INSERT ... ON - // CONFLICT DO UPDATE`, so it reports as `INSERT`. - PlanKind::DmlResultByOp => HandledWrite::Dml(DmlOutcome { - verb: "INSERT", - affected: 0, - }), - PlanKind::Execution - | PlanKind::ArraySlice - | PlanKind::ReturningRows - | PlanKind::SingleDocument - | PlanKind::MultiRow => HandledWrite::Opaque, - }; - return Ok(PreDispatchOutcome::Handled(PreDispatchHandled::Write( - handled, - ))); - } - PreDispatchResult::Proceed { - mutated_fields: Some(fields), - } => { - crate::control::trigger::dml_hook::patch_task_with_mutated_fields( - &mut task, &fields, - ); - } - PreDispatchResult::Proceed { - mutated_fields: None, - } => {} - } - } + // A write whose collection fires a BEFORE, INSTEAD OF or SYNC AFTER + // body never reaches this path: it runs in its statement's + // transaction (`txn_route`). The classification feeds the + // post-dispatch auto-analyze. + let dml_info = crate::control::trigger::dml_hook::classify_dml_write(&task.plan); // Extract truncate restart_identity info before task is moved. // Engine-neutral: `truncate_target` names every truncate-shaped op. @@ -376,52 +142,9 @@ impl NodeDbPgHandler { ))) })? { CloneWriteOutcome::Handled(resp) => { - use crate::control::server::response_shape::compose::{ - ShapeOutcome, shape_payload_no_plan, - }; - use crate::control::server::response_shape::redaction::QueryRedaction; - // A clone write can carry RETURNING rows, which deliver - // stored column values just as a SELECT does. - let redaction = QueryRedaction::for_plan(tenant_id, auth, &task.plan); - // A clone write's RETURNING list names stored columns - // only, never a Control-Plane computed column. - match shape_payload_no_plan( - resp.payload.as_ref(), - plan_kind, - projection, - Some(redaction.ctx(&self.state.redaction)), - None, - ) - .map_err(|e| shape_error_to_pg(&e))? - { - ShapeOutcome::Rows(shaped) => { - // Clone write-path DML result (PointUpdate/PointDelete): - // no client-requested result formats, so text. - let (response, notice) = - crate::control::server::pgwire::handler::shape_encode::shaped_query_response( - shaped, - &[], - ); - if let Some(n) = notice { - self.sessions.push_notice(session_id, n); - } - return Ok(PreDispatchOutcome::Handled(PreDispatchHandled::Rows( - response, - ))); - } - ShapeOutcome::Passthrough => { - let handled = - match payload_to_dml_outcome(resp.payload.as_ref(), plan_kind) - .map_err(|e| error_to_pg(&e))? - { - Some(outcome) => HandledWrite::Dml(outcome), - None => HandledWrite::Opaque, - }; - return Ok(PreDispatchOutcome::Handled(PreDispatchHandled::Write( - handled, - ))); - } - } + return Ok(PreDispatchOutcome::Handled( + self.clone_write_answer(context, &task.plan, &resp)?, + )); } CloneWriteOutcome::Passthrough => {} } @@ -430,8 +153,68 @@ impl NodeDbPgHandler { Ok(PreDispatchOutcome::Proceed(Box::new(PreDispatchProceed { task, dml_info, - old_row, truncate_restart_collection, }))) } } + +impl NodeDbPgHandler { + /// The answer a clone copy-on-write that handled a write gives the + /// statement: its `RETURNING` rows, or its count. + pub(super) fn clone_write_answer( + &self, + context: PreDispatchContext<'_>, + plan: &nodedb_physical::physical_plan::PhysicalPlan, + resp: &crate::bridge::envelope::Response, + ) -> PgWireResult { + use crate::control::server::response_shape::compose::{ + ShapeOutcome, shape_payload_no_plan, + }; + use crate::control::server::response_shape::redaction::QueryRedaction; + let PreDispatchContext { + auth, + tenant_id, + session_id, + plan_kind, + projection, + .. + } = context; + // A clone write can carry RETURNING rows, which deliver stored column + // values as a SELECT does. + let redaction = QueryRedaction::for_plan(tenant_id, auth, plan); + // A clone write's RETURNING list names stored columns only, never a + // Control-Plane computed column. + match shape_payload_no_plan( + resp.payload.as_ref(), + plan_kind, + projection, + Some(redaction.ctx(&self.state.redaction)), + None, + ) + .map_err(|e| shape_error_to_pg(&e))? + { + ShapeOutcome::Rows(shaped) => { + // Clone write-path DML result (PointUpdate/PointDelete): no + // client-requested result formats, so text. + let (response, notice) = + crate::control::server::pgwire::handler::shape_encode::shaped_query_response( + shaped, + &[], + ); + if let Some(n) = notice { + self.sessions.push_notice(session_id, n); + } + Ok(PreDispatchHandled::Rows(response)) + } + ShapeOutcome::Passthrough => { + let handled = match payload_to_dml_outcome(resp.payload.as_ref(), plan_kind) + .map_err(|e| error_to_pg(&e))? + { + Some(outcome) => HandledWrite::Dml(outcome), + None => HandledWrite::Opaque, + }; + Ok(PreDispatchHandled::Write(handled)) + } + } + } +} diff --git a/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs b/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs index 11067e398..c3277c36e 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/gateway_dispatch.rs @@ -106,21 +106,25 @@ impl NodeDbPgHandler { redaction: &redaction, session_id, }; - let gateway = self.state.gateway.get().ok_or_else(|| { + let gateway = self.state.installed_gateway().map_err(|e| { + let (severity, code, message) = super::super::super::types::error_to_sqlstate(&e); PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - "55000".to_owned(), - "gateway not available".to_owned(), + severity.to_owned(), + code.to_owned(), + message, ))) })?; // Forwarding is autocommit only: an in-block statement never reaches // here (`maybe_dispatch_tasks_via_gateway`), so no transaction id. + // A strong session's forwarded reads confirm on the node that serves + // them. Writes ignore the flag: Raft orders them. let gw_ctx = crate::control::gateway::core::QueryContext { tenant_id, trace_id: TraceId::generate(), database_id, txn_id: None, + linearizable: self.sessions.read_consistency(session_id).requires_leader(), }; // A derived implicit-edge write beside the user's own never answers diff --git a/nodedb/src/control/server/pgwire/handler/routing/placement.rs b/nodedb/src/control/server/pgwire/handler/routing/placement.rs index c85625ae0..2ed7cae99 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/placement.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/placement.rs @@ -7,6 +7,11 @@ //! A partition does not notify a deposed leader, so the second claim is //! proven against a quorum rather than assumed from the first. //! +//! A linearizable read that runs here over groups led elsewhere gets the same +//! proof per group: each read confirms its group where it is served (see +//! `control::cluster::linearizable_read`). A read of a group with no replica +//! here cannot be served here with that proof, so such a set is refused. +//! //! A bounded-staleness read makes a weaker claim, and it is checked the same //! way: being a member of the group says nothing about how far behind the //! replica has fallen, so the bound is measured against the leader rather @@ -17,27 +22,32 @@ use nodedb_physical::physical_task::PhysicalTask; use super::super::core::NodeDbPgHandler; -/// Where a set of tasks should execute. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] +/// Where a set of tasks must execute. +#[derive(Debug, Clone, PartialEq, Eq)] pub(super) enum TaskPlacement { /// Run here. Either the deployment has no Raft routing at all, or the - /// read accepts this replica. + /// task set needs no confirmed leader. Local, - /// Run here once a quorum confirms this node still leads `group_id`. - LocalLeader { group_id: u64 }, + /// Run here as a linearizable read. Every read confirms its group where it + /// is served, and this node holds a replica of every group read here. + LocalConfirmed, /// One remote leader owns every task — forward through the gateway. Gateway, /// The read must reach a leader, and this node knows of none to send it - /// to. Serving it here would answer from a replica that may be arbitrarily + /// to. Serving it here will answer from a replica that can be arbitrarily /// far behind. NoLeader, + /// A linearizable read spans groups led by different nodes, and this node + /// holds no replica of at least one of them. No node can serve the whole + /// set with proof. + Unconfirmable, } /// Per-task outcome of [`placement_for_group`], folded across a task set by /// `placement_for_tasks`. #[derive(Debug, Clone, Copy, PartialEq, Eq)] enum GroupPlacement { - /// This node leads the group and the read may be served without proof. + /// This node leads the group and the read can be served without proof. Local, /// This node leads the group by the routing table's account, but the /// read needs that confirmed against a quorum first. The caller attaches @@ -50,7 +60,32 @@ enum GroupPlacement { NoLeader, } -/// Decide how a single task's group should be served, given plain facts +/// One task's group, as `placement_for_tasks` resolves it from the routing +/// table. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum TaskGroup { + /// The routing table maps the task's vShard to no known group. + Unmapped, + Placed { + placement: GroupPlacement, + /// This node holds a replica of the group, as a voter or a learner. + hosts_replica: bool, + }, +} + +/// One task's routing facts, copied out of the routing table. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct RoutedTask { + group_id: u64, + /// The routing table's leader hint for the group. + leader: u64, + /// This node is a voter of the group. + is_member: bool, + /// This node holds a replica of the group, as a voter or a learner. + hosts_replica: bool, +} + +/// Decide how a single task's group must be served, given plain facts /// about it — no routing-table or gate lookups, so this is unit-testable in /// isolation from `SharedState`. /// @@ -68,6 +103,13 @@ fn placement_for_group( needs_confirmed_leader: bool, replica_fresh: bool, ) -> GroupPlacement { + // A hint naming this node is stale once this node is no voter of the + // group: it left the group, so it leads nothing there. + let leader = if leader == my_node && !is_member { + 0 + } else { + leader + }; if leader == my_node { // Leading by the routing table's account. A linearizable read still // has to prove it against a quorum before it is served. @@ -76,7 +118,7 @@ fn placement_for_group( } return GroupPlacement::Local; } - // A replica here may serve a read that does not need the leader — but a + // A replica here can serve a read that does not need the leader — but a // bounded-staleness read only if the replica is actually within its // bound. Too far behind, and it falls through to the leader, which // satisfies any bound by definition. @@ -86,10 +128,10 @@ fn placement_for_group( if leader == 0 { // No leader is known — mid-election, or this node's view is stale. // A read that needs a confirmed leader, or a fresher replica than - // this one just proved it has, waits for one rather than being + // this one proved it has, waits for one rather than being // answered from whatever is local. A write still runs locally: it // is proposed through Raft, which refuses it on a non-leader and - // redirects, so refusing here would only break writes during the + // redirects, so refusing here will only break writes during the // seconds an election takes. let needs_leader_or_fresh = needs_confirmed_leader || consistency.max_staleness().is_some(); if needs_leader_or_fresh { @@ -100,74 +142,132 @@ fn placement_for_group( GroupPlacement::RemoteLeader { leader } } +/// Fold the per-task groups of a task set into one placement. +/// +/// `needs_confirmed_leader` is true for a linearizable read. Such a read +/// never gets `Local`: it runs here only as `LocalConfirmed`, or goes to the +/// one remote leader, or is refused. Every other +/// task set keeps the plain rules: one remote leader forwards, anything else +/// runs here. +fn fold_task_groups( + groups: impl IntoIterator, + needs_confirmed_leader: bool, +) -> TaskPlacement { + let mut led_here = false; + let mut remote_unhosted = false; + let mut remote_leader: Option = None; + let mut split_leaders = false; + + for group in groups { + let TaskGroup::Placed { + placement, + hosts_replica, + } = group + else { + // No group means no leader to prove anything against. + if needs_confirmed_leader { + return TaskPlacement::NoLeader; + } + return TaskPlacement::Local; + }; + match placement { + GroupPlacement::Local if !needs_confirmed_leader => return TaskPlacement::Local, + GroupPlacement::Local | GroupPlacement::LocalLeader => led_here = true, + GroupPlacement::NoLeader => return TaskPlacement::NoLeader, + GroupPlacement::RemoteLeader { leader } => { + remote_unhosted |= !hosts_replica; + match remote_leader { + None => remote_leader = Some(leader), + Some(prev) if prev != leader => split_leaders = true, + Some(_) => {} + } + } + } + } + + match (led_here, remote_leader) { + (false, None) => TaskPlacement::Local, + (false, Some(_)) if !split_leaders => TaskPlacement::Gateway, + (true, None) => TaskPlacement::LocalConfirmed, + // Several leaders, or leaders here and elsewhere: the gateway forwards + // to one node, so the set runs here. + _ if !needs_confirmed_leader => TaskPlacement::Local, + _ if remote_unhosted => TaskPlacement::Unconfirmable, + _ => TaskPlacement::LocalConfirmed, + } +} + impl NodeDbPgHandler { /// Decide where `tasks` run. /// /// `needs_confirmed_leader` is false for a write: it reaches the leader by /// being proposed through Raft, which establishes leadership on its own, so - /// a read-index round in front of it would only add a round trip. + /// a read-index round in front of it will only add a round trip. pub(super) fn placement_for_tasks( &self, tasks: &[PhysicalTask], consistency: ReadConsistency, needs_confirmed_leader: bool, ) -> TaskPlacement { - if self.state.gateway.get().is_none() { - return TaskPlacement::Local; - } let Some(routing) = self.state.cluster_routing.as_ref() else { return TaskPlacement::Local; }; - let routing = routing.read().unwrap_or_else(|p| p.into_inner()); let my_node = self.state.node_id; + // Routing facts are copied out under a short guard. The staleness + // gate locks `MultiRaft`, so it runs only after the guard drops. + let routed: Vec> = { + let routing = routing.read().unwrap_or_else(|p| p.into_inner()); + tasks + .iter() + .map(|task| { + let group_id = routing.group_for_vshard(task.vshard_id.as_u32()).ok()?; + let info = routing.group_info(group_id)?; + let is_member = info.members.contains(&my_node); + Some(RoutedTask { + group_id, + leader: info.leader, + is_member, + hosts_replica: is_member || info.learners.contains(&my_node), + }) + }) + .collect() + }; - let mut remote_leader: Option = None; - let mut local_group: Option = None; - for task in tasks { - let vshard_id = task.vshard_id.as_u32(); - let Ok(group_id) = routing.group_for_vshard(vshard_id) else { - return TaskPlacement::Local; - }; - let Some(info) = routing.group_info(group_id) else { - return TaskPlacement::Local; + let groups = routed.into_iter().map(|routed| { + let Some(RoutedTask { + group_id, + leader, + is_member, + hosts_replica, + }) = routed + else { + return TaskGroup::Unmapped; }; - let leader = info.leader; - let is_member = info.members.contains(&my_node); // Only cheap when consistency carries no bound: `max_staleness()` // returns `None` and the gate is never touched. let replica_fresh = self.replica_satisfies(group_id, consistency); - - match placement_for_group( - leader, - my_node, - is_member, - consistency, - needs_confirmed_leader, - replica_fresh, - ) { - GroupPlacement::Local => return TaskPlacement::Local, - GroupPlacement::LocalLeader => { - local_group = Some(group_id); - } - GroupPlacement::NoLeader => return TaskPlacement::NoLeader, - GroupPlacement::RemoteLeader { leader } => match remote_leader { - None => remote_leader = Some(leader), - // Tasks fan out across leaders — the gateway forwards to - // one node, so this set runs locally instead. - Some(prev) if prev != leader => return TaskPlacement::Local, - _ => {} - }, + // A node with no replica of the group has nothing to run locally: + // a read will see none of its rows, and a proposal will wait for + // an apply that never reaches this node. With no leader to forward + // to, the task waits for one. + let placement = if !hosts_replica && (leader == 0 || leader == my_node) { + GroupPlacement::NoLeader + } else { + placement_for_group( + leader, + my_node, + is_member, + consistency, + needs_confirmed_leader, + replica_fresh, + ) + }; + TaskGroup::Placed { + placement, + hosts_replica, } - } - - match (local_group, remote_leader) { - // Some tasks lead here and others lead elsewhere: no single node - // can serve the set, so it runs locally as it always has. - (Some(_), Some(_)) => TaskPlacement::Local, - (Some(group_id), None) => TaskPlacement::LocalLeader { group_id }, - (None, Some(_)) => TaskPlacement::Gateway, - (None, None) => TaskPlacement::Local, - } + }); + fold_task_groups(groups, needs_confirmed_leader) } /// Whether this node's replica of `group_id` meets `consistency`. @@ -191,7 +291,7 @@ impl NodeDbPgHandler { mod tests { use std::time::Duration; - use super::{GroupPlacement, placement_for_group}; + use super::{GroupPlacement, TaskGroup, TaskPlacement, fold_task_groups, placement_for_group}; use crate::types::ReadConsistency; const LEADER: u64 = 1; @@ -253,4 +353,93 @@ mod tests { let placement = placement_for_group(LEADER, ME, true, ReadConsistency::Strong, true, false); assert_eq!(placement, GroupPlacement::LocalLeader); } + + const THIRD: u64 = 3; + + fn led_here() -> TaskGroup { + TaskGroup::Placed { + placement: GroupPlacement::LocalLeader, + hosts_replica: true, + } + } + + fn led_by(leader: u64, hosts_replica: bool) -> TaskGroup { + TaskGroup::Placed { + placement: GroupPlacement::RemoteLeader { leader }, + hosts_replica, + } + } + + /// Every input a strong read can produce: no combination of leader, + /// membership and freshness serves it here without proof. + #[test] + fn a_strong_read_never_places_a_group_locally() { + for leader in [NO_LEADER, ME, OTHER] { + for is_member in [false, true] { + for fresh in [false, true] { + let placement = placement_for_group( + leader, + ME, + is_member, + ReadConsistency::Strong, + true, + fresh, + ); + assert_ne!(placement, GroupPlacement::Local); + } + } + } + } + + #[test] + fn a_strong_read_over_mixed_leaders_confirms_every_group() { + let placement = fold_task_groups([led_here(), led_by(OTHER, true)], true); + assert_eq!(placement, TaskPlacement::LocalConfirmed); + } + + #[test] + fn a_strong_read_over_several_remote_leaders_confirms_every_group() { + let placement = fold_task_groups([led_by(OTHER, true), led_by(THIRD, true)], true); + assert_eq!(placement, TaskPlacement::LocalConfirmed); + } + + #[test] + fn a_strong_read_over_a_group_with_no_replica_here_is_refused() { + let mixed = fold_task_groups([led_here(), led_by(OTHER, false)], true); + assert_eq!(mixed, TaskPlacement::Unconfirmable); + let fan_out = fold_task_groups([led_by(OTHER, true), led_by(THIRD, false)], true); + assert_eq!(fan_out, TaskPlacement::Unconfirmable); + } + + #[test] + fn a_strong_read_over_an_unmapped_vshard_is_refused() { + let placement = fold_task_groups([led_here(), TaskGroup::Unmapped], true); + assert_eq!(placement, TaskPlacement::NoLeader); + } + + #[test] + fn a_strong_read_led_by_one_remote_node_forwards() { + let placement = fold_task_groups([led_by(OTHER, false), led_by(OTHER, false)], true); + assert_eq!(placement, TaskPlacement::Gateway); + } + + /// A write proves leadership through its proposal, so a mixed or + /// unmapped set still runs here. + #[test] + fn a_write_over_mixed_leaders_runs_locally() { + let write = |groups: Vec| fold_task_groups(groups, false); + let local = TaskGroup::Placed { + placement: GroupPlacement::Local, + hosts_replica: true, + }; + assert_eq!( + write(vec![local, led_by(OTHER, false)]), + TaskPlacement::Local + ); + assert_eq!( + write(vec![led_by(OTHER, false), led_by(THIRD, false)]), + TaskPlacement::Local + ); + assert_eq!(write(vec![TaskGroup::Unmapped]), TaskPlacement::Local); + } } diff --git a/nodedb/src/control/server/pgwire/handler/routing/planning.rs b/nodedb/src/control/server/pgwire/handler/routing/planning.rs index 3b566bac9..08f90cd91 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/planning.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/planning.rs @@ -187,13 +187,13 @@ impl NodeDbPgHandler { tenant_id, database_id, scope.auth(), + self.sessions.tx_id(session_id), ) .await .map_err(StatementSetupError::from)?; // Validate enum-typed column values for INSERT/UPDATE before planning. self.enforce_enum_labels_if_needed(&clean_sql, tenant_id, database_id) - .await .map_err(StatementSetupError::from)?; // Cache key isn't session-knob-scoped, so bypass entirely under a strategy @@ -236,7 +236,6 @@ impl NodeDbPgHandler { }; let (tasks, output_schema, versions) = if !params.is_empty() { - let perm_cache = self.state.permission_cache.read().await; let sec = crate::control::planner::context::PlanSecurityContext { identity, auth: scope.auth(), @@ -244,7 +243,9 @@ impl NodeDbPgHandler { redaction_store: &self.state.redaction, permissions: &self.state.permissions, roles: &self.state.roles, - permission_cache: Some(&*perm_cache), + permission_tree: crate::control::planner::context::PermissionTreeSource::Live( + &self.state.permission_cache, + ), }; let (tasks, output_schema, versions) = self .query_ctx @@ -271,7 +272,6 @@ impl NodeDbPgHandler { (tasks, output_schema, versions) } else { let (planned, output_schema, versions, cache_eligibility) = { - let perm_cache = self.state.permission_cache.read().await; let sec = crate::control::planner::context::PlanSecurityContext { identity, auth: scope.auth(), @@ -279,7 +279,9 @@ impl NodeDbPgHandler { redaction_store: &self.state.redaction, permissions: &self.state.permissions, roles: &self.state.roles, - permission_cache: Some(&*perm_cache), + permission_tree: crate::control::planner::context::PermissionTreeSource::Live( + &self.state.permission_cache, + ), }; self.query_ctx .plan_sql_with_rls_and_versions( @@ -294,7 +296,7 @@ impl NodeDbPgHandler { }; // Strategy overrides aren't in the cache key: caching a resolved PK→surrogate - // binding would preserve stale row identity across later writes. + // binding will preserve stale row identity across later writes. if !bypass_cache && cache_eligibility.is_cacheable() { self.sessions.put_cached_plan( session_id, diff --git a/nodedb/src/control/server/pgwire/handler/routing/pre_dispatch.rs b/nodedb/src/control/server/pgwire/handler/routing/pre_dispatch.rs index fbea2701d..0e28184be 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/pre_dispatch.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/pre_dispatch.rs @@ -19,37 +19,41 @@ use super::result_shaping::ResultShaping; use super::super::super::types::error_to_sqlstate; use super::super::core::NodeDbPgHandler; -/// How long a linearizable read waits for a quorum to confirm this node still -/// leads the group. Several election timeouts (150-300ms), so an ordinary -/// round trip always fits and a partition is reported rather than hung on. -const LEADERSHIP_CONFIRM_TIMEOUT: std::time::Duration = std::time::Duration::from_millis(750); - /// This node has missed a topology transition and must not coordinate work /// until it catches up. fn superseded_topology_view(behind: u64) -> PgWireError { + stale_read(format!( + "this node is {behind} cluster generation(s) behind and is not \ + coordinating queries until it catches up; retry" + )) +} + +/// A retryable refusal of a read that needs a leader. +fn stale_read(message: String) -> PgWireError { PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), nodedb_types::error::sqlstate::STALE_READ_NOT_LEADER.to_owned(), - format!( - "this node is {behind} cluster generation(s) behind and is not \ - coordinating queries until it catches up; retry" - ), + message, ))) } /// No leader is known for a group whose read requires one. fn no_serving_leader() -> PgWireError { - PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - nodedb_types::error::sqlstate::STALE_READ_NOT_LEADER.to_owned(), - "no leader is currently serving this range; retry".to_owned(), - ))) + stale_read("no leader is currently serving this range; retry".to_owned()) +} + +/// A linearizable read that no node can serve with proof. +fn unconfirmable_read() -> PgWireError { + stale_read( + "this read spans ranges led by different nodes, and this node holds no \ + replica of some of them; retry on a node that replicates every range" + .to_owned(), + ) } impl NodeDbPgHandler { - /// Prove against a quorum that this node still leads `group_id`. - /// The routing table is a cached view — a partitioned leader keeps its - /// entry long after a successor is elected. + /// Refuse to coordinate while this node's cluster epoch is behind. Its + /// routing table is then a stale view of who leads what. fn refuse_if_topology_view_superseded(&self) -> PgWireResult<()> { let Some(epoch) = self.state.cluster_epoch.get() else { return Ok(()); @@ -60,28 +64,6 @@ impl NodeDbPgHandler { Ok(()) } - async fn confirm_local_leadership(&self, group_id: u64) -> PgWireResult<()> { - let Some(gate) = self.state.raft_read_gate.get() else { - // Reached only if routed here before `start_raft` published the gate. - return Err(no_serving_leader()); - }; - use crate::control::cluster::read_index::ReadIndexRefusal; - match gate - .confirm_leader(group_id, LEADERSHIP_CONFIRM_TIMEOUT) - .await - { - Ok(_read_index) => Ok(()), - Err(ReadIndexRefusal::NotLeader) => Err(no_serving_leader()), - Err(ReadIndexRefusal::Timeout { waited_ms }) => { - Err(PgWireError::UserError(Box::new(ErrorInfo::new( - "ERROR".to_owned(), - nodedb_types::error::sqlstate::STALE_READ_NOT_LEADER.to_owned(), - format!("no quorum confirmed leadership within {waited_ms}ms; retry"), - )))) - } - } - } - /// Route an implicit-edge dependent predicate through OLLP/Calvin when its /// catalog and session prerequisites require atomic edge maintenance. pub(super) async fn maybe_dispatch_implicit_edge_recon( @@ -98,9 +80,7 @@ impl NodeDbPgHandler { formats: result_formats, } = shaping; let tx_state = self.sessions.transaction_state(session_id); - if tx_state == TransactionState::InBlock - || self.state.calvin_completion_registry.get().is_none() - { + if tx_state == TransactionState::InBlock { return Ok(None); } @@ -137,7 +117,10 @@ impl NodeDbPgHandler { /// Forward an ordinary remote-leader task set through the gateway. /// /// Unresolved multi-step DML stays local so its orchestrator can resolve - /// final plans before authorization. + /// final plans before authorization. A `ClusterArray` plan stays local + /// because this node's array coordinator routes it to the owning shards. + /// Array DDL stays local because the dispatch loop proposes it as a + /// replicated catalog entry. Its task's vShard names no owner. /// /// The caller skips this for an in-block statement. A write forwarded /// here applies durably at once, outside the transaction; the dispatch @@ -159,19 +142,19 @@ impl NodeDbPgHandler { projection, formats: result_formats, } = shaping; - if has_orchestrated_dml(tasks) { + if has_orchestrated_dml(tasks) || has_cluster_array_op(tasks) || has_array_ddl(tasks) { return Ok(None); } self.refuse_if_topology_view_superseded()?; let consistency = consistency_for_tasks(&self.sessions, tasks, session_id); let needs_confirmed_leader = consistency.requires_leader() && !has_replicated_writes(tasks); match self.placement_for_tasks(tasks, consistency, needs_confirmed_leader) { - TaskPlacement::Local => return Ok(None), - TaskPlacement::LocalLeader { group_id } => { - self.confirm_local_leadership(group_id).await?; - return Ok(None); - } + // Each read in the set confirms its group where it is served: the + // local dispatch for a replica read here, the serving node for a + // leg the gateway sends elsewhere. + TaskPlacement::Local | TaskPlacement::LocalConfirmed => return Ok(None), TaskPlacement::NoLeader => return Err(no_serving_leader()), + TaskPlacement::Unconfirmable => return Err(unconfirmable_read()), TaskPlacement::Gateway => {} } @@ -198,6 +181,32 @@ impl NodeDbPgHandler { } } +/// Whether any task is a `ClusterArray` plan. +/// +/// The `ArrayCoordinator` on this node fans such a plan out to the shards +/// that own its cells. The plan's own vShard names no owner, so forwarding +/// it to that vShard's leader is wrong, and the plan has no wire encoding. +/// The dispatch loop runs it here. +fn has_cluster_array_op(tasks: &[PhysicalTask]) -> bool { + tasks.iter().any(|task| { + matches!( + &task.plan, + crate::bridge::envelope::PhysicalPlan::ClusterArray(_) + ) + }) +} + +/// Whether any task is array DDL. +/// +/// The dispatch loop proposes it as a `PutArray` or `DeleteArray` catalog +/// entry, which every node applies. Forwarded to its vShard's leader, it +/// opens the array on that one node and writes no catalog entry. +fn has_array_ddl(tasks: &[PhysicalTask]) -> bool { + tasks + .iter() + .any(|task| crate::control::array_catalog::ddl::is_array_ddl(&task.plan)) +} + fn has_orchestrated_dml(tasks: &[PhysicalTask]) -> bool { tasks.iter().any(|task| { matches!( diff --git a/nodedb/src/control/server/pgwire/handler/routing/streaming.rs b/nodedb/src/control/server/pgwire/handler/routing/streaming.rs index c52b5b697..ad43b5135 100644 --- a/nodedb/src/control/server/pgwire/handler/routing/streaming.rs +++ b/nodedb/src/control/server/pgwire/handler/routing/streaming.rs @@ -171,23 +171,18 @@ impl NodeDbPgHandler { ) => checked, }; - // Single-node fans to local cores directly; cluster routes the scan to - // its owning vShard (local or remote over the L4 QUIC streaming - // transport) via the gateway and merges per-route streams. - let stream = if let Some(gw) = state.gateway.get() { - let ctx = crate::control::gateway::core::QueryContext { - tenant_id: task.tenant_id, - trace_id: crate::types::TraceId::ZERO, - database_id: task.database_id, - txn_id: None, - }; - gw.execute_stream(&ctx, checked_child).await - } else { - crate::control::server::exchange::gather::gather_all_cores_stream_authorized( - &state, - checked_child.into_authorized(), - crate::types::TraceId::ZERO, - ) + // The gateway routes the scan to its owning vShard (local or remote + // over the L4 QUIC streaming transport) and merges per-route streams. + let ctx = crate::control::gateway::core::QueryContext { + tenant_id: task.tenant_id, + trace_id: crate::types::TraceId::ZERO, + database_id: task.database_id, + txn_id: None, + linearizable: self.sessions.read_consistency(session_id).requires_leader(), + }; + let stream = match state.installed_gateway() { + Ok(gateway) => gateway.execute_stream(&ctx, checked_child).await, + Err(error) => Err(error), } .map_err(|e| { let (severity, code, message) = error_to_sqlstate(&e); @@ -198,21 +193,20 @@ impl NodeDbPgHandler { ))) })?; - // Shape the streamed rows to match the SELECT projection, mirroring - // the (now-removed) post-hoc reproject seam: + // Shape the streamed rows to match the SELECT projection: // - named columns -> lazy per-batch shaping + projection // - `SELECT *` -> materialize, then id-first column union // - anything else -> raw single-column envelope passthrough // // Column-level redaction is resolved ONCE here, from the scan plan the // stream actually reads, and handed to every batch's shaping call - // below. Deriving it per batch would let an early batch ship before + // below. Deriving it per batch will let an early batch ship before // the policy was consulted. let redaction = Some(QueryRedaction::for_plan(task.tenant_id, auth, &child_plan)); // The streaming fast path bypasses the dispatch loop's own metering // call, so without this every pgwire autocommit SELECT that streams - // would go unbilled — and a token quota over reads would never + // will go unbilled — and a token quota over reads will never // accumulate anything to enforce against. The guard charges on drop, // when the stream finishes or the client disconnects, so what is // billed is rows actually written. `None` when metering is disabled. diff --git a/nodedb/src/control/server/pgwire/handler/session_explain.rs b/nodedb/src/control/server/pgwire/handler/session_explain.rs index 68988a315..91efc6826 100644 --- a/nodedb/src/control/server/pgwire/handler/session_explain.rs +++ b/nodedb/src/control/server/pgwire/handler/session_explain.rs @@ -78,10 +78,9 @@ impl NodeDbPgHandler { self.state.auth_stores(), database_id, ); - let perm_cache = - crate::control::security::auth_fence::permission_view(&self.state, tenant_id) - .await - .map_err(|e| crate::control::server::pgwire::types::error_map::error_to_pg(&e))?; + crate::control::security::auth_fence::admit_permission_view(&self.state, tenant_id) + .await + .map_err(|e| crate::control::server::pgwire::types::error_map::error_to_pg(&e))?; let sec = crate::control::planner::context::PlanSecurityContext { identity, auth: scope.auth(), @@ -89,7 +88,9 @@ impl NodeDbPgHandler { redaction_store: &self.state.redaction, permissions: &self.state.permissions, roles: &self.state.roles, - permission_cache: Some(&*perm_cache), + permission_tree: crate::control::planner::context::PermissionTreeSource::Live( + &self.state.permission_cache, + ), }; let (tasks, _output_schema) = self .query_ctx diff --git a/nodedb/src/control/server/pgwire/handler/stream_response.rs b/nodedb/src/control/server/pgwire/handler/stream_response.rs index f839ca70b..c831b4f5a 100644 --- a/nodedb/src/control/server/pgwire/handler/stream_response.rs +++ b/nodedb/src/control/server/pgwire/handler/stream_response.rs @@ -74,7 +74,7 @@ pub(crate) fn streaming_multirow_response( // QueryResponse outlives the pgwire handler. Its row stream owns this // scope so descriptor leases remain held while the client polls rows, // including on a mid-stream error, until completion or disconnect. - let _lease_scope = lease_scope; + let lease_scope = lease_scope; // Owned by the lazy response for the same reason the lease scope is: // it charges on drop, which must be when the stream finishes or the // client disconnects — so what gets billed is rows actually written @@ -82,7 +82,13 @@ pub(crate) fn streaming_multirow_response( let mut meter_guard = meter_guard; let mut emitted: usize = 0; let mut batches = stream; - while let Some(batch) = batches.next().await { + // A lease this node loses mid-stream ends the stream with a retryable + // error rather than more rows from a stale descriptor. + while let Some(batch) = lease_scope + .guard(batches.next()) + .await + .unwrap_or_else(|revoked| Some(Err(revoked))) + { let batch = batch.map_err(|e| { let (severity, code, message) = error_to_sqlstate(&e); PgWireError::UserError(Box::new(ErrorInfo::new( @@ -163,7 +169,7 @@ pub(crate) fn streaming_shaped_response( .map(|c| c.display_name.clone()) .collect(); // Cells live in the shaped row maps under unique per-column keys - // (display names may repeat across columns, e.g. `SELECT w.id, b.id`); + // (display names can repeat across columns, e.g. `SELECT w.id, b.id`); // derive the same keys the shaper used so every column reads its own cell. let cell_keys = crate::control::server::response_shape::project::cell_keys(&display_columns); // Advertise each projected column's real catalog type so the streaming @@ -192,7 +198,7 @@ pub(crate) fn streaming_shaped_response( // The lazy QueryResponse owns the admission scope rather than the // handler stack frame; dropping it on wire completion or disconnect // releases the descriptor leases. - let _lease_scope = lease_scope; + let lease_scope = lease_scope; // Owned by the lazy response for the same reason the lease scope is: // it charges on drop, which must be when the stream finishes or the // client disconnects — so what gets billed is rows actually written @@ -200,7 +206,13 @@ pub(crate) fn streaming_shaped_response( let mut meter_guard = meter_guard; let mut emitted: usize = 0; let mut batches = stream; - while let Some(batch) = batches.next().await { + // A lease this node loses mid-stream ends the stream with a retryable + // error rather than more rows from a stale descriptor. + while let Some(batch) = lease_scope + .guard(batches.next()) + .await + .unwrap_or_else(|revoked| Some(Err(revoked))) + { let batch = batch.map_err(|e| { let (severity, code, message) = error_to_sqlstate(&e); PgWireError::UserError(Box::new(ErrorInfo::new( diff --git a/nodedb/src/control/server/pgwire/handler/submit.rs b/nodedb/src/control/server/pgwire/handler/submit.rs index dabaed9a9..d090cd589 100644 --- a/nodedb/src/control/server/pgwire/handler/submit.rs +++ b/nodedb/src/control/server/pgwire/handler/submit.rs @@ -36,7 +36,9 @@ impl NodeDbPgHandler { user_id: Option>, durability: WalDurability, ) -> crate::Result { - let task = checked.into_authorized().into_physical_task(); + // The write lease lives until the submit returns its outcome. + let (authorized, _lease) = checked.into_parts(); + let task = authorized.into_physical_task(); self.submit_to_data_plane(SubmitArgs { tenant_id: task.tenant_id, vshard_id: task.vshard_id, @@ -75,13 +77,11 @@ impl NodeDbPgHandler { user_id, durability, ordering: WriteOrdering::Gate, - // This node both handles and applies the write: `dispatch_local` - // is reached only when no Raft proposer exists (single node) or - // when the plan is not encodable as a replicated entry, so - // exactly one node applies it and exactly one event is emitted. - // The replicated path publishes at its own origin site instead - // — see [`ChangeFeedOwner`]. - change_feed: ChangeFeedOwner::Funnel, + // This node alone applies the write: `dispatch_local` is reached + // only for a staged write or a plan with no replicated entry. The + // replicated path stages its events under its entry instead — + // see [`ChangeFeedOwner`]. + change_feed: ChangeFeedOwner::LocalApply, }, ) .await diff --git a/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs b/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs index 7a937d568..266dd9d04 100644 --- a/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs +++ b/nodedb/src/control/server/pgwire/handler/transaction_cmds/commit.rs @@ -4,8 +4,8 @@ //! //! Thin pgwire shim over the protocol-neutral commit orchestrator in //! `control/server/shared/session/commit.rs`: builds the pgwire Data-Plane -//! dispatch seam (keeping the materialize-freeze gate), drives `run_commit`, -//! and shapes the neutral [`CommitOutcome`] into a pgwire tag or error. +//! dispatch seam, drives `run_commit`, and shapes the neutral +//! [`CommitOutcome`] into a pgwire tag or error. use std::future::Future; use std::pin::Pin; @@ -27,9 +27,7 @@ use super::errors::calvin_cancelled_error; /// pgwire Data-Plane dispatch seam for the neutral transaction orchestrator. /// -/// Wraps `dispatch_task_no_wal`, preserving its materialize-freeze gate so a -/// transaction that began before a clone freeze cannot COMMIT writes during the -/// freeze window. +/// Wraps `dispatch_task_no_wal`. pub(in crate::control::server::pgwire::handler) struct PgwireTxnDp<'a> { pub(in crate::control::server::pgwire::handler) handler: &'a NodeDbPgHandler, } diff --git a/nodedb/src/control/server/pgwire/handler/transaction_savepoint.rs b/nodedb/src/control/server/pgwire/handler/transaction_savepoint.rs index 22aee3455..6cdbbdf22 100644 --- a/nodedb/src/control/server/pgwire/handler/transaction_savepoint.rs +++ b/nodedb/src/control/server/pgwire/handler/transaction_savepoint.rs @@ -51,11 +51,11 @@ impl NodeDbPgHandler { let cmd = match savepoint_ops::parse_deferred_offset(sql_trimmed, upper) { Ok(Some(cmd)) => cmd, Ok(None) => return None, - Err(message) => { + Err(error) => { return Some(Err(PgWireError::UserError(Box::new(ErrorInfo::new( "ERROR".to_owned(), "42601".to_owned(), - message, + error.to_string(), ))))); } }; diff --git a/nodedb/src/control/server/pgwire/types/error_map.rs b/nodedb/src/control/server/pgwire/types/error_map.rs index b29a8d3b6..80c76eff0 100644 --- a/nodedb/src/control/server/pgwire/types/error_map.rs +++ b/nodedb/src/control/server/pgwire/types/error_map.rs @@ -190,6 +190,11 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::AuthorizationStateBehind { .. } => { ("ERROR", sqlstate::STALE_READ_NOT_LEADER, err.to_string()) } + // Nothing was read, and a retry succeeds once a leader confirms a read + // index this node has applied. + crate::Error::LinearizableReadRefused { .. } => { + ("ERROR", sqlstate::STALE_READ_NOT_LEADER, err.to_string()) + } // Nothing was applied, and a retry succeeds once the group's majority // is reachable again. crate::Error::GroupQuorumUnavailable { .. } => { @@ -200,11 +205,16 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::GroupMarksUnavailable { .. } => { ("ERROR", sqlstate::LOCK_NOT_AVAILABLE, err.to_string()) } + // Nothing was captured, and a retry succeeds once the group's + // leadership settles. + crate::Error::BackupCaptureMoved { .. } => { + ("ERROR", sqlstate::LOCK_NOT_AVAILABLE, err.to_string()) + } crate::Error::ConflictRetry { .. } => { ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) } // A cross-shard Calvin OCC abort is a serialization failure — the client - // should retry the whole transaction. + // must retry the whole transaction. crate::Error::CalvinSerializationConflict => { ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) } @@ -214,9 +224,6 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::CalvinParticipantError => { ("ERROR", sqlstate::TRANSACTION_ROLLBACK, err.to_string()) } - crate::Error::SourceFrozen { .. } => { - ("ERROR", sqlstate::SERIALIZATION_FAILURE, err.to_string()) - } // A descriptor changed under the statement and the server's own // retries ran out. The client retries the statement, so it takes // SERIALIZATION_FAILURE (40001), the SQLSTATE drivers retry on. @@ -248,9 +255,6 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str } crate::Error::MemoryExhausted { .. } => ("ERROR", sqlstate::OUT_OF_MEMORY, err.to_string()), crate::Error::Backpressure { .. } => ("ERROR", sqlstate::OUT_OF_MEMORY, err.to_string()), - crate::Error::FanOutExceeded { .. } => { - ("ERROR", sqlstate::STATEMENT_TOO_COMPLEX, err.to_string()) - } // A cross-collection write refused because source and target are not // co-resident is a not-yet-supported operation, NOT a transient/internal // fault — surface FEATURE_NOT_SUPPORTED (0A000) so clients do not retry. @@ -271,7 +275,7 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str ), // OLLP retry exhaustion is retryable only when it exhausted on real // drift. A pre-admission cause keeps ITS OWN sqlstate — telling a client - // to retry a deterministic failure just burns another round trip — and a + // to retry a deterministic failure burns another round trip — and a // refused admission gate is transient like any other load rejection. crate::Error::OllpExhausted { cause, .. } => match cause { OllpExhaustedCause::PredicateDrift => { @@ -295,7 +299,7 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str ("ERROR", sqlstate::BALANCE_VIOLATION, err.to_string()) } // A Data Plane verdict that travelled back as a typed code keeps the - // SQLSTATE it would have had on the direct dispatch path. + // SQLSTATE it has on the direct dispatch path. crate::Error::DataPlane(code) => { crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate(code) } @@ -374,6 +378,15 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str crate::Error::StaleReadNotLeader { .. } => { ("ERROR", sqlstate::STALE_READ_NOT_LEADER, err.to_string()) } + // The write committed and its result is gone, or its outcome is + // unknown. The code base has no class for either, and the standard + // ones (`08007`, `40003`) sit in classes drivers and pools retry. + // Class `XX` is never treated as transient, so no client re-proposes + // a write that can have committed. + crate::Error::CommittedResultUnavailable { .. } + | crate::Error::ProposalOutcomeUnknown { .. } => { + ("ERROR", sqlstate::INTERNAL_ERROR, err.to_string()) + } // Server-side faults and system defects. The client can act on none // of them, and their public codes are internal classes. crate::Error::MaterializedSumResolutionMissing { .. } @@ -392,10 +405,13 @@ pub fn error_to_sqlstate(err: &crate::Error) -> (&'static str, &'static str, Str | crate::Error::Encryption { .. } | crate::Error::Bridge { .. } | crate::Error::VersionCompat { .. } + | crate::Error::RestoreTargetNotEmpty { .. } + | crate::Error::RestoreVerificationFailed { .. } | crate::Error::Internal { .. } | crate::Error::DescriptorVersionAnomaly { .. } | crate::Error::CatalogIntegrityViolation { .. } | crate::Error::CollectionPurgeRowMissing { .. } + | crate::Error::CollectionUnstamped { .. } | crate::Error::CascadeCycle { .. } => ("ERROR", sqlstate::INTERNAL_ERROR, err.to_string()), } } diff --git a/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs b/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs index 112c6fe9a..631b3dc51 100644 --- a/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs +++ b/nodedb/src/control/server/pgwire/types/numeric_sqlstate.rs @@ -17,8 +17,8 @@ pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> // Mirrors the `RejectedConstraint` arm. Ec::CONSTRAINT_VIOLATION => sqlstate::UNIQUE_VIOLATION, // Mirrors the `ConflictRetry` / `CalvinSerializationConflict` / - // `SourceFrozen` / `RetryableSchemaChanged` arms, and `OllpExhausted` - // when it exhausted on drift. + // `RetryableSchemaChanged` arms, and `OllpExhausted` when it exhausted + // on drift. Ec::WRITE_CONFLICT => sqlstate::SERIALIZATION_FAILURE, // Mirrors the `CalvinParticipantError` arm. Ec::TRANSACTION_ROLLBACK => sqlstate::TRANSACTION_ROLLBACK, @@ -51,8 +51,6 @@ pub(crate) fn numeric_code_to_sqlstate(code: nodedb_types::error::ErrorCode) -> Ec::DATA_EXCEPTION => sqlstate::DATA_EXCEPTION, // Mirrors the `InvalidLimitValue` arm. Ec::INVALID_LIMIT_VALUE => sqlstate::INVALID_LIMIT_VALUE, - // Mirrors the `FanOutExceeded` arm. - Ec::FAN_OUT_EXCEEDED => sqlstate::STATEMENT_TOO_COMPLEX, // Mirrors the `RejectedAuthz` arm. Ec::AUTHORIZATION_DENIED => sqlstate::INSUFFICIENT_PRIVILEGE, // Mirrors the `SessionTokenExpired` arm. diff --git a/nodedb/src/control/server/resp/codec.rs b/nodedb/src/control/server/resp/codec.rs index 97a1594eb..27e4e8276 100644 --- a/nodedb/src/control/server/resp/codec.rs +++ b/nodedb/src/control/server/resp/codec.rs @@ -65,8 +65,8 @@ impl RespValue { /// client as the code it actually is (`NOPERM` for an authorization or /// policy denial, `MOVED`, `TIMEOUT`, `BUSY`, …) instead of a generic `ERR` /// that reads like a server fault. `Error::Bridge` already carries a - /// RESP-shaped detail (the gateway mapping, or the local `BUSY` retry - /// hint), so it is passed through rather than mapped twice. + /// RESP-shaped detail (the gateway mapping), so it is passed through + /// rather than mapped twice. pub fn from_error(error: &crate::Error) -> Self { match error { crate::Error::Bridge { detail } => Self::Error(detail.clone()), diff --git a/nodedb/src/control/server/resp/gateway_dispatch.rs b/nodedb/src/control/server/resp/gateway_dispatch.rs index bea9b39b5..a77515cae 100644 --- a/nodedb/src/control/server/resp/gateway_dispatch.rs +++ b/nodedb/src/control/server/resp/gateway_dispatch.rs @@ -2,9 +2,8 @@ //! RESP gateway dispatch helpers. //! -//! Routes KV operations through `Gateway::execute_response` when the gateway is -//! available (cluster-aware routing), falling back to direct local SPSC -//! dispatch on single-node boot. +//! Routes KV operations through `Gateway::execute_response`, which sends each +//! one to the node that owns its vShard. //! //! All helpers return `crate::Result` so the existing sub-handler //! code (`handler_kv`, `handler_hash`, `handler_sorted`) is unchanged. @@ -16,7 +15,6 @@ use crate::control::gateway::GatewayErrorMap; use crate::control::gateway::core::QueryContext; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::request_scope::ClientRequestScope; -use crate::control::server::dispatch_utils; use crate::control::server::shared::clone_write::CloneCheckedOutcome; use crate::control::server::shared::metering::{PlanMeteringInfo, meter_dispatch}; use crate::control::server::shared::quota_admission::admit_quota_for_dispatch; @@ -26,13 +24,10 @@ use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; use super::session::RespSession; -/// Dispatch a read-only KV operation. +/// Dispatch a read-only KV operation through the gateway. /// -/// Routes through the gateway when available (cluster-aware routing), falling -/// back to direct local SPSC dispatch on single-node boot. -/// -/// Bridge/dispatch errors are mapped to `Error::Bridge` with a `BUSY` detail -/// so the RESP handler can return `-BUSY` to the Redis client. +/// A full dispatch queue reaches the Redis client as `-BUSY` +/// (`GatewayErrorMap::to_resp`), which Redis clients retry. pub(super) async fn dispatch_kv( state: &SharedState, session: &RespSession, @@ -66,24 +61,7 @@ pub(super) async fn dispatch_kv( CloneCheckedOutcome::Handled(resp) => return Ok(resp), CloneCheckedOutcome::Proceed(checked) => checked, }; - let result = match state.gateway.get() { - Some(gw) => { - let gw_ctx = QueryContext { - tenant_id: session.tenant_id, - trace_id: TraceId::generate(), - database_id: checked.database_id(), - txn_id: None, - }; - gw.execute_response(&gw_ctx, checked) - .await - .map_err(|e| crate::Error::Bridge { - detail: GatewayErrorMap::to_resp(&e), - }) - } - None => dispatch_utils::dispatch_authorized_to_data_plane(state, checked, TraceId::ZERO) - .await - .map_err(map_busy_error), - }; + let result = execute_through_gateway(state, session, checked).await; if result.is_ok() && let Some(info) = &plan_metering_info { @@ -92,13 +70,10 @@ pub(super) async fn dispatch_kv( result } -/// Dispatch a KV write operation through the gateway or the local Data Plane. +/// Dispatch a KV write operation through the gateway. /// -/// Routes through the gateway when available (cluster-aware routing) — where the -/// gateway owns WAL durability on the target node — falling back to direct local -/// SPSC dispatch on single-node boot. On the local path the WAL append is -/// performed inside the dispatch core, under the write-admission guard and just -/// before the enqueue, so LSN order matches apply order. +/// The gateway routes the write to the node that owns its vShard and owns WAL +/// durability there. pub(super) async fn dispatch_kv_write( state: &SharedState, session: &RespSession, @@ -127,24 +102,7 @@ pub(super) async fn dispatch_kv_write( CloneCheckedOutcome::Handled(resp) => return Ok(resp), CloneCheckedOutcome::Proceed(checked) => checked, }; - let result = match state.gateway.get() { - Some(gw) => { - let gw_ctx = QueryContext { - tenant_id: session.tenant_id, - trace_id: TraceId::generate(), - database_id: checked.database_id(), - txn_id: None, - }; - gw.execute_response(&gw_ctx, checked) - .await - .map_err(|e| crate::Error::Bridge { - detail: GatewayErrorMap::to_resp(&e), - }) - } - None => dispatch_utils::dispatch_authorized_durable_write(state, checked, TraceId::ZERO) - .await - .map_err(map_busy_error), - }; + let result = execute_through_gateway(state, session, checked).await; if result.is_ok() && let Some(info) = &plan_metering_info { @@ -153,6 +111,31 @@ pub(super) async fn dispatch_kv_write( result } +/// Run one checked RESP task through the gateway. +/// +/// A gateway error reaches the client as `Error::Bridge` carrying its RESP +/// rendering. +async fn execute_through_gateway( + state: &SharedState, + session: &RespSession, + checked: crate::control::server::shared::clone_write::CloneCheckedTask, +) -> crate::Result { + let gateway = state.installed_gateway()?; + let gw_ctx = QueryContext { + tenant_id: session.tenant_id, + trace_id: TraceId::generate(), + database_id: checked.database_id(), + txn_id: None, + linearizable: true, + }; + gateway + .execute_response(&gw_ctx, checked) + .await + .map_err(|e| crate::Error::Bridge { + detail: GatewayErrorMap::to_resp(&e), + }) +} + /// Refuse the command when a covering scope's hard quota is already spent. /// /// The sibling of [`meter_resp_dispatch`], run before dispatch rather than @@ -262,7 +245,7 @@ fn authorize_resp_task( // Reads whose results column redaction cannot rewrite (an aggregate over a // redacted column, a graph traversal) are refused on the same seam, so the - // capability is never minted for a plan that would leak them. + // capability is never minted for a plan that will leak them. crate::control::planner::redaction_refusal::refuse_unredactable_plan( &plan, session.tenant_id, @@ -339,30 +322,13 @@ fn resp_auth_scope<'a, 'p>( ClientRequestScope::for_database(identity, stores, database_id, peer_addr) } -/// Map bridge/dispatch errors to a BUSY error for Redis client compatibility. -/// -/// When the SPSC ring buffer is full or the Data Plane core is overloaded, -/// the Redis client receives `-BUSY NodeDB is processing requests, retry later` -/// which Redis clients handle with automatic retry (same as Redis Cluster BUSY). -fn map_busy_error(e: crate::Error) -> crate::Error { - match &e { - crate::Error::Bridge { .. } - | crate::Error::Dispatch { .. } - | crate::Error::DispatchCapacity { .. } => crate::Error::Bridge { - detail: "BUSY NodeDB is processing requests, retry later".into(), - }, - _ => e, - } -} - #[cfg(test)] mod tests { - use crate::bridge::envelope::{Payload, Status}; use crate::control::security::identity::{AuthMethod, DatabaseSet, Role}; use crate::control::security::metering::quota::QuotaManager; use crate::control::security::request_scope::AuthStores; use crate::control::security::scope::grant::ScopeGrantStore; - use crate::types::{Lsn, TenantId}; + use crate::types::TenantId; use super::*; @@ -472,73 +438,13 @@ mod tests { ); } - /// Returns state plus the fake Data-Plane data-side and the backing - /// `TempDir` guard — the caller must keep the guard alive for as long as - /// `state` is in use. - fn metering_fixture() -> ( - Arc, - crate::bridge::dispatch::CoreChannelDataSide, - tempfile::TempDir, - ) { - use crate::bridge::dispatch::Dispatcher; - use crate::wal::WalManager; - - let dir = tempfile::tempdir().expect("create test directory"); - let wal = Arc::new( - WalManager::open_for_testing(&dir.path().join("test.wal")).expect("open test WAL"), - ); - let (dispatcher, mut sides) = Dispatcher::new(1, 64); - let side = sides.pop().expect("one data side"); - let state = SharedState::new(dispatcher, wal).expect("construct shared state"); - (state, side, dir) - } - - /// `metering_config` has no live-mutation path by design — reach in via - /// `Arc::get_mut` while the test is still the sole owner of the freshly - /// constructed state, before any clone escapes into a spawned responder - /// task. Same pattern as `metering::tests::enable_metering`. - fn enable_metering(state: &mut Arc) { - Arc::get_mut(state) - .expect("sole owner in test") - .metering_config - .enabled = true; - } - - /// Fake Data-Plane responder: pops the one dispatched request off `side` - /// and answers it `Ok` with an empty payload, so `dispatch_kv` / - /// `dispatch_kv_write` complete their round trip without a real Data-Plane - /// core (mirrors the responder in `dispatch_utils::dispatch::tests`). - async fn respond_ok_once( - mut side: crate::bridge::dispatch::CoreChannelDataSide, - state: Arc, - ) { - let deadline = std::time::Instant::now() + std::time::Duration::from_secs(5); - let mut handled = false; - while !handled && std::time::Instant::now() < deadline { - if let Ok(request) = side.request_rx.try_pop() { - side.response_tx - .try_push(crate::bridge::dispatch::BridgeResponse { - inner: Response { - request_id: request.inner.request_id, - status: Status::Ok, - attempt: 1, - partial: false, - payload: Payload::empty(), - watermark_lsn: Lsn::new(0), - error_code: None, - read_set_valid: None, - read_version_lsn: Lsn::ZERO, - write_set: Vec::new(), - }, - }) - .expect("fake data-plane response queue has capacity"); - handled = true; - } - state.poll_and_route_responses(); - tokio::task::yield_now().await; - } - assert!(handled, "fake data plane received the dispatched request"); - state.poll_and_route_responses(); + /// A booted one-node cluster with metering on. `metering_config` has no + /// live-mutation path, so it is set before the state is shared. + async fn metered_cluster() -> crate::control::cluster::test_one_node::OneNodeCluster { + crate::control::cluster::test_one_node::boot_with(|state| { + state.metering_config.enabled = true; + }) + .await } fn resp_session_with_identity(identity: AuthenticatedIdentity) -> RespSession { @@ -574,19 +480,16 @@ mod tests { /// A successful RESP KV dispatch records exactly one usage event, /// attributed to the RESP session's selected collection. - #[tokio::test] + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn successful_kv_dispatch_records_one_event_for_session_collection() { - let (mut state, side, _dir) = metering_fixture(); - enable_metering(&mut state); + let cluster = metered_cluster().await; let session = resp_session_with_identity(regular_identity(1)); let plan = kv_get_plan(&session.collection); - let responder = tokio::spawn(respond_ok_once(side, Arc::clone(&state))); - let result = dispatch_kv(&state, &session, plan).await; - responder.await.expect("responder completes"); + let result = dispatch_kv(&cluster.state, &session, plan).await; - assert!(result.is_ok(), "RESP KV dispatch must succeed"); - let events = state.usage_counter.drain(); + assert!(result.is_ok(), "RESP KV dispatch must succeed: {result:?}"); + let events = cluster.state.usage_counter.drain(); assert_eq!( events.len(), 1, @@ -594,15 +497,16 @@ mod tests { ); assert_eq!(events[0].collection, "widgets"); assert_eq!(events[0].engine, "kv"); + cluster.shutdown().await; } /// A denied RESP dispatch — rejected before reaching the Data Plane — /// performed no billable work and must record nothing. - #[tokio::test] + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn denied_kv_dispatch_records_nothing() { - let (mut state, _side, _dir) = metering_fixture(); - enable_metering(&mut state); - state + let cluster = metered_cluster().await; + cluster + .state .blacklist .blacklist_ip("10.0.0.0/8", "test ip ban", "admin", 0) .expect("blacklist CIDR range"); @@ -610,29 +514,32 @@ mod tests { session.peer_addr = "10.1.2.3:54321".into(); let plan = kv_get_plan(&session.collection); - let result = dispatch_kv(&state, &session, plan).await; + let result = dispatch_kv(&cluster.state, &session, plan).await; assert!(result.is_err(), "a blacklisted peer must be denied"); - assert_eq!(state.usage_counter.total_tokens(), 0); + assert_eq!(cluster.state.usage_counter.total_tokens(), 0); + cluster.shutdown().await; } /// Metering disabled (the default) records nothing on a successful RESP /// dispatch — proves this change is inert for the existing RESP suite. - #[tokio::test] + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn metering_disabled_by_default_records_nothing_on_resp_success() { - let (state, side, _dir) = metering_fixture(); - assert!(!state.metering_config.enabled, "default config is disabled"); + let cluster = crate::control::cluster::test_one_node::boot().await; + assert!( + !cluster.state.metering_config.enabled, + "default config is disabled" + ); let session = resp_session_with_identity(regular_identity(3)); let plan = kv_get_plan(&session.collection); - let responder = tokio::spawn(respond_ok_once(side, Arc::clone(&state))); - let result = dispatch_kv(&state, &session, plan).await; - responder.await.expect("responder completes"); + let result = dispatch_kv(&cluster.state, &session, plan).await; assert!( result.is_ok(), - "dispatch must still succeed with metering disabled" + "dispatch must still succeed with metering disabled: {result:?}" ); - assert_eq!(state.usage_counter.total_tokens(), 0); + assert_eq!(cluster.state.usage_counter.total_tokens(), 0); + cluster.shutdown().await; } } diff --git a/nodedb/src/control/server/resp/handler_hash.rs b/nodedb/src/control/server/resp/handler_hash.rs index 0c4dce8ca..9e2bde6f4 100644 --- a/nodedb/src/control/server/resp/handler_hash.rs +++ b/nodedb/src/control/server/resp/handler_hash.rs @@ -129,14 +129,18 @@ pub(super) async fn handle_hset( // Content-addressed cross-engine identity so the merged row keeps the // surrogate its original insert assigned. - let surrogate = match state.surrogate_assigner.assign( + let surrogate = match crate::control::server::surrogate_exchange::assign_surrogate_routed( + state, nodedb_types::CollectionKey::from_bare( nodedb_types::DatabaseId::DEFAULT, &session.collection, ), session.tenant_id, &key, - ) { + crate::types::TraceId::ZERO, + ) + .await + { Ok(s) => s, Err(e) => return RespValue::from_error(&e), }; diff --git a/nodedb/src/control/server/resp/handler_kv/batch.rs b/nodedb/src/control/server/resp/handler_kv/batch.rs index b728a55cc..c137e4291 100644 --- a/nodedb/src/control/server/resp/handler_kv/batch.rs +++ b/nodedb/src/control/server/resp/handler_kv/batch.rs @@ -13,7 +13,7 @@ use super::super::handler::{dispatch_kv, dispatch_kv_write}; use super::super::payload::payload_json; use super::super::redaction::resp_redaction; use super::super::session::RespSession; -use super::surrogate::resp_kv_surrogate; +use super::surrogate::resp_kv_surrogates; use crate::control::server::response_shape::redaction::redact_stored_value_bytes; pub(in crate::control::server::resp) async fn handle_mget( @@ -92,15 +92,12 @@ pub(in crate::control::server::resp) async fn handle_mset( .map(|pair| (pair[0].clone(), pair[1].clone())) .collect(); - // Assign each entry's stable cross-engine surrogate the same way SET - // (`handle_set`) does per key -- otherwise MSET rows would land with - // `Surrogate::ZERO` and be invisible to any surrogate-keyed cross-engine - // read/join. - let surrogates = match entries - .iter() - .map(|(key, _value)| resp_kv_surrogate(state, session, key)) - .collect::, RespValue>>() - { + // Every entry's stable cross-engine surrogate, resolved in one batch the + // way SET (`handle_set`) resolves one key. A row left at + // `Surrogate::ZERO` is invisible to any surrogate-keyed cross-engine read + // or join. + let keys: Vec<&[u8]> = entries.iter().map(|(key, _value)| key.as_slice()).collect(); + let surrogates = match resp_kv_surrogates(state, session, &keys).await { Ok(s) => s, Err(e) => return e, }; diff --git a/nodedb/src/control/server/resp/handler_kv/counters.rs b/nodedb/src/control/server/resp/handler_kv/counters.rs index d4712460b..8a95ee14e 100644 --- a/nodedb/src/control/server/resp/handler_kv/counters.rs +++ b/nodedb/src/control/server/resp/handler_kv/counters.rs @@ -20,10 +20,10 @@ use super::surrogate::resp_kv_surrogate; /// `INCR` / `DECR` / `INCRBY` / `DECRBY` / `INCRBYFLOAT` answer with the row's /// new stored value, which the KV engine holds as the single-value form — the /// column every SQL-side read of that row calls `value`. Masking the answer -/// would report a number the key does not hold, so the command is refused +/// will report a number the key does not hold, so the command is refused /// instead, on the same fail-closed principle the planner applies to an /// aggregate over a redacted column. The refusal happens BEFORE dispatch, so -/// the increment the caller could not observe is never performed either. +/// the increment the caller cannot observe is never performed either. fn refuse_if_counter_is_redacted(state: &SharedState, session: &RespSession) -> Option { let redaction = resp_redaction(state, session)?; redaction @@ -93,7 +93,7 @@ async fn dispatch_incr( if let Some(refusal) = refuse_if_counter_is_redacted(state, session) { return refusal; } - let surrogate = match resp_kv_surrogate(state, session, &key) { + let surrogate = match resp_kv_surrogate(state, session, &key).await { Ok(s) => s, Err(e) => return e, }; @@ -114,7 +114,7 @@ async fn dispatch_incr( Ok(resp) => match payload_field_i64(&resp.payload, "value") { Some(new_val) => RespValue::integer(new_val), // The counter did change; a response we cannot read means we do - // not know its new value, and echoing 0 would report a value the + // not know its new value, and echoing 0 will report a value the // key does not hold. None => RespValue::err("ERR counter response could not be decoded"), }, @@ -144,7 +144,7 @@ pub(in crate::control::server::resp) async fn handle_incrbyfloat( return refusal; } - let surrogate = match resp_kv_surrogate(state, session, &key) { + let surrogate = match resp_kv_surrogate(state, session, &key).await { Ok(s) => s, Err(e) => return e, }; diff --git a/nodedb/src/control/server/resp/handler_kv/strings.rs b/nodedb/src/control/server/resp/handler_kv/strings.rs index 44f9290d9..826e2ae06 100644 --- a/nodedb/src/control/server/resp/handler_kv/strings.rs +++ b/nodedb/src/control/server/resp/handler_kv/strings.rs @@ -128,7 +128,7 @@ pub(in crate::control::server::resp) async fn handle_set( } } - let surrogate = match resp_kv_surrogate(state, session, &key) { + let surrogate = match resp_kv_surrogate(state, session, &key).await { Ok(s) => s, Err(e) => return e, }; @@ -207,7 +207,7 @@ pub(in crate::control::server::resp) async fn handle_exists( match dispatch_kv(state, session, plan).await { Ok(resp) if resp.status == Status::Ok && !resp.payload.is_empty() => count += 1, Ok(_) => {} - // A policy refusal is not an absent key: reporting it as one would + // A policy refusal is not an absent key: reporting it as one will // let EXISTS answer a question the caller is not allowed to ask. Err(e) => return RespValue::from_error(&e), } @@ -230,10 +230,10 @@ pub(in crate::control::server::resp) async fn handle_getset( let new_value = cmd.args[1].clone(); // Resolved once for this command: GETSET returns the row's PREVIOUS stored - // value, which carries exactly the disclosure a GET of it would. + // value, which carries exactly the disclosure a GET of it does. let redaction = resp_redaction(state, session); - let surrogate = match resp_kv_surrogate(state, session, &key) { + let surrogate = match resp_kv_surrogate(state, session, &key).await { Ok(s) => s, Err(e) => return e, }; diff --git a/nodedb/src/control/server/resp/handler_kv/surrogate.rs b/nodedb/src/control/server/resp/handler_kv/surrogate.rs index 7fa6544d1..f7d882c33 100644 --- a/nodedb/src/control/server/resp/handler_kv/surrogate.rs +++ b/nodedb/src/control/server/resp/handler_kv/surrogate.rs @@ -2,29 +2,46 @@ //! Surrogate assignment shared by the KV RESP handlers. +use crate::control::server::surrogate_exchange::assign_surrogates_routed; use crate::control::state::SharedState; +use crate::types::TraceId; use super::super::codec::RespValue; use super::super::session::RespSession; -/// Resolve the stable cross-engine surrogate for a KV atomic op on this -/// session's collection, content-addressed on `(collection, key)` — the same -/// binding a normal insert of that key allocated, so an atomic op on an -/// existing key keeps its identity. -pub(super) fn resp_kv_surrogate( +/// The stable cross-engine surrogates of `keys` in this session's +/// collection, in `keys` order, content-addressed on `(collection, key)`: the +/// binding a normal insert of each key made. An absent key is bound at the +/// collection's home. All keys resolve in one batch through the async routed +/// exchange. +pub(super) async fn resp_kv_surrogates( + state: &SharedState, + session: &RespSession, + keys: &[&[u8]], +) -> Result, RespValue> { + assign_surrogates_routed( + state, + nodedb_types::CollectionKey::from_bare( + crate::types::DatabaseId::DEFAULT, + &session.collection, + ), + session.tenant_id, + keys, + TraceId::ZERO, + ) + .await + .map_err(|e| RespValue::err(format!("ERR {e}"))) +} + +/// [`resp_kv_surrogates`] for one key. A KV atomic op on an existing key keeps +/// that key's identity. +pub(super) async fn resp_kv_surrogate( state: &SharedState, session: &RespSession, key: &[u8], ) -> Result { - state - .surrogate_assigner - .assign( - nodedb_types::CollectionKey::from_bare( - crate::types::DatabaseId::DEFAULT, - &session.collection, - ), - session.tenant_id, - key, - ) - .map_err(|e| RespValue::err(format!("ERR {e}"))) + resp_kv_surrogates(state, session, &[key]) + .await? + .pop() + .ok_or_else(|| RespValue::err("ERR the key's home returned no surrogate")) } diff --git a/nodedb/src/control/server/resp/handler_pubsub.rs b/nodedb/src/control/server/resp/handler_pubsub.rs index 85d43b73d..2e794d058 100644 --- a/nodedb/src/control/server/resp/handler_pubsub.rs +++ b/nodedb/src/control/server/resp/handler_pubsub.rs @@ -73,7 +73,8 @@ pub async fn handle_subscribe( // Exact subscriptions use only their topic buses, so unrelated topic, // tenant, and database traffic cannot create lag or be observed here. - let mut subscriptions: Vec<_> = channels + // A topic dropped after the check above has no bus to subscribe to. + let Some(mut subscriptions) = channels .iter() .map(|channel| { state @@ -81,7 +82,13 @@ pub async fn handle_subscribe( .subscribe(DatabaseId::DEFAULT, tenant_id, channel) }) .collect::>>() - .expect("topics were validated before subscribing"); + else { + return write_rejection( + stream, + RespValue::err("ERR a subscribed topic was dropped; retry the SUBSCRIBE"), + ) + .await; + }; for (i, channel) in channels.iter().enumerate() { let confirm = RespValue::array(vec![ @@ -258,25 +265,10 @@ pub async fn handle_publish( // Durable topics do not track live RESP subscribers. `1` means the // Event Plane accepted the message into the topic's durable buffer. Ok(_) => RespValue::integer(1), - Err(PublishError::RemoteHome { leader_node, .. }) => { - match crate::event::topic::publish::publish_remote( - state, - DatabaseId::DEFAULT, - identity.tenant_id.as_u64(), - &channel, - message, - leader_node, - ) - .await - { - Ok(_) => RespValue::integer(1), - Err(error) => RespValue::err(format!("ERR publish failed: {error}")), - } - } Err(PublishError::TopicNotFound(topic)) => { RespValue::err(format!("ERR no such topic '{topic}'")) } - Err(PublishError::Persistence(error)) | Err(PublishError::RemoteError(error)) => { + Err(PublishError::Persistence(error)) => { RespValue::err(format!("ERR publish failed: {error}")) } } @@ -391,15 +383,12 @@ async fn write_value(stream: &mut ConnStream, response: RespValue) -> crate::Res #[cfg(test)] mod tests { - use std::sync::Arc; - use crate::control::security::identity::{AuthMethod, DatabaseSet}; use crate::control::security::permission::{PermissionStore, collection_target}; use crate::control::security::role::RoleStore; use crate::event::cdc::stream_def::RetentionConfig; use crate::event::topic::TopicDef; use crate::types::TenantId; - use crate::wal::WalManager; use super::*; @@ -448,6 +437,7 @@ mod tests { sequence: 7, event_time: 1, lsn: 7, + epoch: 0, payload: "{\"id\":7,\"unchanged\":true}".into(), }; let channels = ["orders.created".to_string()]; @@ -498,14 +488,11 @@ mod tests { assert_eq!(registry.receiver_count(), before); } - #[tokio::test(flavor = "current_thread")] + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn publish_authorizes_then_appends_to_durable_topic_buffer() { - let directory = tempfile::tempdir().expect("temporary test directory"); - let wal_directory = directory.path().join("wal"); - std::fs::create_dir_all(&wal_directory).expect("create WAL directory"); - let wal = Arc::new(WalManager::open_for_testing(&wal_directory).expect("test WAL")); - let (dispatcher, _) = crate::bridge::dispatch::Dispatcher::new(1, 16); - let state = SharedState::new(dispatcher, wal).expect("test shared state"); + // A publication proposes through the topic's data group. + let cluster = crate::control::cluster::test_one_node::boot().await; + let state = &cluster.state; let identity = identity(); let retention = RetentionConfig { max_events: 10, @@ -520,6 +507,8 @@ mod tests { created_at: 0, last_sequence: 0, last_lsn: 0, + last_epoch: 0, + modification_hlc: nodedb_types::Hlc::ZERO, }; state .credentials @@ -543,7 +532,7 @@ mod tests { }; assert_eq!( - handle_publish(&command, &session, &state).await, + handle_publish(&command, &session, state).await, RespValue::err("NOPERM this user has no permissions to access this channel") ); assert_eq!(buffer.total_pushed(), 0); @@ -551,7 +540,7 @@ mod tests { state .permissions .grant( - &collection_target(identity.tenant_id, "topic:orders"), + &collection_target(DatabaseId::DEFAULT, identity.tenant_id, "topic:orders"), "user:alice", Permission::Write, "test", @@ -559,7 +548,7 @@ mod tests { ) .expect("in-memory grant"); assert_eq!( - handle_publish(&command, &session, &state).await, + handle_publish(&command, &session, state).await, RespValue::integer(1) ); assert_eq!(buffer.total_pushed(), 1); @@ -567,7 +556,7 @@ mod tests { state .permissions .grant( - &collection_target(identity.tenant_id, "topic:missing"), + &collection_target(DatabaseId::DEFAULT, identity.tenant_id, "topic:missing"), "user:alice", Permission::Write, "test", @@ -579,8 +568,9 @@ mod tests { args: vec![b"missing".to_vec(), b"accepted".to_vec()], }; assert_eq!( - handle_publish(&missing_topic, &session, &state).await, + handle_publish(&missing_topic, &session, state).await, RespValue::err("ERR no such topic 'missing'") ); + cluster.shutdown().await; } } diff --git a/nodedb/src/control/server/resp/handler_sorted.rs b/nodedb/src/control/server/resp/handler_sorted.rs index ce2141017..8beb63619 100644 --- a/nodedb/src/control/server/resp/handler_sorted.rs +++ b/nodedb/src/control/server/resp/handler_sorted.rs @@ -41,8 +41,10 @@ pub(super) async fn handle_zadd( // In RESP mode, the sorted index name = session.collection. // The args are: score1 member1 [score2 member2 ...] let index_name = session.collection.clone(); - let mut added = 0i64; + // Every pair is parsed before any write, so an invalid score refuses the + // whole command. + let mut members: Vec<(f64, Vec)> = Vec::with_capacity(cmd.argc() / 2); let mut i = 0; while i + 1 < cmd.argc() { let score_str = match cmd.arg_str(i) { @@ -53,8 +55,30 @@ pub(super) async fn handle_zadd( Ok(v) => v, Err(_) => return RespValue::err("ERR value is not a valid float"), }; - let member = cmd.args[i + 1].clone(); + members.push((score, cmd.args[i + 1].clone())); + i += 2; + } + + // Every member's identity, in one batch at the index collection's home. + let keys: Vec<&[u8]> = members + .iter() + .map(|(_, member)| member.as_slice()) + .collect(); + let surrogates = match crate::control::server::surrogate_exchange::assign_surrogates_routed( + state, + nodedb_types::CollectionKey::from_bare(crate::types::DatabaseId::DEFAULT, &index_name), + session.tenant_id, + &keys, + crate::types::TraceId::ZERO, + ) + .await + { + Ok(s) => s, + Err(e) => return RespValue::from_error(&e), + }; + let mut added = 0i64; + for ((score, member), surrogate) in members.into_iter().zip(surrogates) { // Write to the underlying KV collection as a MessagePack document // containing the score and member. The sorted index auto-maintenance // in KvEngine::put will update the order-statistic tree. @@ -64,14 +88,6 @@ pub(super) async fn handle_zadd( })) .unwrap_or_default(); - let surrogate = match state.surrogate_assigner.assign( - nodedb_types::CollectionKey::from_bare(crate::types::DatabaseId::DEFAULT, &index_name), - session.tenant_id, - &member, - ) { - Ok(s) => s, - Err(e) => return RespValue::from_error(&e), - }; let plan = PhysicalPlan::Kv(KvOp::Put { collection: QualifiedCollection::new(crate::types::DatabaseId::DEFAULT, &index_name), key: member, @@ -87,8 +103,6 @@ pub(super) async fn handle_zadd( Ok(_) => added += 1, Err(e) => return RespValue::from_error(&e), } - - i += 2; } RespValue::integer(added) diff --git a/nodedb/src/control/server/resp/listener.rs b/nodedb/src/control/server/resp/listener.rs index 502681f94..93d2baae0 100644 --- a/nodedb/src/control/server/resp/listener.rs +++ b/nodedb/src/control/server/resp/listener.rs @@ -9,6 +9,7 @@ use std::net::SocketAddr; use std::sync::Arc; +use futures::future::BoxFuture; use tokio::io::{AsyncReadExt, AsyncWriteExt}; use tokio::net::TcpListener; use tokio::sync::Semaphore; @@ -175,7 +176,19 @@ impl RespListener { } /// Handle a single RESP connection. -async fn handle_connection( +/// +/// The connection future is boxed, once per connection. Every command path +/// nests inside it, and unboxed it can overflow the compiler's layout depth +/// limit in the listener's connection task. +fn handle_connection( + stream: ConnStream, + peer: SocketAddr, + state: &SharedState, +) -> BoxFuture<'_, crate::Result<()>> { + Box::pin(serve_connection(stream, peer, state)) +} + +async fn serve_connection( mut stream: ConnStream, peer: SocketAddr, state: &SharedState, diff --git a/nodedb/src/control/server/response_shape/calvin_fold.rs b/nodedb/src/control/server/response_shape/calvin_fold.rs index 9cbfd05de..380d34b50 100644 --- a/nodedb/src/control/server/response_shape/calvin_fold.rs +++ b/nodedb/src/control/server/response_shape/calvin_fold.rs @@ -57,7 +57,7 @@ pub enum CalvinTaskOutcome { pub enum CalvinFoldError { /// Two tasks reported verbs that cannot share one tag. Verb(DmlFoldError), - /// A task's payload could not be read as its plan kind requires. + /// A task's payload cannot be read as its plan kind requires. Shape(crate::Error), } @@ -71,7 +71,7 @@ pub struct CalvinBatchFold { /// /// A derived implicit-edge write beside the user's own never answers the /// statement: it folds as opaque, as it never deposits. The rows are taken -/// once rather than accumulated, which would repeat the identical payload per +/// once rather than accumulated, which will repeat the identical payload per /// task. pub fn fold_calvin_batch( plans: &[&PhysicalPlan], @@ -240,7 +240,7 @@ mod tests { key: Vec::new(), value: Vec::new(), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), returning: None, rls_filters: Vec::new(), provenance: None, @@ -272,7 +272,7 @@ mod tests { let point_delete = PhysicalPlan::Document(DocumentOp::PointDelete { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "items"), document_id: "a".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: None, pk_bytes: Vec::new(), returning: None, rls_filters: Vec::new(), diff --git a/nodedb/src/control/server/response_shape/compose/array_slice.rs b/nodedb/src/control/server/response_shape/compose/array_slice.rs index 685ddc7ce..c61bc6482 100644 --- a/nodedb/src/control/server/response_shape/compose/array_slice.rs +++ b/nodedb/src/control/server/response_shape/compose/array_slice.rs @@ -11,7 +11,7 @@ use super::kernel::{empty_shaped, shape_decoded_rows}; /// NOTICE text for an `AS OF SYSTEM TIME` cutoff older than the oldest /// retained tile version. This is the canonical definition, surfaced to /// every protocol via [`ShapedRows::notice`]. -const TRUNCATED_BEFORE_HORIZON_NOTICE: &str = "AS OF SYSTEM TIME cutoff is older than the oldest retained tile version; \ +pub(crate) const TRUNCATED_BEFORE_HORIZON_NOTICE: &str = "AS OF SYSTEM TIME cutoff is older than the oldest retained tile version; \ results may be incomplete"; /// Shape an `ArrayOp::Slice` response: decode the `ArraySliceResponse` diff --git a/nodedb/src/control/server/response_shape/compose/materialized.rs b/nodedb/src/control/server/response_shape/compose/materialized.rs index 9f4c97ed4..ab8b1c10f 100644 --- a/nodedb/src/control/server/response_shape/compose/materialized.rs +++ b/nodedb/src/control/server/response_shape/compose/materialized.rs @@ -4,7 +4,8 @@ //! //! `shape_response_materialized` is the canonical SELECT-read shaping used by //! every protocol entrypoint. It performs the full per-payload shaping order -//! (`apply_kv_wrap` -> `translate_search_response` -> decode -> scan-envelope +//! (`apply_kv_wrap` -> `apply_walk_wrap` -> `translate_search_response` -> +//! decode -> scan-envelope //! unwrap -> optional SELECT-list projection) as a single call, producing an //! already-shaped, already-projected [`ShapeOutcome`]. Every SELECT-read //! producer — pgwire's non-streaming dispatch, native's dispatch loop — calls @@ -32,6 +33,7 @@ use super::super::request::MaterializedShapeRequest; use super::super::returning::shape_returning_rows; use super::super::schema::OutputSchema; use super::super::types::{PlanKind, ShapedRows}; +use super::super::walk::apply_walk_wrap; use super::array_slice::shape_array_slice; use super::kernel::{empty_shaped, shape_decoded_rows, single_result_row}; @@ -76,9 +78,9 @@ pub fn shape_response_materialized( | PlanKind::MultiRow => {} } - // Seam-1 order, exactly as pgwire's `dispatch_task_loop` applies it - // (apply_kv_wrap -> translate_search_response) before any decode/shape step. - let wrapped = apply_kv_wrap(plan, payload); + // The plan-dependent transforms run before any decode or shape step: + // KV wrap, then walk wrap, then search translation. + let wrapped = apply_walk_wrap(plan, &apply_kv_wrap(plan, payload))?; let translated = translate_search_response(&wrapped, plan, state, database_id, tenant_id); let shaped = match plan_kind { diff --git a/nodedb/src/control/server/response_shape/mod.rs b/nodedb/src/control/server/response_shape/mod.rs index fd8b7ca7a..8b361d9b8 100644 --- a/nodedb/src/control/server/response_shape/mod.rs +++ b/nodedb/src/control/server/response_shape/mod.rs @@ -13,6 +13,7 @@ pub mod returning; pub mod schema; pub mod stamp; pub mod types; +pub mod walk; pub use cell::{row_to_wire_json, value_to_wire_json}; pub use redaction::{redact_decoded_value, redact_envelope_row, redact_stored_value_bytes}; diff --git a/nodedb/src/control/server/response_shape/project.rs b/nodedb/src/control/server/response_shape/project.rs index b976cac52..969db72af 100644 --- a/nodedb/src/control/server/response_shape/project.rs +++ b/nodedb/src/control/server/response_shape/project.rs @@ -32,19 +32,24 @@ pub fn push_flat_rows(value: Value, out: &mut Vec) -> crate::Result<( { let mut inner: ShapedRow = inner.into_iter().collect(); // The envelope carries the row's storage key, which is - // internal. A body with no `id` field carries identity - // nowhere else, so the key renders to an identity at this - // boundary. `or_insert` leaves a declared primary key as the + // internal. A body with no `id` field renders its identity at + // this boundary, by the rule the Data Plane's row shaping + // applies: the body's `_rowid`, else the storage key. A copy + // under a new surrogate keeps its `_rowid`, so it keeps its + // identity. `or_insert` leaves a declared primary key as the // authority. if let Some(Value::String(key)) = map.remove("id") { - let identity = crate::engine::document::store::StorageKey::parse(&key) + let storage_key = crate::engine::document::store::StorageKey::parse(&key) .ok_or_else(|| crate::Error::Internal { detail: format!("scan envelope id is not a storage key: '{key}'"), - })? - .to_identity(); + })?; + let identity = match inner.get(nodedb_types::ROWID_COLUMN) { + Some(Value::Integer(rowid)) => rowid.to_string(), + _ => storage_key.to_identity().into_string(), + }; inner .entry("id".to_string()) - .or_insert(Value::String(identity.into_string())); + .or_insert(Value::String(identity)); } out.push(inner); return Ok(()); @@ -99,9 +104,9 @@ pub fn is_scan_wrapper_json(map: &serde_json::Map) -> /// Unique per-column cell keys for a shaped column list. /// -/// SQL output column names may legally repeat — `SELECT w.id, b.id` displays +/// SQL output column names can legally repeat — `SELECT w.id, b.id` displays /// both columns as `id` — but a shaped row is a JSON map and cannot hold two -/// cells under one key: inserting the second cell would overwrite the first, +/// cells under one key: inserting the second cell will overwrite the first, /// making both wire columns render the last value. The shaper therefore /// stores the first occurrence of a name under the name itself and each later /// duplicate under `_` (n = 1, 2, …), skipping any candidate that diff --git a/nodedb/src/control/server/response_shape/returning.rs b/nodedb/src/control/server/response_shape/returning.rs index 4483fc880..280bf49dd 100644 --- a/nodedb/src/control/server/response_shape/returning.rs +++ b/nodedb/src/control/server/response_shape/returning.rs @@ -2,7 +2,10 @@ //! Shaping for DML `RETURNING` responses. //! -//! A `RETURNING` payload is a [`RowsPayload`]: the Data Plane's own column list +//! A timeseries ingest that rejected rows reports them beside its rows, and +//! the shaper raises them as the rejected-lines notice. +//! +//! A `RETURNING` payload is a `RowsPayload`: the Data Plane's own column list //! plus typed `Value` cells. For `RETURNING *` that column list is //! derived from the STORED row, so a schemaless collection can carry fields no //! catalog column declares — the list is only knowable once the rows exist. @@ -36,7 +39,7 @@ use nodedb_types::{NativeCell, NdbDateTime, NodeDbError, Value}; -use crate::data::executor::response_codec::{RowsPayload, decode_payload_to_json}; +use crate::data::executor::response_codec::{ReturningRowsReply, decode_payload_to_json}; use crate::control::sequence::SequenceAccess; @@ -68,12 +71,12 @@ pub fn shape_returning_rows( }); } - let rp = match zerompk::from_msgpack::(payload) { + let rp = match zerompk::from_msgpack::(payload) { Ok(rp) => rp, Err(e) => { // Bytes that are not a `RowsPayload` yield no column list at all, // so there is none that honours what was announced: a substitute - // single-column row would be unparseable to the client that holds + // single-column row will be unparseable to the client that holds // the RowDescription, which is the very failure this module // exists to prevent. Fail the statement instead. if announced.is_some() { @@ -94,9 +97,23 @@ pub fn shape_returning_rows( } }; - let RowsPayload { columns, rows } = rp; + let ReturningRowsReply { + columns, + rows, + rejected, + } = rp; + // A row set has no place for a rejected row: the rows the ingest + // rejected reach the client as the rejected-lines notice. + if let Some(rejection) = rejected.filter(|rejection| rejection.lines > 0) { + crate::control::server::shared::session::statement_notice::raise( + crate::control::server::shared::sql::staging_predicates::rejected_lines_notice( + &rejection.collection, + rejection.lines, + ), + ); + } let mut rows = rows_keyed_by_column(&columns, rows); - // `RETURNING` delivers stored column values to the client just as a SELECT + // `RETURNING` delivers stored column values to the client as a SELECT // does, so the same redaction applies — and it runs on the payload's own // names, before any projection renames or drops them. redact_rows(redaction.as_ref(), &mut rows); @@ -221,7 +238,7 @@ fn retype_cell(ct: DdlColType, cell: &mut Value) { } /// The announced columns with no rows — a write that matched nothing still -/// answers with the result set the client was promised, just an empty one. +/// answers with the result set the client was promised, only an empty one. fn empty_announced(schema: &OutputSchema) -> ShapedRows { ShapedRows::from_rows( schema @@ -249,11 +266,46 @@ mod tests { use super::*; use crate::control::server::response_shape::cell::value_to_wire_json; use crate::control::server::response_shape::schema::OutputColumn; + use crate::control::server::shared::session::{conn_scope, statement_notice}; + use crate::data::executor::response_codec::{ + IngestRejection, RejectingRowsPayload, RowsPayload, + }; fn text(s: &str) -> Value { Value::String(s.to_string()) } + /// A timeseries ingest whose install rejected rows answers them beside + /// its row set. The shaper keeps the rows and raises the rejected-lines + /// notice. A plain row set raises none. + #[tokio::test] + async fn rows_an_ingest_rejected_raise_the_rejected_lines_notice() { + conn_scope::scoped(async { + let rejecting = zerompk::to_msgpack_vec(&RejectingRowsPayload { + columns: vec!["value".into()], + rows: vec![vec![NativeCell(Value::Float(1.5))]], + rejected: IngestRejection { + collection: "metrics".into(), + lines: 2, + }, + }) + .expect("encode rejecting payload"); + let shaped = shape_returning_rows(&rejecting, None, None, None).expect("shape"); + assert_eq!(shaped.rows.len(), 1); + let notices = statement_notice::take(); + assert_eq!(notices.len(), 1); + assert!( + notices[0].contains("2 line(s)") && notices[0].contains("metrics"), + "{notices:?}" + ); + + let plain = payload(&["id"], &[&[Some("a")]]); + shape_returning_rows(&plain, None, None, None).expect("shape"); + assert!(statement_notice::take().is_empty()); + }) + .await; + } + fn typed_payload(columns: &[&str], rows: Vec>) -> Vec { let rp = RowsPayload { columns: columns.iter().map(|c| (*c).to_string()).collect(), diff --git a/nodedb/src/control/server/response_shape/types/dml_outcome.rs b/nodedb/src/control/server/response_shape/types/dml_outcome.rs index 3923d547f..fbf32f7fb 100644 --- a/nodedb/src/control/server/response_shape/types/dml_outcome.rs +++ b/nodedb/src/control/server/response_shape/types/dml_outcome.rs @@ -8,7 +8,8 @@ //! count payload becomes a [`DmlOutcome`]; pgwire and native both call them. use crate::control::server::shared::sql::staging_predicates::{ - StagedTagKind, extract_kv_conflict_op, require_affected_count, + StagedTagKind, extract_ingest_rejections, extract_kv_conflict_op, rejected_lines_notice, + require_affected_count, }; use super::PlanKind; @@ -103,6 +104,29 @@ impl StatementTag { } } +/// The outcome a write an INSTEAD OF trigger body replaced contributes: the +/// statement ran and changed nothing. A count-bearing plan keeps its verb with +/// a zero count, so the fold stays on one verb. `None` for a plan whose +/// outcome folds as opaque. +pub(crate) fn replaced_write_outcome(plan_kind: super::PlanKind) -> Option { + use super::PlanKind; + match plan_kind { + PlanKind::DmlResult(verb) => Some(DmlOutcome { verb, affected: 0 }), + // The verb is resolved at apply time and no apply happened. The + // statement is an `INSERT ... ON CONFLICT DO UPDATE`, so it reports + // as `INSERT`. + PlanKind::DmlResultByOp => Some(DmlOutcome { + verb: "INSERT", + affected: 0, + }), + PlanKind::Execution + | PlanKind::ArraySlice + | PlanKind::ReturningRows + | PlanKind::SingleDocument + | PlanKind::MultiRow => None, + } +} + /// The count-bearing outcome of a staged write, from the neutral /// [`StagedTagKind`] the staging gate decided. Verb mapping: `INSERT` / /// `UPDATE` / `DELETE` by kind, and for a KV `InsertOnConflictUpdate` the @@ -129,7 +153,7 @@ pub(crate) fn staged_dml_outcome(kind: StagedTagKind, affected: usize) -> DmlOut // sole SQL surface (`SELECT KV_INCR(..)` and friends, in // `ddl/neutral/kv_atomic/`) reads `StagedWriteOutcome::payload` // directly, and both dispatch loops fold `RawPayload` as opaque. This - // arm exists only so the match stays exhaustive against a new + // arm exists only so the match stays exhaustive against a further // `PhysicalPlan::Kv` caller; it names the tag a function-call // `SELECT` renders. StagedTagKind::RawPayload => "SELECT", @@ -149,13 +173,29 @@ pub(crate) fn staged_dml_outcome(kind: StagedTagKind, affected: usize) -> DmlOut /// affected exactly 1 row" shortcut: a point delete or a conflicting /// `ON CONFLICT DO NOTHING` insert is the same plan whether it touched a row /// or not, so assuming 1 here reported rows that were never there. +/// +/// A timeseries ingest answer that rejected lines raises a statement notice +/// with the collection and the count: pgwire sends it as a +/// `NoticeResponse`, and the native protocol adds it to `warnings`. pub(crate) fn dml_outcome_from_payload( payload: &[u8], verb: &'static str, ) -> crate::Result { + // A verb that reports no count reads none. A staged TRUNCATE, in a + // session transaction or a Calvin transaction, stages no count, and + // both protocols answer it bare. + if !DmlOutcome::verb_carries_count(verb) { + return Ok(DmlOutcome { verb, affected: 0 }); + } let affected = require_affected_count(payload).map_err(|e| crate::Error::Internal { detail: format!("{verb} response is missing its affected count: {e}"), })?; + if let Some((collection, rejected)) = extract_ingest_rejections(payload) { + crate::control::server::shared::session::statement_notice::raise(rejected_lines_notice( + &collection, + rejected, + )); + } Ok(DmlOutcome { verb, affected }) } @@ -373,4 +413,15 @@ mod tests { assert_eq!(staged, outcome("TRUNCATE", 0)); assert!(!staged.carries_count()); } + + /// A TRUNCATE payload with no count, as a Calvin TRUNCATE stages it, + /// answers the bare tag. A count-bearing verb still requires one. + #[test] + fn a_count_less_truncate_payload_answers_the_bare_tag() { + assert_eq!( + dml_outcome_from_payload(&[], "TRUNCATE").expect("truncate"), + outcome("TRUNCATE", 0) + ); + assert!(dml_outcome_from_payload(&[], "DELETE").is_err()); + } } diff --git a/nodedb/src/control/server/response_shape/types/mod.rs b/nodedb/src/control/server/response_shape/types/mod.rs index 0e5cdd436..96176689b 100644 --- a/nodedb/src/control/server/response_shape/types/mod.rs +++ b/nodedb/src/control/server/response_shape/types/mod.rs @@ -8,7 +8,8 @@ pub mod shaped; pub use dml_outcome::{DmlFoldError, DmlOutcome, FoldedTag, StatementTag}; pub(crate) use dml_outcome::{ - dml_outcome_by_op, dml_outcome_from_payload, payload_to_dml_outcome, staged_dml_outcome, + dml_outcome_by_op, dml_outcome_from_payload, payload_to_dml_outcome, replaced_write_outcome, + staged_dml_outcome, }; pub use plan_kind::{PlanKind, describe_plan}; pub use shaped::{DdlColType, ShapedRow, ShapedRows}; diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/array.rs b/nodedb/src/control/server/response_shape/types/plan_kind/array.rs index 23ec59f37..a8b39a0cc 100644 --- a/nodedb/src/control/server/response_shape/types/plan_kind/array.rs +++ b/nodedb/src/control/server/response_shape/types/plan_kind/array.rs @@ -2,10 +2,21 @@ //! `ArrayOp` classification. -use nodedb_physical::physical_plan::ArrayOp; +use nodedb_physical::physical_plan::{ArrayOp, ClusterArrayOp}; use super::kind::PlanKind; +/// A cluster array op answers with the payload its local `ArrayOp` +/// counterpart answers with, so it shapes the same way. +pub(super) fn describe_cluster_array(op: &ClusterArrayOp) -> PlanKind { + match op { + ClusterArrayOp::Slice { .. } => PlanKind::ArraySlice, + ClusterArrayOp::Agg { .. } => PlanKind::MultiRow, + ClusterArrayOp::Put { .. } => PlanKind::DmlResult("INSERT"), + ClusterArrayOp::Delete { .. } => PlanKind::DmlResult("DELETE"), + } +} + pub(super) fn describe_array(op: &ArrayOp) -> PlanKind { match op { ArrayOp::Slice { .. } => PlanKind::ArraySlice, @@ -25,7 +36,7 @@ pub(super) fn describe_array(op: &ArrayOp) -> PlanKind { // Array DDL: `{"opened": 1}` / `{"dropped": 1}` status, not a row count. ArrayOp::OpenArray { .. } | ArrayOp::DropArray { .. } - | ArrayOp::RestoreArrayDrop { .. } + | ArrayOp::RekeyArray { .. } | ArrayOp::PurgeArrayDrop { .. } // Internal roaring bitmap for cross-engine prefilter, never a client row. | ArrayOp::SurrogateBitmapScan { .. } => PlanKind::Execution, diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs b/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs index cf9b6fad7..d2fca71c2 100644 --- a/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs +++ b/nodedb/src/control/server/response_shape/types/plan_kind/describe.rs @@ -6,7 +6,7 @@ use crate::bridge::envelope::PhysicalPlan; -use super::array::describe_array; +use super::array::{describe_array, describe_cluster_array}; use super::columnar_family::{describe_columnar, describe_spatial, describe_timeseries}; use super::crdt::describe_crdt; use super::document::describe_document; @@ -28,14 +28,12 @@ pub fn describe_plan(plan: &PhysicalPlan) -> PlanKind { PhysicalPlan::Timeseries(op) => describe_timeseries(op), PhysicalPlan::Spatial(op) => describe_spatial(op), PhysicalPlan::Array(op) => describe_array(op), + PhysicalPlan::ClusterArray(op) => describe_cluster_array(op), PhysicalPlan::Query(op) => describe_query(op), // Control-plane catalog, session and cluster ops. No `MetaOp` is a // client DML: none reports a row count, each answers its own caller. PhysicalPlan::Meta(_) - // Never dispatched through the plan-shaping path: the pgwire cluster - // array router classifies these itself (`routing/cluster_array.rs`). - | PhysicalPlan::ClusterArray(_) // Event-plane forwarding, answered by its own dispatcher. | PhysicalPlan::ClusterEvent(_) => PlanKind::Execution, } @@ -81,7 +79,7 @@ mod tests { }) } - /// A `MERGE ... RETURNING` payload is real target rows — `Execution` would + /// A `MERGE ... RETURNING` payload is real target rows — `Execution` will /// pass them unredacted. #[test] fn merge_with_returning_is_returning_rows() { @@ -111,7 +109,7 @@ mod tests { document_id: "d".into(), value: Vec::new(), if_absent: false, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), returning: spec(), rls_filters: Vec::new(), resolved_sum_targets: Vec::new(), @@ -121,7 +119,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "c"), document_id: "d".into(), value: Vec::new(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), pk_bytes: Vec::new(), returning: spec(), rls_filters: Vec::new(), @@ -141,7 +139,7 @@ mod tests { document_id: "d".into(), value: Vec::new(), on_conflict_updates: Vec::new(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: spec(), rls_filters: Vec::new(), @@ -173,7 +171,7 @@ mod tests { key: b"k".to_vec(), value: Vec::new(), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), returning: spec(), rls_filters: Vec::new(), }), @@ -182,7 +180,7 @@ mod tests { key: b"k".to_vec(), value: Vec::new(), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), returning: spec(), rls_filters: Vec::new(), }), @@ -192,7 +190,7 @@ mod tests { value: Vec::new(), ttl_ms: 0, updates: Vec::new(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: spec(), rls_filters: Vec::new(), @@ -202,7 +200,7 @@ mod tests { key: b"k".to_vec(), value: Vec::new(), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), returning: spec(), rls_filters: Vec::new(), provenance: None, @@ -240,7 +238,7 @@ mod tests { key: b"k".to_vec(), value: Vec::new(), ttl_ms: 0, - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), returning: None, rls_filters: Vec::new(), provenance: None, @@ -267,7 +265,7 @@ mod tests { value: Vec::new(), ttl_ms: 0, updates: Vec::new(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), rls_write_check: nodedb_types::RlsWriteCheck::pending_injection(), returning: None, rls_filters: Vec::new(), @@ -301,7 +299,7 @@ mod tests { collection: QualifiedCollection::new(DatabaseId::DEFAULT, "notes"), document_id: "d1".into(), fields_json: "{}".into(), - surrogate: nodedb_types::Surrogate::ZERO, + surrogate: nodedb_types::Surrogate::new(1), partial: matches!(verb, CrdtWriteVerb::Update), verb, returning: None, @@ -335,6 +333,7 @@ mod tests { cells_msgpack: Vec::new(), wal_lsn: 0, provenance: None, + vshard_id: 0, }); assert!(matches!( describe_plan(&plan), diff --git a/nodedb/src/control/server/response_shape/types/plan_kind/graph.rs b/nodedb/src/control/server/response_shape/types/plan_kind/graph.rs index 10489a26a..0795dc56d 100644 --- a/nodedb/src/control/server/response_shape/types/plan_kind/graph.rs +++ b/nodedb/src/control/server/response_shape/types/plan_kind/graph.rs @@ -24,7 +24,8 @@ pub(super) fn describe_graph(op: &GraphOp) -> PlanKind { | GraphOp::WccSuperstep(_) | GraphOp::TemporalNeighbors { .. } | GraphOp::TemporalAlgorithm { .. } - | GraphOp::Stats { .. } => PlanKind::MultiRow, + | GraphOp::Stats { .. } + | GraphOp::NodePresenceRead { .. } => PlanKind::MultiRow, GraphOp::EdgePut { .. } | GraphOp::EdgePutBatch { .. } => PlanKind::DmlResult("INSERT"), @@ -38,5 +39,11 @@ pub(super) fn describe_graph(op: &GraphOp) -> PlanKind { GraphOp::SetNodeLabels { .. } | GraphOp::RemoveNodeLabels { .. } => { PlanKind::DmlResult("UPDATE") } + + // These ride a document delete or TRUNCATE whose own task names the + // statement's tag. + GraphOp::NodeEdgeGuard { .. } + | GraphOp::NodePresenceGuard { .. } + | GraphOp::TruncateEdges { .. } => PlanKind::DmlResult("DELETE"), } } diff --git a/nodedb/src/control/server/response_shape/walk.rs b/nodedb/src/control/server/response_shape/walk.rs new file mode 100644 index 000000000..6c4f8faba --- /dev/null +++ b/nodedb/src/control/server/response_shape/walk.rs @@ -0,0 +1,108 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Graph walk response shaping: turn the node names a walk reached into +//! rows before the protocol layer shapes them. +//! +//! A `Hop` answers a msgpack array of every reached node name. A `Path` +//! answers a msgpack array `[src, …, dst]`. The Data Plane and the +//! cross-shard walk coordinators (`graph_dispatch::walk_reads`) both answer +//! this shape. A bare name is not a row, and the generic row flattener +//! (`push_flat_rows`) drops every scalar. So each name becomes one +//! `{node}` row here, in the order the walk gave it. + +use std::collections::HashMap; + +use crate::bridge::envelope::PhysicalPlan; +use crate::data::executor::response_codec::decode_payload_value; +use nodedb_physical::physical_plan::GraphOp; +use nodedb_types::Value; + +/// The column a walk row names its node under. +pub const WALK_NODE_COLUMN: &str = "node"; + +/// When `plan` is a `Hop` or a `Path`, return its node names as `{node}` +/// rows, msgpack-encoded. Every other plan's payload is returned unchanged. +/// +/// A walk payload that is not an array of names is an error: a walk row +/// is never dropped. +pub fn apply_walk_wrap(plan: &PhysicalPlan, payload: &[u8]) -> crate::Result> { + let is_walk = matches!( + plan, + PhysicalPlan::Graph(GraphOp::Hop { .. } | GraphOp::Path { .. }) + ); + if !is_walk || payload.is_empty() { + return Ok(payload.to_vec()); + } + let Value::Array(names) = decode_payload_value(payload)? else { + return Err(crate::Error::Codec { + detail: "graph walk response is not an array of node names".into(), + }); + }; + let rows = names + .into_iter() + .map(|name| match name { + Value::String(node) => { + let mut row = HashMap::with_capacity(1); + row.insert(WALK_NODE_COLUMN.to_string(), Value::String(node)); + Ok(Value::Object(row)) + } + other => Err(crate::Error::Codec { + detail: format!("graph walk response holds a non-name entry: {other:?}"), + }), + }) + .collect::>>()?; + nodedb_types::value_to_msgpack(&Value::Array(rows)).map_err(|e| crate::Error::Codec { + detail: format!("graph walk rows encode: {e}"), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::server::response_shape::compose::{ShapeOutcome, shape_payload_no_plan}; + use crate::control::server::response_shape::types::PlanKind; + use nodedb_graph::{Direction, GraphTraversalOptions}; + + fn hop() -> PhysicalPlan { + PhysicalPlan::Graph(GraphOp::Hop { + collection: None, + start_nodes: vec!["a".into()], + depth: 2, + edge_label: None, + direction: Direction::Out, + options: GraphTraversalOptions::default(), + rls_filters: Vec::new(), + frontier_bitmap: None, + }) + } + + /// Every reached name becomes one row, in walk order, and the shaped + /// rows keep all of them. + #[test] + fn every_reached_name_is_a_row() { + let names = vec!["a".to_string(), "c".to_string(), "b".to_string()]; + let payload = zerompk::to_msgpack_vec(&names).expect("names encode"); + let wrapped = apply_walk_wrap(&hop(), &payload).expect("walk wraps"); + let Ok(ShapeOutcome::Rows(shaped)) = + shape_payload_no_plan(&wrapped, PlanKind::MultiRow, None, None, None) + else { + panic!("a walk shapes into rows"); + }; + let got: Vec = shaped + .rows + .iter() + .filter_map(|row| row.get(WALK_NODE_COLUMN).cloned()) + .collect(); + assert_eq!( + got, + names.into_iter().map(Value::String).collect::>() + ); + } + + /// A walk payload that is not a name array is refused, never dropped. + #[test] + fn a_non_name_entry_is_refused() { + let payload = zerompk::to_msgpack_vec(&vec![1u32]).expect("encode"); + assert!(apply_walk_wrap(&hop(), &payload).is_err()); + } +} diff --git a/nodedb/src/control/server/shared/authorization/requirements/collect.rs b/nodedb/src/control/server/shared/authorization/requirements/collect.rs index 656e3001c..b2a72725a 100644 --- a/nodedb/src/control/server/shared/authorization/requirements/collect.rs +++ b/nodedb/src/control/server/shared/authorization/requirements/collect.rs @@ -98,6 +98,7 @@ fn collect_requirements(plan: &PhysicalPlan, out: &mut Vec add_collection_requirement(collection.as_str(), required_permission(plan), out), PhysicalPlan::Meta( @@ -105,15 +106,6 @@ fn collect_requirements(plan: &PhysicalPlan, out: &mut Vec add_collection_requirement(name, required_permission(plan), out), - PhysicalPlan::Meta(MetaOp::RenameCollection { - old_collection, - new_collection, - .. - }) => { - let permission = required_permission(plan); - add_collection_requirement(old_collection.as_str(), permission, out); - add_collection_requirement(new_collection.as_str(), permission, out); - } PhysicalPlan::Meta(MetaOp::TransactionBatch { plans, .. }) | PhysicalPlan::Meta(MetaOp::ResolveTxn { plans, .. }) | PhysicalPlan::Meta(MetaOp::RecordCalvinWriteVersions { plans, .. }) diff --git a/nodedb/src/control/server/shared/authorization/service.rs b/nodedb/src/control/server/shared/authorization/service.rs index 1bfa6b560..9da0cbfcf 100644 --- a/nodedb/src/control/server/shared/authorization/service.rs +++ b/nodedb/src/control/server/shared/authorization/service.rs @@ -19,7 +19,7 @@ use super::capability::{AuthorizedCollection, AuthorizedTaskSet}; use super::error::AuthorizationError; use super::requirements::{AuthorizationRequirement, plan_requirements}; -/// Ensure an identity may select `database_id`. +/// Ensure an identity can select `database_id`. pub fn authorize_database( identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -427,7 +427,7 @@ mod tests { let roles = RoleStore::new(); permissions .grant( - "collection:9:orders", + "collection:0:9:orders", "user:reader", Permission::Read, "admin", diff --git a/nodedb/src/control/server/shared/backup_metering.rs b/nodedb/src/control/server/shared/backup_metering.rs new file mode 100644 index 000000000..4abe0c1aa --- /dev/null +++ b/nodedb/src/control/server/shared/backup_metering.rs @@ -0,0 +1,176 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Quota admission and metering for whole-tenant backup and restore. +//! +//! A backup or restore has no `PhysicalPlan`, so each describes itself under +//! the synthetic collection marker `tenant:`. A scope grant caps it only +//! when written against that marker, or against `*`. Every door runs the same +//! pair: the admission before the first read or write, the charge on the +//! success path, so the charge never refuses anything. +//! +//! * The backup permission covers a backup and a restore: the tenant COPY path +//! and `BACKUP` / `RESTORE DATABASE` alike. +//! * A `RESTORE DATABASE` also counts against the write quota of every +//! collection it restores, under the collection's qualified name as DML and +//! COPY charge it, so a write grant on one collection caps the restore. A +//! write grant on `*` covers every collection and is charged once per row. +//! A write grant on the tenant's `tenant:` marker caps the tenant's +//! whole restore and is charged the tenant's rows. + +use nodedb_types::calvin::EngineTag; + +use crate::control::backup::CollectionRows; +use crate::control::security::identity::Permission; +use crate::control::security::permission::parse_permission; +use crate::control::security::request_scope::RequestAuthScope; +use crate::control::state::SharedState; +use crate::types::DatabaseId; +use nodedb_types::QualifiedCollection; + +use super::metering::{PlanMeteringInfo, meter_dispatch}; +use super::quota_admission::admit_quota_for_dispatch; + +fn tenant_marker(tenant_id: u64, permission: Permission) -> PlanMeteringInfo { + PlanMeteringInfo::for_collection( + format!("tenant:{tenant_id}"), + EngineTag::Meta, + "sql", + permission, + ) +} + +/// Refuse a backup or restore of `tenant_id` whose covering scope already +/// spent its hard cap. +pub(crate) fn admit_backup_restore_quota( + state: &SharedState, + scope: &RequestAuthScope<'_>, + tenant_id: u64, +) -> crate::Result<()> { + if !state.metering_config.enabled { + return Ok(()); + } + admit_quota_for_dispatch(state, scope, &tenant_marker(tenant_id, Permission::Backup)) +} + +/// Meter one completed backup or restore of `tenant_id`. `rows: None` charges +/// one unit. +pub(crate) fn meter_backup_restore( + state: &SharedState, + scope: &RequestAuthScope<'_>, + tenant_id: u64, + rows: Option, +) { + if !state.metering_config.enabled { + return; + } + meter_dispatch( + state, + scope, + &tenant_marker(tenant_id, Permission::Backup), + rows, + ); +} + +/// The name a restored collection's write quota is admitted and charged +/// under. A collection of a database this cluster lacks has no qualified name +/// before the restore creates the database, so only a `*` grant covers it. +fn restored_collection(collection: &CollectionRows) -> String { + match collection.database_id { + Some(database_id) => { + QualifiedCollection::new(DatabaseId::new(database_id), &collection.collection) + .as_str() + .to_string() + } + None => "*".to_string(), + } +} + +fn collection_write(collection: &CollectionRows) -> PlanMeteringInfo { + PlanMeteringInfo::for_collection( + restored_collection(collection), + EngineTag::Meta, + "sql", + Permission::Write, + ) +} + +/// Refuse a restore into `tenant_id` when a write scope covering its +/// `tenant:` marker, or any of its `collections`, already spent its hard +/// cap. +pub(crate) fn admit_restore_write_quota( + state: &SharedState, + scope: &RequestAuthScope<'_>, + tenant_id: u64, + collections: &[CollectionRows], +) -> crate::Result<()> { + if !state.metering_config.enabled { + return Ok(()); + } + admit_quota_for_dispatch(state, scope, &tenant_marker(tenant_id, Permission::Write))?; + for collection in collections { + admit_quota_for_dispatch(state, scope, &collection_write(collection))?; + } + Ok(()) +} + +/// Charge the rows a restore verified in each of `collections` of +/// `tenant_id` to the write quota: every covering collection or `*` grant per +/// collection, then every grant on the tenant's marker once for the total. +pub(crate) fn meter_restore_writes( + state: &SharedState, + scope: &RequestAuthScope<'_>, + tenant_id: u64, + collections: &[CollectionRows], +) { + if !state.metering_config.enabled { + return; + } + for collection in collections { + meter_dispatch( + state, + scope, + &collection_write(collection), + Some(collection.rows), + ); + } + let rows = collections.iter().map(|collection| collection.rows).sum(); + charge_marker_grants(state, scope, &format!("tenant:{tenant_id}"), rows); +} + +/// Charge `rows` to every held write scope that grants exactly `marker`. A +/// `*` grant also covers the marker, but the per-collection charge already +/// charged it for these rows. +fn charge_marker_grants( + state: &SharedState, + scope: &RequestAuthScope<'_>, + marker: &str, + rows: u64, +) { + if scope.identity().is_internal_service() { + return; + } + let cost = state + .metering_config + .operation_costs + .get("sql") + .copied() + .unwrap_or(1); + let tokens = cost.saturating_mul(rows.max(1)); + let auth = scope.auth(); + let now_secs = crate::control::security::time::now_secs(); + for scope_name in state.scope_grants.effective_scopes(&auth.id, &auth.org_ids) { + let grants_marker = + state + .scope_defs + .resolve(&scope_name) + .into_iter() + .any(|(permission, collection)| { + parse_permission(&permission) == Some(Permission::Write) && collection == marker + }); + if grants_marker { + state + .quota_manager + .record_usage(&scope_name, &auth.id, tokens, now_secs); + } + } +} diff --git a/nodedb/src/control/server/shared/check_constraint/mod.rs b/nodedb/src/control/server/shared/check_constraint/mod.rs index 875adeff7..c2a167366 100644 --- a/nodedb/src/control/server/shared/check_constraint/mod.rs +++ b/nodedb/src/control/server/shared/check_constraint/mod.rs @@ -2,14 +2,16 @@ //! Protocol-neutral runtime enforcement for general CHECK constraints. //! -//! Relocated from the pgwire `ddl::collection::check_constraint` module (now -//! deleted): this is runtime write-path enforcement, not a DDL handler, so it -//! does not live under `ddl/`. See [`enforce::enforce_check_constraints`] for -//! the evaluation strategy. +//! This is runtime write-path enforcement, not a DDL handler, so it does not +//! live under `ddl/`. See [`enforce::enforce_check_constraints`] for the +//! evaluation strategy and [`statement`] for the per-statement entry points +//! every protocol calls before planning. mod enforce; mod simple; +pub mod statement; mod subquery; pub use enforce::enforce_check_constraints; +pub use statement::{enforce_statement_checks, enforce_statement_enum_labels}; pub(crate) use subquery::validate_in_subquery_check; diff --git a/nodedb/src/control/server/shared/check_constraint/statement.rs b/nodedb/src/control/server/shared/check_constraint/statement.rs new file mode 100644 index 000000000..ab5fafdf1 --- /dev/null +++ b/nodedb/src/control/server/shared/check_constraint/statement.rs @@ -0,0 +1,464 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Statement-level CHECK and enum-label enforcement for the SQL write path, +//! shared by every protocol. +//! +//! Runs on an INSERT or UPDATE statement before planning: the collection and +//! the column/value pairs come from the SQL text. An UPDATE's current row is +//! merged under its SET values, so a cross-field CHECK sees the whole row. A +//! statement whose collection fires a BEFORE, INSTEAD OF or SYNC AFTER body +//! is checked by the transaction route instead, against the row its BEFORE +//! bodies left. + +use std::collections::HashMap; + +use nodedb_sql::parser::preprocess::lex::{ + find_ascii_case_insensitive, find_ascii_case_insensitive_from, +}; +use nodedb_types::{DatabaseId, strip_prefix_ascii_case_insensitive}; + +use crate::control::security::audit::ArcAuditEmitter; +use crate::control::security::auth_context::AuthContext; +use crate::control::security::identity::{AuthenticatedIdentity, Permission}; +use crate::control::server::shared::authorization::authorize_collection; +use crate::control::server::shared::ddl::result::DdlError; +use crate::control::state::SharedState; +use crate::control::trigger::statement_txn::fires_joined_body; +use crate::control::trigger::{DmlEvent, TriggerScope}; +use crate::types::{TenantId, TxnId}; + +use super::enforce::enforce_check_constraints; + +/// Enforce the general CHECK constraints of the INSERT or UPDATE in `sql`. +/// +/// `txn_id` names the session's open transaction: an UPDATE's current row is +/// read as it left it. Any other statement passes. +pub async fn enforce_statement_checks( + state: &SharedState, + identity: &AuthenticatedIdentity, + tenant_id: TenantId, + database_id: DatabaseId, + auth: &AuthContext, + txn_id: Option, + sql: &str, +) -> Result<(), DdlError> { + let Some((coll_name, is_insert)) = extract_collection_from_sql(sql) else { + return Ok(()); + }; + + // CHECK evaluation is on the write path. Authorize its target before + // catalog lookup or an OLD-row read so unauthorized SQL cannot probe + // collection metadata or row existence. + let audit = ArcAuditEmitter(std::sync::Arc::clone(&state.audit)); + authorize_collection( + identity, + database_id, + &coll_name, + Permission::Write, + &state.permissions, + &state.roles, + &audit, + ) + .map_err(|error| DdlError::new("42501", error.resource().to_owned()))?; + + // A write that fires a BEFORE, INSTEAD OF or SYNC AFTER body runs + // through the transaction route, which checks the NEW row as the BEFORE + // bodies left it. + let event = if is_insert { + DmlEvent::Insert + } else { + DmlEvent::Update + }; + if fires_joined_body( + state, + TriggerScope { + database_id, + tenant_id, + }, + nodedb_types::QualifiedCollection::new(database_id, &coll_name).as_str(), + event, + ) { + return Ok(()); + } + + let catalog = state.credentials.catalog(); + let coll = match catalog.get_collection(database_id, tenant_id.as_u64(), &coll_name) { + Ok(Some(c)) => c, + _ => return Ok(()), + }; + if coll.check_constraints.is_empty() { + return Ok(()); + } + + let mut fields = if is_insert { + extract_insert_fields(sql).map_err(|e| DdlError::new("42601", e))? + } else { + extract_update_fields(sql).map_err(|e| DdlError::new("42601", e))? + }; + if fields.is_empty() { + return Ok(()); + } + + // For UPDATE: merge SET values over the current row for cross-field CHECK. + if !is_insert && let Some(doc_id) = extract_where_id(sql) { + let old = crate::control::trigger::dml_hook::fetch_old_row( + state, + identity, + database_id, + auth, + &nodedb_types::QualifiedCollection::new(database_id, &coll_name), + &doc_id, + txn_id, + ) + .await + .map_err(|error| { + let (_, sqlstate, message) = + crate::control::server::pgwire::types::error_to_sqlstate(&error); + DdlError::new(sqlstate, message) + })?; + let mut merged = old; + for (k, v) in &fields { + merged.insert(k.clone(), v.clone()); + } + fields = merged; + } + + enforce_check_constraints( + state, + identity, + database_id, + &coll.check_constraints, + &fields, + ) + .await +} + +/// Validate the enum-typed column values of the INSERT or UPDATE in `sql` +/// against the custom type registry. Any other statement, and a collection +/// with no enum-typed column, passes. +pub fn enforce_statement_enum_labels( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + sql: &str, +) -> Result<(), DdlError> { + let Some((coll_name, is_insert)) = extract_collection_from_sql(sql) else { + return Ok(()); + }; + + let catalog = state.credentials.catalog(); + let coll = match catalog.get_collection(database_id, tenant_id.as_u64(), &coll_name) { + Ok(Some(c)) => c, + _ => return Ok(()), + }; + // Quick path: no user-defined types means nothing to validate. + if coll.fields.is_empty() { + return Ok(()); + } + + let fields = if is_insert { + extract_insert_fields(sql) + } else { + extract_update_fields(sql) + }; + // Unparseable text is the planner's to refuse. + let Ok(fields) = fields else { + return Ok(()); + }; + + for (field_name, type_name) in &coll.fields { + let Some(nodedb_types::Value::String(label)) = fields.get(field_name.as_str()) else { + continue; + }; + if let Err(msg) = state.custom_type_registry.validate_enum_label( + database_id.as_u64(), + tenant_id.as_u64(), + type_name, + label, + ) { + return Err(DdlError::new("22P02", msg)); + } + } + Ok(()) +} + +/// Extract the collection name and operation type from an INSERT or UPDATE SQL +/// statement. Returns `None` for any other statement kind. +fn extract_collection_from_sql(sql: &str) -> Option<(String, bool)> { + if let Some(after) = strip_prefix_ascii_case_insensitive(sql, "INSERT INTO ") { + let after = after.trim_start(); + let end = after + .find(|c: char| c.is_whitespace() || c == '(') + .unwrap_or(after.len()); + Some((after[..end].to_lowercase(), true)) + } else if let Some(after) = strip_prefix_ascii_case_insensitive(sql, "UPDATE ") { + let after = after.trim_start(); + let end = after + .find(|c: char| c.is_whitespace()) + .unwrap_or(after.len()); + Some((after[..end].to_lowercase(), false)) + } else { + None + } +} + +/// Extract column/value pairs from `INSERT INTO x (col1, col2) VALUES (val1, val2)`. +fn extract_insert_fields(sql: &str) -> Result, String> { + let cols_start = sql.find('(').ok_or_else(|| { + let preview: String = sql.chars().take(60).collect(); + format!("missing '(' in INSERT: {preview}") + })?; + let cols_end = sql[cols_start + 1..] + .find(')') + .map(|p| cols_start + 1 + p) + .ok_or_else(|| "missing ')' after column list in INSERT".to_string())?; + let cols: Vec<&str> = sql[cols_start + 1..cols_end] + .split(',') + .map(|s| s.trim()) + .collect(); + + let values_pos = find_ascii_case_insensitive(sql, "VALUES") + .ok_or_else(|| "missing VALUES keyword in INSERT".to_string())? + + 6; + let vals_start = sql[values_pos..] + .find('(') + .map(|p| values_pos + p + 1) + .ok_or_else(|| "missing '(' after VALUES in INSERT".to_string())?; + + let mut depth = 1i32; + let mut vals_end = vals_start; + for (i, ch) in sql[vals_start..].char_indices() { + match ch { + '(' => depth += 1, + ')' => { + depth -= 1; + if depth == 0 { + vals_end = vals_start + i; + break; + } + } + _ => {} + } + } + if depth != 0 { + return Err("unmatched parentheses in VALUES clause".to_string()); + } + + let vals = split_top_level_commas(&sql[vals_start..vals_end]); + let mut fields = HashMap::new(); + for (i, col) in cols.iter().enumerate() { + if let Some(val_str) = vals.get(i) { + let col_name = col.trim_matches('"').trim_matches('`').to_lowercase(); + let val = parse_sql_literal(val_str.trim()); + fields.insert(col_name, val); + } + } + + Ok(fields) +} + +/// Extract column/value pairs from `UPDATE x SET col1 = val1, col2 = val2 WHERE ...`. +fn extract_update_fields(sql: &str) -> Result, String> { + let set_pos = find_ascii_case_insensitive(sql, " SET ") + .ok_or_else(|| "missing SET keyword in UPDATE".to_string())? + + 5; + + let where_pos = find_ascii_case_insensitive_from(sql, " WHERE ", set_pos).unwrap_or(sql.len()); + let assignments_str = &sql[set_pos..where_pos]; + + let mut fields = HashMap::new(); + for assignment in split_top_level_commas(assignments_str) { + let assignment = assignment.trim(); + if let Some(eq_pos) = assignment.find('=') { + let col = assignment[..eq_pos] + .trim() + .trim_matches('"') + .trim_matches('`') + .to_lowercase(); + let val_str = assignment[eq_pos + 1..].trim(); + let val = parse_sql_literal(val_str); + fields.insert(col, val); + } + } + + Ok(fields) +} + +/// Extract document ID from a `WHERE id = 'value'` clause. +/// +/// Only matches standalone `id` with word boundaries — `userid`, `order_id` etc. won't match. +fn extract_where_id(sql: &str) -> Option { + let where_pos = find_ascii_case_insensitive(sql, " WHERE ")?; + let after = &sql[where_pos + 7..]; + // Find standalone "ID" with word boundary checks. + let mut search_start = 0; + loop { + let abs_pos = find_ascii_case_insensitive_from(after, "ID", search_start)?; + + // Check word boundary before: must be start or non-alphanumeric/underscore. + if abs_pos > 0 { + let prev = after.as_bytes()[abs_pos - 1]; + if prev.is_ascii_alphanumeric() || prev == b'_' { + search_start = abs_pos + 2; + continue; + } + } + // Check word boundary after: must be end or non-alphanumeric/underscore. + let end_pos = abs_pos + 2; + if end_pos < after.len() { + let next = after.as_bytes()[end_pos]; + if next.is_ascii_alphanumeric() || next == b'_' { + search_start = end_pos; + continue; + } + } + + let after_id = after[end_pos..].trim_start(); + let Some(val_str) = after_id.strip_prefix('=') else { + search_start = end_pos; + continue; + }; + let val_str = val_str.trim_start(); + + if let Some(inner) = val_str.strip_prefix('\'') { + let end = inner.find('\'')?; + return Some(inner[..end].to_string()); + } + if let Some(inner) = val_str.strip_prefix('"') { + let end = inner.find('"')?; + return Some(inner[..end].to_string()); + } + let end = val_str + .find(|c: char| c.is_whitespace() || c == ';') + .unwrap_or(val_str.len()); + return Some(val_str[..end].to_string()); + } +} + +/// Split a string on commas, respecting parentheses and string quotes. +fn split_top_level_commas(s: &str) -> Vec<&str> { + let mut parts = Vec::new(); + let mut depth = 0i32; + let mut in_single_quote = false; + let mut in_double_quote = false; + let mut last = 0; + + for (i, ch) in s.char_indices() { + match ch { + '\'' if !in_double_quote => in_single_quote = !in_single_quote, + '"' if !in_single_quote => in_double_quote = !in_double_quote, + '(' if !in_single_quote && !in_double_quote => depth += 1, + ')' if !in_single_quote && !in_double_quote => depth -= 1, + ',' if depth == 0 && !in_single_quote && !in_double_quote => { + parts.push(&s[last..i]); + last = i + 1; + } + _ => {} + } + } + parts.push(&s[last..]); + parts +} + +/// Parse a SQL literal string into a Value (best-effort). +fn parse_sql_literal(s: &str) -> nodedb_types::Value { + let s = s.trim(); + + if s.eq_ignore_ascii_case("NULL") { + return nodedb_types::Value::Null; + } + if s.eq_ignore_ascii_case("TRUE") { + return nodedb_types::Value::Bool(true); + } + if s.eq_ignore_ascii_case("FALSE") { + return nodedb_types::Value::Bool(false); + } + if let Some(inner) = s + .strip_prefix('\'') + .and_then(|value| value.strip_suffix('\'')) + { + return nodedb_types::Value::String(inner.replace("''", "'")); + } + if let Ok(i) = s.parse::() { + return nodedb_types::Value::Integer(i); + } + if let Ok(f) = s.parse::() { + return nodedb_types::Value::Float(f); + } + nodedb_types::Value::String(s.to_string()) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn extract_where_id_basic() { + let sql = "UPDATE orders SET amount = 5 WHERE id = 'o1'"; + assert_eq!(extract_where_id(sql), Some("o1".to_string())); + } + + #[test] + fn extract_where_id_no_match_userid() { + // "userid" must NOT match — only standalone "id". + let sql = "UPDATE orders SET amount = 5 WHERE userid = 'u1'"; + assert_eq!(extract_where_id(sql), None); + } + + #[test] + fn extract_where_id_no_match_order_id() { + let sql = "UPDATE orders SET amount = 5 WHERE order_id = 'x'"; + assert_eq!(extract_where_id(sql), None); + } + + #[test] + fn extract_where_id_after_unicode_value_preserves_original_offsets() { + let sql = "UPDATE orders SET note = 'ǰ' WHERE id = 'o1'"; + assert_eq!(extract_where_id(sql), Some("o1".to_string())); + } + + #[test] + fn extract_insert_fields_basic() { + let fields = extract_insert_fields("INSERT INTO t (a, b) VALUES ('hello', 42)").unwrap(); + assert_eq!( + fields.get("a"), + Some(&nodedb_types::Value::String("hello".into())) + ); + assert_eq!(fields.get("b"), Some(&nodedb_types::Value::Integer(42))); + } + + #[test] + fn extract_insert_fields_error_on_bad_sql() { + let result = extract_insert_fields("INSERT INTO t no_parens"); + assert!(result.is_err()); + } + + #[test] + fn extract_insert_fields_with_unicode_before_values_preserves_original_offsets() { + let fields = extract_insert_fields("INSERT INTO tffff (a) VALUES (42)").unwrap(); + assert_eq!(fields.get("a"), Some(&nodedb_types::Value::Integer(42))); + } + + #[test] + fn malformed_insert_preview_respects_utf8_boundaries() { + let sql = format!("INSERT INTO {}é no_parens", "a".repeat(47)); + assert_eq!(sql.find('é'), Some(59)); + assert!(extract_insert_fields(&sql).is_err()); + } + + #[test] + fn extract_update_fields_basic() { + let fields = extract_update_fields("UPDATE t SET x = 10, y = 'hi' WHERE id = '1'").unwrap(); + assert_eq!(fields.get("x"), Some(&nodedb_types::Value::Integer(10))); + assert_eq!( + fields.get("y"), + Some(&nodedb_types::Value::String("hi".into())) + ); + } + + #[test] + fn extract_update_fields_with_unicode_before_set_preserves_original_offsets() { + let fields = extract_update_fields("UPDATE tǰ SET x = 10 WHERE id = '1'").unwrap(); + assert_eq!(fields.get("x"), Some(&nodedb_types::Value::Integer(10))); + } +} diff --git a/nodedb/src/control/server/shared/clone_read/dispatch.rs b/nodedb/src/control/server/shared/clone_read/dispatch.rs index 240c4291d..0ffc86b98 100644 --- a/nodedb/src/control/server/shared/clone_read/dispatch.rs +++ b/nodedb/src/control/server/shared/clone_read/dispatch.rs @@ -4,8 +4,10 @@ //! (target task + every source-side task the chain walk produced) is //! authorized together — the source tasks are derived here and were never //! part of the caller's own pre-authorized task list — dispatched to the -//! Data Plane, then merged into one `Response` with tombstoned source rows -//! filtered out. +//! Data Plane, then merged into one `Response` with tombstoned source rows, +//! and source rows whose primary key the target holds, filtered out. + +use std::collections::HashSet; use crate::bridge::envelope::Response; use crate::control::local_dispatch::reject_data_plane_error; @@ -13,10 +15,11 @@ use crate::control::security::audit::AuditEmitter; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::permission::PermissionStore; use crate::control::security::role::RoleStore; -use crate::control::server::dispatch_utils::dispatch_to_data_plane; +use crate::control::server::dispatch_utils::dispatch_to_data_plane_with_txn; use crate::control::server::response_shape::kv::apply_kv_wrap; use crate::control::server::response_shape::types::{PlanKind, describe_plan}; use crate::control::server::shared::authorization::authorize_task_set; +use crate::control::server::shared::plan_util::extract_collection; use crate::control::state::SharedState; use crate::types::TraceId; use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; @@ -25,6 +28,7 @@ use nodedb_physical::physical_task::PhysicalTask; use super::merge::{ filter_kv_tombstoned_rows, merge_msgpack_arrays, unwrap_single_row, wrap_single_map_as_array, }; +use super::shadow::{drop_shadowed_rows, plain_row_scan, row_keys}; use crate::control::clone::resolver::filter_tombstoned_rows; /// Inputs for [`dispatch_augmented`] — one struct because the augmented @@ -70,11 +74,17 @@ pub(super) async fn dispatch_augmented( let trace_id = TraceId::ZERO; + let row_scan = plain_row_scan(&target_task.plan); + let declared_primary_key = match row_scan { + Some(_) => collection_primary_key(state, &target_task)?, + None => None, + }; + let target_plan = row_scan.map(|_| target_task.plan.clone()); let target_resp = dispatch_one(state, target_task, trace_id).await?; reject_data_plane_error(&target_resp)?; // Suppressed source surrogates: tombstones plus copy-ups (a copy-up - // leaves the superseded source row in place, so merging it back would + // leaves the superseded source row in place, so merging it back will // double it). Each catalog handle is a temporary, dropped before the // next `.await`, so nothing non-`Send` is held across a suspend point. let mut tombstoned = state @@ -87,10 +97,29 @@ pub(super) async fn dispatch_augmented( .catalog() .list_clone_copyups(target_collection_key)?, ); - let kv_tombstoned = state + let mut kv_tombstoned = state .credentials .catalog() .list_kv_clone_tombstones(target_collection_key)?; + // A read inside a transaction also hides the source rows the + // transaction's own buffered tombstones and copy-ups hide. + let buffered = crate::control::server::shared::session::ddl_buffer::buffered_clone_suppressions( + target_collection_key, + ); + tombstoned.extend(buffered.surrogates); + kv_tombstoned.extend(buffered.kv_keys); + + // A plain row scan also hides every source row whose primary key a target + // row holds, with or without its mapping or tombstone. + let shadowed = match (row_scan, &target_plan) { + (Some(shape), Some(plan)) => row_keys( + shape, + plan, + target_resp.payload.as_ref(), + declared_primary_key.as_deref(), + ), + _ => HashSet::new(), + }; let mut merged_payload = wrap_single_map_as_array(target_resp.payload.as_ref().to_vec()); @@ -134,6 +163,24 @@ pub(super) async fn dispatch_augmented( } }; + let source_payload = match row_scan { + Some(shape) => drop_shadowed_rows( + shape, + &source_payload, + declared_primary_key.as_deref(), + &shadowed, + ) + .ok_or_else(|| crate::Error::Storage { + engine: "clone_merge".into(), + detail: format!( + "source rows of '{target_collection_key}' are not a msgpack array \ + (len={})", + source_payload.len() + ), + })?, + None => source_payload, + }; + merged_payload = merge_msgpack_arrays(&merged_payload, &source_payload)?; } @@ -167,28 +214,52 @@ pub(super) fn empty_response(state: &SharedState) -> Response { } } +/// Dispatch one task under the transaction it carries. The target task is the +/// caller's own, so it reads that transaction's staged target writes: its +/// copy-ups and inserts. A source task carries no transaction: a write +/// through the clone never stages into the source. async fn dispatch_one( state: &SharedState, task: PhysicalTask, trace_id: TraceId, ) -> crate::Result { - dispatch_to_data_plane( + dispatch_to_data_plane_with_txn( state, task.tenant_id, task.database_id, task.vshard_id, task.plan, trace_id, + task.txn_id, ) .await } +/// The clone collection's declared primary key column, `None` for the default +/// `id`. Target and source share it: a clone copies the source's descriptor. +fn collection_primary_key( + state: &SharedState, + task: &PhysicalTask, +) -> crate::Result> { + let Some(qualified) = extract_collection(&task.plan) else { + return Ok(None); + }; + let key = nodedb_types::CollectionKey::from_qualified_str(task.database_id, qualified)?; + Ok(state + .credentials + .catalog() + .get_collection(task.database_id, task.tenant_id.as_u64(), key.name())? + .and_then(|coll| coll.declared_primary_key)) +} + /// The source surrogate a rewritten document point-read fetches. /// A point-get answers with the row body alone, no surrogate — deciding /// suppression from the plan keeps it consistent with scans. fn point_read_surrogate(plan: &PhysicalPlan) -> Option { match plan { - PhysicalPlan::Document(DocumentOp::PointGet { surrogate, .. }) => Some(surrogate.as_u32()), + PhysicalPlan::Document(DocumentOp::PointGet { surrogate, .. }) => { + surrogate.map(|s| s.as_u32()) + } _ => None, } } diff --git a/nodedb/src/control/server/shared/clone_read/entry.rs b/nodedb/src/control/server/shared/clone_read/entry.rs index 18e2e3e14..22f3549c3 100644 --- a/nodedb/src/control/server/shared/clone_read/entry.rs +++ b/nodedb/src/control/server/shared/clone_read/entry.rs @@ -59,10 +59,11 @@ pub(in crate::control::server) async fn maybe_intercept_clone_read( roles, emitter, } = params; - // If the task carries `system_as_of_ms`, derive query_lsn from that - // wall-clock time; otherwise fall back to the current WAL LSN. + // With `system_as_of_ms`, query_lsn is the highest LSN committed by that + // time, and a time before the oldest retained anchor is refused. Without + // it, query_lsn is the current WAL LSN. let (query_lsn, query_ms) = if let Some(as_of_ms) = extract_system_as_of_ms(Some(&task.plan)) { - let lsn = state.ms_to_lsn(as_of_ms); + let lsn = state.ms_to_lsn(as_of_ms)?; (lsn, Some(as_of_ms)) } else { let lsn = state.wal.next_lsn(); @@ -74,7 +75,7 @@ pub(in crate::control::server) async fn maybe_intercept_clone_read( query_ms, }; - let Some(outcome) = resolve_read(state, task.clone(), tenant_id, &resolve_params)? else { + let Some(outcome) = resolve_read(state, task.clone(), tenant_id, &resolve_params).await? else { return Ok(CloneReadOutcome::Passthrough); }; diff --git a/nodedb/src/control/server/shared/clone_read/mod.rs b/nodedb/src/control/server/shared/clone_read/mod.rs index 07b999816..0e0cc4af9 100644 --- a/nodedb/src/control/server/shared/clone_read/mod.rs +++ b/nodedb/src/control/server/shared/clone_read/mod.rs @@ -9,6 +9,7 @@ mod dispatch; mod entry; mod merge; +mod shadow; mod temporal; pub(in crate::control::server) use entry::{ diff --git a/nodedb/src/control/server/shared/clone_read/shadow.rs b/nodedb/src/control/server/shared/clone_read/shadow.rs new file mode 100644 index 000000000..d8c306c75 --- /dev/null +++ b/nodedb/src/control/server/shared/clone_read/shadow.rs @@ -0,0 +1,282 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Primary-key shadowing for clone reads. +//! +//! A source row whose primary key the clone target already holds is hidden, +//! whether or not a copy-up mapping or a KV tombstone records it. A target row +//! written before its mapping or tombstone (a copy-up that failed between the +//! two, or a materializer copy) then reads once. +//! +//! Both result sets reach this node in full, so the keys are compared on the +//! rows themselves. A document row's key is its identity under the rule +//! INSERT applies ([`RowIdentity::of_stored_row`]): the declared primary key's +//! body value, else the `id` body value, else the storage surrogate. The rule +//! reads only the row and the replicated collection descriptor, so every node +//! derives the same keys, with no surrogate binding and no RPC. A KV row +//! carries its key itself. +//! +//! Only a plain row scan qualifies. A projected, computed, or aggregated row +//! does not carry the full body, and its `id` can be a user column. + +use std::collections::HashSet; + +use nodedb_physical::physical_plan::{DocumentOp, KvOp, PhysicalPlan, QueryOp}; +use nodedb_query::msgpack_scan; +use nodedb_types::{RowIdentity, StorageKey}; + +use crate::control::server::response_shape::kv::apply_kv_wrap; + +/// Which key a plain row scan's rows carry. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(super) enum RowScan { + /// `{id, data}` rows: `id` is the storage key, `data` the full body. + Document, + /// Rows carrying their KV `key`. + Kv, +} + +/// The row shape of `plan` when it returns whole rows, else `None`. +/// +/// Gather and post-process wrappers qualify only when they keep every row +/// whole: no projection, computed column, or window function. +pub(super) fn plain_row_scan(plan: &PhysicalPlan) -> Option { + match plan { + PhysicalPlan::Query(QueryOp::Exchange(op)) => plain_row_scan(&op.child), + PhysicalPlan::Query(QueryOp::PostProcess { + input, + projection, + computed_columns, + window_functions, + .. + }) if projection.is_empty() + && computed_columns.is_empty() + && window_functions.is_empty() => + { + plain_row_scan(input) + } + PhysicalPlan::Document(DocumentOp::Scan { + projection, + computed_columns, + window_functions, + .. + }) if projection.is_empty() + && computed_columns.is_empty() + && window_functions.is_empty() => + { + Some(RowScan::Document) + } + PhysicalPlan::Kv(KvOp::Get { .. } | KvOp::BatchGet { .. }) => Some(RowScan::Kv), + PhysicalPlan::Kv(KvOp::Scan { + projection, + computed_columns, + .. + }) if projection.is_empty() && computed_columns.is_empty() => Some(RowScan::Kv), + _ => None, + } +} + +/// Every row's key in a msgpack array of plain scan rows. +/// +/// `plan` shapes a KV point read's bare row first. `declared_primary_key` is +/// the collection's declared key column, `None` for the default `id`. +pub(super) fn row_keys( + shape: RowScan, + plan: &PhysicalPlan, + payload: &[u8], + declared_primary_key: Option<&str>, +) -> HashSet { + let wrapped = super::merge::wrap_single_map_as_array(apply_kv_wrap(plan, payload)); + let mut keys = HashSet::new(); + for_each_row(&wrapped, |row| { + if let Some(key) = row_key(shape, row, declared_primary_key) { + keys.insert(key); + } + }); + keys +} + +/// Drop every row of a msgpack array whose key is in `shadowed`. Returns the +/// payload unchanged when nothing is dropped, and `None` only when a non-empty +/// payload is not a msgpack array. +pub(super) fn drop_shadowed_rows( + shape: RowScan, + payload: &[u8], + declared_primary_key: Option<&str>, + shadowed: &HashSet, +) -> Option> { + if shadowed.is_empty() || payload.is_empty() { + return Some(payload.to_vec()); + } + let (count, _) = msgpack_scan::array_header(payload, 0)?; + let mut kept: Vec<&[u8]> = Vec::with_capacity(count); + let walked = for_each_row(payload, |row| { + let hidden = + row_key(shape, row, declared_primary_key).is_some_and(|k| shadowed.contains(&k)); + if !hidden { + kept.push(row); + } + }); + if walked != count { + return None; + } + if kept.len() == count { + return Some(payload.to_vec()); + } + let mut buf = Vec::with_capacity(payload.len()); + msgpack_scan::write_array_header(&mut buf, kept.len()); + for row in kept { + buf.extend_from_slice(row); + } + Some(buf) +} + +/// Call `visit` with each element of a msgpack array. Returns how many +/// elements were visited, fewer than the header count on a malformed payload. +fn for_each_row<'a>(payload: &'a [u8], mut visit: impl FnMut(&'a [u8])) -> usize { + let Some((count, mut offset)) = msgpack_scan::array_header(payload, 0) else { + return 0; + }; + for visited in 0..count { + let Some(next) = msgpack_scan::skip_value(payload, offset) else { + return visited; + }; + visit(&payload[offset..next]); + offset = next; + } + count +} + +/// One row's key. A document row without a parseable storage key, or a KV row +/// without `key`, has none and is never shadowed. +fn row_key(shape: RowScan, row: &[u8], declared_primary_key: Option<&str>) -> Option { + match shape { + RowScan::Kv => msgpack_scan::extract_field(row, 0, "key") + .and_then(|(start, _)| msgpack_scan::read_str(row, start)) + .map(str::to_string), + RowScan::Document => { + let storage_key = msgpack_scan::extract_field(row, 0, "id") + .and_then(|(start, _)| msgpack_scan::read_str(row, start)) + .and_then(StorageKey::parse)?; + let (start, end) = msgpack_scan::extract_field(row, 0, "data")?; + Some( + RowIdentity::of_stored_row(&row[start..end], declared_primary_key, storage_key) + .into_string(), + ) + } + } +} + +#[cfg(test)] +mod tests { + use super::*; + use nodedb_types::{QualifiedCollection, Value}; + + fn body(fields: &[(&str, &str)]) -> Vec { + let obj = fields + .iter() + .map(|(k, v)| ((*k).to_string(), Value::String((*v).to_string()))) + .collect(); + nodedb_types::value_to_msgpack(&Value::Object(obj)).unwrap() + } + + /// `{id: , data: }` rows, as a plain document scan + /// returns them. + fn scan_rows(rows: &[(u32, Vec)]) -> Vec { + let mut buf = Vec::new(); + msgpack_scan::write_array_header(&mut buf, rows.len()); + for (surrogate, data) in rows { + msgpack_scan::write_map_header(&mut buf, 2); + msgpack_scan::write_kv_str(&mut buf, "id", &format!("{surrogate:08x}")); + msgpack_scan::write_kv_raw(&mut buf, "data", data); + } + buf + } + + fn doc_scan(projection: Vec) -> PhysicalPlan { + PhysicalPlan::Document(DocumentOp::Scan { + collection: QualifiedCollection::new(nodedb_types::DatabaseId::new(1025), "docs"), + limit: usize::MAX, + offset: 0, + sort_keys: Vec::new(), + filters: Vec::new(), + distinct: false, + projection, + computed_columns: Vec::new(), + window_functions: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + prefilter: None, + }) + } + + /// A target row with no copy-up mapping hides its source twin by primary + /// key, with no surrogate binding anywhere: the two rows carry different + /// surrogates. A source row the target does not hold stays. + #[test] + fn target_row_without_mapping_reads_once() { + let plan = doc_scan(Vec::new()); + let shape = plain_row_scan(&plan).unwrap(); + let target = scan_rows(&[(50, body(&[("id", "d1"), ("content", "old1")]))]); + let source = scan_rows(&[ + (10, body(&[("id", "d1"), ("content", "old1")])), + (11, body(&[("id", "d2"), ("content", "old2")])), + ]); + + let shadowed = row_keys(shape, &plan, &target, None); + assert_eq!(shadowed, HashSet::from(["d1".to_string()])); + let kept = drop_shadowed_rows(shape, &source, None, &shadowed).unwrap(); + assert_eq!( + row_keys(shape, &plan, &kept, None), + HashSet::from(["d2".to_string()]) + ); + } + + /// A declared primary key column names the row, not `id`. + #[test] + fn declared_primary_key_names_the_row() { + let plan = doc_scan(Vec::new()); + let target = scan_rows(&[(50, body(&[("sku", "a"), ("id", "x")]))]); + let source = scan_rows(&[(10, body(&[("sku", "a"), ("id", "y")]))]); + let shadowed = row_keys(RowScan::Document, &plan, &target, Some("sku")); + let kept = drop_shadowed_rows(RowScan::Document, &source, Some("sku"), &shadowed).unwrap(); + assert!(row_keys(RowScan::Document, &plan, &kept, Some("sku")).is_empty()); + } + + /// Projected rows are never shadowed: their `id` can be a user column. + #[test] + fn projected_scan_is_not_a_plain_row_scan() { + assert_eq!(plain_row_scan(&doc_scan(vec!["id".into()])), None); + assert_eq!( + plain_row_scan(&doc_scan(Vec::new())), + Some(RowScan::Document) + ); + } + + #[test] + fn kv_rows_shadow_by_key() { + let mut target = Vec::new(); + msgpack_scan::write_array_header(&mut target, 1); + target.extend_from_slice(&msgpack_scan::build_str_map(&[ + ("key", "k1"), + ("value", "new"), + ])); + let mut source = Vec::new(); + msgpack_scan::write_array_header(&mut source, 2); + source.extend_from_slice(&msgpack_scan::build_str_map(&[ + ("key", "k1"), + ("value", "old"), + ])); + source.extend_from_slice(&msgpack_scan::build_str_map(&[ + ("key", "k2"), + ("value", "old"), + ])); + + let plan = doc_scan(Vec::new()); + let shadowed = row_keys(RowScan::Kv, &plan, &target, None); + let kept = drop_shadowed_rows(RowScan::Kv, &source, None, &shadowed).unwrap(); + assert_eq!( + row_keys(RowScan::Kv, &plan, &kept, None), + HashSet::from(["k2".to_string()]) + ); + } +} diff --git a/nodedb/src/control/server/shared/clone_write/document.rs b/nodedb/src/control/server/shared/clone_write/document.rs index 3469da3f1..aab2035ed 100644 --- a/nodedb/src/control/server/shared/clone_write/document.rs +++ b/nodedb/src/control/server/shared/clone_write/document.rs @@ -20,19 +20,23 @@ use nodedb_physical::physical_plan::{DocumentOp, PhysicalPlan}; use nodedb_physical::physical_task::PhysicalTask; use super::entry::CloneWriteOutcome; -use super::probes::{fetch_source_row, probe_row_in_target}; +use super::probes::{ProbeTarget, fetch_source_row, probe_row_in_target}; use super::util::{strip_db_prefix, synthetic_affected_response, write_err}; +use crate::control::server::shared::session::ddl_buffer::buffered_clone_suppressions; +use crate::types::TxnId; /// The clone-relevant shape of one document write. enum DocWriteKind<'a> { Update { document_id: &'a str, /// Target-side surrogate carried by the plan, used to probe the clone. - surrogate: Surrogate, + /// `None` when the key has no target binding. + surrogate: Option, }, Delete { document_id: &'a str, - surrogate: Surrogate, + /// See `Update::surrogate`. + surrogate: Option, }, /// Insert / put / upsert. Carries one id per row the statement writes. Insert { document_ids: Vec<&'a str> }, @@ -108,41 +112,46 @@ fn classify(plan: &PhysicalPlan) -> Option> { } } -/// Resolve the surrogate the source database bound to `document_id`. +/// Resolve the surrogate the source database bound to `document_id`, at the +/// source collection's home through the async routed exchange. /// /// `None` means the source never held that primary key. -fn source_surrogate( +async fn source_surrogate( state: &SharedState, tenant_id: TenantId, source_db_id: DatabaseId, source_coll_qualified: &str, document_id: &str, ) -> crate::Result> { - state - .surrogate_assigner - .lookup( - nodedb_types::CollectionKey::from_qualified_str(source_db_id, source_coll_qualified)?, - tenant_id, - document_id.as_bytes(), - ) - .map_err(|e| write_err(format!("clone write source surrogate lookup: {e}"))) + crate::control::server::surrogate_exchange::lookup_surrogate_routed( + state, + nodedb_types::CollectionKey::from_qualified_str(source_db_id, source_coll_qualified)?, + tenant_id, + document_id.as_bytes(), + crate::types::TraceId::ZERO, + ) + .await + .map_err(|e| write_err(format!("clone write source surrogate lookup: {e}"))) } /// Point a `PointUpdate` plan at the surrogate the copy-up bound in the target /// database. The plan resolved the pk read-only, so before the copy-up there -/// was no target binding to carry and the plan holds `Surrogate::ZERO`. +/// was no target binding to carry and the plan holds `None`. fn retarget_point_update(plan: &mut PhysicalPlan, target: Surrogate) { if let PhysicalPlan::Document(DocumentOp::PointUpdate { surrogate, .. }) = plan { - *surrogate = target; + *surrogate = Some(target); } } -/// Handle Document CoW write interception. +/// Handle Document CoW write interception. Inside transaction `txn_id` the +/// presence probes read its overlay, and a source row the transaction already +/// tombstoned or copied up counts as gone from the source. pub(super) async fn intercept_doc_clone_write( state: &SharedState, task: &mut PhysicalTask, identity: &AuthenticatedIdentity, tenant_id: TenantId, + txn_id: Option, ) -> crate::Result { let Some(write) = classify(&task.plan) else { return Ok(CloneWriteOutcome::Passthrough); @@ -184,6 +193,17 @@ pub(super) async fn intercept_doc_clone_write( source_db_id, origin.source_collection.as_str(), ); + // Source rows this transaction's buffered tombstones and copy-ups hide. + let hidden = buffered_clone_suppressions( + &crate::control::planner::sql_plan_convert::convert::db_qualified(db_id, coll_name), + ) + .surrogates; + let target = ProbeTarget { + tenant_id, + db_id, + collection_qualified: write.collection_qualified, + txn_id, + }; match write.kind { DocWriteKind::Insert { document_ids } => { @@ -194,18 +214,21 @@ pub(super) async fn intercept_doc_clone_write( source_db_id, &source_coll_qualified, document_id, - )? + ) + .await? else { continue; }; // A binding without a live row makes the tombstone a read no-op — // not worth a probe round trip per inserted key. perform_clone_tombstone(TombstoneParams { + tenant_id, state, target_db_id: db_id, target_collection: coll_name, source_surrogate: surrogate, }) + .await .map_err(|e| write_err(format!("clone insert tombstone: {e}")))?; } Ok(CloneWriteOutcome::Passthrough) @@ -215,17 +238,10 @@ pub(super) async fn intercept_doc_clone_write( document_id, surrogate, } => { - let row_in_target = probe_row_in_target( - state, - identity, - tenant_id, - db_id, - write.collection_qualified, - document_id, - surrogate, - ) - .await - .map_err(|e| write_err(format!("clone write probe: {e}")))?; + let row_in_target = + probe_row_in_target(state, identity, target, document_id, surrogate) + .await + .map_err(|e| write_err(format!("clone write probe: {e}")))?; let src = source_surrogate( state, @@ -233,17 +249,20 @@ pub(super) async fn intercept_doc_clone_write( source_db_id, &source_coll_qualified, document_id, - )?; + ) + .await?; // Tombstone regardless of target residency — after DELETE the clone // must never show the source copy again. if let Some(src) = src { perform_clone_tombstone(TombstoneParams { + tenant_id, state, target_db_id: db_id, target_collection: coll_name, source_surrogate: src, }) + .await .map_err(|e| write_err(format!("clone tombstone: {e}")))?; } @@ -253,7 +272,8 @@ pub(super) async fn intercept_doc_clone_write( // The source read decides rows-affected (1 or 0) — a resolved surrogate // is not evidence the row exists, since a surrogate outlives its row. - let source_row = match src { + // A row this transaction already hid is gone from the clone. + let source_row = match src.filter(|src| !hidden.contains(&src.as_u32())) { Some(src) => fetch_source_row( state, identity, @@ -279,17 +299,10 @@ pub(super) async fn intercept_doc_clone_write( document_id, surrogate, } => { - let row_in_target = probe_row_in_target( - state, - identity, - tenant_id, - db_id, - write.collection_qualified, - document_id, - surrogate, - ) - .await - .map_err(|e| write_err(format!("clone write probe: {e}")))?; + let row_in_target = + probe_row_in_target(state, identity, target, document_id, surrogate) + .await + .map_err(|e| write_err(format!("clone write probe: {e}")))?; if row_in_target { return Ok(CloneWriteOutcome::Passthrough); @@ -301,10 +314,15 @@ pub(super) async fn intercept_doc_clone_write( source_db_id, &source_coll_qualified, document_id, - )? + ) + .await? else { return Ok(CloneWriteOutcome::Passthrough); }; + // A row this transaction already hid updates nothing. + if hidden.contains(&src.as_u32()) { + return Ok(CloneWriteOutcome::Passthrough); + } let source_row_bytes = fetch_source_row( state, @@ -335,7 +353,7 @@ pub(super) async fn intercept_doc_clone_write( .map_err(|e| write_err(format!("clone copyup: {e}")))?; // The copied-up row lives under a target surrogate the plan - // could not know: hand it to the passthrough dispatch so the + // cannot know: hand it to the passthrough dispatch so the // UPDATE lands on that row instead of an unbound key. retarget_point_update(&mut task.plan, target_surrogate); Ok(CloneWriteOutcome::Passthrough) diff --git a/nodedb/src/control/server/shared/clone_write/entry.rs b/nodedb/src/control/server/shared/clone_write/entry.rs index 014a73794..1c0c351b1 100644 --- a/nodedb/src/control/server/shared/clone_write/entry.rs +++ b/nodedb/src/control/server/shared/clone_write/entry.rs @@ -5,7 +5,7 @@ //! of those protocols claims is refused outright on a `Shadowed`/ //! `Materializing` clone (see [`refuse_unsupported_clone_write`]). -use nodedb_types::{CloneStatus, CollectionType, TenantId}; +use nodedb_types::{CloneStatus, CollectionType, Lsn, TenantId}; use crate::bridge::envelope::Response; use crate::control::security::identity::AuthenticatedIdentity; @@ -15,7 +15,8 @@ use crate::control::state::SharedState; use nodedb_physical::physical_plan::{DocumentOp, KvOp, PhysicalPlan}; use nodedb_physical::physical_task::PhysicalTask; -use super::util::{strip_db_prefix, write_err}; +use super::util::{strip_db_prefix, synthetic_affected_response, write_err}; +use crate::types::TxnId; /// Outcome of write-path clone interception. pub(in crate::control::server) enum CloneWriteOutcome { @@ -25,6 +26,29 @@ pub(in crate::control::server) enum CloneWriteOutcome { Handled(Response), } +/// Outcome of clone interception for a write inside a transaction. +pub(in crate::control::server) enum TxnCloneWriteOutcome { + /// Stage the write, retargeted when a copy-up moved its row. + Passthrough, + /// The copy-on-write answered the write: it stages nothing. Its + /// tombstones commit with the transaction. + Handled(Response), + /// Stage the write, which now names only the rows the clone's target + /// holds, and add `hidden` to its count: the source-only rows the + /// copy-on-write's tombstones hid. + Narrowed { hidden: u64 }, +} + +/// One copy-on-write step, before the caller's mode decides how the write +/// applies. +enum CloneStep { + Passthrough, + Handled(Response), + /// A KV delete on a shadowed clone: its tombstones are recorded, and the + /// keys the target holds still need removing. + KvDelete(super::kv::KvCloneDelete), +} + /// Intercept a single write task for a cloned collection. /// /// Called once per task from [`super::gate::intercept_and_authorize`], before @@ -35,6 +59,59 @@ pub(in crate::control::server) async fn maybe_intercept_clone_write( identity: &AuthenticatedIdentity, tenant_id: TenantId, ) -> crate::Result { + match intercept(state, task, identity, tenant_id, None).await? { + CloneStep::Passthrough => Ok(CloneWriteOutcome::Passthrough), + CloneStep::Handled(resp) => Ok(CloneWriteOutcome::Handled(resp)), + CloneStep::KvDelete(delete) => Ok(CloneWriteOutcome::Handled( + super::kv::apply_kv_clone_delete(state, task, delete).await?, + )), + } +} + +/// Intercept a single write task for a cloned collection inside transaction +/// `txn_id`, before the write stages. +/// +/// The probes read the transaction's overlay. The tombstones and copy-up +/// mappings are catalog entries: the connection's transaction buffers them +/// (`cow_entry::replicate_async`), so they commit with the write and a rollback +/// drops them. A copy-up's target row is an exact copy of its source row, so +/// it applies at once: a rolled-back transaction leaves the clone reading the +/// same row. A KV delete stages its removal of the target's rows instead of +/// applying it. +pub(in crate::control::server) async fn maybe_intercept_clone_write_in_txn( + state: &SharedState, + task: &mut PhysicalTask, + identity: &AuthenticatedIdentity, + tenant_id: TenantId, + txn_id: TxnId, +) -> crate::Result { + match intercept(state, task, identity, tenant_id, Some(txn_id)).await? { + CloneStep::Passthrough => Ok(TxnCloneWriteOutcome::Passthrough), + CloneStep::Handled(resp) => Ok(TxnCloneWriteOutcome::Handled(resp)), + CloneStep::KvDelete(delete) => match delete.narrowed { + None => Ok(TxnCloneWriteOutcome::Handled(synthetic_affected_response( + state.next_request_id(), + Lsn::new(0), + delete.source_only_hidden, + ))), + Some(narrowed) => { + task.plan = *narrowed; + Ok(TxnCloneWriteOutcome::Narrowed { + hidden: delete.source_only_hidden, + }) + } + }, + } +} + +/// Route a write by plan shape to its engine's copy-on-write protocol. +async fn intercept( + state: &SharedState, + task: &mut PhysicalTask, + identity: &AuthenticatedIdentity, + tenant_id: TenantId, + txn_id: Option, +) -> crate::Result { // Classify first: the shape check borrows the plan, and the document // arm needs it mutably (a copy-up retargets the plan's surrogate). enum Shape { @@ -62,18 +139,28 @@ pub(in crate::control::server) async fn maybe_intercept_clone_write( ) => Shape::KvInsert, _ => Shape::None, }; - match shape { + let outcome = match shape { Shape::Document => { - super::document::intercept_doc_clone_write(state, task, identity, tenant_id).await + super::document::intercept_doc_clone_write(state, task, identity, tenant_id, txn_id) + .await? } Shape::KvMutate => { - super::kv::intercept_kv_clone_write(state, task, identity, tenant_id).await + return super::kv::intercept_kv_clone_write(state, task, identity, tenant_id, txn_id) + .await + .map(|step| match step { + super::kv::KvCloneStep::Passthrough => CloneStep::Passthrough, + super::kv::KvCloneStep::Delete(delete) => CloneStep::KvDelete(delete), + }); } Shape::KvInsert => { - super::kv_insert::intercept_kv_clone_insert(state, task, tenant_id).await + super::kv_insert::intercept_kv_clone_insert(state, task, tenant_id).await? } - Shape::None => refuse_unsupported_clone_write(state, task, tenant_id), - } + Shape::None => refuse_unsupported_clone_write(state, task, tenant_id)?, + }; + Ok(match outcome { + CloneWriteOutcome::Passthrough => CloneStep::Passthrough, + CloneWriteOutcome::Handled(resp) => CloneStep::Handled(resp), + }) } /// Refuse a write shape none of `document`/`kv`/`kv_insert` claims when it diff --git a/nodedb/src/control/server/shared/clone_write/gate.rs b/nodedb/src/control/server/shared/clone_write/gate.rs index f4ce5404a..31dff271e 100644 --- a/nodedb/src/control/server/shared/clone_write/gate.rs +++ b/nodedb/src/control/server/shared/clone_write/gate.rs @@ -7,10 +7,14 @@ //! no way to produce a [`CloneCheckedTask`] itself, so an entry point that //! forgets either clone hook fails to compile instead of silently bypassing it. +use nodedb_cluster::{DescriptorId, DescriptorKind}; use nodedb_physical::physical_task::PhysicalTask; use nodedb_types::{DatabaseId, TenantId}; use crate::bridge::envelope::{PhysicalPlan, Response}; +use crate::control::gateway::version_set::touched_collections; +use crate::control::lease::QueryLeaseScope; +use crate::control::planner::descriptor_set::DescriptorVersionSet; use crate::control::security::audit::AuditEmitter; use crate::control::security::identity::{AuthenticatedIdentity, Permission, required_permission}; use crate::control::security::permission::PermissionStore; @@ -28,37 +32,138 @@ use super::entry::{CloneWriteOutcome, maybe_intercept_clone_write}; /// /// Boxed so this stays small relative to [`Response`]: `AuthorizedTask` wraps /// a `PhysicalTask`, whose `PhysicalPlan` is the largest enum in the crate, so -/// an unboxed field here would force [`CloneCheckedOutcome`] to size itself to +/// an unboxed field here will force [`CloneCheckedOutcome`] to size itself to /// the bigger of the two variants either way. -pub struct CloneCheckedTask(Box); +/// +/// A write carries a descriptor lease on every collection it touches, from +/// authorization until the caller drops the lease after the write's outcome. +/// A descriptor drain waits for that lease on every node, so no write to a +/// drained collection is in flight once the drain returns. +pub struct CloneCheckedTask { + task: Box, + lease: QueryLeaseScope, +} impl CloneCheckedTask { - /// Unwrap into the inner authorization capability, for a caller (e.g. Raft - /// replication) that dispatches through a lower-level path than the ones - /// gated on this type. - pub fn into_authorized(self) -> AuthorizedTask { - *self.0 + /// Unwrap into the authorization capability and the write's descriptor + /// lease. The caller holds the lease until the dispatch returns its + /// outcome; dropping it earlier lets a drain pass a write still in flight. + pub fn into_parts(self) -> (AuthorizedTask, QueryLeaseScope) { + (*self.task, self.lease) } pub fn tenant_id(&self) -> TenantId { - self.0.tenant_id() + self.task.tenant_id() } pub fn database_id(&self) -> DatabaseId { - self.0.database_id() + self.task.database_id() } pub fn vshard_id(&self) -> VShardId { - self.0.vshard_id() + self.task.vshard_id() } pub fn txn_id(&self) -> Option { - self.0.txn_id() + self.task.txn_id() } pub fn plan(&self) -> &PhysicalPlan { - self.0.plan() + self.task.plan() + } +} + +/// Take a descriptor lease on every collection a write-class `task` touches, +/// at the version the planner leases. +/// +/// Every write entry point takes this lease, SQL or not, so the descriptor +/// drain gates them all: a collection under drain refuses the write as +/// `RetryableSchemaChanged`, and the drain waits for writes already holding +/// the lease. A read, and a collection the catalog does not name, take none. +pub async fn write_lease( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + plan: &PhysicalPlan, +) -> crate::Result { + if required_permission(plan) == Permission::Read { + return Ok(QueryLeaseScope::empty()); + } + let mut versions = DescriptorVersionSet::new(); + record_collections( + state, + tenant_id, + database_id, + touched_collections(plan), + &mut versions, + )?; + // Array DDL and cell writes lease the array's descriptor, so an array + // drain gates them like a collection drain gates row writes. + let array = match plan { + PhysicalPlan::Array(op) => Some(op.primary_array()), + PhysicalPlan::ClusterArray(op) => Some(op.array_id()), + _ => None, + }; + if let Some(array) = array { + versions.record(array_descriptor(array), ARRAY_DESCRIPTOR_VERSION); + } + state.acquire_plan_lease_scope(&versions).await +} + +/// Arrays carry no descriptor version. Every array lease and drain uses this +/// one, so a drain on an array covers every write to it. +pub const ARRAY_DESCRIPTOR_VERSION: u64 = 1; + +/// The descriptor an array's lease and drain name. +pub fn array_descriptor(array: &nodedb_array::types::ArrayId) -> DescriptorId { + DescriptorId::new( + array.database_id.as_u64(), + array.tenant_id.as_u64(), + DescriptorKind::Array, + array.name.clone(), + ) +} + +/// Take a descriptor lease on each named collection (database-qualified or +/// bare), for a write whose collections are known without a plan. +pub async fn collections_write_lease( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collections: impl IntoIterator, +) -> crate::Result { + let mut versions = DescriptorVersionSet::new(); + record_collections(state, tenant_id, database_id, collections, &mut versions)?; + state.acquire_plan_lease_scope(&versions).await +} + +/// Record the descriptor of every named collection the catalog holds. +fn record_collections( + state: &SharedState, + tenant_id: TenantId, + database_id: DatabaseId, + collections: impl IntoIterator, + versions: &mut DescriptorVersionSet, +) -> crate::Result<()> { + let catalog = state.credentials.catalog(); + for qualified in collections { + let key = nodedb_types::CollectionKey::from_qualified_str(database_id, &qualified) + .unwrap_or_else(|_| nodedb_types::CollectionKey::from_bare(database_id, &qualified)); + let Some(stored) = catalog.get_collection(database_id, tenant_id.as_u64(), key.name())? + else { + continue; + }; + versions.record( + DescriptorId::new( + database_id.as_u64(), + tenant_id.as_u64(), + DescriptorKind::Collection, + stored.name.clone(), + ), + stored.descriptor_version.max(1), + ); } + Ok(()) } /// Outcome of [`intercept_and_authorize`]. @@ -105,6 +210,8 @@ pub async fn intercept_and_authorize( roles, emitter, } = params; + // Taken before the clone-write hook, which can complete the write itself. + let lease = write_lease(state, task.tenant_id, task.database_id, &task.plan).await?; if let CloneWriteOutcome::Handled(resp) = maybe_intercept_clone_write(state, &mut task, identity, tenant_id).await? { @@ -141,9 +248,10 @@ pub async fn intercept_and_authorize( .ok_or_else(|| crate::Error::Internal { detail: "authorization returned an empty capability set".into(), })?; - Ok(CloneCheckedOutcome::Proceed(CloneCheckedTask(Box::new( - authorized, - )))) + Ok(CloneCheckedOutcome::Proceed(CloneCheckedTask { + task: Box::new(authorized), + lease, + })) } /// Intercept, authorize, and dispatch one task to the Data Plane in one call — diff --git a/nodedb/src/control/server/shared/clone_write/kv.rs b/nodedb/src/control/server/shared/clone_write/kv.rs index 1fefc79da..747205f30 100644 --- a/nodedb/src/control/server/shared/clone_write/kv.rs +++ b/nodedb/src/control/server/shared/clone_write/kv.rs @@ -4,201 +4,86 @@ use std::sync::Arc; -use nodedb_types::{CloneStatus, Lsn, TenantId}; +use nodedb_types::{CloneStatus, TenantId}; +use crate::bridge::envelope::Response; use crate::control::clone::copyup::{KvCopyUpParams, perform_kv_clone_copyup}; use crate::control::clone::tombstone::{KvTombstoneParams, perform_kv_clone_tombstone}; use crate::control::security::audit::ArcAuditEmitter; use crate::control::security::identity::{AuthenticatedIdentity, Permission}; use crate::control::server::shared::authorization::authorize_collection; +use crate::control::server::shared::session::ddl_buffer::buffered_clone_suppressions; use crate::control::server::shared::sql::staging_predicates::require_affected_count; use crate::control::state::SharedState; +use crate::types::TxnId; use nodedb_physical::physical_plan::{KvOp, PhysicalPlan}; use nodedb_physical::physical_task::PhysicalTask; -use super::entry::CloneWriteOutcome; -use super::probes::{dispatch_data_plane_raw, fetch_kv_source_value, probe_kv_key_in_target}; +use super::probes::{ + ProbeTarget, RawTarget, dispatch_data_plane_raw, fetch_kv_source_value, probe_kv_key_in_target, +}; use super::util::{strip_db_prefix, synthetic_affected_response, write_err}; -/// Handle KV CoW write interception (FieldSet / Delete). +/// What the copy-on-write decided for a KV delete on a shadowed clone. Every +/// key's tombstone is already recorded. +pub(super) struct KvCloneDelete { + /// The delete narrowed to the keys the clone's target holds, which it + /// still removes. `None` when the target holds none of them. Boxed: a + /// plan dwarfs every other step the enums carrying this hold. + pub narrowed: Option>, + /// Source-only keys the tombstones hid. The delete removed these rows + /// from the clone's view, and the tombstones report no count of their own. + pub source_only_hidden: u64, +} + +/// One KV copy-on-write step. +pub(super) enum KvCloneStep { + /// Dispatch the write as planned. + Passthrough, + /// A delete on a shadowed clone. The caller's mode removes the target's + /// rows. + Delete(KvCloneDelete), +} + +/// Handle KV CoW write interception (FieldSet / Delete). Inside transaction +/// `txn_id` the presence probes read its overlay, and a key the transaction +/// already tombstoned counts as absent from the source. pub(super) async fn intercept_kv_clone_write( state: &SharedState, task: &PhysicalTask, identity: &AuthenticatedIdentity, tenant_id: TenantId, -) -> crate::Result { - let (collection_qualified, kv_key, is_delete) = match &task.plan { + txn_id: Option, +) -> crate::Result { + let (collection_qualified, kv_key) = match &task.plan { PhysicalPlan::Kv(KvOp::FieldSet { collection, key, .. - }) => (collection.as_str(), key.clone(), false), + }) => (collection.as_str(), key.clone()), PhysicalPlan::Kv(KvOp::Delete { collection, keys, - rls_write_check, returning, + rls_write_check, rls_filters, - // The tombstone path answers a count, not a sync ack. A Lite KV - // push that lands here moves its stream mark on its own, after - // this returns `Handled`. - provenance: _, + .. }) => { - // Delete may have multiple keys; handle each. We serialize here - // (one tombstone per key) and return Handled with synthetic OK. - let collection_qualified = collection.as_str(); - let db_id = task.database_id; - let coll_name = strip_db_prefix(db_id, collection_qualified); - - let catalog = state.credentials.catalog(); - - let desc = catalog - .get_collection(db_id, tenant_id.as_u64(), coll_name) - .map_err(|e| write_err(format!("clone kv delete: get_collection: {e}")))?; - let Some(desc) = desc else { - return Ok(CloneWriteOutcome::Passthrough); - }; - let Some(ref origin) = desc.cloned_from else { - return Ok(CloneWriteOutcome::Passthrough); - }; - match desc.clone_status { - CloneStatus::Materialized => return Ok(CloneWriteOutcome::Passthrough), - CloneStatus::Shadowed | CloneStatus::Materializing { .. } => {} - } - // A row this delete hides only by tombstone lives in the source, - // so the clone has no stored pre-image to project for it. The - // reply below is a synthesized count, which the RETURNING renderer - // cannot decode as rows; refusing beats answering the wrong shape. - if returning.is_some() { - return Err(crate::Error::BadRequest { - detail: "RETURNING is not supported on a DELETE against a shadowed clone: \ - rows hidden by tombstone have no stored pre-image to project. \ - Materialize the clone first, or SELECT the rows before deleting." - .to_string(), - }); - } - - let emitter = ArcAuditEmitter(Arc::clone(&state.audit)); - authorize_collection( + return intercept_kv_clone_delete( + state, + task, identity, - origin.source_database, - &origin.source_collection, - Permission::Read, - &state.permissions, - &state.roles, - &emitter, - )?; - - // Split each key into one of two paths: - // • key absent in target (source-only) → record a tombstone - // so future scans hide the source row. - // • key present in target (already copied up or written - // in this clone) → dispatch a real KV Delete to remove - // the target row, then ALSO record a tombstone so any - // surviving source row remains hidden after deletion. - // - // Tombstoning unconditionally for target-resident keys is - // safe: the source row (if any) must always be hidden in - // this clone after the user has issued a DELETE. - // `source_only_hidden` counts keys the tombstone alone removed - // from this clone's view: absent from the target but present in - // the source. Those are rows this DELETE removed just as much as - // the target-resident ones, and the tombstone write reports no - // count of its own, so the source read is what makes the total - // honest. A key in neither target nor source removed nothing. - let mut keys_to_dispatch: Vec> = Vec::new(); - let mut source_only_hidden = 0u64; - let source_db_id = origin.source_database; - let source_coll_qualified = - crate::control::planner::sql_plan_convert::convert::db_qualified( - source_db_id, - origin.source_collection.as_str(), - ); - for key in keys { - let key_str = String::from_utf8_lossy(key).into_owned(); - let key_in_target = probe_kv_key_in_target( - state, - identity, + KvDeleteTarget { tenant_id, - db_id, - collection_qualified, - key, - ) - .await - .map_err(|e| write_err(format!("clone kv delete probe: {e}")))?; - - if !key_in_target { - let source_value = fetch_kv_source_value( - state, - identity, - tenant_id, - source_db_id, - &source_coll_qualified, - key, - ) - .await - .map_err(|e| write_err(format!("clone kv delete source probe: {e}")))?; - if source_value.is_some() { - source_only_hidden += 1; - } - } - - perform_kv_clone_tombstone(KvTombstoneParams { - state, - target_db_id: db_id, - target_collection: coll_name, - kv_key: key_str, - }) - .map_err(|e| write_err(format!("clone kv tombstone: {e}")))?; - - if key_in_target { - keys_to_dispatch.push(key.clone()); - } - } - - if !keys_to_dispatch.is_empty() { - // Dispatch a real Delete for keys that exist in target. - let delete_plan = PhysicalPlan::Kv(KvOp::Delete { - collection: nodedb_types::QualifiedCollection::from_stored( - collection_qualified.to_string(), - ), - keys: keys_to_dispatch, - // The narrowed delete is the same statement's write, so - // it carries the same compiled predicate: dropping it - // here would launder a governed delete into an - // ungoverned one for exactly the keys that resolve to - // real target rows. - rls_write_check: rls_write_check.clone(), - // Same statement, same projection and read gate. - returning: returning.clone(), - rls_filters: rls_filters.clone(), - provenance: None, - }); - let vshard_id = - nodedb_types::CollectionKey::from_qualified_str(db_id, collection_qualified)? - .vshard(); - let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, db_id, delete_plan) - .await - .map_err(|e| write_err(format!("clone kv delete dispatch: {e}")))?; - - // Total = keys removed from the target + keys the tombstones - // hid in the source. Re-wrap so the client sees one count for - // the one statement it issued. - let dispatched = require_affected_count(resp.payload.as_ref()) - .map_err(|e| write_err(format!("clone kv delete count: {e}")))?; - return Ok(CloneWriteOutcome::Handled(synthetic_affected_response( - state.next_request_id(), - resp.watermark_lsn, - dispatched + source_only_hidden, - ))); - } - - let synthetic_resp = synthetic_affected_response( - state.next_request_id(), - Lsn::new(0), - source_only_hidden, - ); - return Ok(CloneWriteOutcome::Handled(synthetic_resp)); + collection, + keys, + has_returning: returning.is_some(), + rls_write_check, + rls_filters, + txn_id, + }, + ) + .await; } - _ => return Ok(CloneWriteOutcome::Passthrough), + _ => return Ok(KvCloneStep::Passthrough), }; // FieldSet path: check clone status, copy-up if needed. @@ -211,14 +96,14 @@ pub(super) async fn intercept_kv_clone_write( .get_collection(db_id, tenant_id.as_u64(), coll_name) .map_err(|e| write_err(format!("clone kv write: get_collection: {e}")))?; let Some(desc) = desc else { - return Ok(CloneWriteOutcome::Passthrough); + return Ok(KvCloneStep::Passthrough); }; let Some(ref origin) = desc.cloned_from else { - return Ok(CloneWriteOutcome::Passthrough); + return Ok(KvCloneStep::Passthrough); }; match desc.clone_status { - CloneStatus::Materialized => return Ok(CloneWriteOutcome::Passthrough), + CloneStatus::Materialized => return Ok(KvCloneStep::Passthrough), CloneStatus::Shadowed | CloneStatus::Materializing { .. } => {} } @@ -233,15 +118,15 @@ pub(super) async fn intercept_kv_clone_write( &emitter, )?; - // FieldSet is not a delete. - let _ = is_delete; - let key_in_target = probe_kv_key_in_target( state, identity, - tenant_id, - db_id, - collection_qualified, + ProbeTarget { + tenant_id, + db_id, + collection_qualified, + txn_id, + }, &kv_key, ) .await @@ -249,7 +134,19 @@ pub(super) async fn intercept_kv_clone_write( if key_in_target { // Row exists in target — let the normal FieldSet proceed. - return Ok(CloneWriteOutcome::Passthrough); + return Ok(KvCloneStep::Passthrough); + } + + let kv_key_str = String::from_utf8_lossy(&kv_key).into_owned(); + // A key this transaction already tombstoned is gone from the clone: the + // FieldSet finds no row, as the committed tombstone will make it find. + let qualified = + crate::control::planner::sql_plan_convert::convert::db_qualified(db_id, coll_name); + if buffered_clone_suppressions(&qualified) + .kv_keys + .contains(&kv_key_str) + { + return Ok(KvCloneStep::Passthrough); } // Fetch source KV row and copy it up to target. @@ -271,11 +168,9 @@ pub(super) async fn intercept_kv_clone_write( let Some(source_value) = source_value else { // Row absent in source — let normal FieldSet run (no-op or error from DP). - return Ok(CloneWriteOutcome::Passthrough); + return Ok(KvCloneStep::Passthrough); }; - let kv_key_str = String::from_utf8_lossy(&kv_key).into_owned(); - perform_kv_clone_copyup(KvCopyUpParams { state, tenant_id, @@ -291,13 +186,221 @@ pub(super) async fn intercept_kv_clone_write( // now-superseded source row. The copy-up wrote the row to the target // and the FieldSet will overwrite it; the source copy must be hidden. perform_kv_clone_tombstone(KvTombstoneParams { + tenant_id, state, target_db_id: db_id, target_collection: coll_name, kv_key: kv_key_str, }) + .await .map_err(|e| write_err(format!("clone kv tombstone after copyup: {e}")))?; // Fall through: let the original FieldSet dispatch to the target. - Ok(CloneWriteOutcome::Passthrough) + Ok(KvCloneStep::Passthrough) +} + +/// The clone collection, keys and write gate a KV delete names. +struct KvDeleteTarget<'a> { + tenant_id: TenantId, + collection: &'a nodedb_types::QualifiedCollection, + keys: &'a [Vec], + has_returning: bool, + rls_write_check: &'a nodedb_types::RlsWriteCheck, + rls_filters: &'a [u8], + txn_id: Option, +} + +/// Record a tombstone for every key a KV delete on a shadowed clone names, +/// and sort the keys by where they live. +async fn intercept_kv_clone_delete( + state: &SharedState, + task: &PhysicalTask, + identity: &AuthenticatedIdentity, + target: KvDeleteTarget<'_>, +) -> crate::Result { + let KvDeleteTarget { + tenant_id, + collection, + keys, + has_returning, + rls_write_check, + rls_filters, + txn_id, + } = target; + let collection_qualified = collection.as_str(); + let db_id = task.database_id; + let coll_name = strip_db_prefix(db_id, collection_qualified); + + let catalog = state.credentials.catalog(); + + let desc = catalog + .get_collection(db_id, tenant_id.as_u64(), coll_name) + .map_err(|e| write_err(format!("clone kv delete: get_collection: {e}")))?; + let Some(desc) = desc else { + return Ok(KvCloneStep::Passthrough); + }; + let Some(ref origin) = desc.cloned_from else { + return Ok(KvCloneStep::Passthrough); + }; + match desc.clone_status { + CloneStatus::Materialized => return Ok(KvCloneStep::Passthrough), + CloneStatus::Shadowed | CloneStatus::Materializing { .. } => {} + } + // A row this delete hides only by tombstone lives in the source, so the + // clone has no stored pre-image to project for it. The reply is a + // synthesized count, which the RETURNING renderer cannot decode as rows; + // refusing beats answering the wrong shape. + if has_returning { + return Err(crate::Error::BadRequest { + detail: "RETURNING is not supported on a DELETE against a shadowed clone: \ + rows hidden by tombstone have no stored pre-image to project. \ + Materialize the clone first, or SELECT the rows before deleting." + .to_string(), + }); + } + + let emitter = ArcAuditEmitter(Arc::clone(&state.audit)); + authorize_collection( + identity, + origin.source_database, + &origin.source_collection, + Permission::Read, + &state.permissions, + &state.roles, + &emitter, + )?; + + // Split each key into one of two paths: + // • key absent in target (source-only) → record a tombstone so future + // scans hide the source row. + // • key present in target (already copied up or written in this clone) + // → the target row is removed, and a tombstone is ALSO recorded so any + // surviving source row remains hidden after deletion. + // + // Tombstoning unconditionally for target-resident keys is safe: the + // source row (if any) must always be hidden in this clone after the user + // has issued a DELETE. `source_only_hidden` counts keys the tombstone + // alone removed from this clone's view: absent from the target but present + // in the source, and not already tombstoned by this transaction. A key in + // neither target nor source removed nothing. + let mut keys_in_target: Vec> = Vec::new(); + let mut source_only_hidden = 0u64; + let source_db_id = origin.source_database; + let source_coll_qualified = crate::control::planner::sql_plan_convert::convert::db_qualified( + source_db_id, + origin.source_collection.as_str(), + ); + let qualified = + crate::control::planner::sql_plan_convert::convert::db_qualified(db_id, coll_name); + let already_hidden = buffered_clone_suppressions(&qualified).kv_keys; + for key in keys { + let key_str = String::from_utf8_lossy(key).into_owned(); + let key_in_target = probe_kv_key_in_target( + state, + identity, + ProbeTarget { + tenant_id, + db_id, + collection_qualified, + txn_id, + }, + key, + ) + .await + .map_err(|e| write_err(format!("clone kv delete probe: {e}")))?; + + if !key_in_target && !already_hidden.contains(&key_str) { + let source_value = fetch_kv_source_value( + state, + identity, + tenant_id, + source_db_id, + &source_coll_qualified, + key, + ) + .await + .map_err(|e| write_err(format!("clone kv delete source probe: {e}")))?; + if source_value.is_some() { + source_only_hidden += 1; + } + } + + perform_kv_clone_tombstone(KvTombstoneParams { + tenant_id, + state, + target_db_id: db_id, + target_collection: coll_name, + kv_key: key_str, + }) + .await + .map_err(|e| write_err(format!("clone kv tombstone: {e}")))?; + + if key_in_target { + keys_in_target.push(key.clone()); + } + } + + let narrowed = (!keys_in_target.is_empty()).then(|| { + Box::new(PhysicalPlan::Kv(KvOp::Delete { + collection: collection.clone(), + keys: keys_in_target, + // The narrowed delete is the same statement's write, so it carries + // the same compiled predicate: dropping it here will launder a + // governed delete into an ungoverned one for exactly the keys that + // resolve to real target rows. + rls_write_check: rls_write_check.clone(), + // `has_returning` refused a projection above. + returning: None, + rls_filters: rls_filters.to_vec(), + // The narrowed delete answers a count, not a sync ack. + provenance: None, + })) + }); + Ok(KvCloneStep::Delete(KvCloneDelete { + narrowed, + source_only_hidden, + })) +} + +/// Outside a transaction: remove the target's rows of `delete` at once and +/// answer the statement's whole count. +pub(super) async fn apply_kv_clone_delete( + state: &SharedState, + task: &PhysicalTask, + delete: KvCloneDelete, +) -> crate::Result { + let KvCloneDelete { + narrowed, + source_only_hidden, + } = delete; + let Some(narrowed) = narrowed else { + return Ok(synthetic_affected_response( + state.next_request_id(), + crate::types::Lsn::new(0), + source_only_hidden, + )); + }; + let resp = dispatch_data_plane_raw( + state, + RawTarget { + tenant_id: task.tenant_id, + vshard_id: task.vshard_id, + database_id: task.database_id, + txn_id: None, + }, + *narrowed, + ) + .await + .map_err(|e| write_err(format!("clone kv delete dispatch: {e}")))?; + + // Total = keys removed from the target + keys the tombstones hid in the + // source. Re-wrap so the client sees one count for the one statement it + // issued. + let dispatched = require_affected_count(resp.payload.as_ref()) + .map_err(|e| write_err(format!("clone kv delete count: {e}")))?; + Ok(synthetic_affected_response( + state.next_request_id(), + resp.watermark_lsn, + dispatched + source_only_hidden, + )) } diff --git a/nodedb/src/control/server/shared/clone_write/kv_insert.rs b/nodedb/src/control/server/shared/clone_write/kv_insert.rs index 91eae9870..f19268729 100644 --- a/nodedb/src/control/server/shared/clone_write/kv_insert.rs +++ b/nodedb/src/control/server/shared/clone_write/kv_insert.rs @@ -44,7 +44,7 @@ fn classify(plan: &PhysicalPlan) -> Option<(&str, Vec<&[u8]>)> { } } -/// Hide the source rows an insert into a clone would otherwise duplicate. +/// Hide the source rows an insert into a clone will otherwise duplicate. pub(super) async fn intercept_kv_clone_insert( state: &SharedState, task: &PhysicalTask, @@ -75,11 +75,13 @@ pub(super) async fn intercept_kv_clone_insert( for key in keys { let kv_key = String::from_utf8_lossy(key).into_owned(); perform_kv_clone_tombstone(KvTombstoneParams { + tenant_id, state, target_db_id: db_id, target_collection: coll_name, kv_key, }) + .await .map_err(|e| write_err(format!("clone kv insert tombstone: {e}")))?; } diff --git a/nodedb/src/control/server/shared/clone_write/mod.rs b/nodedb/src/control/server/shared/clone_write/mod.rs index a41b84927..0316692e4 100644 --- a/nodedb/src/control/server/shared/clone_write/mod.rs +++ b/nodedb/src/control/server/shared/clone_write/mod.rs @@ -23,8 +23,12 @@ mod util; // being fully `pub` (matching `dispatch_authorized_to_data_plane` and friends, // which integration tests call directly) does not weaken the guarantee: the // type can still only be constructed by `intercept_and_authorize`. -pub(in crate::control::server) use entry::{CloneWriteOutcome, maybe_intercept_clone_write}; +pub(in crate::control::server) use entry::{ + CloneWriteOutcome, TxnCloneWriteOutcome, maybe_intercept_clone_write, + maybe_intercept_clone_write_in_txn, +}; pub use gate::{ - CloneCheckedOutcome, CloneCheckedTask, InterceptAndAuthorizeParams, intercept_and_authorize, - intercept_authorize_and_dispatch, + ARRAY_DESCRIPTOR_VERSION, CloneCheckedOutcome, CloneCheckedTask, InterceptAndAuthorizeParams, + array_descriptor, collections_write_lease, intercept_and_authorize, + intercept_authorize_and_dispatch, write_lease, }; diff --git a/nodedb/src/control/server/shared/clone_write/probes.rs b/nodedb/src/control/server/shared/clone_write/probes.rs index 071feb778..8489dab95 100644 --- a/nodedb/src/control/server/shared/clone_write/probes.rs +++ b/nodedb/src/control/server/shared/clone_write/probes.rs @@ -11,23 +11,43 @@ use nodedb_types::{DatabaseId, QualifiedCollection, Surrogate, TenantId}; use crate::bridge::envelope::{Priority, Request, Response, Status}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; -use crate::types::{ReadConsistency, RequestId, TraceId, VShardId}; +use crate::types::{ReadConsistency, RequestId, TraceId, TxnId, VShardId}; use nodedb_physical::physical_plan::{DocumentOp, KvOp, PhysicalPlan}; +/// The clone collection a presence probe reads, and the transaction it runs +/// in. +#[derive(Clone, Copy)] +pub(super) struct ProbeTarget<'a> { + pub tenant_id: TenantId, + pub db_id: DatabaseId, + pub collection_qualified: &'a str, + /// The transaction whose overlay the probe reads, `None` outside one. + pub txn_id: Option, +} + /// Probe whether `document_id` exists in target storage. /// /// Issues a synchronous PointGet to the local Data Plane and returns `true` -/// if the row is present. Uses `Surrogate::ZERO` when the catalog has no -/// registered surrogate for the PK — the handler will return "not found". +/// if the row is present. A key with no target binding (`None`) names no +/// target row, so it is absent without a read. Inside transaction `txn_id` +/// the probe reads the transaction's overlay: a row it staged is present, +/// and a row it deleted is absent. pub(super) async fn probe_row_in_target( state: &SharedState, identity: &AuthenticatedIdentity, - tenant_id: TenantId, - db_id: DatabaseId, - collection_qualified: &str, + target: ProbeTarget<'_>, document_id: &str, - surrogate: Surrogate, + surrogate: Option, ) -> crate::Result { + if surrogate.is_none() { + return Ok(false); + } + let ProbeTarget { + tenant_id, + db_id, + collection_qualified, + txn_id, + } = target; let plan = PhysicalPlan::Document(DocumentOp::PointGet { collection: QualifiedCollection::from_stored(collection_qualified.to_string()), document_id: document_id.to_string(), @@ -40,7 +60,17 @@ pub(super) async fn probe_row_in_target( let plan = with_caller_rls(state, identity, tenant_id, db_id, plan)?; let vshard_id = nodedb_types::CollectionKey::from_qualified_str(db_id, collection_qualified)?.vshard(); - let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, db_id, plan).await?; + let resp = dispatch_data_plane_raw( + state, + RawTarget { + tenant_id, + vshard_id, + database_id: db_id, + txn_id, + }, + plan, + ) + .await?; Ok(!resp.payload.is_empty() && resp.status == Status::Ok) } @@ -59,7 +89,7 @@ pub(super) async fn fetch_source_row( let plan = PhysicalPlan::Document(DocumentOp::PointGet { collection: QualifiedCollection::from_stored(source_coll_qualified.to_string()), document_id: document_id.to_string(), - surrogate, + surrogate: Some(surrogate), pk_bytes: document_id.as_bytes().to_vec(), rls_filters: Vec::new(), system_time: nodedb_types::SystemTimeScope::Current, @@ -69,7 +99,17 @@ pub(super) async fn fetch_source_row( let vshard_id = nodedb_types::CollectionKey::from_qualified_str(source_db_id, source_coll_qualified)? .vshard(); - let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, source_db_id, plan).await?; + let resp = dispatch_data_plane_raw( + state, + RawTarget { + tenant_id, + vshard_id, + database_id: source_db_id, + txn_id: None, + }, + plan, + ) + .await?; if resp.payload.is_empty() || resp.status != Status::Ok { return Ok(None); } @@ -79,15 +119,19 @@ pub(super) async fn fetch_source_row( /// Probe whether `kv_key` exists in target KV storage. /// /// Issues a KvOp::Get to the local Data Plane and returns `true` if the key -/// is present. +/// is present. Inside a transaction the probe reads its overlay. pub(super) async fn probe_kv_key_in_target( state: &SharedState, identity: &AuthenticatedIdentity, - tenant_id: TenantId, - db_id: DatabaseId, - collection_qualified: &str, + target: ProbeTarget<'_>, kv_key: &[u8], ) -> crate::Result { + let ProbeTarget { + tenant_id, + db_id, + collection_qualified, + txn_id, + } = target; let plan = PhysicalPlan::Kv(KvOp::Get { collection: QualifiedCollection::from_stored(collection_qualified.to_string()), key: kv_key.to_vec(), @@ -99,7 +143,17 @@ pub(super) async fn probe_kv_key_in_target( let plan = with_caller_rls(state, identity, tenant_id, db_id, plan)?; let vshard_id = nodedb_types::CollectionKey::from_qualified_str(db_id, collection_qualified)?.vshard(); - let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, db_id, plan).await?; + let resp = dispatch_data_plane_raw( + state, + RawTarget { + tenant_id, + vshard_id, + database_id: db_id, + txn_id, + }, + plan, + ) + .await?; Ok(!resp.payload.is_empty() && resp.status == Status::Ok) } @@ -120,14 +174,24 @@ pub(super) async fn fetch_kv_source_value( rls_filters: Vec::new(), // Copy-up reads must see every binding in the source — the // post-copy target write reflects the latest source state, and - // a missed source row would silently drop data on the clone. + // a missed source row will silently drop data on the clone. surrogate_ceiling: None, }); let plan = with_caller_rls(state, identity, tenant_id, source_db_id, plan)?; let vshard_id = nodedb_types::CollectionKey::from_qualified_str(source_db_id, source_coll_qualified)? .vshard(); - let resp = dispatch_data_plane_raw(state, tenant_id, vshard_id, source_db_id, plan).await?; + let resp = dispatch_data_plane_raw( + state, + RawTarget { + tenant_id, + vshard_id, + database_id: source_db_id, + txn_id: None, + }, + plan, + ) + .await?; if resp.payload.is_empty() || resp.status != Status::Ok { return Ok(None); } @@ -138,7 +202,7 @@ pub(super) async fn fetch_kv_source_value( /// /// Copy-up reads a row out of the source collection and writes it into the /// clone, where the source's policies no longer govern it. Without this the -/// clone would launder policy-excluded rows into readable ones. Presence probes +/// clone will launder policy-excluded rows into readable ones. Presence probes /// carry the same filters so a row the caller cannot see is not reported as /// present either. /// @@ -176,15 +240,29 @@ fn with_caller_rls( Ok(plan) } +/// Where [`dispatch_data_plane_raw`] sends a plan. +pub(super) struct RawTarget { + pub tenant_id: TenantId, + pub vshard_id: VShardId, + pub database_id: DatabaseId, + /// The transaction whose overlay a read consults, `None` outside one. + pub txn_id: Option, +} + /// Dispatch a plan directly to the local Data Plane, bypassing WAL and Raft. -/// Used only for read probes inside the clone write helper. +/// Used for the read probes and the autocommit KV delete inside the clone +/// write helper. pub(super) async fn dispatch_data_plane_raw( state: &SharedState, - tenant_id: TenantId, - vshard_id: VShardId, - database_id: DatabaseId, + target: RawTarget, plan: PhysicalPlan, ) -> crate::Result { + let RawTarget { + tenant_id, + vshard_id, + database_id, + txn_id, + } = target; let req_id = RequestId::new( state .request_id_counter @@ -207,9 +285,10 @@ pub(super) async fn dispatch_data_plane_raw( user_roles: Vec::new(), user_id: None, statement_digest: None, - txn_id: None, + txn_id, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: crate::bridge::envelope::Admission::Exempt( crate::bridge::envelope::ExemptReason::Read, ), diff --git a/nodedb/src/control/server/shared/cluster_array_dispatch.rs b/nodedb/src/control/server/shared/cluster_array_dispatch.rs index 527cdebd5..6e065ed54 100644 --- a/nodedb/src/control/server/shared/cluster_array_dispatch.rs +++ b/nodedb/src/control/server/shared/cluster_array_dispatch.rs @@ -1,27 +1,32 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Protocol-neutral `ClusterArray` plan dispatch, shared by pgwire and native. +//! Protocol-neutral `ClusterArray` plan dispatch, shared by every transport. //! //! `ClusterArrayOp` plans are handled entirely on the Control Plane by the -//! `ArrayCoordinator` — they must never reach the SPSC bridge or the -//! trigger/DML machinery. Each protocol's dispatch loop intercepts a -//! `PhysicalPlan::ClusterArray` task right after its own in-transaction -//! routing gate, calls [`execute_cluster_array`], then renders the -//! [`ClusterArrayShaped`] outcome in its own wire format. pgwire's adapter -//! lives in `pgwire::handler::routing::cluster_array`; native's lives in -//! `native::dispatch::cluster_array`. +//! `ArrayCoordinator`. They never reach the gateway, the SPSC bridge or the +//! trigger/DML machinery. Each transport's dispatch loop intercepts a +//! `PhysicalPlan::ClusterArray` task after authorization: +//! +//! - pgwire and native call [`execute_cluster_array`] and render the +//! [`ClusterArrayShaped`] outcome in their own wire format. The adapters +//! live in `pgwire::handler::routing::cluster_array` and +//! `native::dispatch::cluster_array`. +//! - HTTP and WebSocket RPC call [`run_cluster_array`] and shape the raw +//! payload the way they shape a gateway payload. use std::sync::Arc; use nodedb_physical::physical_plan::{ClusterArrayOp, PhysicalPlan}; +use nodedb_physical::physical_task::PhysicalTask; use crate::control::cluster::ClusterArrayExecutor; use crate::control::security::auth_context::AuthContext; -use crate::control::server::dispatch_utils::publish_cluster_array_change_events; use crate::control::server::response_shape::compose::{self, ShapeOutcome}; use crate::control::server::response_shape::redaction::QueryRedaction; use crate::control::server::response_shape::schema::OutputSchema; -use crate::control::server::response_shape::types::{DmlOutcome, PlanKind, ShapedRows}; +use crate::control::server::response_shape::types::{ + DmlOutcome, PlanKind, ShapedRows, describe_plan, +}; use crate::control::server::shared::authorization::AuthorizedTask; use crate::control::server::shared::sql::staging_predicates::require_affected_count; use crate::control::state::SharedState; @@ -35,28 +40,39 @@ pub(crate) enum ClusterArrayShaped { Affected(DmlOutcome), } +/// Whether `plan` is a `ClusterArray` plan. Every transport tests this after +/// authorization and runs a match through [`run_cluster_array`] or +/// [`execute_cluster_array`], never through the gateway. +pub(crate) fn is_cluster_array(plan: &PhysicalPlan) -> bool { + matches!(plan, PhysicalPlan::ClusterArray(_)) +} + /// Execute one authorized `ClusterArrayOp` via the `ArrayCoordinator` and -/// shape its payload into protocol-neutral rows or a count-bearing outcome. +/// return its raw payload. /// -/// On a successful `Put`/`Delete` (writes; `Slice`/`Agg` are reads and -/// publish nothing), publishes a CDC change event keyed by the op's own -/// `wal_lsn` — this path never touches the SPSC bridge, so there is no -/// Data-Plane `Response::watermark_lsn` to read the LSN from the way the -/// normal dispatch funnel does (see `publish_cluster_array_change_events`'s -/// own doc comment). -pub(crate) async fn execute_cluster_array( +/// The payload has the shape the local `ArrayOp` counterpart answers with, +/// so a transport shapes it with `describe_plan` of the task's plan, exactly +/// as it shapes a gateway payload. +/// +/// A `Put`/`Delete` publishes no change event here: each shard's committed +/// array write publishes as its replicas apply it. +pub(crate) async fn run_cluster_array( state: &Arc, - auth: &AuthContext, authorized: AuthorizedTask, - projection: Option<&OutputSchema>, -) -> crate::Result { - // Read before the task is consumed: an in-transaction `Slice`/`Agg` - // carries the session's transaction id, and each shard folds that - // transaction's staged cells into its result. - let txn_id = authorized.txn_id(); - let task = authorized.into_physical_task(); - let tenant_id = task.tenant_id; - let database_id = task.database_id; +) -> crate::Result> { + run_trusted_cluster_array(state, authorized.into_physical_task()).await +} + +/// [`run_cluster_array`] for a task whose authority comes from an already +/// admitted operation, such as a read inside a trigger or procedure body's +/// system transaction. +pub(crate) async fn run_trusted_cluster_array( + state: &Arc, + task: PhysicalTask, +) -> crate::Result> { + // An in-transaction `Slice`/`Agg` carries the transaction id, and each + // shard folds that transaction's staged cells into its result. + let txn_id = task.txn_id; let PhysicalPlan::ClusterArray(cluster_op) = task.plan else { return Err(crate::Error::Internal { detail: "authorized task is not a ClusterArray operation".to_owned(), @@ -93,39 +109,29 @@ pub(crate) async fn execute_cluster_array( array = %cluster_op.array_id().name, "cluster array dispatch" ); - let payload_bytes = executor.execute(&cluster_op, txn_id).await?; - - // Publish CDC change event(s) for a successful write. `Slice`/`Agg` are - // reads and publish nothing; `Put`/`Delete` carry their own - // Control-Plane-allocated `wal_lsn` since there is no Data-Plane - // `Response::watermark_lsn` on this coordinator-only path. - let write_lsn = match &cluster_op { - ClusterArrayOp::Put { wal_lsn, .. } | ClusterArrayOp::Delete { wal_lsn, .. } => { - Some(*wal_lsn) - } - ClusterArrayOp::Slice { .. } | ClusterArrayOp::Agg { .. } => None, - }; - if let Some(lsn) = write_lsn { - publish_cluster_array_change_events(state, tenant_id, database_id, &cluster_op, lsn); - } + executor.execute(&cluster_op, txn_id).await +} - let cluster_plan_kind = match &cluster_op { - ClusterArrayOp::Slice { .. } => PlanKind::ArraySlice, - ClusterArrayOp::Agg { .. } => PlanKind::MultiRow, - // The coordinator reports `{"inserted": n}` / `{"deleted": n}`, the - // same count map the local array handlers emit. - ClusterArrayOp::Put { .. } => PlanKind::DmlResult("INSERT"), - ClusterArrayOp::Delete { .. } => PlanKind::DmlResult("DELETE"), - }; - // This coordinator path never builds a `PhysicalPlan`, so the source +/// Execute one authorized `ClusterArrayOp` through [`run_cluster_array`] and +/// shape its payload into protocol-neutral rows or a count-bearing outcome. +pub(crate) async fn execute_cluster_array( + state: &Arc, + auth: &AuthContext, + authorized: AuthorizedTask, + projection: Option<&OutputSchema>, +) -> crate::Result { + let tenant_id = authorized.tenant_id(); + let cluster_plan_kind = describe_plan(authorized.plan()); + // The coordinator path builds no `PhysicalPlan` per shard, so the source // collection comes straight off the op's array name. A single source // means bare-key matching, which is what an array's cell rows carry. - let array_name = match &cluster_op { - ClusterArrayOp::Slice { array_id, .. } - | ClusterArrayOp::Agg { array_id, .. } - | ClusterArrayOp::Put { array_id, .. } - | ClusterArrayOp::Delete { array_id, .. } => array_id.name.clone(), + let PhysicalPlan::ClusterArray(cluster_op) = authorized.plan() else { + return Err(crate::Error::Internal { + detail: "authorized task is not a ClusterArray operation".to_owned(), + }); }; + let array_name = cluster_op.array_id().name.clone(); + let payload_bytes = run_cluster_array(state, authorized).await?; let redaction = QueryRedaction::for_collections(tenant_id, auth, vec![(String::new(), array_name)]); // A cluster array plan projects attribute names only, never a diff --git a/nodedb/src/control/server/shared/ddl/catalog.rs b/nodedb/src/control/server/shared/ddl/catalog.rs index 2eecf15bb..ab0c02851 100644 --- a/nodedb/src/control/server/shared/ddl/catalog.rs +++ b/nodedb/src/control/server/shared/ddl/catalog.rs @@ -2,39 +2,31 @@ //! Protocol-neutral propose-and-apply helper for parent-replicated DDL. //! -//! Neutral twin of the pgwire `catalog_propose::propose_and_apply`: it runs the -//! same three-step ritual (build entry → propose through the metadata raft group -//! → local apply when the proposer reports `LocalOnly`), but yields a -//! protocol-neutral [`DdlError`] instead of a pgwire `PgWireError` so the -//! neutral family handlers carry no pgwire types. -//! -//! Every neutral `CREATE` / `ALTER` handler routes its catalog write through -//! this helper, which makes the step-3 local-apply omission unrepresentable. +//! Neutral twin of the pgwire `catalog_propose::propose_and_apply`: it +//! proposes through the metadata proposer, but yields a protocol-neutral +//! [`DdlError`] instead of a pgwire `PgWireError` so the neutral family +//! handlers carry no pgwire types. use crate::control::catalog_entry::CatalogEntry; -use crate::control::catalog_entry::apply::local::apply_locally_if_needed; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::propose_outcome::ProposeOutcome; use crate::control::state::SharedState; use super::result::DdlError; -/// Propose `entry` through the metadata raft group and, when the proposer -/// reports [`ProposeOutcome::LocalOnly`], apply the entry locally so the -/// primary row and the companion `StoredOwner` row both land in redb. +/// Propose `entry`, on any runtime flavor. On return it applied on this node, +/// with both post-apply lanes awaited, or it is held for COMMIT. /// -/// Callers gate their own single-node-only side effects (in-memory registry -/// refresh) on `needs_local_apply`. A `Buffered` outcome belongs to an open -/// transaction: nothing durable is applied. The Data-Plane registration a -/// collection CREATE/ALTER dispatches next is deliberately NOT gated on the -/// outcome — the transaction encodes its own writes against the shape it sees, -/// and ROLLBACK puts the Data Plane back (`session::ddl_rollback`). -pub fn propose_and_apply( +/// A `Buffered` outcome belongs to an open transaction: nothing durable is +/// applied. The Data-Plane registration a collection CREATE/ALTER dispatches +/// next is deliberately NOT gated on the outcome — the transaction encodes its +/// own writes against the shape it sees, and ROLLBACK puts the Data Plane back +/// (`session::ddl_rollback`). +pub async fn propose_and_apply_async( state: &SharedState, entry: &CatalogEntry, ) -> Result { - let outcome = propose_catalog_entry(state, entry) - .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - apply_locally_if_needed(state, entry, outcome); - Ok(outcome) + propose_catalog_entry_async(state, entry) + .await + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e)) } diff --git a/nodedb/src/control/server/shared/ddl/index_registry.rs b/nodedb/src/control/server/shared/ddl/index_registry.rs index 473581153..01d9167f9 100644 --- a/nodedb/src/control/server/shared/ddl/index_registry.rs +++ b/nodedb/src/control/server/shared/ddl/index_registry.rs @@ -4,11 +4,10 @@ //! //! Every `CREATE [] INDEX` registers its index here and every drop //! removes it, through the metadata raft group so all nodes list and resolve -//! the same set. The single-node fallback writes the catalog row directly, -//! mirroring [`super::owner`]. +//! the same set. use crate::control::catalog_entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::catalog::{IndexKind, StoredIndexRecord}; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId}; @@ -26,7 +25,7 @@ pub struct IndexRegistration<'a> { } /// Register an index so it is listable and droppable by name. -pub fn propose_index_record( +pub async fn propose_index_record( state: &SharedState, registration: &IndexRegistration<'_>, ) -> Result<(), DdlError> { @@ -39,21 +38,15 @@ pub fn propose_index_record( fields: registration.fields.clone(), is_active: true, }; - let entry = CatalogEntry::PutIndexRecord(Box::new(record.clone())); - let outcome = propose_catalog_entry(state, &entry) + let entry = CatalogEntry::PutIndexRecord(Box::new(record)); + propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - state - .credentials - .catalog() - .put_index_record(&record) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } Ok(()) } /// Remove an index's identity record. -pub fn propose_delete_index_record( +pub async fn propose_delete_index_record( state: &SharedState, database_id: DatabaseId, tenant_id: TenantId, @@ -66,14 +59,8 @@ pub fn propose_delete_index_record( name: name.to_string(), collection: collection.to_string(), }; - let outcome = propose_catalog_entry(state, &entry) + propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - state - .credentials - .catalog() - .delete_index_record(database_id.as_u64(), tenant_id.as_u64(), name) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/alert/alter.rs b/nodedb/src/control/server/shared/ddl/neutral/alert/alter.rs index 2d6b7a8c7..bc54e48fd 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/alert/alter.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/alert/alter.rs @@ -23,7 +23,7 @@ fn err(sqlstate: &str, message: String) -> DdlError { DdlError::new(sqlstate, message) } -pub fn alter_alert( +pub async fn alter_alert( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -45,7 +45,7 @@ pub fn alter_alert( _ => return Err(err("42601", "expected ENABLE or DISABLE".to_string())), } - super::replicate::propose_put(state, &def)?; + super::replicate::propose_put(state, &def).await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/alert/create.rs b/nodedb/src/control/server/shared/ddl/neutral/alert/create.rs index 8cb1fe73b..8a421962b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/alert/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/alert/create.rs @@ -45,7 +45,7 @@ pub struct CreateAlertRequest<'a> { } /// Handle `CREATE ALERT`. Converts raw strings to `AlertCondition` and `Vec`. -pub fn create_alert( +pub async fn create_alert( state: &SharedState, identity: &AuthenticatedIdentity, req: &CreateAlertRequest<'_>, @@ -126,10 +126,10 @@ pub fn create_alert( }; // Replicate the row and the registry install to every node. - super::replicate::propose_put(state, &def)?; + super::replicate::propose_put(state, &def).await?; // Emit CRDT sync delta for Lite visibility. Handler-scoped: the apply - // path runs on every node, so emitting there would duplicate the delta. + // path runs on every node, so emitting there will duplicate the delta. { let delta_payload = zerompk::to_msgpack_vec(&def).unwrap_or_default(); let delta = crate::event::crdt_sync::types::OutboundDelta { diff --git a/nodedb/src/control/server/shared/ddl/neutral/alert/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/alert/drop.rs index 3756969fa..4845b1b5a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/alert/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/alert/drop.rs @@ -25,7 +25,7 @@ fn err(sqlstate: &str, message: String) -> DdlError { } /// Existence check used by the `DROP ALERT IF EXISTS` short-circuit in the -/// neutral router. Mirrors the pgwire `exists::alert_exists` helper. +/// neutral router. pub fn alert_exists( state: &SharedState, identity: &AuthenticatedIdentity, @@ -39,7 +39,7 @@ pub fn alert_exists( .is_some() } -pub fn drop_alert( +pub async fn drop_alert( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -61,10 +61,10 @@ pub fn drop_alert( return Err(err("42704", format!("alert '{name}' does not exist"))); } - super::replicate::propose_delete(state, database_id.as_u64(), tenant_id, &name)?; + super::replicate::propose_delete(state, database_id.as_u64(), tenant_id, &name).await?; // Emit CRDT tombstone delta. Handler-scoped: the apply path runs on every - // node, so emitting there would duplicate the delta. + // node, so emitting there will duplicate the delta. { let delta = crate::event::crdt_sync::types::OutboundDelta { database_id, diff --git a/nodedb/src/control/server/shared/ddl/neutral/alert/replicate.rs b/nodedb/src/control/server/shared/ddl/neutral/alert/replicate.rs index ddac16443..0a2a91363 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/alert/replicate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/alert/replicate.rs @@ -7,32 +7,23 @@ //! An alert created on one node evaluates on all. use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::catalog_entry::post_apply::alert_rule as post_apply; use crate::control::state::SharedState; use crate::event::alert::types::AlertDef; use super::super::super::result::DdlError; -use super::super::replicate::propose_and_apply; +use super::super::replicate::propose_and_apply_async; /// Propose the full alert record. CREATE and ALTER both re-put the row. /// /// The leader validates before proposing, so apply never rejects. -pub(super) fn propose_put(state: &SharedState, def: &AlertDef) -> Result<(), DdlError> { +pub(super) async fn propose_put(state: &SharedState, def: &AlertDef) -> Result<(), DdlError> { let entry = CatalogEntry::PutAlertRule(Box::new(def.clone())); - propose_and_apply(state, &entry, || { - state - .credentials - .catalog() - .put_alert_rule(def) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - post_apply::put(def, state); - Ok(()) - }) + propose_and_apply_async(state, &entry).await } /// Propose removal of the alert row, the registry entry, and the hysteresis /// state on every node. -pub(super) fn propose_delete( +pub(super) async fn propose_delete( state: &SharedState, database_id: u64, tenant_id: u64, @@ -43,13 +34,5 @@ pub(super) fn propose_delete( tenant_id, name: name.to_string(), }; - propose_and_apply(state, &entry, || { - state - .credentials - .catalog() - .delete_alert_rule(database_id, tenant_id, name) - .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; - post_apply::delete(database_id, tenant_id, name, state); - Ok(()) - }) + propose_and_apply_async(state, &entry).await } diff --git a/nodedb/src/control/server/shared/ddl/neutral/alert/show.rs b/nodedb/src/control/server/shared/ddl/neutral/alert/show.rs index 5720a65d7..bf9abed89 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/alert/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/alert/show.rs @@ -2,14 +2,11 @@ //! Protocol-neutral `SHOW ALERTS` and `SHOW ALERT STATUS ON ` DDL handlers. //! -//! Ported from the pgwire `ddl::alert::show` handlers. The registry read, the -//! hysteresis-state read, the condition/window/status formatting, and the exact -//! column set are preserved verbatim; only the result construction changed from -//! pgwire `Response` / `QueryResponse` to the protocol-neutral -//! [`DdlResult::Rows`] over [`ShapedRows`]. The mixed text/`int8` column OIDs are -//! reproduced by building `column_types` manually so the RowDescription stays -//! byte-identical (the `int8` cells are emitted as their decimal text form, the -//! same bytes the pgwire `DataRowEncoder::encode_field(&i64)` produced). +//! The registry read, the hysteresis-state read, the condition/window/status +//! formatting, and the exact column set run here. The result is the +//! protocol-neutral [`DdlResult::Rows`] over [`ShapedRows`]. The mixed +//! text/`int8` column OIDs come from building `column_types` manually (the +//! `int8` cells are emitted as their decimal text form). use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs b/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs index d0dbaa05f..56c58afb9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/apikey/create.rs @@ -2,11 +2,9 @@ //! Protocol-neutral `CREATE API KEY` handler. //! -//! Ported from the pgwire `ddl::apikey::create_api_key` handler. The -//! permission checks, owner-subset validation, key preparation, catalog +//! The permission checks, owner-subset validation, key preparation, catalog //! propose / single-node fallback, `install_replicated_key`, and `audit_record` -//! side effects are preserved verbatim; only the result construction changed -//! from pgwire `Response` / `QueryResponse` to the protocol-neutral +//! side effects run here. The result is the protocol-neutral //! [`DdlResult`] over [`ShapedRows`]. use serde_json::{Map, Value as JsonValue}; @@ -25,7 +23,7 @@ use super::parse::{ /// CREATE API KEY FOR [EXPIRES ] [WITH SCOPES ...] [WITH DATABASES (db1, db2)] /// /// Returns the full API key (shown once). Requires admin or self. -pub fn create_api_key( +pub async fn create_api_key( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -124,15 +122,9 @@ pub fn create_api_key( accessible_databases, }); let entry = crate::control::catalog_entry::CatalogEntry::PutApiKey(Box::new(stored.clone())); - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - let catalog = state.credentials.catalog(); - catalog - .put_api_key(&stored) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - state.api_keys.install_replicated_key(&stored); - } state.audit_record( AuditEvent::PrivilegeChange, diff --git a/nodedb/src/control/server/shared/ddl/neutral/apikey/manage.rs b/nodedb/src/control/server/shared/ddl/neutral/apikey/manage.rs index 886e55c1d..9e3d75771 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/apikey/manage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/apikey/manage.rs @@ -3,12 +3,9 @@ //! Protocol-neutral `REVOKE API KEY` / `LIST API KEYS` / `SHOW API KEYS` //! handlers. //! -//! Ported from the pgwire `ddl::apikey::revoke_api_key` / `list_api_keys` -//! handlers. The ownership checks, local pre-check, catalog propose / -//! single-node fallback, `revoke_key`, and `audit_record` side effects are -//! preserved verbatim; only the result construction changed from pgwire -//! `Response` / `QueryResponse` / `Tag` to the protocol-neutral [`DdlResult`] -//! over [`ShapedRows`]. +//! The ownership checks, local pre-check, catalog propose / +//! single-node fallback, `revoke_key`, and `audit_record` side effects run +//! here. The result is the protocol-neutral [`DdlResult`] over [`ShapedRows`]. use serde_json::{Map, Value as JsonValue}; @@ -22,7 +19,7 @@ use super::super::auth_support::require_tenant_admin; use super::parse::err; /// REVOKE API KEY -pub fn revoke_api_key( +pub async fn revoke_api_key( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -53,35 +50,20 @@ pub fn revoke_api_key( let entry = crate::control::catalog_entry::CatalogEntry::RevokeApiKey { key_id: key_id.to_string(), }; - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - let revoked = if outcome.needs_local_apply() { - let catalog = state.credentials.catalog(); - state - .api_keys - .revoke_key(key_id, Some(catalog)) - .map_err(|e| DdlError::from_error(&e))? - } else { - // Cluster mode: trust the committed log index — the - // in-memory cache update runs in a spawned tokio task and - // may not be visible yet. - true - }; - if revoked { - state.audit_record( - AuditEvent::PrivilegeChange, - Some(identity.tenant_id), - &identity.username, - &format!("revoked API key '{key_id}'"), - ); - Ok(vec![DdlResult::Status { - command: "REVOKE API KEY".to_string(), - rows_affected: None, - }]) - } else { - Err(err("42704", format!("API key '{key_id}' not found"))) - } + state.audit_record( + AuditEvent::PrivilegeChange, + Some(identity.tenant_id), + &identity.username, + &format!("revoked API key '{key_id}'"), + ); + Ok(vec![DdlResult::Status { + command: "REVOKE API KEY".to_string(), + rows_affected: None, + }]) } /// LIST API KEYS [FOR ] / SHOW API KEYS [FOR ] diff --git a/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs b/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs index 0d8174447..2923001d1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/apikey/parse.rs @@ -2,9 +2,8 @@ //! Shared parsing helpers for the protocol-neutral API-key DDL family. //! -//! Ported from the pgwire `ddl::apikey` helpers. The scope / database / owner -//! resolution logic is preserved verbatim; only the error construction changed -//! from pgwire `sqlstate_error` to the protocol-neutral [`DdlError`]. +//! The scope / database / owner resolution logic reports errors as the +//! protocol-neutral [`DdlError`]. use nodedb_types::id::DatabaseId; use smallvec::SmallVec; @@ -14,8 +13,7 @@ use crate::control::state::SharedState; use super::super::super::result::DdlError; -/// Construct a [`DdlError`] with the given SQLSTATE and message, preserving the -/// exact codes and messages the pgwire handlers produced. +/// Construct a [`DdlError`] with the given SQLSTATE and message. pub(super) fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/auth_key.rs b/nodedb/src/control/server/shared/ddl/neutral/auth_key.rs index 57db25aef..4331e2088 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/auth_key.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/auth_key.rs @@ -8,11 +8,9 @@ //! LIST AUTH KEYS [FOR AUTH USER 'x'] //! ``` //! -//! Ported from the pgwire `ddl::auth_key_ddl` handlers. The superuser gate, -//! token creation / rotation, listing, and `audit_record` side effects are -//! preserved verbatim; only the result construction changed from pgwire -//! `Response` / `QueryResponse` to the protocol-neutral [`DdlResult`] over -//! [`ShapedRows`]. +//! The superuser gate, token creation / rotation, listing, and +//! `audit_record` side effects run here. The result is the protocol-neutral +//! [`DdlResult`] over [`ShapedRows`]. use serde_json::{Map, Value as JsonValue}; @@ -22,8 +20,7 @@ use crate::control::state::SharedState; use super::super::result::{DdlError, DdlResult}; -/// Construct a [`DdlError`], preserving the exact SQLSTATE codes and messages -/// the pgwire handlers produced. +/// Construct a [`DdlError`] from a SQLSTATE code and a message. fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/auth_support.rs b/nodedb/src/control/server/shared/ddl/neutral/auth_support.rs index 0993eb445..5b3fe6905 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/auth_support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/auth_support.rs @@ -4,9 +4,7 @@ //! the tenant-admin gate, single-tag status construction, role-name parsing, //! and the `IF [NOT] EXISTS` token strippers. //! -//! Folded in verbatim from the pgwire `require_tenant_admin` / `parse_role` -//! helpers and the `parse_utils` strippers; only the result/error type changed -//! from pgwire `PgWireError` to the protocol-neutral [`DdlError`]. +//! Errors are the protocol-neutral [`DdlError`]. use crate::control::security::identity::{AuthenticatedIdentity, Role}; @@ -22,9 +20,7 @@ pub(super) fn status(command: &str) -> Vec { /// Require that the identity is superuser or tenant_admin. /// -/// Folded in verbatim from the pgwire `require_tenant_admin` helper: it does -/// NOT emit an audit record on denial and returns SQLSTATE 42501 with the -/// identical message. +/// It does NOT emit an audit record on denial and returns SQLSTATE 42501. pub(super) fn require_tenant_admin( identity: &AuthenticatedIdentity, action: &str, diff --git a/nodedb/src/control/server/shared/ddl/neutral/auth_user.rs b/nodedb/src/control/server/shared/ddl/neutral/auth_user.rs index 525f2f81b..f577c7e7c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/auth_user.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/auth_user.rs @@ -9,11 +9,9 @@ //! SHOW AUTH USERS //! ``` //! -//! Ported from the pgwire `ddl::auth_user_ddl` handlers. The superuser gate, -//! status parsing, auth-user store mutations, and `audit_record` side effects -//! are preserved verbatim; only the result construction changed from pgwire -//! `Response` / `QueryResponse` / `Tag` to the protocol-neutral [`DdlResult`] -//! over [`ShapedRows`]. +//! The superuser gate, status parsing, auth-user store mutations, and +//! `audit_record` side effects run here. The result is the protocol-neutral +//! [`DdlResult`] over [`ShapedRows`]. use serde_json::{Map, Value as JsonValue}; @@ -24,8 +22,7 @@ use crate::control::state::SharedState; use super::super::result::{DdlError, DdlResult}; -/// Construct a [`DdlError`], preserving the exact SQLSTATE codes and messages -/// the pgwire handlers produced. +/// Construct a [`DdlError`] from a SQLSTATE code and a message. fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/blacklist.rs b/nodedb/src/control/server/shared/ddl/neutral/blacklist.rs index 770ea9f62..74cdfeb78 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/blacklist.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/blacklist.rs @@ -15,11 +15,9 @@ //! the system catalog and reloaded at boot, so without a removal command an //! operator who bans a range has no way back short of editing storage. //! -//! Ported from the pgwire `ddl::blacklist_ddl` handlers. The superuser gate, -//! blacklist-registry mutations, `WITH KILL SESSIONS` session termination, and -//! `audit_record` side effects are preserved verbatim; only the result -//! construction changed from pgwire `Response` / `QueryResponse` / `Tag` to the -//! protocol-neutral [`DdlResult`] over [`ShapedRows`]. +//! The superuser gate, blacklist-registry mutations, `WITH KILL SESSIONS` +//! session termination, and `audit_record` side effects run here. The result +//! is the protocol-neutral [`DdlResult`] over [`ShapedRows`]. use serde_json::{Map, Value as JsonValue}; @@ -29,8 +27,7 @@ use crate::control::state::SharedState; use super::super::result::{DdlError, DdlResult}; -/// Construct a [`DdlError`], preserving the exact SQLSTATE codes and messages -/// the pgwire handlers produced. +/// Construct a [`DdlError`] from a SQLSTATE code and a message. fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/bulk.rs b/nodedb/src/control/server/shared/ddl/neutral/bulk.rs index 33cbda221..68e9e534f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/bulk.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/bulk.rs @@ -6,10 +6,8 @@ //! reads the file under the operator's UID and streams content to //! the server. //! -//! Ported from the pgwire `ddl::bulk` handler; only the result -//! construction changed from a pgwire `Response` error to the -//! protocol-neutral [`DdlError`]. The SQLSTATE and message are preserved -//! verbatim. +//! The result is the protocol-neutral [`DdlError`] carrying the SQLSTATE and +//! message. use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; diff --git a/nodedb/src/control/server/shared/ddl/neutral/change_stream/alter.rs b/nodedb/src/control/server/shared/ddl/neutral/change_stream/alter.rs index 6bed09704..73289de85 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/change_stream/alter.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/change_stream/alter.rs @@ -2,10 +2,8 @@ //! Protocol-neutral `ALTER CHANGE STREAM` DDL handler. //! -//! Ported from the pgwire `ddl::change_stream::alter` handler. The tenant-admin -//! gate and the action matching are preserved verbatim; only the error -//! construction changed from pgwire `PgWireError` to the protocol-neutral -//! [`DdlError`]. +//! The tenant-admin gate and the action matching run here. Errors are the +//! protocol-neutral [`DdlError`]. //! //! Syntax: //! ```sql diff --git a/nodedb/src/control/server/shared/ddl/neutral/change_stream/create.rs b/nodedb/src/control/server/shared/ddl/neutral/change_stream/create.rs index a72684551..4885c6581 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/change_stream/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/change_stream/create.rs @@ -2,12 +2,9 @@ //! Protocol-neutral `CREATE CHANGE STREAM` DDL handler. //! -//! Ported from the pgwire `ddl::change_stream::create` handler. All non-return -//! logic (WITH-clause parsing, `ChangeStreamDef` build, `propose_and_apply` + -//! `LocalOnly` local registry refresh, webhook / kafka task startup, and -//! the `audit_record` call) is preserved verbatim; only the result construction -//! changed from pgwire `Response` / `PgWireError` to the protocol-neutral -//! [`DdlResult`] / [`DdlError`]. +//! The WITH-clause parsing, `ChangeStreamDef` build, `propose_and_apply`, +//! webhook / kafka task startup, and the `audit_record` call run here. +//! The result is the protocol-neutral [`DdlResult`] / [`DdlError`]. //! //! Syntax: //! ```sql @@ -24,14 +21,14 @@ use crate::event::cdc::stream_def::{ use crate::event::webhook::WebhookConfig; use crate::types::DatabaseId; -use super::super::super::catalog::propose_and_apply; +use super::super::super::catalog::propose_and_apply_async; use super::super::super::result::{DdlError, DdlResult}; use super::super::auth_support::{require_tenant_admin, status}; /// Handle `CREATE CHANGE STREAM ON [WITH (...)]` /// /// `with_clause_raw` is the raw text inside the outer `WITH (...)` parens, or empty. -pub fn create_change_stream( +pub async fn create_change_stream( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -134,8 +131,8 @@ pub fn create_change_stream( .as_secs(); // Capture the creating principal's roles onto the subscription record. - // The webhook and Kafka delivery tasks this stream may own run on the - // Event Plane, where no request identity exists and none may be resolved + // The webhook and Kafka delivery tasks this stream can own run on the + // Event Plane, where no request identity exists and none can be resolved // across the Data→Event bus, so the scope their column redaction is keyed // on has to be resolved here and carried by the definition itself. let subscriber_roles = @@ -159,17 +156,15 @@ pub fn create_change_stream( owner: identity.username.clone(), created_at: now, subscriber_roles, + modification_hlc: nodedb_types::Hlc::ZERO, }; let has_webhook = def.webhook.is_configured(); let webhook_config = def.webhook.clone(); let kafka_config = def.kafka.clone(); - let entry = crate::control::catalog_entry::CatalogEntry::PutChangeStream(Box::new(def.clone())); - let outcome = propose_and_apply(state, &entry)?; - if outcome.needs_local_apply() { - state.stream_registry.register(def.clone()); - } + let entry = crate::control::catalog_entry::CatalogEntry::PutChangeStream(Box::new(def)); + propose_and_apply_async(state, &entry).await?; if has_webhook { state diff --git a/nodedb/src/control/server/shared/ddl/neutral/change_stream/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/change_stream/drop.rs index db30f3e3e..f60958483 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/change_stream/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/change_stream/drop.rs @@ -2,12 +2,9 @@ //! Protocol-neutral `DROP CHANGE STREAM` DDL handler. //! -//! Ported from the pgwire `ddl::change_stream::drop` handler. The catalog path -//! (`propose_catalog_entry` + `LocalOnly` local delete / registry / -//! cdc-router cleanup, the local-only webhook task stop, and the `audit_record` -//! call) is preserved verbatim; only the result construction changed from pgwire -//! `Response` / `PgWireError` to the protocol-neutral [`DdlResult`] / -//! [`DdlError`]. +//! The catalog path (`propose_catalog_entry`, the local-only webhook task +//! stop, and the `audit_record` call) runs here. The result is the +//! protocol-neutral [`DdlResult`] / [`DdlError`]. use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; @@ -18,8 +15,8 @@ use super::super::super::result::{DdlError, DdlResult}; use super::super::auth_support::{require_tenant_admin, status}; /// Existence check used by the neutral router's `DROP CHANGE STREAM IF EXISTS` -/// short-circuit. Folded in verbatim from the pgwire `change_stream_exists` -/// guard helper: checks the in-memory stream registry for the identity tenant. +/// short-circuit. It checks the in-memory stream registry for the identity +/// tenant. pub fn change_stream_exists( state: &SharedState, identity: &AuthenticatedIdentity, @@ -31,7 +28,7 @@ pub fn change_stream_exists( } /// Handle `DROP CHANGE STREAM [IF EXISTS] ` -pub fn drop_change_stream( +pub async fn drop_change_stream( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -79,20 +76,11 @@ pub fn drop_change_stream( database_id: database_id.as_u64(), tenant_id, name: name.clone(), + target_hlc: nodedb_types::Hlc::ZERO, }; - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - let _ = catalog - .delete_change_stream(database_id, tenant_id, &name) - .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; - state - .stream_registry - .unregister(database_id, tenant_id, &name); - state - .cdc_router - .remove_buffer(database_id, tenant_id, &name); - } // Stop webhook delivery task if one was running for this stream. // Only the proposing node had a webhook task active; followers diff --git a/nodedb/src/control/server/shared/ddl/neutral/change_stream/show.rs b/nodedb/src/control/server/shared/ddl/neutral/change_stream/show.rs index 1316326a7..d8958676c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/change_stream/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/change_stream/show.rs @@ -2,10 +2,8 @@ //! Protocol-neutral `SHOW CHANGE STREAMS` DDL handler. //! -//! Ported from the pgwire `ddl::change_stream::show` handler. The tenant -//! scoping and the per-stream field extraction are preserved verbatim; only the -//! result construction changed from a pgwire `QueryResponse` to the -//! protocol-neutral [`DdlResult::Rows`]. +//! The tenant scoping and the per-stream field extraction run here. The +//! result is the protocol-neutral [`DdlResult::Rows`]. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/health.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/health.rs index 4824b0371..dd1502a41 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/health.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/health.rs @@ -2,9 +2,7 @@ //! Protocol-neutral peer health DDL command: SHOW PEER HEALTH. //! -//! Ported from the pgwire `ddl::cluster::health` handler. The topology / -//! circuit-breaker reads are preserved verbatim; only the result -//! construction changed from pgwire `Response` / `QueryResponse` to the +//! The topology / circuit-breaker reads run here. The result is the //! protocol-neutral `DdlResult` over `ShapedRows`. use serde_json::{Map, Value as JsonValue}; @@ -14,7 +12,7 @@ use crate::control::server::response_shape::types::{DdlColType, ShapedRows}; use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; -use super::support::{ddl_err, node_state_str}; +use super::support::{cluster_not_started, ddl_err, node_state_str}; /// SHOW PEER HEALTH — circuit breaker state for all known peers. /// @@ -32,19 +30,12 @@ pub fn show_peer_health( let transport = match &state.cluster_transport { Some(t) => t, - None => { - return Err(ddl_err( - "55000", - "cluster mode not enabled (single-node instance)", - )); - } + None => return Err(cluster_not_started("cluster transport")), }; let topo = match &state.cluster_topology { Some(t) => t, - None => { - return Err(ddl_err("55000", "cluster topology not available")); - } + None => return Err(cluster_not_started("cluster topology")), }; let topo = topo.read().unwrap_or_else(|p| p.into_inner()); diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/migration.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/migration.rs index 8002aa2a3..1501888d5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/migration.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/migration.rs @@ -1,11 +1,8 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Protocol-neutral migration DDL command: SHOW MIGRATIONS. -//! -//! Ported from the pgwire `ddl::cluster::migration` handler. The migration -//! tracker read is preserved verbatim; only the result construction changed -//! from pgwire `Response` / `QueryResponse` to the protocol-neutral -//! `DdlResult` over `ShapedRows`. +//! Protocol-neutral migration DDL command: SHOW MIGRATIONS. It reads the +//! migration tracker `wire_cluster_handle` installs, which the rebalancer's +//! migration executor reports to. use serde_json::{Map, Value as JsonValue}; @@ -30,15 +27,12 @@ pub fn show_migrations( )); } - let tracker = match &state.migration_tracker { - Some(t) => t, - None => { - return Err(ddl_err( - "55000", - "cluster mode not enabled (single-node instance)", - )); - } - }; + let tracker = state.migration_tracker.as_ref().ok_or_else(|| { + ddl_err( + "XX000", + "the migration tracker is not installed: this node's cluster is not wired", + ) + })?; let snapshots = tracker.snapshot(); @@ -82,3 +76,54 @@ pub fn show_migrations( rows, ))]) } + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::cluster::test_one_node; + use crate::control::security::identity::{DatabaseSet, Role}; + use crate::types::TenantId; + + fn superuser() -> AuthenticatedIdentity { + AuthenticatedIdentity::new_internal_service( + 0, + "show_migrations_test", + TenantId::new(1), + vec![Role::Superuser], + true, + None, + DatabaseSet::All, + ) + } + + fn row_count(results: &[DdlResult]) -> usize { + match results { + [DdlResult::Rows(shaped)] => shaped.rows.len(), + _ => panic!("SHOW MIGRATIONS returns one row set"), + } + } + + /// A booted node's tracker is the one its migration executor reports + /// to, and SHOW MIGRATIONS lists what it holds. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn show_migrations_lists_the_executors_migrations() { + let cluster = test_one_node::boot().await; + let state = &cluster.state; + let identity = superuser(); + + let empty = show_migrations(state, &identity).expect("SHOW MIGRATIONS on a booted node"); + assert_eq!(row_count(&empty), 0); + + let mut migration = nodedb_cluster::MigrationState::new(7, 1, 1, 1, 2, 500_000); + migration.start_base_copy(10); + state + .migration_tracker + .as_ref() + .expect("wire_cluster_handle installs the tracker") + .record(uuid::Uuid::new_v4(), &migration); + + let listed = show_migrations(state, &identity).expect("SHOW MIGRATIONS"); + assert_eq!(row_count(&listed), 1); + cluster.shutdown().await; + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/mod.rs index 06d04ef61..17cb256cf 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/mod.rs @@ -10,6 +10,7 @@ mod migration; mod raft; mod ranges; mod rebalance_cmd; +mod restore_points; mod routing_hint; mod schema_version; mod support; @@ -20,6 +21,7 @@ pub use migration::show_migrations; pub use raft::{alter_raft_group, show_raft_group, show_raft_groups}; pub use ranges::show_ranges; pub use rebalance_cmd::rebalance; +pub use restore_points::{create_restore_point, show_restore_points}; pub use routing_hint::show_routing; pub use schema_version::show_schema_version; pub use topology::{remove_node, show_cluster, show_node, show_nodes}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/raft.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/raft.rs index 08362b1d4..850f43b52 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/raft.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/raft.rs @@ -3,10 +3,8 @@ //! Protocol-neutral raft group DDL commands: SHOW RAFT GROUPS, SHOW RAFT //! GROUP, ALTER RAFT GROUP. //! -//! Ported from the pgwire `ddl::cluster::raft` handlers. The raft-status / -//! routing reads and the `ALTER RAFT GROUP` `ConfChange` propose are -//! preserved verbatim; only the result construction changed from pgwire -//! `Response` / `QueryResponse` to the protocol-neutral `DdlResult` over +//! The raft-status / routing reads and the `ALTER RAFT GROUP` `ConfChange` +//! propose run here. The result is the protocol-neutral `DdlResult` over //! `ShapedRows`. use serde_json::{Map, Value as JsonValue}; @@ -16,7 +14,7 @@ use crate::control::server::response_shape::types::{DdlColType, ShapedRows}; use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; -use super::support::ddl_err; +use super::support::{cluster_not_started, ddl_err}; /// SHOW RAFT GROUPS — list all Raft groups with leader, term, and status. /// @@ -34,12 +32,7 @@ pub fn show_raft_groups( let status_fn = match state.raft_status_fn.get() { Some(f) => f, - None => { - return Err(ddl_err( - "55000", - "cluster mode not enabled (single-node instance)", - )); - } + None => return Err(cluster_not_started("raft status function")), }; let statuses = status_fn(); @@ -132,12 +125,7 @@ pub fn show_raft_group( let status_fn = match state.raft_status_fn.get() { Some(f) => f, - None => { - return Err(ddl_err( - "55000", - "cluster mode not enabled (single-node instance)", - )); - } + None => return Err(cluster_not_started("raft status function")), }; let statuses = status_fn(); @@ -244,15 +232,9 @@ pub fn alter_raft_group( } }; - let proposer = match state.raft_proposer.get() { - Some(p) => p, - None => { - return Err(ddl_err( - "55000", - "cluster mode not enabled (single-node instance)", - )); - } - }; + let proposer = state + .sync_raft_proposer() + .map_err(|e| DdlError::from_error_in_context("raft proposer", &e))?; let change = nodedb_cluster::ConfChange { change_type, @@ -265,9 +247,7 @@ pub fn alter_raft_group( // Find a vShard that maps to this group to propose through Raft. let routing = match &state.cluster_routing { Some(r) => r, - None => { - return Err(ddl_err("55000", "cluster routing not available")); - } + None => return Err(cluster_not_started("cluster routing table")), }; let routing = routing.read().unwrap_or_else(|p| p.into_inner()); diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/ranges.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/ranges.rs index c7de6a4b5..367271c7b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/ranges.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/ranges.rs @@ -2,18 +2,16 @@ //! Protocol-neutral `SHOW RANGES` — vshard distribution across the cluster. //! -//! Ported from the pgwire `ddl::cluster::ranges` handler. The routing / -//! per-vshard-metrics reads are preserved verbatim; only the result -//! construction changed from pgwire `Response` / `QueryResponse` to the +//! The routing / per-vshard-metrics reads run here. The result is the //! protocol-neutral `DdlResult` over `ShapedRows`. //! //! `qps` and `p99_latency_ms` are `float8` columns. Unlike every other //! migrated column (rendered as `JsonValue::String` decimal/text), these are //! carried as `JsonValue::Number` so the shared pgwire `ddl_encode.rs` can //! encode them through pgwire's native `f64` text path (ryu + -//! `extra_float_digits`) — the exact path the original handler's -//! `encoder.encode_field(&f64)` used. Pre-rendering via `f64::to_string()` -//! would diverge (e.g. `1.0` → pgwire `"1.0"` vs Rust `"1"`). +//! `extra_float_digits`) — the path `encoder.encode_field(&f64)` +//! uses. Pre-rendering via `f64::to_string()` +//! will diverge (e.g. `1.0` → pgwire `"1.0"` vs Rust `"1"`). //! //! Both source values are always finite: `qps` is an integer centihertz //! counter divided by 100.0, and `p99_latency_ms` is an integer microsecond @@ -28,7 +26,7 @@ use crate::control::server::response_shape::types::{DdlColType, ShapedRows}; use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; -use super::support::ddl_err; +use super::support::{cluster_not_started, ddl_err}; fn float_cell(v: f64) -> JsonValue { Number::from_f64(v) @@ -56,12 +54,7 @@ pub fn show_ranges( let routing = match &state.cluster_routing { Some(r) => r, - None => { - return Err(ddl_err( - "55000", - "cluster mode not enabled (single-node instance)", - )); - } + None => return Err(cluster_not_started("cluster routing table")), }; let columns = vec![ diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/rebalance_cmd.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/rebalance_cmd.rs index a9086ecd4..df65a5977 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/rebalance_cmd.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/rebalance_cmd.rs @@ -2,10 +2,8 @@ //! Protocol-neutral rebalance DDL command: REBALANCE. //! -//! Ported from the pgwire `ddl::cluster::rebalance_cmd` handler. The -//! routing / topology reads and the `compute_plan` call are preserved -//! verbatim; only the result construction changed from pgwire `Response` / -//! `QueryResponse` to the protocol-neutral `DdlResult` over `ShapedRows`. +//! The routing / topology reads and the `compute_plan` call run here. +//! The result is the protocol-neutral `DdlResult` over `ShapedRows`. use serde_json::{Map, Value as JsonValue}; @@ -14,7 +12,7 @@ use crate::control::server::response_shape::types::{DdlColType, ShapedRows}; use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; -use super::support::ddl_err; +use super::support::{cluster_not_started, ddl_err}; /// REBALANCE — compute and display a rebalance plan. /// @@ -33,18 +31,11 @@ pub fn rebalance( let routing = match &state.cluster_routing { Some(r) => r, - None => { - return Err(ddl_err( - "55000", - "cluster mode not enabled (single-node instance)", - )); - } + None => return Err(cluster_not_started("cluster routing table")), }; let topo = match &state.cluster_topology { Some(t) => t, - None => { - return Err(ddl_err("55000", "cluster topology not available")); - } + None => return Err(cluster_not_started("cluster topology")), }; let routing = routing.read().unwrap_or_else(|p| p.into_inner()); diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/restore_points.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/restore_points.rs new file mode 100644 index 000000000..5638c5c18 --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/restore_points.rs @@ -0,0 +1,77 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! `CREATE RESTORE POINT` and `SHOW RESTORE POINTS`: cluster restore points. +//! Superuser only. + +use serde_json::{Map, Value as JsonValue}; + +use crate::control::pitr::restore_point::{ + RestorePointError, create_restore_point as create_point, list_restore_points, +}; +use crate::control::security::catalog::restore_points::StoredRestorePoint; +use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::response_shape::types::{DdlColType, ShapedRows}; +use crate::control::state::SharedState; + +use super::super::super::result::{DdlError, DdlResult}; +use super::support::ddl_err; + +/// CREATE RESTORE POINT — take one consistent point across every Raft group. +pub async fn create_restore_point( + state: &SharedState, + identity: &AuthenticatedIdentity, +) -> Result, DdlError> { + require_superuser(identity, "create a restore point")?; + let point = create_point(state).await.map_err(point_err)?; + Ok(vec![rows(&[point])]) +} + +/// SHOW RESTORE POINTS — every cluster restore point, oldest first. +pub fn show_restore_points( + state: &SharedState, + identity: &AuthenticatedIdentity, +) -> Result, DdlError> { + require_superuser(identity, "list restore points")?; + let points = list_restore_points(state).map_err(point_err)?; + Ok(vec![rows(&points)]) +} + +fn require_superuser(identity: &AuthenticatedIdentity, action: &str) -> Result<(), DdlError> { + if identity.is_superuser { + return Ok(()); + } + Err(ddl_err( + "42501", + format!("permission denied: only superuser can {action}"), + )) +} + +fn point_err(e: RestorePointError) -> DdlError { + match e { + RestorePointError::Node(inner) => DdlError::internal(inner.to_string()), + } +} + +fn rows(points: &[StoredRestorePoint]) -> DdlResult { + let columns = vec![ + "id".to_string(), + "hlc".to_string(), + "created_at_ms".to_string(), + ]; + let column_types = vec![DdlColType::Int8, DdlColType::Int8, DdlColType::Int8]; + let rows = points + .iter() + .map(|point| { + let mut row = Map::new(); + for (name, value) in [ + ("id", point.id), + ("hlc", point.hlc), + ("created_at_ms", point.created_at_ms), + ] { + row.insert(name.to_string(), JsonValue::String(value.to_string())); + } + row + }) + .collect(); + DdlResult::Rows(ShapedRows::from_json_rows(columns, column_types, rows)) +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/routing_hint.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/routing_hint.rs index 58291ac80..ce7c7c60d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/routing_hint.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/routing_hint.rs @@ -4,9 +4,7 @@ //! node address mapping so smart clients can cache it and route writes //! directly to the leaseholder, skipping the gateway hop. //! -//! Ported from the pgwire `ddl::cluster::routing_hint` handler. The -//! routing / topology reads are preserved verbatim; only the result -//! construction changed from pgwire `Response` / `QueryResponse` to the +//! The routing / topology reads run here. The result is the //! protocol-neutral `DdlResult` over `ShapedRows`. //! //! Result columns: `vshard_id`, `group_id`, `leaseholder_node_id`, @@ -19,23 +17,18 @@ use crate::control::server::response_shape::types::{DdlColType, ShapedRows}; use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; -use super::support::ddl_err; +use super::support::cluster_not_started; /// SHOW ROUTING — full vshard → leaseholder → address table. /// -/// Any authenticated user may call this (smart-client libs need it). +/// Any authenticated user can call this (smart-client libs need it). pub fn show_routing( state: &SharedState, _identity: &AuthenticatedIdentity, ) -> Result, DdlError> { let routing = match &state.cluster_routing { Some(r) => r, - None => { - return Err(ddl_err( - "55000", - "cluster mode not enabled (single-node instance)", - )); - } + None => return Err(cluster_not_started("cluster routing table")), }; let columns = vec![ diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/schema_version.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/schema_version.rs index 47143629f..641217e3b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/schema_version.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/schema_version.rs @@ -3,10 +3,8 @@ //! Protocol-neutral `SHOW SCHEMA VERSION` — current descriptor version //! visible on this node. //! -//! Ported from the pgwire `ddl::cluster::schema_version` handler. The -//! schema-version / metadata-cache reads are preserved verbatim; only the -//! result construction changed from pgwire `Response` / `QueryResponse` to -//! the protocol-neutral `DdlResult` over `ShapedRows`. +//! The schema-version / metadata-cache reads run here. The result is the +//! protocol-neutral `DdlResult` over `ShapedRows`. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/support.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/support.rs index 2636c1a6d..f72c57d69 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/support.rs @@ -7,15 +7,22 @@ use super::super::super::result::DdlError; /// Build a [`DdlError`] from an ANSI SQLSTATE code and a message. /// -/// Preserves the exact SQLSTATE / message the pgwire cluster handlers -/// produced (via `sqlstate_error`), so error parity stays byte-identical -/// after the migration off the pgwire router. +/// The SQLSTATE and message reach the client unchanged. pub(super) fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } -/// Render a [`nodedb_cluster::NodeState`] the same way the pgwire -/// `topology.rs` handler did — moved verbatim off `pub(super)` in that file. +/// The error for a cluster handle a handler reads before `start_raft` +/// installed it. Every node runs a cluster, a one-node cluster included, and +/// boot runs `start_raft` before any listener opens. +pub(super) fn cluster_not_started(what: &str) -> DdlError { + DdlError::new( + "XX000", + format!("the {what} is not installed: start_raft has not run on this node"), + ) +} + +/// Render a [`nodedb_cluster::NodeState`] as its lowercase name. pub(super) fn node_state_str(state: nodedb_cluster::NodeState) -> &'static str { match state { nodedb_cluster::NodeState::Joining => "joining", diff --git a/nodedb/src/control/server/shared/ddl/neutral/cluster/topology.rs b/nodedb/src/control/server/shared/ddl/neutral/cluster/topology.rs index e46c11976..ef54bc0e2 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/cluster/topology.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/cluster/topology.rs @@ -3,11 +3,9 @@ //! Protocol-neutral cluster topology DDL commands: SHOW NODES, SHOW NODE, //! REMOVE NODE, SHOW CLUSTER. //! -//! Ported from the pgwire `ddl::cluster::topology` handlers. The topology / -//! routing / raft-status reads and the `REMOVE NODE` `set_state` side-effect -//! are preserved verbatim; only the result construction changed from pgwire -//! `Response` / `QueryResponse` to the protocol-neutral `DdlResult` over -//! `ShapedRows`. +//! The topology / routing / raft-status reads and the `REMOVE NODE` +//! `set_state` side-effect run here. The result is the protocol-neutral +//! `DdlResult` over `ShapedRows`. use serde_json::{Map, Value as JsonValue}; @@ -16,7 +14,7 @@ use crate::control::server::response_shape::types::{DdlColType, ShapedRows}; use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; -use super::support::{ddl_err, node_state_str}; +use super::support::{cluster_not_started, ddl_err, node_state_str}; /// SHOW NODES — list all cluster members with state. /// @@ -75,21 +73,7 @@ pub fn show_nodes( rows.push(row); } } - None => { - // Single-node mode: show this node as the only member. - let mut row = Map::new(); - row.insert( - "node_id".to_string(), - JsonValue::String((state.node_id as i64).to_string()), - ); - row.insert( - "address".to_string(), - JsonValue::String("local".to_string()), - ); - row.insert("state".to_string(), JsonValue::String("active".to_string())); - row.insert("raft_groups".to_string(), JsonValue::String(String::new())); - rows.push(row); - } + None => return Err(cluster_not_started("cluster topology")), } Ok(vec![DdlResult::Rows(ShapedRows::from_json_rows( @@ -150,26 +134,7 @@ pub fn show_node( ), ] } - None => { - // Single-node mode: show self info if node_id matches. - if node_id != state.node_id { - return Err(ddl_err( - "42704", - format!( - "node {node_id} not found (single-node instance, this node is {})", - state.node_id - ), - )); - } - let wal_lsn = state.wal.next_lsn().as_u64().saturating_sub(1); - vec![ - ("node_id".to_string(), state.node_id.to_string()), - ("address".to_string(), "local".to_string()), - ("state".to_string(), "active".to_string()), - ("mode".to_string(), "single-node".to_string()), - ("wal_lsn".to_string(), wal_lsn.to_string()), - ] - } + None => return Err(cluster_not_started("cluster topology")), }; let mut rows = Vec::new(); @@ -208,12 +173,7 @@ pub fn remove_node( let topo = match &state.cluster_topology { Some(t) => t, - None => { - return Err(ddl_err( - "55000", - "cluster mode not enabled (single-node instance)", - )); - } + None => return Err(cluster_not_started("cluster topology")), }; let mut topo = topo.write().unwrap_or_else(|p| p.into_inner()); @@ -251,13 +211,14 @@ pub fn show_cluster( let mut props = vec![("node_id", state.node_id.to_string())]; - if let Some(topo) = &state.cluster_topology { + let Some(topo) = &state.cluster_topology else { + return Err(cluster_not_started("cluster topology")); + }; + { let topo = topo.read().unwrap_or_else(|p| p.into_inner()); props.push(("nodes_total", topo.node_count().to_string())); props.push(("nodes_active", topo.active_nodes().len().to_string())); props.push(("topology_version", topo.version().to_string())); - } else { - props.push(("mode", "single-node".to_string())); } if let Some(routing) = &state.cluster_routing { diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/add_column.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/add_column.rs index d2facea4d..2f3acef1e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/add_column.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/add_column.rs @@ -3,12 +3,10 @@ //! `ALTER {TABLE,COLLECTION} ADD [COLUMN] ` — append a column //! to a strict-document / columnar collection's schema. //! -//! Ported verbatim from the pgwire `ddl::collection::alter::add_column` -//! handler; only the result type changed from pgwire `PgWireResult` / -//! `Response` to the protocol-neutral [`DdlResult`] / [`DdlError`]. The +//! The result type is the protocol-neutral [`DdlResult`] / [`DdlError`]. The //! multi-version add (`added_at_version` stamp + `schema.version` bump), -//! duplicate-column check, propose + register, and audit are unchanged, as -//! is the `ALTER TABLE` command tag. +//! duplicate-column check, propose + register, and audit run here, and the +//! command tag is `ALTER TABLE`. use nodedb_types::DatabaseId; @@ -108,8 +106,8 @@ pub(super) async fn alter_table_add_column( // Offload the durable catalog commit (redb `fsync`) off the // Tokio worker so this online ALTER never stalls concurrent // INSERTs on the same runtime. - super::support::propose_and_apply_async(state, entry).await?; - Some(updated) + let outcome = super::support::propose_and_apply_async(state, entry).await?; + Some((updated, outcome)) } else { None } @@ -123,8 +121,8 @@ pub(super) async fn alter_table_add_column( } }; - if let Some(ref coll) = updated { - super::super::register::dispatch_register_from_stored(state, coll) + if let Some((ref coll, outcome)) = updated { + super::super::register::register_proposed_collection(state, outcome, coll) .await .map_err(|e| DdlError::from_error(&e))?; super::strict_schema::recompile_rls_policies(state, coll)?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/alter_type.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/alter_type.rs index cce1c7638..f9d2ee1b7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/alter_type.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/alter_type.rs @@ -3,13 +3,12 @@ //! `ALTER COLLECTION ALTER COLUMN TYPE ` — change a //! column's declared type in a strict-document collection's schema. //! -//! Ported verbatim from the pgwire `ddl::collection::alter::alter_type` -//! handler; only the result type changed to the protocol-neutral -//! [`DdlResult`] / [`DdlError`]. The full-equality gate rejects any type -//! change that requires re-encoding existing rows, including a parameter -//! change (`VECTOR(384)` to `VECTOR(768)`) that shares a discriminant with -//! the current type. Version bump, persist, and audit are unchanged, as is -//! the `ALTER COLLECTION` command tag. +//! The result type is the protocol-neutral [`DdlResult`] / [`DdlError`]. The +//! full-equality gate rejects any type change that requires re-encoding +//! existing rows, including a parameter change (`VECTOR(384)` to +//! `VECTOR(768)`) that shares a discriminant with the current type. Version +//! bump, persist, and audit run here, and the command tag is +//! `ALTER COLLECTION`. use std::str::FromStr; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/dispatch.rs index 6c2e559dc..8d29450a2 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/dispatch.rs @@ -4,10 +4,9 @@ //! protocol-neutral handler. //! //! This is the total match over `AlterCollectionOp` — no `_ =>` fallthrough — -//! so the neutral router owns every `ALTER COLLECTION` sub-command. Ported from -//! the pgwire `router::ast::alter::dispatch_alter_collection`; the ADD COLUMN -//! `col_def` string assembly (` [NOT NULL] [DEFAULT ...]`) is -//! preserved verbatim. `SetOnConflict` continues to route to the +//! so the neutral router owns every `ALTER COLLECTION` sub-command. The ADD +//! COLUMN `col_def` string assembly is ` [NOT NULL] [DEFAULT ...]`. +//! `SetOnConflict` continues to route to the //! `conflict_policy` family; every other variant routes to its sibling handler //! in this directory. @@ -83,6 +82,7 @@ pub async fn dispatch_alter_collection( AlterCollectionOp::OwnerTo { new_owner } => { super::ownership::alter_collection_owner(state, identity, database_id, name, new_owner) + .await } AlterCollectionOp::SetRetention { value } => { @@ -93,10 +93,12 @@ pub async fn dispatch_alter_collection( name, value, ) + .await } AlterCollectionOp::SetAppendOnly => { super::enforcement::alter_collection_set_append_only(state, identity, database_id, name) + .await } AlterCollectionOp::SetLastValueCache { enabled } => { @@ -107,6 +109,7 @@ pub async fn dispatch_alter_collection( name, *enabled, ) + .await } AlterCollectionOp::SetLegalHold { enabled, tag } => { @@ -118,6 +121,7 @@ pub async fn dispatch_alter_collection( *enabled, tag, ) + .await } AlterCollectionOp::AddMaterializedSum { diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/drop_column.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/drop_column.rs index 575aa2eca..15c32e6a7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/drop_column.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/drop_column.rs @@ -3,11 +3,10 @@ //! `ALTER COLLECTION DROP COLUMN ` — remove a column from a //! strict-document collection's schema. //! -//! Ported verbatim from the pgwire `ddl::collection::alter::drop_column` -//! handler; only the result type changed to the protocol-neutral -//! [`DdlResult`] / [`DdlError`]. The dropped-column bookkeeping -//! (`dropped_columns` push + version bump), primary-key guard, persist, -//! and audit are unchanged, as is the `ALTER COLLECTION` command tag. +//! The result type is the protocol-neutral [`DdlResult`] / [`DdlError`]. The +//! dropped-column bookkeeping (`dropped_columns` push + version bump), +//! primary-key guard, persist, and audit run here, and the command tag is +//! `ALTER COLLECTION`. use nodedb_types::DatabaseId; @@ -72,7 +71,7 @@ pub(super) async fn alter_collection_drop_column( // The embedding-model row is keyed by column name and outlives the column // otherwise. A re-added column then inherits the old dimensions and // `strict_dimensions`. - drop_vector_model_row(state, database_id, tenant_id.as_u64(), name, column_name)?; + drop_vector_model_row(state, database_id, tenant_id.as_u64(), name, column_name).await?; state.audit_record( AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/enforcement.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/enforcement.rs index 6fa8c8654..e42be8891 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/enforcement.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/enforcement.rs @@ -3,12 +3,11 @@ //! `ALTER COLLECTION ... SET {RETENTION,LEGAL_HOLD,APPEND_ONLY,LAST_VALUE_CACHE}` //! — non-schema enforcement knobs propagated through `CatalogEntry::PutCollection`. //! -//! Ported verbatim from the pgwire `ddl::collection::alter::enforcement` -//! handlers; only the result type changed to the protocol-neutral -//! [`DdlResult`] / [`DdlError`]. The retention-period validation, legal-hold -//! add/remove bookkeeping, append-only / last-value-cache guards, the -//! `PutCollection` propose, and the `schema_version` bump are unchanged, as is -//! the `ALTER COLLECTION` command tag. +//! The result type is the protocol-neutral [`DdlResult`] / [`DdlError`]. The +//! retention-period validation, legal-hold add/remove bookkeeping, +//! append-only / last-value-cache guards, the `PutCollection` propose, and +//! the `schema_version` bump run here, and the command tag is +//! `ALTER COLLECTION`. use nodedb_types::DatabaseId; @@ -19,7 +18,7 @@ use crate::control::state::SharedState; use super::support::{err, load_active_collection, status}; /// ALTER COLLECTION SET RETENTION = '' -pub(super) fn alter_collection_set_retention( +pub(super) async fn alter_collection_set_retention( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -33,11 +32,11 @@ pub(super) fn alter_collection_set_retention( .map_err(|e| err("22023", e.to_string()))?; coll.retention_period = Some(value.to_string()); - persist_and_bump(state, &coll) + persist_and_bump(state, &coll).await } /// ALTER COLLECTION SET LEGAL_HOLD = TRUE|FALSE TAG '' -pub(super) fn alter_collection_set_legal_hold( +pub(super) async fn alter_collection_set_legal_hold( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -76,11 +75,11 @@ pub(super) fn alter_collection_set_legal_hold( } } - persist_and_bump(state, &coll) + persist_and_bump(state, &coll).await } /// ALTER COLLECTION SET APPEND_ONLY -pub(super) fn alter_collection_set_append_only( +pub(super) async fn alter_collection_set_append_only( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -96,11 +95,11 @@ pub(super) fn alter_collection_set_append_only( )); } coll.append_only = true; - persist_and_bump(state, &coll) + persist_and_bump(state, &coll).await } /// ALTER COLLECTION SET LAST_VALUE_CACHE = TRUE|FALSE -pub(super) fn alter_collection_set_last_value_cache( +pub(super) async fn alter_collection_set_last_value_cache( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -117,15 +116,15 @@ pub(super) fn alter_collection_set_last_value_cache( )); } coll.lvc_enabled = enabled; - persist_and_bump(state, &coll) + persist_and_bump(state, &coll).await } -fn persist_and_bump( +async fn persist_and_bump( state: &SharedState, coll: &crate::control::security::catalog::StoredCollection, ) -> Result, DdlError> { let entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(coll.clone())); - super::support::propose_and_apply(state, &entry)?; + super::support::propose_and_apply_async(state, entry).await?; state.schema_version.bump(); Ok(status("ALTER COLLECTION")) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/materialized_sum.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/materialized_sum.rs index e9aa64422..52092af8d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/materialized_sum.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/materialized_sum.rs @@ -12,7 +12,7 @@ //! are applied. Declaring only the binding leaves a strict target with no //! `balance` field, and every maintenance write into it is a full document //! write that the Binary Tuple encoder rejects on a field the schema does not -//! carry — the balance would never land, and neither would the source row that +//! carry — the balance will never land, and neither will the source row that //! caused it. Column and binding are mutated into one `StoredCollection` and //! proposed as a single `PutCollection`, so no committed state ever holds a //! binding whose column does not exist. @@ -76,6 +76,10 @@ pub(super) async fn add_materialized_sum( let catalog = state.credentials.catalog(); let mut coll = load_active_collection(state, database_id, tenant_id, target_collection)?; + crate::control::server::shared::ddl::neutral::collection::enforcement::validate_sum_target( + &coll, + ) + .map_err(|e| err(e.sqlstate(), e.to_string()))?; if coll .materialized_sums @@ -99,13 +103,14 @@ pub(super) async fn add_materialized_sum( declare_target_column(&mut coll, target_column, target_column_type)?; coll.materialized_sums.push(def); let entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(coll.clone())); - super::support::propose_and_apply_async(state, entry).await?; + let outcome = super::support::propose_and_apply_async(state, entry).await?; // The Data Plane's in-memory shape is what the Binary Tuple encoder // consults, so it must learn the new column before the first source write // arrives; without this the binding is durable and the very next - // maintenance write is still rejected on an unknown field. - super::super::register::dispatch_register_from_stored(state, &coll) + // maintenance write is still rejected on an unknown field. A durable + // apply registered the target and its sources in its post-apply. + super::super::register::register_proposed_collection(state, outcome, &coll) .await .map_err(|e| DdlError::from_error(&e))?; @@ -115,9 +120,11 @@ pub(super) async fn add_materialized_sum( // derived for the source. Registering only the target leaves the source // asserting it drives no binding, so every co-resident write into it folds // nothing and the total silently stays where it was. - super::super::register::dispatch_register_for_sum_sources(state, &coll) - .await - .map_err(|e| DdlError::from_error(&e))?; + if outcome.is_buffered() { + super::super::register::dispatch_register_for_sum_sources(state, &coll) + .await + .map_err(|e| DdlError::from_error(&e))?; + } state.schema_version.bump(); @@ -181,7 +188,7 @@ fn declare_target_column( Ok(()) } -/// Refuse a binding that would make some collection both a materialized-sum +/// Refuse a binding that will make some collection both a materialized-sum /// source and a materialized-sum target. /// /// Maintenance of a materialized sum writes the target row through a plain @@ -215,7 +222,7 @@ fn validate_binding_depth( return Err(chain_error(target_collection, source_collection)); } - // The new target already feeds another collection: it would become both a + // The new target already feeds another collection: it will become both a // sink (of this binding) and a source (of that one). if let Some(downstream) = existing .iter() @@ -294,7 +301,7 @@ mod tests { /// needs to: the resolution a write carries is keyed on the /// `(target collection, join value)` PAIR, so each binding resolves, defers /// and folds against its own target row independently. There is nothing left - /// for a DDL-time guard to protect, and refusing the shape here would reject + /// for a DDL-time guard to protect, and refusing the shape here will reject /// a schema the engine now maintains correctly. #[test] fn a_second_binding_on_the_same_source_and_join_column_is_accepted() { @@ -304,7 +311,7 @@ mod tests { #[test] fn extending_an_existing_target_into_a_source_is_rejected() { - // entries -> accounts already exists; accounts -> ledger would chain. + // entries -> accounts already exists; accounts -> ledger will chain. let existing = vec![binding("entries", "accounts")]; let error = validate_binding_depth(&existing, "ledger", "accounts") .expect_err("a two-hop chain must be refused"); @@ -322,7 +329,7 @@ mod tests { #[test] fn feeding_an_existing_source_is_rejected() { - // entries -> accounts already exists; raw_events -> entries would chain. + // entries -> accounts already exists; raw_events -> entries will chain. let existing = vec![binding("entries", "accounts")]; let error = validate_binding_depth(&existing, "entries", "raw_events") .expect_err("a two-hop chain must be refused"); @@ -351,7 +358,7 @@ mod tests { #[test] fn longer_chain_is_rejected_at_every_extension_point() { let existing = vec![binding("a", "b"), binding("c", "d")]; - // b -> c would join the two independent edges into a -> b -> c -> d. + // b -> c will join the two independent edges into a -> b -> c -> d. assert!(validate_binding_depth(&existing, "c", "b").is_err()); } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/ownership.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/ownership.rs index 81a61fe70..50bd6a664 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/ownership.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/ownership.rs @@ -2,13 +2,12 @@ //! `ALTER COLLECTION OWNER TO ` — transfer collection ownership. //! -//! Ported verbatim from the pgwire `ddl::ownership` handler; only the result -//! type changed to the protocol-neutral [`DdlResult`] / [`DdlError`]. +//! The result type is the protocol-neutral [`DdlResult`] / [`DdlError`]. //! //! The ownership change is applied by mutating the parent `StoredCollection` //! and re-proposing it (NOT via a standalone `PutOwner`): the `OWNERS` redb //! table is rewritten from `stored.owner` by the `PutCollection` `post_apply` -//! on every node, so a separate `PutOwner` would be silently overwritten the +//! on every node, so a separate `PutOwner` will be silently overwritten the //! next time anyone re-proposed the collection. The authorization gate, new- //! owner existence check, the propose + single-node fallback //! (`put_collection` + `install_replicated_owner`), and the audit are @@ -17,7 +16,7 @@ use nodedb_types::DatabaseId; use crate::control::catalog_entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::audit::AuditEvent; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::result::{DdlError, DdlResult}; @@ -26,7 +25,7 @@ use crate::control::state::SharedState; use super::support::{err, load_active_collection, status}; /// ALTER COLLECTION OWNER TO -pub(super) fn alter_collection_owner( +pub(super) async fn alter_collection_owner( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -68,29 +67,15 @@ pub(super) fn alter_collection_owner( // `PutCollection` rewrites it from `stored.owner` on every // node — so the only way to keep an owner change durable // through subsequent ALTER COLLECTION calls is to also mutate - // the parent record. A separate `PutOwner` would be silently + // the parent record. A separate `PutOwner` will be silently // overwritten the next time anyone re-proposed the collection. - let catalog = state.credentials.catalog(); let mut stored = load_active_collection(state, database_id, identity.tenant_id.as_u64(), collection)?; stored.owner = new_owner.to_string(); - let entry = CatalogEntry::PutCollection(Box::new(stored.clone())); - let outcome = propose_catalog_entry(state, &entry) + let entry = CatalogEntry::PutCollection(Box::new(stored)); + propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - catalog - .put_collection(database_id, &stored) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - state.permissions.install_replicated_owner( - &crate::control::security::catalog::StoredOwner { - database_id: stored.database_id.as_u64(), - object_type: "collection".into(), - object_name: stored.name.clone(), - tenant_id: stored.tenant_id, - owner_username: stored.owner.clone(), - }, - ); - } state.audit_record( AuditEvent::PrivilegeChange, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/rename_column.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/rename_column.rs index 6ba9c4ec3..be690282e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/rename_column.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/rename_column.rs @@ -3,11 +3,9 @@ //! `ALTER COLLECTION RENAME COLUMN TO ` — rename a //! column in a strict-document collection's schema. //! -//! Ported verbatim from the pgwire `ddl::collection::alter::rename_column` -//! handler; only the result type changed to the protocol-neutral -//! [`DdlResult`] / [`DdlError`]. The duplicate-name guard, positional -//! rename + version bump, persist, and audit are unchanged, as is the -//! `ALTER COLLECTION` command tag. +//! The result type is the protocol-neutral [`DdlResult`] / [`DdlError`]. The +//! duplicate-name guard, positional rename + version bump, persist, and +//! audit run here, and the command tag is `ALTER COLLECTION`. use nodedb_types::DatabaseId; @@ -79,7 +77,8 @@ pub(super) async fn alter_collection_rename_column( name, old_name, new_name, - )?; + ) + .await?; state.audit_record( AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/strict_schema.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/strict_schema.rs index c4699df0f..35803e6eb 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/strict_schema.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/strict_schema.rs @@ -9,12 +9,9 @@ //! `PutCollection` entry, replicate it through the metadata raft group, //! refresh the Data Plane register, and bump the schema version. //! -//! Ported verbatim from the pgwire -//! `ddl::collection::alter::strict_schema` module; only the result type -//! changed from pgwire `PgWireResult` / `sqlstate_error` to the -//! protocol-neutral [`DdlError`]. The catalog lookup, engine gate, schema -//! (de)serialization, propose + register + version-bump ordering, and the -//! SQLSTATE codes / messages are unchanged. +//! The error type is the protocol-neutral [`DdlError`]. The catalog lookup, +//! engine gate, schema (de)serialization, propose + register + version-bump +//! ordering, and the SQLSTATE codes / messages run here. use nodedb_types::DatabaseId; @@ -73,8 +70,8 @@ pub(super) fn write_schema_back( /// spelling drives the column's advertised wire OID and the range accepted on /// write, which is exactly why `ALTER COLUMN TYPE` — whose only supported use /// *is* an alias change such as `INT` → `BIGINT` — has to update it. Leaving -/// it stale would make the alter a silent no-op for the case it exists to -/// serve, and would keep rejecting writes the new type allows. +/// it stale will make the alter a silent no-op for the case it exists to +/// serve, and will keep rejecting writes the new type allows. pub(super) fn retype_field(coll: &mut StoredCollection, column: &str, new_type: &str) { for (name, type_str) in coll.fields.iter_mut() { if name.eq_ignore_ascii_case(column) { @@ -106,19 +103,20 @@ pub(super) fn add_field(coll: &mut StoredCollection, column: &str, declared_type .push((column.to_string(), declared_type.to_string())); } -/// Replicate the mutated collection through the metadata raft group, -/// refresh this node's Data Plane register so the in-memory shape -/// catches up with the new schema, recompile the collection's RLS -/// policies against it, then bump `schema_version`. +/// Replicate the mutated collection through the metadata raft group, and +/// register the new schema on this node's Data Plane. A durable apply +/// registers it in its post-apply, and a buffered one registers it here. +/// Then recompile the collection's RLS policies against it and bump +/// `schema_version`. pub(super) async fn persist_schema_change( state: &SharedState, updated: &StoredCollection, ) -> Result<(), DdlError> { let entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(updated.clone())); - super::support::propose_and_apply(state, &entry)?; + let outcome = super::support::propose_and_apply_async(state, entry).await?; - super::super::register::dispatch_register_from_stored(state, updated) + super::super::register::register_proposed_collection(state, outcome, updated) .await .map_err(|e| DdlError::from_error(&e))?; recompile_rls_policies(state, updated)?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/support.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/support.rs index 81f13cc47..921870ee5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/support.rs @@ -2,31 +2,26 @@ //! Shared helpers for the protocol-neutral `ALTER COLLECTION` handlers. //! -//! Provides the [`DdlError`] constructor (preserving the exact SQLSTATE codes -//! and messages the pgwire handlers produced), the single-row `ALTER`-status -//! result builder, and the neutral `propose_and_apply` mirror of the pgwire -//! `ddl::catalog_propose::propose_and_apply` (same propose + local-apply -//! ordering). A propose error keeps its own SQLSTATE under a -//! `"metadata propose"` prefix. +//! Provides the [`DdlError`] constructor, the single-row `ALTER`-status +//! result builder, and the neutral `propose_and_apply`. A propose error keeps its own +//! SQLSTATE under a `"metadata propose"` prefix. use nodedb_types::DatabaseId; use crate::control::catalog_entry::CatalogEntry; -use crate::control::catalog_entry::apply::local::apply_locally_if_needed; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::propose_outcome::ProposeOutcome; use crate::control::security::catalog::StoredCollection; use crate::control::server::shared::ddl::result::{DdlError, DdlResult}; use crate::control::state::SharedState; -/// Construct a [`DdlError`], preserving the exact SQLSTATE codes and messages -/// the pgwire handlers produced. +/// Construct a [`DdlError`] from a SQLSTATE code and a message. pub(super) fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } /// Build the single `Status` result every `ALTER` sub-command returns. `command` -/// is the pgwire command tag, preserved verbatim (`ALTER TABLE` for ADD COLUMN, +/// is the pgwire command tag (`ALTER TABLE` for ADD COLUMN, /// `ALTER COLLECTION` for every other sub-command). pub(super) fn status(command: &str) -> Vec { vec![DdlResult::Status { @@ -58,51 +53,21 @@ pub(super) fn load_active_collection( .ok_or_else(|| err("42P01", format!("collection '{name}' does not exist"))) } -/// Neutral mirror of the pgwire `ddl::catalog_propose::propose_and_apply`. +/// Propose `entry` from an async handler, including online DDL that runs +/// concurrently with ingest. On return it applied on this node, primary row +/// and companion `StoredOwner` row both, or it is held for COMMIT. It awaits +/// the entry's post-apply Data Plane work on any runtime flavor. /// -/// Propose `entry` through the metadata raft group and, when the proposer -/// reports `LocalOnly`, apply the entry locally so -/// the primary row and the companion `StoredOwner` row both land in redb. -pub(super) fn propose_and_apply( - state: &SharedState, - entry: &CatalogEntry, -) -> Result { - let outcome = propose_catalog_entry(state, entry) - .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - apply_locally_if_needed(state, entry, outcome); - Ok(outcome) -} - -/// Async variant of [`propose_and_apply`] for online DDL that runs -/// concurrently with ingest. -/// -/// The local catalog apply (`LocalOnly` path) performs a redb write -/// transaction whose `commit()` issues an `fsync`; that fsync can take tens -/// of milliseconds. Running it inline on the Tokio worker would monopolise -/// the worker for the duration of the flush, stalling every concurrent -/// `INSERT` task scheduled on it — an online `ALTER` must never block the -/// write path. Moving the blocking commit onto a `spawn_blocking` thread -/// keeps the worker free to service writes while the catalog is made durable. -/// -/// Proposal ordering is unchanged: the durable apply still completes before -/// this call returns, so the cross-core schema-register barrier that follows -/// observes the applied schema. +/// The apply's redb commit issues an `fsync`. The proposer runs that wait off +/// the Tokio worker's task queue on a multi-thread runtime, so an online +/// `ALTER` never stalls the `INSERT` tasks scheduled on it. The durable apply +/// completes before this call returns, so the cross-core schema-register +/// barrier that follows observes the applied schema. pub(super) async fn propose_and_apply_async( state: &SharedState, entry: CatalogEntry, ) -> Result { - let outcome = propose_catalog_entry(state, &entry) - .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - // Clone only the cheap `Arc` handle (not `SharedState`) so - // the blocking closure owns exactly what the apply needs. - let catalog = state.credentials.catalog().clone(); - tokio::task::spawn_blocking(move || { - crate::control::catalog_entry::apply::apply_to(&entry, &catalog) - }) + propose_catalog_entry_async(state, &entry) .await - .map_err(|e| DdlError::internal(format!("catalog apply join: {e}")))? - .map_err(|e| DdlError::from_error_in_context("catalog apply", &e))?; - } - Ok(outcome) + .map_err(|e| DdlError::from_error_in_context("metadata propose", &e)) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/vector_model.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/vector_model.rs index b3885a5d8..4f12271ab 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/alter/vector_model.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/alter/vector_model.rs @@ -20,7 +20,7 @@ use crate::control::state::SharedState; use super::super::super::vector_replicate::{propose_delete_model, propose_put_model}; /// Drop `column`'s embedding-model row on every node. -pub(super) fn drop_vector_model_row( +pub(super) async fn drop_vector_model_row( state: &SharedState, database_id: DatabaseId, tenant_id: u64, @@ -31,14 +31,14 @@ pub(super) fn drop_vector_model_row( if !model_row_exists(state, db, tenant_id, collection, column)? { return Ok(()); } - propose_delete_model(state, db, tenant_id, collection, column) + propose_delete_model(state, db, tenant_id, collection, column).await } /// Re-key `old_column`'s embedding-model row onto `new_column` on every node. /// /// The write lands before the delete, so an interrupted rename leaves the row /// readable under one of the two names. The reverse order can lose it. -pub(super) fn move_vector_model_row( +pub(super) async fn move_vector_model_row( state: &SharedState, database_id: DatabaseId, tenant_id: u64, @@ -57,8 +57,8 @@ pub(super) fn move_vector_model_row( }; entry.column = new_column.to_string(); - propose_put_model(state, &entry)?; - propose_delete_model(state, db, tenant_id, collection, old_column) + propose_put_model(state, &entry).await?; + propose_delete_model(state, db, tenant_id, collection, old_column).await } fn model_row_exists( @@ -95,6 +95,7 @@ mod tests { Arc::new(WalManager::open_for_testing(&dir.path().join(name)).expect("open test WAL")); let (dispatcher, _data_sides) = Dispatcher::new(1, 64); let state = SharedState::new(dispatcher, wal).expect("construct shared state"); + crate::bootstrap::state_wiring::install_gateway(&state).expect("install gateway"); (dir, state) } @@ -113,20 +114,22 @@ mod tests { } } - #[tokio::test] + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn a_rename_moves_the_model_row_instead_of_duplicating_it() { - let (_dir, state) = test_state("vector-model-move.wal"); + let cluster = crate::control::cluster::test_one_node::boot().await; + let state = &cluster.state; let catalog = state.credentials.catalog(); catalog.put_vector_model(&model("embedding")).expect("seed"); move_vector_model_row( - &state, + state, DatabaseId::DEFAULT, TENANT, COLLECTION, "embedding", "vector", ) + .await .expect("move the row"); let db = DatabaseId::DEFAULT.as_u64(); @@ -143,6 +146,7 @@ mod tests { .expect("the row lands under the new column"); assert_eq!(moved.metadata.dimensions, 384); assert_eq!(moved.column, "vector"); + cluster.shutdown().await; } #[tokio::test] @@ -157,6 +161,7 @@ mod tests { "quantity", "amount", ) + .await .expect("no row to move"); let db = DatabaseId::DEFAULT.as_u64(); @@ -169,13 +174,15 @@ mod tests { ); } - #[tokio::test] + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn a_drop_removes_the_model_row() { - let (_dir, state) = test_state("vector-model-drop.wal"); + let cluster = crate::control::cluster::test_one_node::boot().await; + let state = &cluster.state; let catalog = state.credentials.catalog(); catalog.put_vector_model(&model("embedding")).expect("seed"); - drop_vector_model_row(&state, DatabaseId::DEFAULT, TENANT, COLLECTION, "embedding") + drop_vector_model_row(state, DatabaseId::DEFAULT, TENANT, COLLECTION, "embedding") + .await .expect("drop the row"); assert!( @@ -189,5 +196,6 @@ mod tests { .expect("read") .is_none() ); + cluster.shutdown().await; } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/csv_import.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/csv_import.rs index 8a2f1ac42..88dcc64c8 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/csv_import.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/csv_import.rs @@ -2,8 +2,7 @@ //! CSV import for `COPY FROM`. //! -//! Relocated verbatim from the pgwire `ddl::collection::copy_from::csv_import` -//! module (now deleted). `plan_and_dispatch` returns the protocol-neutral +//! `plan_and_dispatch` returns the protocol-neutral //! [`DdlError`] directly (it is the neutral collection-DML helper), so this //! module's own file-read/parse errors are built as `DdlError` at their call //! sites to keep one error type end to end. @@ -101,7 +100,7 @@ pub(super) async fn import_csv( }; // Inject a unique row number as id if the field map has no id. // This prevents duplicate-key errors on schemaless collections - // where all rows would otherwise receive the same empty-string id. + // where all rows will otherwise receive the same empty-string id. if !fields.contains_key("id") { fields.insert( "id".to_string(), @@ -122,6 +121,8 @@ pub(super) async fn import_csv( ctx.database_id, &sql, ctx.txn_ctx, + // Each row fires its collection's triggers, as an INSERT does. + true, ) .await .map_err(|e| wrap_row_error(e, *ln, "CSV"))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/entry.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/entry.rs index fe76a4f37..a6d73daa4 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/entry.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/entry.rs @@ -2,11 +2,9 @@ //! Entry point: `copy_from_file`, path validation, and engine-support check. //! -//! Relocated verbatim from the pgwire `ddl::collection::copy_from::entry` -//! module (now deleted) except for the result type, which is [`DdlResult`] / -//! [`DdlError`] throughout instead of pgwire `Response` / `PgWireResult`. Each -//! error site builds a `DdlError` directly via [`ddl_err`] so the whole call -//! chain speaks one error type. +//! The result type is [`DdlResult`] / [`DdlError`] throughout. Each error +//! site builds a `DdlError` directly via [`ddl_err`] so the whole call chain +//! speaks one error type. use nodedb_types::DatabaseId; use std::path::Path; @@ -20,6 +18,9 @@ use crate::control::server::shared::ddl::result::{DdlError, DdlResult}; use crate::control::server::shared::session::DmlTxnCtx; use crate::control::state::SharedState; +use crate::control::trigger::statement_txn::{fires_joined_body, in_block, with_statement_txn}; +use crate::control::trigger::{DmlEvent, TriggerScope}; + use super::csv_import::{CsvOptions, import_csv}; use super::import_ctx::ImportCtx; use super::json_import::{import_json_array, import_ndjson}; @@ -85,30 +86,52 @@ pub async fn copy_from_file( let tenant_id = identity.tenant_id; - let import_ctx = ImportCtx { + // Each row fires its collection's BEFORE, INSTEAD OF and SYNC AFTER + // bodies, which join the statement's transaction. Outside a transaction + // block a COPY into such a collection runs in an implicit one: its rows + // and the bodies' writes commit together, or none of them do. + let implicit = !in_block(txn_ctx) + && fires_joined_body( + state, + TriggerScope { + database_id, + tenant_id, + }, + collection, + DmlEvent::Insert, + ); + let row_count = with_statement_txn( state, identity, - tenant_id, - database_id, txn_ctx, - }; - - let row_count = match resolved_format { - CopyFormat::Ndjson => import_ndjson(&import_ctx, collection, path).await?, - CopyFormat::JsonArray => import_json_array(&import_ctx, collection, path).await?, - CopyFormat::Csv => { - import_csv( - &import_ctx, - collection, - path, - CsvOptions { - delimiter: delimiter.unwrap_or(','), - has_header: header, - }, - ) - .await? - } - }; + implicit, + async |txn_ctx: &DmlTxnCtx<'_>| { + let import_ctx = ImportCtx { + state, + identity, + tenant_id, + database_id, + txn_ctx, + }; + match resolved_format { + CopyFormat::Ndjson => import_ndjson(&import_ctx, collection, path).await, + CopyFormat::JsonArray => import_json_array(&import_ctx, collection, path).await, + CopyFormat::Csv => { + import_csv( + &import_ctx, + collection, + path, + CsvOptions { + delimiter: delimiter.unwrap_or(','), + has_header: header, + }, + ) + .await + } + } + }, + ) + .await?; // The count is baked into `command` (not `rows_affected`) because the // native and HTTP encoders (`ddl_result_to_native`, `ddl_results_to_json`) @@ -200,7 +223,7 @@ fn check_engine_support( /// Row-level import helpers (`import_csv`, `import_ndjson`, `import_json_array`) /// call `plan_and_dispatch`, which returns a protocol-neutral [`DdlError`] (not /// a pgwire `PgWireError`), so this wraps the same type — only the message is -/// decorated with the row number, matching the original pgwire behavior. +/// decorated with the row number. pub(super) fn wrap_row_error(e: DdlError, line_no: usize, fmt: &str) -> DdlError { DdlError::new( e.sqlstate, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/json_import.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/json_import.rs index 332531957..bf1d83965 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/json_import.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_from/json_import.rs @@ -2,8 +2,7 @@ //! NDJSON and JSON array import for `COPY FROM`. //! -//! Relocated verbatim from the pgwire `ddl::collection::copy_from::json_import` -//! module (now deleted). `plan_and_dispatch` returns the protocol-neutral +//! `plan_and_dispatch` returns the protocol-neutral //! [`DdlError`] directly (it is the neutral collection-DML helper), so this //! module's own file-read/parse errors are built as `DdlError` at their call //! sites to keep one error type end to end. @@ -66,6 +65,8 @@ pub(super) async fn import_ndjson( ctx.database_id, &sql, ctx.txn_ctx, + // Each row fires its collection's triggers, as an INSERT does. + true, ) .await .map_err(|e| wrap_row_error(e, *line_no, "NDJSON"))?; @@ -123,6 +124,8 @@ pub(super) async fn import_json_array( ctx.database_id, &sql, ctx.txn_ctx, + // Each row fires its collection's triggers, as an INSERT does. + true, ) .await .map_err(|e| wrap_row_error(e, line_no, "JSON array"))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/entry.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/entry.rs index 07b65fdee..88b51bcb7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/entry.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/entry.rs @@ -2,9 +2,7 @@ //! Entry point: `copy_to_file`, path validation, scan, and atomic file write. //! -//! Relocated verbatim from the pgwire `ddl::collection::copy_to::entry` -//! module (now deleted) except for the result type, which is [`DdlResult`] / -//! [`DdlError`] throughout instead of pgwire `Response` / `PgWireResult`. +//! The result type is [`DdlResult`] / [`DdlError`] throughout. use nodedb_types::DatabaseId; use std::path::Path; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/format.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/format.rs index b552612ac..4eea69e17 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/format.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/copy_to/format.rs @@ -1,9 +1,6 @@ // SPDX-License-Identifier: BUSL-1.1 //! Format serializers for COPY TO: NDJSON, JSON array, CSV. -//! -//! Relocated verbatim from the pgwire `ddl::collection::copy_to::format` -//! module (now deleted). use nodedb_sql::ddl_ast::statement::CopyFormat; use sonic_rs; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/create/build.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/create/build.rs index 836260ad6..0f6278dec 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/create/build.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/create/build.rs @@ -2,17 +2,15 @@ //! Shared implementation behind `CREATE COLLECTION` and `CREATE TABLE`. //! -//! Relocated verbatim from the pgwire `pgwire::ddl::collection::create::build` -//! module (now deleted). The two surface DDLs differ in only five places: the +//! The two surface DDLs differ in only five places: the //! error label ("collection" vs "table"), whether an empty column list is //! allowed, the default `CollectionType` when no engine is named (schemaless //! vs strict), the audit-log verb, and the response tag. Everything in //! between — name validation, duplicate check, engine validation, schema //! construction, vector-primary parsing, flag validation, `StoredCollection` //! assembly, propose+apply, SERIAL sequence auto-creation, vector-field -//! auto-config — is identical, and is preserved verbatim here; only the -//! result construction changed from pgwire `Response` / `PgWireError` to the -//! protocol-neutral [`DdlResult`] / [`DdlError`]. +//! auto-config — is identical and runs here. The +//! result is the protocol-neutral [`DdlResult`] / [`DdlError`]. //! //! [`build_and_persist`] is the single body; [`Variant`] supplies the five //! differences declaratively. Name/flag validation lives in @@ -21,20 +19,25 @@ use nodedb_types::DatabaseId; +use crate::control::propose_outcome::ProposeOutcome; use crate::control::security::audit::AuditEvent; use crate::control::security::catalog::StoredCollection; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; -use super::super::super::super::catalog::propose_and_apply; +use super::super::super::super::catalog::propose_and_apply_async; use super::super::super::super::result::{DdlError, DdlResult}; use super::super::enforcement::{ parse_and_validate_balanced_clause, resolve_custom_type_columns, validate_hash_chain_flags, + validate_hash_chain_storage, }; use super::engine_option::validate_engine_name; use super::request::CreateCollectionRequest; -use super::build_flags::{err, resolve_crdt_flag, validate_crdt_signing_storage, validate_name}; +use super::build_flags::{ + declare_hash_chain_columns, err, resolve_crdt_flag, validate_crdt_signing_storage, + validate_name, +}; use super::build_post_create::{create_serial_sequences, log_vector_fields}; use super::build_primary_engine::resolve_primary_engine; use crate::control::server::shared::ddl::neutral::column_default::validate_column_defaults; @@ -58,6 +61,18 @@ pub struct Variant { pub default_strict: bool, } +/// A proposed CREATE COLLECTION / CREATE TABLE. +pub struct CreatedCollection { + /// The client response. + pub results: Vec, + /// What the proposer did with the `PutCollection` entry. A durable outcome + /// already registered the collection on every local core. A buffered one + /// registered nothing. + pub outcome: ProposeOutcome, + /// The collection the entry carries. + pub collection: StoredCollection, +} + /// Shared body. Validates the request, builds the /// `StoredCollection`, replicates it through the metadata raft /// group, and runs the post-create side effects (SERIAL sequence @@ -68,7 +83,7 @@ pub async fn build_and_persist( req: &CreateCollectionRequest<'_>, database_id: DatabaseId, variant: &Variant, -) -> Result, DdlError> { +) -> Result { let CreateCollectionRequest { name, engine, @@ -97,20 +112,6 @@ pub async fn build_and_persist( let tenant_id = identity.tenant_id; - // Metadata Raft serializes clustered DDL. Without it, hold an exclusive - // per-name lifecycle guard across validation, any predecessor reclaim, - // catalog creation, and Data Plane registration. - let mut local_lifecycle = if state.metadata_raft.get().is_none() { - Some( - state - .quiesce - .acquire_lifecycle(database_id.as_u64(), tenant_id.as_u64(), name) - .await, - ) - } else { - None - }; - // A materialized-view definition durably owns its same-name target even if // a crash occurred between definition and target registration. let catalog = state.credentials.catalog(); @@ -126,7 +127,7 @@ pub async fn build_and_persist( } // Check if the object already exists. A catalog-read fault must abort the - // CREATE — proceeding as if no row exists could build a fresh collection + // CREATE — proceeding as if no row exists can build a fresh collection // over a soft-deleted incarnation's still-present storage. let existing = catalog .get_collection(database_id, tenant_id.as_u64(), name) @@ -153,9 +154,9 @@ pub async fn build_and_persist( // row sits below it and is shadowed on replay, while every row the // new collection writes sits at or above it and survives. let purge_lsn = state.wal.next_lsn().as_u64(); - // Fail closed: if the hard-purge could not remove the old + // Fail closed: if the hard-purge cannot remove the old // catalog row, ABORT the CREATE rather than build a new - // collection over un-purged data (which would resurrect the + // collection over un-purged data (which will resurrect the // stale rows). Surface as an internal error to the client. let purge_result = crate::control::server::shared::ddl::neutral::collection::purge::hard_purge_collection( @@ -164,18 +165,10 @@ pub async fn build_and_persist( tenant_id.as_u64(), name, purge_lsn, - local_lifecycle.is_some(), + false, ) .await; if let Err(failure) = purge_result { - // Only disarm when a durable retry record owns the drain. Otherwise - // let the guard release the in-memory hold so this same-name CREATE - // can be retried against the durable inactive catalog row. - if failure.retry_queued - && let Some(guard) = local_lifecycle.take() - { - guard.disarm(); - } return Err(DdlError::from_error(&failure.error)); } } @@ -205,14 +198,19 @@ pub async fn build_and_persist( tenant_id.as_u64(), ); - let (collection_type, columnar_schema_columns) = nodedb_sql::ddl_ast::build_collection_type( - canonical_engine, - &resolved_columns, - options, - bitemporal_flag, - variant.default_strict, - ) - .map_err(|e| err("42601", e.to_string()))?; + let (mut collection_type, columnar_schema_columns) = + nodedb_sql::ddl_ast::build_collection_type( + canonical_engine, + &resolved_columns, + options, + bitemporal_flag, + variant.default_strict, + ) + .map_err(|e| err("42601", e.to_string()))?; + declare_hash_chain_columns( + &mut collection_type, + flags.iter().any(|f| f == "HASH_CHAIN"), + )?; let mut fields = expanded_columns.clone(); if fields.is_empty() && !columnar_schema_columns.is_empty() { @@ -238,6 +236,7 @@ pub async fn build_and_persist( .map_err(|e| err(e.sqlstate(), e.to_string()))?; let crdt = resolve_crdt_flag(options, &collection_type)?; + validate_hash_chain_storage(hash_chain, crdt).map_err(|e| err(e.sqlstate(), e.to_string()))?; validate_crdt_signing_storage( crdt_signing_required, crdt, @@ -247,9 +246,9 @@ pub async fn build_and_persist( // physically TEXT, so the resolved list is the one to check. // // A schemaless collection is deliberately not checked: its field list is - // advisory, a write may carry any field whether or not it appears there, + // advisory, a write can carry any field whether or not it appears there, // and the commit-time check reads whatever the row actually holds. Refusing - // a BALANCED column that is merely absent from that list would reject + // a BALANCED column that is merely absent from that list will reject // `CREATE COLLECTION x WITH BALANCED ON (...)` — a declaration with no // column list at all, which is the ordinary schemaless spelling. let balanced = @@ -279,6 +278,7 @@ pub async fn build_and_persist( constraint_version: 0, crdt_signing_required, modification_hlc: nodedb_types::Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, fields, field_defs: Vec::new(), event_defs: Vec::new(), @@ -314,11 +314,13 @@ pub async fn build_and_persist( declared_primary_key, }; + // The apply of a new incarnation clears the name's storage before the + // engine registers, on every node, this one included. let entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(coll.clone())); - propose_and_apply(state, &entry)?; + let outcome = propose_and_apply_async(state, &entry).await?; log_vector_fields(name, &coll.fields); - create_serial_sequences(state, identity, database_id, name, &serial_fields, now)?; + create_serial_sequences(state, identity, database_id, name, &serial_fields, now).await?; state.audit_record( AuditEvent::AdminAction, @@ -327,8 +329,12 @@ pub async fn build_and_persist( &format!("created {} '{name}'", variant.label), ); - Ok(vec![DdlResult::Status { - command: variant.response_tag.to_string(), - rows_affected: None, - }]) + Ok(CreatedCollection { + results: vec![DdlResult::Status { + command: variant.response_tag.to_string(), + rows_affected: None, + }], + outcome, + collection: coll, + }) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/create/build_flags.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/create/build_flags.rs index 6d635c843..f26bc0cd8 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/create/build_flags.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/create/build_flags.rs @@ -2,8 +2,7 @@ //! Name / flag validation for `build_and_persist`: collection-name shape, //! the `WITH (crdt=...)` boolean, and `SIGNED_DELTAS` ⇒ CRDT + authenticated -//! WAL. Relocated verbatim from the pgwire -//! `pgwire::ddl::collection::create::build` module (now deleted). +//! WAL. use super::super::super::super::result::DdlError; @@ -11,6 +10,49 @@ pub(super) fn err(sqlstate: &str, message: String) -> DdlError { DdlError::new(sqlstate, message) } +/// Declare the hash-chain system columns on a strict `HASH_CHAIN` collection. +/// +/// A strict row stores only declared columns, and the chain writes +/// `_chain_hash` and `_chain_seq` into every row. They are nullable because a +/// transaction encodes its staged rows before the install links them. A user +/// column with either name is refused. Other collection types are unchanged. +pub(super) fn declare_hash_chain_columns( + collection_type: &mut nodedb_types::CollectionType, + hash_chain: bool, +) -> Result<(), DdlError> { + use crate::types::hash_chain::{CHAIN_HASH_FIELD, CHAIN_SEQ_FIELD}; + use nodedb_types::columnar::{ColumnDef, ColumnType}; + + let nodedb_types::CollectionType::Document(nodedb_types::DocumentMode::Strict(schema)) = + collection_type + else { + return Ok(()); + }; + if !hash_chain { + return Ok(()); + } + if let Some(column) = schema + .columns + .iter() + .find(|column| column.name == CHAIN_HASH_FIELD || column.name == CHAIN_SEQ_FIELD) + { + return Err(err( + "42939", + format!( + "column '{}' is a hash-chain system column; HASH_CHAIN declares it", + column.name + ), + )); + } + schema + .columns + .push(ColumnDef::nullable(CHAIN_HASH_FIELD, ColumnType::String)); + schema + .columns + .push(ColumnDef::nullable(CHAIN_SEQ_FIELD, ColumnType::Int64)); + Ok(()) +} + /// Parse a `WITH (crdt=...)` option value as a boolean, accepting /// `"true"`/`"false"` case-insensitively. Any other value is a /// user error surfaced as a typed DDL error (SQLSTATE 42601). @@ -30,7 +72,7 @@ fn parse_crdt_flag(value: &str) -> Result { /// A missing `crdt` option defaults to `false`. CRDT (Loro) storage is a /// document-engine capability, so `crdt=true` is rejected with SQLSTATE /// 42601 on any non-document collection rather than persisting a flag no -/// engine would honor. +/// engine will honor. pub(super) fn resolve_crdt_flag( options: &[(String, String)], collection_type: &nodedb_types::CollectionType, @@ -88,8 +130,7 @@ pub(super) fn validate_name(name: &str, label: &str) -> Result<(), DdlError> { #[cfg(test)] mod tests { - //! Collection name validation tests. Relocated verbatim from the pgwire - //! `pgwire::ddl::collection::create::tests` module (now deleted). + //! Collection name validation tests. use super::{resolve_crdt_flag, validate_crdt_signing_storage}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/create/build_post_create.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/create/build_post_create.rs index e3f9a1fc4..2f16febac 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/create/build_post_create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/create/build_post_create.rs @@ -1,15 +1,13 @@ // SPDX-License-Identifier: BUSL-1.1 //! Post-create side effects for `build_and_persist`: vector-field -//! auto-config logging and `SERIAL` sequence auto-creation. Relocated -//! verbatim from the pgwire `pgwire::ddl::collection::create::build` module -//! (now deleted). +//! auto-config logging and `SERIAL` sequence auto-creation. use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; use crate::types::DatabaseId; -use super::super::super::super::catalog::propose_and_apply; +use super::super::super::super::catalog::propose_and_apply_async; use super::super::super::super::result::DdlError; /// INFO-log every detected vector field so operators can see what @@ -28,9 +26,9 @@ pub(super) fn log_vector_fields(collection_name: &str, fields: &[(String, String } /// Materialise one `StoredSequence` per `SERIAL` column, via the same -/// propose+apply path as `CREATE SEQUENCE`, gated the same way: shared-registry -/// install only on `needs_local_apply`, so a `Buffered` outcome cannot leak it. -pub(super) fn create_serial_sequences( +/// propose path as `CREATE SEQUENCE`. A `Buffered` outcome installs nothing +/// in the shared registry until COMMIT. +pub(super) async fn create_serial_sequences( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -47,16 +45,12 @@ pub(super) fn create_serial_sequences( identity.username.clone(), ); seq_def.created_at = now; - // Route the auto-created sequence through the proposer + - // local apply path so the OWNERS row lands alongside the - // sequence row — the same architectural guarantee CREATE - // SEQUENCE has, applied to SERIAL columns. - let seq_entry = - crate::control::catalog_entry::CatalogEntry::PutSequence(Box::new(seq_def.clone())); - let outcome = propose_and_apply(state, &seq_entry)?; - if outcome.needs_local_apply() { - let _ = state.sequence_registry.create(seq_def); - } + // Route the auto-created sequence through the proposer so the + // OWNERS row lands alongside the sequence row — the same + // architectural guarantee CREATE SEQUENCE has, applied to SERIAL + // columns. + let seq_entry = crate::control::catalog_entry::CatalogEntry::PutSequence(Box::new(seq_def)); + propose_and_apply_async(state, &seq_entry).await?; tracing::info!( collection = %collection_name, field = %field_name, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/create/build_primary_engine.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/create/build_primary_engine.rs index 23adaea8a..b0fef945d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/create/build_primary_engine.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/create/build_primary_engine.rs @@ -1,8 +1,7 @@ // SPDX-License-Identifier: BUSL-1.1 //! Resolve the `WITH (primary='vector', vector_field=...)` access-path -//! config for `build_and_persist`. Relocated verbatim from the pgwire -//! `pgwire::ddl::collection::create::build` module (now deleted). +//! config for `build_and_persist`. use super::super::super::super::result::DdlError; use super::build_flags::err; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/create/engine_option/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/create/engine_option/mod.rs index 43a819679..86e3c02c7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/create/engine_option/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/create/engine_option/mod.rs @@ -6,13 +6,8 @@ //! `TYPE `, `WITH (profile='...')`, or bare `WITH (vector_field='...')` //! without an explicit engine — are rejected hard with a helpful SQLSTATE error. //! -//! Relocated from `pgwire::ddl::collection::create::engine_option` (now -//! deleted): `validate_engine_name` had exactly one caller -//! (`create::build::build_and_persist`), which moved here too, so keeping this -//! module on the pgwire side would have left neutral importing back across the -//! pgwire boundary for no other reason. Only the error envelope changed -//! (`PgWireResult` → `Result<_, DdlError>`); every SQLSTATE code and message is -//! byte-identical. +//! `validate_engine_name` has one caller (`create::build::build_and_persist`). +//! Errors use `Result<_, DdlError>`. pub mod parse; pub mod validate; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/create/engine_option/parse.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/create/engine_option/parse.rs index 545b5167e..30082b8dc 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/create/engine_option/parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/create/engine_option/parse.rs @@ -4,9 +4,7 @@ //! engine-name suggestions for the deprecated `TYPE ` axis. //! //! `parse_engine_option` has no production caller today (the typed-AST path -//! uses `validate_engine_name` instead) — it is kept, with its test suite, -//! exactly as it was on the pgwire side; only the error envelope changed from -//! `PgWireResult` to `Result<_, DdlError>`. +//! uses `validate_engine_name` instead). Errors use `Result<_, DdlError>`. use nodedb_sql::parser::preprocess::lex::find_ascii_case_insensitive; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/create/handler.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/create/handler.rs index a370d4f86..3f09e2f97 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/create/handler.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/create/handler.rs @@ -2,8 +2,7 @@ //! The `create_collection` handler. //! -//! Relocated verbatim from the pgwire `pgwire::ddl::collection::create::handler` -//! module (now deleted). Thin wrapper over [`super::build::build_and_persist`] — +//! Thin wrapper over [`super::build::build_and_persist`] — //! the entire validation + storage + replication body is shared with the //! [`super::table::create_table`] path; the only collection-specific knobs are //! the labels and the schemaless-by-default engine mapping. @@ -13,8 +12,8 @@ use nodedb_types::DatabaseId; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; -use super::super::super::super::result::{DdlError, DdlResult}; -use super::build::{Variant, build_and_persist}; +use super::super::super::super::result::DdlError; +use super::build::{CreatedCollection, Variant, build_and_persist}; use super::request::CreateCollectionRequest; /// CREATE COLLECTION [( , ...)] [WITH (engine='...')] @@ -30,7 +29,7 @@ pub async fn create_collection( identity: &AuthenticatedIdentity, req: &CreateCollectionRequest<'_>, database_id: DatabaseId, -) -> Result, DdlError> { +) -> Result { build_and_persist( state, identity, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/create/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/create/mod.rs index c59a645ad..f0cda961f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/create/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/create/mod.rs @@ -2,7 +2,7 @@ //! `CREATE COLLECTION` / `CREATE TABLE` DDL — split by concern. //! -//! Relocated from `pgwire::ddl::collection::create` (now deleted): +//! Modules: //! - [`build`] — the shared `build_and_persist` body + `Variant` //! - [`build_flags`] — name / flag validation for `build` //! - [`build_primary_engine`] — vector-primary resolution for `build` diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/create/request.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/create/request.rs index 8121b5b37..77a7cb970 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/create/request.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/create/request.rs @@ -2,9 +2,6 @@ //! Parsed request struct shared by `CREATE COLLECTION` and `CREATE TABLE`. //! -//! Relocated verbatim from the pgwire -//! `pgwire::ddl::collection::create::request` module (now deleted). The -//! pgwire typed-AST router (`router/ast/async_ops.rs`) does not call this: //! `CreateCollection` / `CreateTable` are handled by the neutral router //! directly. diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/create/table.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/create/table.rs index f919b9f34..63f21e1d4 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/create/table.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/create/table.rs @@ -2,8 +2,7 @@ //! `CREATE TABLE` DDL handler — strict-default Postgres-style syntax. //! -//! Relocated verbatim from the pgwire `pgwire::ddl::collection::create::table` -//! module (now deleted). Thin wrapper over [`super::build::build_and_persist`] — +//! Thin wrapper over [`super::build::build_and_persist`] — //! the entire validation + storage + replication body is shared with the //! [`super::handler::create_collection`] path; the only TABLE-specific knobs //! are the labels, the mandatory column list, and the strict-by-default engine @@ -14,8 +13,8 @@ use nodedb_types::DatabaseId; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; -use super::super::super::super::result::{DdlError, DdlResult}; -use super::build::{Variant, build_and_persist}; +use super::super::super::super::result::DdlError; +use super::build::{CreatedCollection, Variant, build_and_persist}; use super::request::CreateCollectionRequest; /// Handle `CREATE [IF NOT EXISTS] TABLE () [WITH (engine='...')]`. @@ -32,7 +31,7 @@ pub async fn create_table( identity: &AuthenticatedIdentity, req: &CreateCollectionRequest<'_>, database_id: DatabaseId, -) -> Result, DdlError> { +) -> Result { build_and_persist( state, identity, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/describe.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/describe.rs index 62731497a..74157522a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/describe.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/describe.rs @@ -2,10 +2,8 @@ //! Protocol-neutral DESCRIBE COLLECTION and SHOW COLLECTIONS DDL. //! -//! Ported from the pgwire `ddl::collection::describe` handlers. The catalog -//! reads, row ordering, and error paths are preserved verbatim; only the result -//! construction changed from pgwire `Response` / `QueryResponse` to the -//! protocol-neutral `DdlResult` over `ShapedRows`. +//! The catalog reads, row ordering, and error paths run here. The result is +//! the protocol-neutral `DdlResult` over `ShapedRows`. use nodedb_types::DatabaseId; use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs index 0d6573484..ffcf28728 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/insert.rs @@ -2,9 +2,7 @@ //! INSERT INTO dispatch for schemaless, KV, and columnar collections. //! -//! Relocated verbatim from the pgwire `ddl::collection::insert` handler (now -//! deleted) except for the result type, which is [`DdlError`] / [`DdlResult`] -//! instead of pgwire `Response` / `PgWireResult`. +//! The result type is [`DdlError`] / [`DdlResult`]. use nodedb_physical::physical_plan::VectorOp; use nodedb_types::DatabaseId; @@ -16,11 +14,14 @@ use crate::control::server::shared::session::{DmlTxnCtx, PendingFieldInference}; use crate::control::state::SharedState; use super::indexed_vector_fields::indexed_vector_fields; +use super::parse::ParsedInsert; use super::parse::{ authorize_write_target, dispatch_plan, extract_vector_fields, fields_to_insert_sql, parse_write_statement, plan_and_dispatch, }; use super::triggers::{fire_before_triggers, fire_instead_triggers, fire_sync_after_triggers}; +use crate::control::trigger::statement_txn::{fires_joined_body, in_block, with_statement_txn}; +use crate::control::trigger::{DmlEvent, SyncFire, TriggerScope}; /// INSERT INTO (col1, col2, ...) VALUES (val1, val2, ...) pub async fn insert_document( @@ -39,36 +40,64 @@ pub async fn insert_document( return Some(Err(error)); } + // A write that fires a BEFORE, INSTEAD OF or SYNC AFTER body runs in its + // statement's transaction together with the bodies. + let implicit = !in_block(txn_ctx) + && fires_joined_body( + state, + TriggerScope { + database_id, + tenant_id: identity.tenant_id, + }, + &parsed.coll_name, + DmlEvent::Insert, + ); + Some( + with_statement_txn( + state, + identity, + txn_ctx, + implicit, + async |ctx: &DmlTxnCtx<'_>| { + insert_parsed(state, identity, database_id, &parsed, ctx).await + }, + ) + .await, + ) +} + +/// Insert one parsed document on the statement's transaction `txn_ctx`. +async fn insert_parsed( + state: &SharedState, + identity: &AuthenticatedIdentity, + database_id: DatabaseId, + parsed: &ParsedInsert, + txn_ctx: &DmlTxnCtx<'_>, +) -> Result, DdlError> { let tenant_id = identity.tenant_id; - // Fire INSTEAD OF INSERT triggers — if handled, skip normal dispatch. - if let Some(result) = fire_instead_triggers( + let fire = SyncFire { state, identity, - database_id, - tenant_id, - &parsed.coll_name, - &parsed.fields, - "INSERT", - ) - .await + scope: TriggerScope { + database_id, + tenant_id, + }, + cascade_depth: 0, + txn: txn_ctx, + }; + + // Fire INSTEAD OF INSERT triggers — if handled, skip normal dispatch. + if let Some(result) = + fire_instead_triggers(fire, &parsed.coll_name, &parsed.fields, "INSERT").await { - return Some(result); + return result; } - // Fire BEFORE INSERT triggers — may reject via RAISE EXCEPTION, may mutate NEW fields. - let fields = match fire_before_triggers( - state, - identity, - database_id, - tenant_id, - &parsed.coll_name, - &parsed.fields, - ) - .await - { + // Fire BEFORE INSERT triggers — can reject via RAISE EXCEPTION, can mutate NEW fields. + let fields = match fire_before_triggers(fire, &parsed.coll_name, &parsed.fields).await { Ok(f) => f, - Err(e) => return Some(e), + Err(e) => return e, }; // Auto-generate sequence values for fields with sequence_name where the @@ -101,11 +130,11 @@ pub async fn insert_document( fields.insert(field_def.name.clone(), typed_val); } Err(e) => { - return Some(Err(DdlError::from_error( + return Err(DdlError::from_error( &crate::control::sequence::error_map::sequence_error_to_error( seq_name, e, ), - ))); + )); } } } @@ -127,10 +156,10 @@ pub async fn insert_document( ) { let (_severity, code, message) = error_code_to_sqlstate(&violation); - return Some(Err(DdlError::new(code, message))); + return Err(DdlError::new(code, message)); } - // General CHECK constraints (Control Plane enforcement, may have subqueries). + // General CHECK constraints (Control Plane enforcement, can have subqueries). if !coll_def.check_constraints.is_empty() && let Err(e) = crate::control::server::shared::check_constraint::enforce_check_constraints( @@ -142,7 +171,7 @@ pub async fn insert_document( ) .await { - return Some(Err(e)); + return Err(e); } } @@ -166,7 +195,7 @@ pub async fn insert_document( type_name, label, ) { - return Some(Err(ddl_err("22P02", msg))); + return Err(ddl_err("22P02", msg)); } } } @@ -175,8 +204,8 @@ pub async fn insert_document( // Build SQL from fields and route through nodedb-sql → sql_plan_convert. // This ensures all engine-type routing goes through the shared EngineRules. // The statement is REBUILT from `fields`, so the author's `RETURNING` list - // has to be re-attached here or the planner would never see it and the - // clause would be silently dropped. + // has to be re-attached here or the planner will never see it and the + // clause will be silently dropped. let mut insert_sql = fields_to_insert_sql(&parsed.coll_name, &fields); if let Some(ref columns) = parsed.returning_clause { insert_sql.push_str(" RETURNING "); @@ -189,16 +218,19 @@ pub async fn insert_document( database_id, &insert_sql, txn_ctx, + // The statement fired its INSTEAD OF, BEFORE and SYNC AFTER bodies + // around this write itself. + false, ) .await { Ok(rows) => rows, - Err(e) => return Some(Err(e)), + Err(e) => return Err(e), }; // Track field names in catalog for schemaless collections. Learned fields // are part of the replicated descriptor, so they go out through the - // metadata path with a stamped version; a bare `put_collection` would leave + // metadata path with a stamped version; a bare `put_collection` will leave // the local record byte-different at the same version and wedge the applier. if parsed .collection_type @@ -206,7 +238,7 @@ pub async fn insert_document( .is_none_or(|ct| ct.is_schemaless()) { // Inside a transaction the merge is deferred to COMMIT. Bumping the - // descriptor now would move the version out from under this + // descriptor now will move the version out from under this // transaction's own buffered writes and drain against the lease this // very session is holding for them, which can never clear. let pending = PendingFieldInference { @@ -223,33 +255,26 @@ pub async fn insert_document( pending.database_id, pending.tenant_id, &pending.collection, + None, &pending.fields, ) + .await { - return Some(Err(DdlError::from_error_in_context( + return Err(DdlError::from_error_in_context( "record inferred schema fields", &e, - ))); + )); } } // Fire SYNC AFTER INSERT triggers. - if let Some(err) = fire_sync_after_triggers( - state, - identity, - database_id, - tenant_id, - &parsed.coll_name, - &fields, - ) - .await - { - return Some(err); + if let Some(err) = fire_sync_after_triggers(fire, &parsed.coll_name, &fields).await { + return err; } // Dispatch VectorInsert for the numeric-array fields no vector index // covers. The document write above already indexed the covered ones, and - // a second insert would append a second HNSW node for the same row. + // a second insert will append a second HNSW node for the same row. let indexed = match indexed_vector_fields( state, database_id, @@ -258,7 +283,7 @@ pub async fn insert_document( parsed.collection_type.as_ref(), ) { Ok(indexed) => indexed, - Err(e) => return Some(Err(e)), + Err(e) => return Err(e), }; let vec_vshard = nodedb_types::CollectionKey::from_bare(database_id, &parsed.coll_name).vshard(); @@ -283,23 +308,27 @@ pub async fn insert_document( ) && entry.metadata.strict_dimensions && entry.metadata.dimensions != dim { - return Some(Err(ddl_err( + return Err(ddl_err( "23514", format!( "strict_dimensions: vector has {} dimensions, model '{}' requires {}", dim, entry.metadata.model, entry.metadata.dimensions ), - ))); + )); } } - let surrogate = match state.surrogate_assigner.assign( + let surrogate = match crate::control::server::surrogate_exchange::assign_surrogate_routed( + state, nodedb_types::CollectionKey::from_bare(database_id, &parsed.coll_name), tenant_id, parsed.doc_id.as_bytes(), - ) { + crate::types::TraceId::ZERO, + ) + .await + { Ok(s) => s, Err(e) => { - return Some(Err(DdlError::from_error_in_context("surrogate assign", &e))); + return Err(DdlError::from_error_in_context("surrogate assign", &e)); } }; let vec_plan = crate::bridge::envelope::PhysicalPlan::Vector(VectorOp::Insert { @@ -312,22 +341,26 @@ pub async fn insert_document( provenance: None, }); - if let Some(err) = dispatch_plan(state, identity, database_id, vec_vshard, vec_plan).await { - return Some(err); + // The vector write joins the statement's transaction with the + // document write, so both commit or neither does. + if let Some(err) = + dispatch_plan(state, identity, database_id, vec_vshard, vec_plan, txn_ctx).await + { + return err; } } if !returned_rows.is_empty() { - return Some(Ok(returned_rows)); + return Ok(returned_rows); } // A single-document `{ ... }` insert without RETURNING always applies // exactly one row — the Postgres `INSERT ` tag needs a real // count, not a bare `INSERT` (which real `psql` cannot parse). - Some(Ok(vec![DdlResult::Status { + Ok(vec![DdlResult::Status { command: "INSERT".to_string(), rows_affected: Some(1), - }])) + }]) } /// The `(field, sql_type)` pairs a schemaless write contributes to the diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse.rs index ccf08c8f5..20fc59444 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse.rs @@ -4,14 +4,16 @@ //! helpers. mod dispatch; +mod dispatch_write; mod encoding; mod statement; mod types; -pub(in crate::control::server::shared::ddl::neutral::collection) use dispatch::{ - authorize_write_target, dispatch_plan, plan_and_dispatch, +pub(in crate::control::server::shared::ddl::neutral::collection) use dispatch::plan_and_dispatch; +pub(in crate::control::server::shared::ddl::neutral::collection) use dispatch_write::{ + authorize_write_target, dispatch_plan, }; pub(in crate::control::server::shared::ddl::neutral::collection) use encoding::fields_to_insert_sql; pub(super) use encoding::fields_to_upsert_sql; pub(super) use statement::parse_write_statement; -pub(super) use types::extract_vector_fields; +pub(super) use types::{ParsedInsert, extract_vector_fields}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs index 92f0373ff..73730c139 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch.rs @@ -1,125 +1,45 @@ // SPDX-License-Identifier: BUSL-1.1 +//! Plan and dispatch one rebuilt collection-DML statement. + use std::sync::Arc; use crate::control::planner::context::PlanSecurityContext; use crate::control::security::audit::ArcAuditEmitter; -use crate::control::security::identity::{AuthenticatedIdentity, Permission}; +use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::request_scope::RequestAuthScope; -use crate::control::sequence::SessionSequenceAccess; use crate::control::server::pgwire::types::error_to_sqlstate; -use crate::control::server::response_shape::compose::{ShapeOutcome, shape_response_materialized}; -use crate::control::server::response_shape::redaction::QueryRedaction; -use crate::control::server::response_shape::request::MaterializedShapeRequest; -use crate::control::server::response_shape::types::{PlanKind, ShapedRows}; -use crate::control::server::shared::authorization::{ - AuthorizationError, AuthorizedTaskSet, authorize_collection, authorize_task_set, +use crate::control::server::response_shape::types::ShapedRows; +use crate::control::server::shared::authorization::{AuthorizedTaskSet, authorize_task_set}; +use crate::control::server::shared::clone_write::{ + CloneCheckedOutcome, InterceptAndAuthorizeParams, intercept_and_authorize, }; use crate::control::server::shared::ddl::result::{DdlError, DdlResult}; use crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate; use crate::control::server::shared::returning; -use crate::control::server::shared::session::{ - DmlTxnCtx, InTxnRoute, StagingGateError, route_in_tx_write, +use crate::control::server::shared::session::DmlTxnCtx; +use crate::control::server::shared::txn_route::{ + StatementEvents, TxnTaskContext, TxnTaskOutcome, route_txn_task, }; use crate::control::server::shared::write_admission::all_writes_bufferable; use crate::control::state::SharedState; use crate::types::TraceId; +use super::dispatch_write::{ + ReturningShape, authorization_error_to_ddl, dispatch_staged, error_to_ddl, staging_error_to_ddl, +}; use super::types::ddl_err; -/// Dispatch a write plan on the durable route, returning an error response on -/// failure. `None` means the write applied. -pub(in crate::control::server::shared::ddl::neutral::collection) async fn dispatch_plan( - state: &SharedState, - identity: &AuthenticatedIdentity, - database_id: crate::types::DatabaseId, - vshard_id: crate::types::VShardId, - plan: crate::bridge::envelope::PhysicalPlan, -) -> Option, DdlError>> { - let task = nodedb_physical::physical_task::PhysicalTask { - tenant_id: identity.tenant_id, - database_id, - vshard_id, - plan, - post_set_op: nodedb_physical::physical_task::PostSetOp::None, - txn_id: None, - }; - let emitter = ArcAuditEmitter(Arc::clone(&state.audit)); - let checked = match crate::control::server::shared::clone_write::intercept_and_authorize( - crate::control::server::shared::clone_write::InterceptAndAuthorizeParams { - state, - task, - identity, - tenant_id: identity.tenant_id, - permissions: &state.permissions, - roles: &state.roles, - emitter: &emitter, - }, - ) - .await - { - Ok(crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(_)) => { - return None; - } - Ok(crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed(checked)) => { - checked - } - Err(error) => { - let (_, sqlstate, message) = error_to_sqlstate(&error); - return Some(Err(ddl_err(sqlstate, message))); - } - }; - - // The durable route: Raft in cluster mode, else the funnel's `AppendHere`. - match crate::control::server::dispatch_utils::dispatch_authorized_durable_write( - state, - checked, - TraceId::ZERO, - ) - .await - { - Err(error) => { - let (_, sqlstate, message) = error_to_sqlstate(&error); - Some(Err(ddl_err(sqlstate, message))) - } - // A refusal arrives as an error status inside an `Ok` response. - Ok(response) if response.status == crate::bridge::envelope::Status::Error => { - Some(Err(match response.error_code.as_deref() { - Some(code) => { - let (_, sqlstate, message) = error_code_to_sqlstate(code); - ddl_err(sqlstate, message) - } - None => DdlError::internal("unknown data plane error"), - })) - } - Ok(_) => None, - } -} - -/// Authorize a write target before triggers, sequences, or catalog reads run. -pub(in crate::control::server::shared::ddl::neutral::collection) fn authorize_write_target( - state: &SharedState, - identity: &AuthenticatedIdentity, - database_id: crate::types::DatabaseId, - collection: &str, -) -> Result<(), DdlError> { - let emitter = ArcAuditEmitter(Arc::clone(&state.audit)); - authorize_collection( - identity, - database_id, - collection, - Permission::Write, - &state.permissions, - &state.roles, - &emitter, - ) - .map_err(authorization_error_to_ddl) -} - /// Plan SQL through nodedb-sql, authorize the final task set, and dispatch it. /// /// Returns the rows a `RETURNING` clause on `sql` produced, empty when the -/// statement carries none. The rows are decoded from the Data Plane's own +/// statement carries none. +/// +/// Inside a transaction block every task takes the shared `txn_route`: its +/// BEFORE, INSTEAD OF and SYNC AFTER bodies join the transaction when +/// `fire_triggers` is set (a caller that fired them around the statement +/// itself clears it), a shadowed clone takes its copy-on-write steps, and +/// the write stages. The rows are decoded from the Data Plane's own /// response — the STORED post-image — and are redacted before they leave, so /// this path answers `RETURNING` exactly as the pgwire planner does rather than /// echoing back the values the caller submitted. @@ -130,6 +50,7 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a database_id: crate::types::DatabaseId, sql: &str, txn_ctx: &DmlTxnCtx<'_>, + fire_triggers: bool, ) -> Result, DdlError> { // The clause is stripped from the rebuilt statement before planning. The // planner resolves the item text against the planned target and attaches @@ -142,20 +63,19 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a // This is a client statement — the object-literal `INSERT INTO c { … }` and // `UPSERT` forms land here after being rewritten to standard SQL — so it // plans under the requester's own scope, the same one it is authorized and - // metered as below. Planning it as the system would apply no row policy to - // it: read filters would not be injected, and the write gates would decide + // metered as below. Planning it as the system will apply no row policy to + // it: read filters will not be injected, and the write gates will decide // nothing, on a transport a client can reach directly. // // Injection happens inside planning, before the task set is consumed: // implicit-edge extraction, authorization, staging, and dispatch all read - // `tasks` after this point, and injecting later would hand them + // `tasks` after this point, and injecting later will hand them // un-injected copies. let (mut tasks, output_schema, versions) = { let scope = RequestAuthScope::for_database(identity, state.auth_stores(), database_id); - let permission_cache = - crate::control::security::auth_fence::permission_view(state, tenant_id) - .await - .map_err(|error| DdlError::from_error(&error))?; + crate::control::security::auth_fence::admit_permission_view(state, tenant_id) + .await + .map_err(|error| DdlError::from_error(&error))?; let sec = PlanSecurityContext { identity, auth: scope.auth(), @@ -163,7 +83,9 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a redaction_store: &state.redaction, permissions: &state.permissions, roles: &state.roles, - permission_cache: Some(&*permission_cache), + permission_tree: crate::control::planner::context::PermissionTreeSource::Live( + &state.permission_cache, + ), }; let query_ctx = crate::control::planner::context::QueryContext::for_state(state); let (tasks, output_schema, versions, _) = query_ctx @@ -249,11 +171,12 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a // Admission follows final authorization so an implicit-edge target denied // by policy does not consume a descriptor lease. The scope remains live // through expansion's successors, transaction staging, and dispatch below. - let plan_lease_scope = - Arc::new(state.acquire_plan_lease_scope(&versions).map_err(|error| { + let plan_lease_scope = Arc::new(state.acquire_plan_lease_scope(&versions).await.map_err( + |error| { let (_, sqlstate, message) = error_to_sqlstate(&error); ddl_err(sqlstate, message) - })?); + }, + )?); // A statement dispatched to Calvin as autocommit escapes the transaction // buffer entirely: it applies durably at statement time and ROLLBACK cannot @@ -271,7 +194,6 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a let sum_read_vshards = crate::control::planner::calvin::read_vshards_of(&sum_target_reads) .map_err(|error| DdlError::from_error(&error))?; if !in_txn_block - && state.sequencer_inbox.get().is_some() && matches!( crate::control::planner::calvin::classify_dispatch(&tasks, &sum_read_vshards), crate::control::planner::calvin::DispatchClass::MultiShard { .. } @@ -293,7 +215,7 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a })?; // A cross-shard Calvin dispatch returns no per-task payload here, so // there is no stored row to project. Refused rather than answered with - // an empty row set, which would read as "the write matched nothing". + // an empty row set, which will read as "the write matched nothing". if has_returning { return Err(ddl_err( "0A000", @@ -303,121 +225,93 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a return Ok(Vec::new()); } + // The row images the statement's cross-shard balances were settled from + // join the transaction's read set, so COMMIT's conflict check covers + // them. + if in_txn_block && !sum_target_reads.is_empty() { + txn_ctx + .sessions + .record_read_entries(txn_ctx.session_id, sum_target_reads); + } + + let auth_scope = RequestAuthScope::for_database(identity, state.auth_stores(), database_id); + let route_ctx = in_txn_block.then(|| TxnTaskContext { + state, + identity, + auth: auth_scope.auth(), + txn: txn_ctx, + lease_scope: &plan_lease_scope, + fire_triggers, + }); + let returning_shape = ReturningShape { + state, + identity, + database_id, + txn_ctx, + output_schema: &output_schema, + }; + let mut statement_events = StatementEvents::default(); let mut returned_rows: Option = None; - let statement_buffer_start = txn_ctx.sessions.buffered_task_count(txn_ctx.session_id); for (task, initial_authorized) in tasks.into_iter().zip(authorized_tasks.into_tasks()) { - let routed = route_in_tx_write( - state, - txn_ctx.sessions, - txn_ctx.session_id, - task, - |staged| async move { - let emitter = ArcAuditEmitter(Arc::clone(&state.audit)); - match crate::control::server::shared::clone_write::intercept_and_authorize( - crate::control::server::shared::clone_write::InterceptAndAuthorizeParams { - state, - task: staged, - identity, - tenant_id, - permissions: &state.permissions, - roles: &state.roles, - emitter: &emitter, - }, - ) - .await? + drop(initial_authorized); + let task = match route_ctx.as_ref() { + Some(route) => { + let plan = task.plan.clone(); + match route_txn_task(route, task, &mut statement_events, |staged| { + dispatch_staged(state, identity, staged) + }) + .await + .map_err(staging_error_to_ddl)? { - crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled( - resp, - ) => Ok(resp), - crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed( - checked, - ) => { - crate::control::server::dispatch_utils::dispatch_authorized_to_data_plane( - state, - checked, - TraceId::ZERO, - ) - .await + TxnTaskOutcome::Dispatch(task) => *task, + TxnTaskOutcome::Staged(staged) => { + if has_returning { + for rows in &staged.returning_rows { + returning_shape.fold(&plan, rows, &mut returned_rows)?; + } + } + continue; } - } - }, - ) - .await; - - if txn_ctx.sessions.buffered_task_count(txn_ctx.session_id) > statement_buffer_start - && !txn_ctx.sessions.attach_tx_lease_scope_since( - txn_ctx.session_id, - statement_buffer_start, - Arc::clone(&plan_lease_scope), - ) - { - return Err(DdlError::internal( - "internal error: failed to retain descriptor leases for buffered transaction tasks", - )); - } - - let task = match routed { - Ok(InTxnRoute::Read(task) | InTxnRoute::Autocommit(task)) => *task, - Ok(InTxnRoute::Buffered) | Ok(InTxnRoute::Staged(_)) => { - drop(initial_authorized); - // A buffered/staged write produces its rows at COMMIT, not - // here, so the clause cannot be answered on this path. Refused - // through the shared rule so this transport's message is the - // one the pgwire and native loops give for the same limitation. - if has_returning { - let (_, sqlstate, message) = - error_to_sqlstate(&returning::in_transaction_returning_unsupported()); - return Err(ddl_err(sqlstate, message)); - } - continue; - } - Err(StagingGateError::Dispatch(error)) => { - let (_, sqlstate, message) = error_to_sqlstate(&error); - return Err(ddl_err(sqlstate, message)); - } - Err(StagingGateError::Rejected { code }) => { - return Err(match code { - Some(code) => { - let (_, sqlstate, message) = error_code_to_sqlstate(&code); - ddl_err(sqlstate, message) + TxnTaskOutcome::CloneHandled(resp) => { + if has_returning { + returning_shape.fold( + &plan, + resp.payload.as_bytes(), + &mut returned_rows, + )?; + } + continue; } - None => DdlError::internal("unknown data plane error"), - }); + TxnTaskOutcome::Buffered | TxnTaskOutcome::InsteadOf => continue, + } } + None => task, }; - drop(initial_authorized); let emitter = ArcAuditEmitter(Arc::clone(&state.audit)); - let response = match crate::control::server::shared::clone_write::intercept_and_authorize( - crate::control::server::shared::clone_write::InterceptAndAuthorizeParams { - state, - task: task.clone(), - identity, - tenant_id, - permissions: &state.permissions, - roles: &state.roles, - emitter: &emitter, - }, - ) + let response = match intercept_and_authorize(InterceptAndAuthorizeParams { + state, + task: task.clone(), + identity, + tenant_id, + permissions: &state.permissions, + roles: &state.roles, + emitter: &emitter, + }) .await - .map_err(|error| { - let (_, sqlstate, message) = error_to_sqlstate(&error); - ddl_err(sqlstate, message) - })? { - crate::control::server::shared::clone_write::CloneCheckedOutcome::Handled(resp) => resp, + .map_err(|error| error_to_ddl(&error))? + { + CloneCheckedOutcome::Handled(resp) => resp, // A write takes the durable route: Raft in cluster mode, else the // funnel's `AppendHere`. A read takes the read route. - crate::control::server::shared::clone_write::CloneCheckedOutcome::Proceed(checked) => { + CloneCheckedOutcome::Proceed(checked) => { crate::control::server::dispatch_utils::dispatch_authorized_task_by_class( state, checked, TraceId::ZERO, ) .await - .map_err(|error| { - let (_, sqlstate, message) = error_to_sqlstate(&error); - ddl_err(sqlstate, message) - })? + .map_err(|error| error_to_ddl(&error))? } }; @@ -431,45 +325,16 @@ pub(in crate::control::server::shared::ddl::neutral::collection) async fn plan_a }); } - // Shape the STORED rows the write returned, redacted for the caller — - // the same choke point the pgwire dispatch loop uses, so a redaction - // policy masks identically on both transports. if has_returning { - let scope = RequestAuthScope::for_database(identity, state.auth_stores(), database_id); - let redaction = QueryRedaction::for_plan(tenant_id, scope.auth(), &task.plan); - // A RETURNING expression is evaluated here, per returned row, and - // a sequence accessor in it resolves against this session's - // `currval` map exactly as the pgwire loop's does. - let sequences = SessionSequenceAccess::for_session( - state, - txn_ctx.sessions.sequence_values(txn_ctx.session_id), - database_id, - tenant_id, - ); - let outcome = shape_response_materialized(MaterializedShapeRequest { - payload: response.payload.as_bytes(), - plan: &task.plan, - plan_kind: PlanKind::ReturningRows, - // The statement's announced `RETURNING` columns, so this - // transport renders a returned cell exactly as pgwire does. - projection: Some(&output_schema), - state, - database_id, - tenant_id, - redaction: Some(redaction.ctx(&state.redaction)), - sequences: Some(&sequences), - }) - .map_err(|error| DdlError::from_error(&crate::Error::from(error)))?; - // Folded rather than pushed: a statement is ONE result set, however - // many tasks it planned to. - if let ShapeOutcome::Rows(shaped) = outcome { - match returned_rows { - Some(ref mut accumulated) => accumulated.append(shaped), - None => returned_rows = Some(shaped), - } - } + returning_shape.fold(&task.plan, response.payload.as_bytes(), &mut returned_rows)?; } } + if let Some(route) = route_ctx.as_ref() { + statement_events + .fire(route) + .await + .map_err(|error| error_to_ddl(&error))?; + } Ok(returned_rows.map(DdlResult::Rows).into_iter().collect()) } @@ -483,16 +348,10 @@ fn authorize_final_task_set( .map_err(authorization_error_to_ddl) } -fn authorization_error_to_ddl(error: AuthorizationError) -> DdlError { - DdlError::new( - nodedb_types::error::sqlstate::INSUFFICIENT_PRIVILEGE, - error.resource().to_owned(), - ) -} - #[cfg(test)] mod tests { use super::*; + use crate::control::server::shared::authorization::AuthorizationError; use crate::types::TenantId; #[test] diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch_write.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch_write.rs new file mode 100644 index 000000000..ef17d5e4b --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/parse/dispatch_write.rs @@ -0,0 +1,264 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! Single-write dispatch for the protocol-neutral collection DML: a write +//! plan on the statement's transaction, the staging requests a transaction +//! sends, and the `RETURNING` rows a write answers with. + +use std::sync::Arc; + +use crate::bridge::envelope::{PhysicalPlan, Response}; +use crate::control::security::audit::ArcAuditEmitter; +use crate::control::security::identity::{AuthenticatedIdentity, Permission}; +use crate::control::sequence::SessionSequenceAccess; +use crate::control::server::pgwire::types::error_to_sqlstate; +use crate::control::server::response_shape::compose::{ShapeOutcome, shape_response_materialized}; +use crate::control::server::response_shape::redaction::QueryRedaction; +use crate::control::server::response_shape::request::MaterializedShapeRequest; +use crate::control::server::response_shape::schema::OutputSchema; +use crate::control::server::response_shape::types::{PlanKind, ShapedRows}; +use crate::control::server::shared::authorization::{AuthorizationError, authorize_collection}; +use crate::control::server::shared::clone_write::{ + CloneCheckedOutcome, InterceptAndAuthorizeParams, intercept_and_authorize, write_lease, +}; +use crate::control::server::shared::ddl::result::{DdlError, DdlResult}; +use crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate; +use crate::control::server::shared::session::{ + DmlTxnCtx, InTxnRoute, StagingGateError, TransactionState, route_in_tx_write, +}; +use crate::control::state::SharedState; +use crate::types::TraceId; + +use super::types::ddl_err; + +/// Dispatch a write plan on the statement's transaction `txn_ctx`: inside a +/// transaction block it stages or buffers there and commits with the +/// statement, outside one it takes the durable route. Returns an error +/// response on failure. `None` means the write applied or joined the +/// transaction. +pub(in crate::control::server::shared::ddl::neutral::collection) async fn dispatch_plan( + state: &SharedState, + identity: &AuthenticatedIdentity, + database_id: crate::types::DatabaseId, + vshard_id: crate::types::VShardId, + plan: PhysicalPlan, + txn_ctx: &DmlTxnCtx<'_>, +) -> Option, DdlError>> { + let task = nodedb_physical::physical_task::PhysicalTask { + tenant_id: identity.tenant_id, + database_id, + vshard_id, + plan, + post_set_op: nodedb_physical::physical_task::PostSetOp::None, + txn_id: None, + }; + let sessions = txn_ctx.sessions; + let session_id = txn_ctx.session_id; + if sessions.transaction_state(session_id) == TransactionState::InBlock { + let lease = match write_lease(state, task.tenant_id, database_id, &task.plan).await { + Ok(lease) => Arc::new(lease), + Err(error) => return Some(Err(error_to_ddl(&error))), + }; + let buffer_start = sessions.buffered_task_count(session_id); + let routed = route_in_tx_write(state, sessions, session_id, task, |staged| { + dispatch_staged(state, identity, staged) + }) + .await; + if sessions.buffered_task_count(session_id) > buffer_start + && !sessions.attach_tx_lease_scope_since(session_id, buffer_start, lease) + { + return Some(Err(DdlError::internal( + "internal error: failed to retain descriptor leases for buffered transaction tasks", + ))); + } + return match routed { + Ok(InTxnRoute::Buffered | InTxnRoute::Staged(_)) => None, + // A write the transaction cannot hold will apply at once and + // survive its rollback. + Ok(InTxnRoute::Autocommit(_) | InTxnRoute::Read(_)) => { + let (_, sqlstate, message) = + error_to_sqlstate(&crate::Error::CrossShardInExplicitTransaction); + Some(Err(ddl_err(sqlstate, message))) + } + Err(error) => Some(Err(staging_error_to_ddl(error))), + }; + } + + let emitter = ArcAuditEmitter(Arc::clone(&state.audit)); + let checked = match intercept_and_authorize(InterceptAndAuthorizeParams { + state, + task, + identity, + tenant_id: identity.tenant_id, + permissions: &state.permissions, + roles: &state.roles, + emitter: &emitter, + }) + .await + { + Ok(CloneCheckedOutcome::Handled(_)) => return None, + Ok(CloneCheckedOutcome::Proceed(checked)) => checked, + Err(error) => return Some(Err(error_to_ddl(&error))), + }; + + // The durable route: Raft in cluster mode, else the funnel's `AppendHere`. + match crate::control::server::dispatch_utils::dispatch_authorized_durable_write( + state, + checked, + TraceId::ZERO, + ) + .await + { + Err(error) => Some(Err(error_to_ddl(&error))), + // A refusal arrives as an error status inside an `Ok` response. + Ok(response) if response.status == crate::bridge::envelope::Status::Error => { + Some(Err(match response.error_code.as_deref() { + Some(code) => { + let (_, sqlstate, message) = error_code_to_sqlstate(code); + ddl_err(sqlstate, message) + } + None => DdlError::internal("unknown data plane error"), + })) + } + Ok(_) => None, + } +} + +/// Send one staging request of a transaction to the Data Plane: clone-checked +/// and authorized, as every dispatch is. +pub(super) async fn dispatch_staged( + state: &SharedState, + identity: &AuthenticatedIdentity, + staged: nodedb_physical::physical_task::PhysicalTask, +) -> crate::Result { + let emitter = ArcAuditEmitter(Arc::clone(&state.audit)); + match intercept_and_authorize(InterceptAndAuthorizeParams { + state, + task: staged, + identity, + tenant_id: identity.tenant_id, + permissions: &state.permissions, + roles: &state.roles, + emitter: &emitter, + }) + .await? + { + CloneCheckedOutcome::Handled(resp) => Ok(resp), + CloneCheckedOutcome::Proceed(checked) => { + crate::control::server::dispatch_utils::dispatch_authorized_to_data_plane( + state, + checked, + TraceId::ZERO, + ) + .await + } + } +} + +/// Authorize a write target before triggers, sequences, or catalog reads run. +pub(in crate::control::server::shared::ddl::neutral::collection) fn authorize_write_target( + state: &SharedState, + identity: &AuthenticatedIdentity, + database_id: crate::types::DatabaseId, + collection: &str, +) -> Result<(), DdlError> { + let emitter = ArcAuditEmitter(Arc::clone(&state.audit)); + authorize_collection( + identity, + database_id, + collection, + Permission::Write, + &state.permissions, + &state.roles, + &emitter, + ) + .map_err(authorization_error_to_ddl) +} + +pub(super) fn authorization_error_to_ddl(error: AuthorizationError) -> DdlError { + DdlError::new( + nodedb_types::error::sqlstate::INSUFFICIENT_PRIVILEGE, + error.resource().to_owned(), + ) +} + +/// The DDL error for `error`. +pub(super) fn error_to_ddl(error: &crate::Error) -> DdlError { + let (_, sqlstate, message) = error_to_sqlstate(error); + ddl_err(sqlstate, message) +} + +/// The DDL error for a write its transaction refused. +pub(super) fn staging_error_to_ddl(error: StagingGateError) -> DdlError { + match error { + StagingGateError::Dispatch(error) => error_to_ddl(&error), + StagingGateError::Rejected { code: Some(code) } => { + let (_, sqlstate, message) = error_code_to_sqlstate(&code); + ddl_err(sqlstate, message) + } + StagingGateError::Rejected { code: None } => DdlError::internal("unknown data plane error"), + } +} + +/// Where a statement's `RETURNING` rows are shaped: the STORED rows a write +/// answered with, redacted for the caller through the same choke point the +/// pgwire dispatch loop uses, so a redaction policy masks identically on +/// every transport. +pub(super) struct ReturningShape<'a> { + pub state: &'a SharedState, + pub identity: &'a AuthenticatedIdentity, + pub database_id: crate::types::DatabaseId, + pub txn_ctx: &'a DmlTxnCtx<'a>, + /// The statement's announced `RETURNING` columns, so this transport + /// renders a returned cell exactly as pgwire does. + pub output_schema: &'a OutputSchema, +} + +impl ReturningShape<'_> { + /// Shape `payload`, the `RETURNING` rows `plan`'s write answered with, + /// and fold them into `rows`: a statement is ONE result set, however + /// many tasks it planned to. + pub(super) fn fold( + &self, + plan: &PhysicalPlan, + payload: &[u8], + rows: &mut Option, + ) -> Result<(), DdlError> { + let tenant_id = self.identity.tenant_id; + let scope = crate::control::security::request_scope::RequestAuthScope::for_database( + self.identity, + self.state.auth_stores(), + self.database_id, + ); + let redaction = QueryRedaction::for_plan(tenant_id, scope.auth(), plan); + // A RETURNING expression is evaluated here, per returned row, and a + // sequence accessor in it resolves against this session's `currval` + // map exactly as the pgwire loop's does. + let sequences = SessionSequenceAccess::for_session( + self.state, + self.txn_ctx + .sessions + .sequence_values(self.txn_ctx.session_id), + self.database_id, + tenant_id, + ); + let outcome = shape_response_materialized(MaterializedShapeRequest { + payload, + plan, + plan_kind: PlanKind::ReturningRows, + projection: Some(self.output_schema), + state: self.state, + database_id: self.database_id, + tenant_id, + redaction: Some(redaction.ctx(&self.state.redaction)), + sequences: Some(&sequences), + }) + .map_err(|error| DdlError::from_error(&crate::Error::from(error)))?; + if let ShapeOutcome::Rows(shaped) = outcome { + match rows { + Some(accumulated) => accumulated.append(shaped), + None => *rows = Some(shaped), + } + } + Ok(()) + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/triggers.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/triggers.rs index b9292a1e3..558290c71 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/triggers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/triggers.rs @@ -1,84 +1,36 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Trigger-firing helpers shared by the protocol-neutral INSERT/UPSERT DML -//! handlers. -//! -//! Relocated verbatim from the pgwire `ddl::collection::insert_parse` module -//! (now deleted) except for the result type, which is [`DdlError`] / -//! [`DdlResult`] instead of pgwire `Response` / `PgWireResult`. +//! Trigger-firing helpers of the protocol-neutral INSERT DML handler. Every +//! body joins the statement's transaction, so the statement's write and its +//! bodies' writes commit together. -use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::result::{DdlError, DdlResult}; -use crate::control::state::SharedState; +use crate::control::trigger::SyncFire; /// Fire SYNC AFTER INSERT triggers, returning an error response on failure. +/// Each body joins the statement's transaction `fire.txn`. pub(super) async fn fire_sync_after_triggers( - state: &SharedState, - identity: &AuthenticatedIdentity, - database_id: nodedb_types::DatabaseId, - tenant_id: nodedb_types::TenantId, + fire: SyncFire<'_>, coll_name: &str, fields: &std::collections::HashMap, ) -> Option, DdlError>> { use crate::control::security::catalog::trigger_types::TriggerExecutionMode; if let Err(e) = crate::control::trigger::fire::fire_after_insert( crate::control::trigger::fire::FireAfterInsertParams { - state, - identity, - database_id, - tenant_id, + state: fire.state, + identity: fire.identity, + database_id: fire.scope.database_id, + tenant_id: fire.scope.tenant_id, collection: coll_name, new_fields: fields, - cascade_depth: 0, + cascade_depth: fire.cascade_depth, mode_filter: Some(TriggerExecutionMode::Sync), - // SYNC AFTER triggers fire in the Control-Plane write path, which - // has no source-write LSN/HWM identity; cross-shard origination for - // this path is a tracked follow-up. - cross_shard_origin: None, - on_error: crate::control::trigger::fire_common::FireErrorPolicy::Abort, - only_trigger: None, - }, - ) - .await - .into_result() - { - return Some(Err(DdlError::from_error_in_context("trigger error", &e))); - } - None -} - -/// Fire SYNC AFTER UPDATE triggers, returning an error response on failure. -/// -/// Used by the UPSERT DSL when the probe finds a pre-existing row — -/// without this, AFTER UPDATE subscribers would silently miss overwrite -/// events because the UPSERT handler historically fired only AFTER INSERT. -pub(super) async fn fire_sync_after_update_triggers( - state: &SharedState, - identity: &AuthenticatedIdentity, - database_id: nodedb_types::DatabaseId, - tenant_id: nodedb_types::TenantId, - coll_name: &str, - old_fields: &std::collections::HashMap, - new_fields: &std::collections::HashMap, -) -> Option, DdlError>> { - use crate::control::security::catalog::trigger_types::TriggerExecutionMode; - if let Err(e) = crate::control::trigger::fire_after::fire_after_update( - crate::control::trigger::fire_after::FireAfterUpdateParams { - state, - identity, - database_id, - tenant_id, - collection: coll_name, - old_fields, - new_fields, - cascade_depth: 0, - mode_filter: Some(TriggerExecutionMode::Sync), - // SYNC AFTER triggers run in the Control-Plane write path (no - // source-write LSN/HWM identity); cross-shard origination here is a - // tracked follow-up. + // A SYNC body stages into the statement's transaction on this + // node, which routes each write to its vShard leader. cross_shard_origin: None, on_error: crate::control::trigger::fire_common::FireErrorPolicy::Abort, only_trigger: None, + joined: Some(fire.txn), }, ) .await @@ -91,24 +43,13 @@ pub(super) async fn fire_sync_after_update_triggers( /// Fire INSTEAD OF INSERT triggers, returning the result. pub(super) async fn fire_instead_triggers( - state: &SharedState, - identity: &AuthenticatedIdentity, - database_id: nodedb_types::DatabaseId, - tenant_id: nodedb_types::TenantId, + fire: SyncFire<'_>, coll_name: &str, fields: &std::collections::HashMap, tag: &str, ) -> Option, DdlError>> { - match crate::control::trigger::fire_instead::fire_instead_of_insert( - state, - identity, - database_id, - tenant_id, - coll_name, - fields, - 0, - ) - .await + match crate::control::trigger::fire_instead::fire_instead_of_insert(fire, coll_name, fields) + .await { Ok(crate::control::trigger::fire_instead::InsteadOfResult::Handled) => { // The INSTEAD OF trigger replaced a single-document statement, so @@ -126,25 +67,12 @@ pub(super) async fn fire_instead_triggers( /// Fire BEFORE INSERT triggers, returning mutated fields or an error. pub(super) async fn fire_before_triggers( - state: &SharedState, - identity: &AuthenticatedIdentity, - database_id: nodedb_types::DatabaseId, - tenant_id: nodedb_types::TenantId, + fire: SyncFire<'_>, coll_name: &str, fields: &std::collections::HashMap, ) -> Result, Result, DdlError>> { - match crate::control::trigger::fire_before::fire_before_insert( - state, - identity, - database_id, - tenant_id, - coll_name, - fields, - 0, - ) - .await - { + match crate::control::trigger::fire_before::fire_before_insert(fire, coll_name, fields).await { Ok(f) => Ok(f), Err(e) => Err(Err(DdlError::from_error_in_context( "BEFORE trigger error", diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/upsert.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/upsert.rs index 24a930c4c..4fb7c4ccc 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/dml/upsert.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/dml/upsert.rs @@ -2,26 +2,22 @@ //! UPSERT INTO dispatch for schemaless and KV collections. //! -//! Relocated verbatim from the pgwire `ddl::collection::upsert` handler (now -//! deleted) except for the result type, which is [`DdlError`] / [`DdlResult`] -//! instead of pgwire `Response` / `PgWireResult`. +//! The result type is [`DdlError`] / [`DdlResult`]. use nodedb_types::DatabaseId; use crate::control::security::identity::AuthenticatedIdentity; -use crate::control::security::request_scope::RequestAuthScope; use crate::control::server::shared::ddl::result::{DdlError, DdlResult}; use crate::control::server::shared::ddl::sqlstate::error_code_to_sqlstate; use crate::control::server::shared::session::DmlTxnCtx; use crate::control::state::SharedState; +use super::parse::ParsedInsert; use super::parse::{ authorize_write_target, fields_to_upsert_sql, parse_write_statement, plan_and_dispatch, }; -use super::triggers::{ - fire_before_triggers, fire_instead_triggers, fire_sync_after_triggers, - fire_sync_after_update_triggers, -}; +use crate::control::trigger::statement_txn::{fires_joined_body, in_block, with_statement_txn}; +use crate::control::trigger::{DmlEvent, TriggerScope}; /// UPSERT INTO (col1, col2, ...) VALUES (val1, val2, ...) /// @@ -43,39 +39,57 @@ pub async fn upsert_document( return Some(Err(error)); } - let tenant_id = identity.tenant_id; - - // Fire INSTEAD OF INSERT triggers (upsert treated as INSERT for triggers). - if let Some(result) = fire_instead_triggers( - state, - identity, + // A write that fires a BEFORE, INSTEAD OF or SYNC AFTER body runs in its + // statement's transaction together with the bodies. An UPSERT fires the + // INSERT family before its probe and either family after it. + let scope = TriggerScope { database_id, - tenant_id, - &parsed.coll_name, - &parsed.fields, - "UPSERT", + tenant_id: identity.tenant_id, + }; + let implicit = !in_block(txn_ctx) + && [DmlEvent::Insert, DmlEvent::Update] + .into_iter() + .any(|event| fires_joined_body(state, scope, &parsed.coll_name, event)); + Some( + with_statement_txn( + state, + identity, + txn_ctx, + implicit, + async |ctx: &DmlTxnCtx<'_>| { + upsert_parsed(state, identity, database_id, &parsed, ctx).await + }, + ) + .await, ) - .await - { - return Some(result); - } +} - // Fire BEFORE INSERT triggers — may mutate NEW fields. - let mut fields = match fire_before_triggers( - state, - identity, +/// Upsert one parsed document on the statement's transaction `txn_ctx`. +/// +/// The write takes the shared transaction route with its triggers: the route +/// reads the row the upsert finds, fires the INSERT or the UPDATE family's +/// INSTEAD OF and BEFORE bodies to match, checks the NEW row the BEFORE +/// bodies left against the CHECK constraints, and fires the matching SYNC +/// AFTER bodies once the write staged. +async fn upsert_parsed( + state: &SharedState, + identity: &AuthenticatedIdentity, + database_id: DatabaseId, + parsed: &ParsedInsert, + txn_ctx: &DmlTxnCtx<'_>, +) -> Result, DdlError> { + let tenant_id = identity.tenant_id; + let scope = TriggerScope { database_id, tenant_id, - &parsed.coll_name, - &parsed.fields, - ) - .await - { - Ok(f) => f, - Err(e) => return Some(e), }; + let fires_row_body = [DmlEvent::Insert, DmlEvent::Update] + .into_iter() + .any(|event| fires_joined_body(state, scope, &parsed.coll_name, event)); + let mut fields = parsed.fields.clone(); - // Enforce type guards and CHECK constraints (after BEFORE trigger). + // Inject defaults and enforce type guards and CHECK constraints. The + // route checks a write that fires a row body, after its BEFORE bodies. let catalog = state.credentials.catalog(); if let Ok(Some(coll_def)) = catalog.get_collection(database_id, tenant_id.as_u64(), &parsed.coll_name) @@ -90,11 +104,12 @@ pub async fn upsert_document( ) { let (_severity, code, message) = error_code_to_sqlstate(&violation); - return Some(Err(DdlError::new(code.to_owned(), message))); + return Err(DdlError::new(code.to_owned(), message)); } - // General CHECK constraints (Control Plane enforcement, may have subqueries). - if !coll_def.check_constraints.is_empty() + // General CHECK constraints (Control Plane enforcement, can have subqueries). + if !fires_row_body + && !coll_def.check_constraints.is_empty() && let Err(e) = crate::control::server::shared::check_constraint::enforce_check_constraints( state, @@ -105,7 +120,7 @@ pub async fn upsert_document( ) .await { - return Some(Err(e)); + return Err(e); } } @@ -126,55 +141,12 @@ pub async fn upsert_document( type_name, label, ) { - return Some(Err(ddl_err("22P02", msg))); + return Err(ddl_err("22P02", msg)); } } } } - // Probe for an existing row BEFORE dispatch so the correct AFTER - // trigger class fires: UPSERT onto an existing primary key is an - // UPDATE from every downstream consumer's perspective (AFTER UPDATE - // triggers, CDC, materialized views). Probing ahead of dispatch is - // safe because the document primary key acts as the upsert key and - // the probe + dispatch + AFTER-fire all run serially on this - // connection. - let pk_for_probe = fields - .get("id") - .or_else(|| fields.get("document_id")) - .or_else(|| fields.get("key")) - .map(|v| match v { - nodedb_types::Value::String(s) => s.clone(), - nodedb_types::Value::Integer(i) => i.to_string(), - other => format!("{other:?}"), - }); - let old_fields = if let Some(ref pk) = pk_for_probe { - // The neutral DDL entry point receives an explicit selected database; - // keep `$auth.database_id` identical to the task being probed. - let scope = RequestAuthScope::for_database(identity, state.auth_stores(), database_id); - let row = crate::control::trigger::dml_hook::fetch_old_row( - state, - identity, - database_id, - scope.auth(), - &nodedb_types::QualifiedCollection::new(database_id, &parsed.coll_name), - pk, - ) - .await - .map_err(|error| { - let (_, sqlstate, message) = - crate::control::server::pgwire::types::error_to_sqlstate(&error); - DdlError::new(sqlstate.to_owned(), message) - }); - match row { - Ok(row) if row.is_empty() => None, - Ok(row) => Some(row), - Err(error) => return Some(Err(error)), - } - } else { - None - }; - // Build SQL and route through nodedb-sql → EngineRules → sql_plan_convert. // // The statement is REBUILT from `fields`, so the author's `RETURNING` list @@ -193,55 +165,26 @@ pub async fn upsert_document( database_id, &upsert_sql, txn_ctx, + // The route fires the write's row bodies for the family the row it + // finds makes the upsert. + true, ) .await { Ok(rows) => rows, - Err(e) => return Some(Err(e)), + Err(e) => return Err(e), }; - // Fire the AFTER trigger family that matches the actual mutation: - // AFTER UPDATE when a prior row existed, AFTER INSERT otherwise. - // Firing AFTER INSERT unconditionally would silently skip AFTER - // UPDATE subscribers on overwrites — the exact bug this routing - // fixes. - if let Some(ref old) = old_fields { - if let Some(err) = fire_sync_after_update_triggers( - state, - identity, - database_id, - tenant_id, - &parsed.coll_name, - old, - &fields, - ) - .await - { - return Some(err); - } - } else if let Some(err) = fire_sync_after_triggers( - state, - identity, - database_id, - tenant_id, - &parsed.coll_name, - &fields, - ) - .await - { - return Some(err); - } - if !returned_rows.is_empty() { - return Some(Ok(returned_rows)); + return Ok(returned_rows); } // A single-document `{ ... }` upsert without RETURNING always applies // exactly one row — carry a real count rather than a bare tag. - Some(Ok(vec![DdlResult::Status { + Ok(vec![DdlResult::Status { command: "UPSERT".to_string(), rows_affected: Some(1), - }])) + }]) } /// Build a [`DdlError`] from an ANSI SQLSTATE code and a message. diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/drop.rs index 6a479f455..30903e56e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/drop.rs @@ -22,11 +22,9 @@ //! slice so the `IF EXISTS` and spelling-synonym contracts cannot be //! lost by an off-by-one index into the tokens. //! -//! Ported from the pgwire `ddl::collection::drop` handler. The catalog -//! (propose + single-node fallback), cascade dependent enumeration, soft -//! vs hard delete, implicit-sequence sweep, and audit pair are preserved -//! verbatim; only the result construction changed from pgwire `Response` -//! / `Tag` to the protocol-neutral `DdlResult` / `DdlError`. +//! The catalog (propose + single-node fallback), cascade dependent +//! enumeration, soft vs hard delete, implicit-sequence sweep, and audit pair +//! run here. The result is the protocol-neutral `DdlResult` / `DdlError`. use nodedb_types::DatabaseId; @@ -67,7 +65,7 @@ pub struct DropCollectionRequest<'a> { /// without ownership or admin rights gets `42501` (permission denied) /// regardless of whether the target actually exists — this prevents /// using error-code differences to probe collection existence. -pub fn drop_collection( +pub async fn drop_collection( state: &SharedState, identity: &AuthenticatedIdentity, req: &DropCollectionRequest<'_>, @@ -160,7 +158,7 @@ pub fn drop_collection( } // PURGE requires admin — it bypasses the retention safety net, - // which an owner alone should not be able to invoke. + // which an owner alone cannot invoke. if purge && !is_admin { return Err(err( "42501", @@ -180,7 +178,7 @@ pub fn drop_collection( // // The two idempotency branches (already-deleted, already-purged) // short-circuit with a success tag and skip the audit pair + - // propose — re-running a drop that's already a no-op should not + // propose — re-running a drop that's already a no-op must not // spawn extra raft rounds or audit noise. The `if_exists` case // joins them on the absent-name branch. { @@ -246,6 +244,9 @@ pub fn drop_collection( database_id: database_id.as_u64(), tenant_id: tenant_id.as_u64(), name: name.to_string(), + // Frozen by the proposer's stamp. + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, } } else { crate::control::catalog_entry::CatalogEntry::DeactivateCollection { @@ -258,101 +259,31 @@ pub fn drop_collection( modification_hlc: nodedb_types::Hlc::ZERO, } }; - // Without metadata Raft, acquire the per-name lifecycle guard before the - // catalog mutation and hold it through local reclaim. - let mut local_lifecycle = if state.metadata_raft.get().is_none() { - Some( - state - .quiesce - .try_acquire_lifecycle(database_id.as_u64(), tenant_id.as_u64(), name) - .ok_or_else(|| err("55006", format!("collection '{name}' lifecycle is busy")))?, - ) - } else { - None - }; - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + let outcome = crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|error| DdlError::from_error(&error))?; - if outcome.needs_local_apply() { - let catalog = state.credentials.catalog(); - if purge { - let purge_lsn = state.wal.next_lsn().as_u64(); - let purge_result = tokio::task::block_in_place(|| { - tokio::runtime::Handle::current().block_on(async { - crate::control::server::shared::ddl::neutral::collection::purge::hard_purge_collection( - state, - database_id.as_u64(), - tenant_id.as_u64(), - name, - purge_lsn, - local_lifecycle.is_some(), - ) - .await - }) - }); - if let Err(failure) = purge_result { - // Disarm only when a durable retry record owns the drain; on a - // no-retry failure the guard's unwind Drop releases the hold so - // a same-name CREATE is not wedged behind an orphaned drain. - if failure.retry_queued - && let Some(guard) = local_lifecycle.take() - { - guard.disarm(); - } - panic!("local collection reclaim failed: {}", failure.error); - } - state.permissions.install_replicated_remove_owner( - "collection", - database_id.as_u64(), - tenant_id.as_u64(), - name, - ); - state - .permissions - .remove_grants_for_target(&format!("collection:{}:{name}", tenant_id.as_u64())); - } else { - // Local apply is only reached without a metadata raft group, where - // the entry is never stamped. The sentinel version leaves the - // row's existing ordering metadata in place. - crate::control::catalog_entry::apply::collection::deactivate( - database_id.as_u64(), - tenant_id.as_u64(), - name, - crate::control::catalog_entry::apply::collection::DeactivateStamp { - descriptor_version: 0, - modification_hlc: nodedb_types::Hlc::ZERO, - }, - catalog, - ) - .map_err(|error| { - DdlError::from_error_in_context("catalog deactivate failed", &error) - })?; - } - } - // Cascade: drop implicit sequences (SERIAL/BIGSERIAL fields create {coll}_{field}_seq). + // Cascade: drop implicit sequences (SERIAL/BIGSERIAL fields create + // {coll}_{field}_seq). Each delete replicates, so every node's catalog + // and sequence registry lose it. let catalog = state.credentials.catalog(); - if let Ok(seqs) = catalog.load_sequences_for_tenant(database_id.as_u64(), tenant_id.as_u64()) { - let prefix = format!("{name}_"); - let suffix = "_seq"; - for seq in &seqs { - if seq.name.starts_with(&prefix) && seq.name.ends_with(suffix) { - catalog - .delete_sequence(database_id.as_u64(), tenant_id.as_u64(), &seq.name) - .map_err(|e| { - DdlError::from_error_in_context( - &format!("failed to drop sequence '{}'", seq.name), - &e, - ) - })?; - // Best-effort: registry removal is non-critical since catalog - // is the source of truth and the sequence won't be reloaded. - let _ = state.sequence_registry.remove( - database_id.as_u64(), - tenant_id.as_u64(), - &seq.name, - ); - } - } + let seqs = catalog + .load_sequences_for_tenant(database_id.as_u64(), tenant_id.as_u64()) + .map_err(|e| DdlError::from_error_in_context("failed to list sequences", &e))?; + let prefix = format!("{name}_"); + for seq in seqs + .iter() + .filter(|seq| seq.name.starts_with(&prefix) && seq.name.ends_with("_seq")) + { + let entry = crate::control::catalog_entry::CatalogEntry::DeleteSequence { + database_id: database_id.as_u64(), + tenant_id: tenant_id.as_u64(), + name: seq.name.clone(), + // Frozen by the proposer's stamp. + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, + }; + super::super::replicate::propose_and_apply_async(state, &entry).await?; } // Emit a second audit record with the completion status so the diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/enforcement.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/enforcement.rs index 05273014c..88b7a79d0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/enforcement.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/enforcement.rs @@ -9,6 +9,9 @@ //! parse-then-check entry point DDL uses //! - [`validate_hash_chain_flags`] — HASH_CHAIN implies an append-only //! collection +//! - [`validate_hash_chain_storage`] — HASH_CHAIN is refused on CRDT storage +//! - [`validate_sum_target`] — a HASH_CHAIN collection is never a +//! materialized-sum target //! - [`find_materialized_sum_bindings`] — cross-collection //! materialized_sum lookup //! - [`build_generated_column_specs`] — extract generated-column @@ -48,7 +51,7 @@ pub enum EnforcementDeclError { /// The commit-time check reads `group_key` and `entry_type` as strings; a /// column of any other type yields no entry at all, so the constraint - /// would silently never fire. + /// will silently never fire. #[error( "BALANCED ON: {field} column '{column}' must be a text column for the \ balance check to read it, got '{declared_type}'" @@ -71,12 +74,26 @@ pub enum EnforcementDeclError { declared_type: String, }, - /// A hash chain exists to make retroactive modification detectable, and - /// `verify_chain` walks entries in order: removing or rewriting a chained - /// row reports the SUCCESSOR's link as broken, blaming an untampered row. - /// The chain is only meaningful on a collection that cannot be modified. + /// A hash chain exists to make retroactive modification detectable, so + /// every rewrite or removal of a chained row reads as tampering. The chain + /// is only meaningful on a collection that cannot be modified. #[error("HASH_CHAIN requires APPEND_ONLY")] HashChainRequiresAppendOnly, + + /// A CRDT collection's rows are written by merged sync deltas, which + /// materialize outside the chain. A chain over them cannot cover them. + #[error("HASH_CHAIN cannot be combined with crdt=true")] + HashChainOnCrdt, + + /// A materialized sum rewrites its target row on every source write. A + /// chained row admits no rewrite. + #[error("'{collection}' declares HASH_CHAIN and cannot be a materialized-sum target")] + HashChainAsSumTarget { collection: String }, + + /// CONVERT re-encodes every row, which rewrites the contents each link + /// covers. + #[error("'{collection}' declares HASH_CHAIN and cannot be converted")] + HashChainConvert { collection: String }, } impl EnforcementDeclError { @@ -89,6 +106,9 @@ impl EnforcementDeclError { | Self::MissingField { .. } | Self::InvalidColumnName { .. } | Self::HashChainRequiresAppendOnly => "42601", + Self::HashChainOnCrdt + | Self::HashChainAsSumTarget { .. } + | Self::HashChainConvert { .. } => "42P16", Self::UnknownColumn { .. } => "42703", Self::NonTextKeyColumn { .. } | Self::NonNumericAmountColumn { .. } => "42804", } @@ -106,7 +126,7 @@ impl From for crate::Error { /// `HASH_CHAIN` is only sound on an append-only collection. /// /// Rejecting the contradictory combination is deliberate: silently switching -/// on `APPEND_ONLY` would impose a restriction the user never asked for. +/// on `APPEND_ONLY` will impose a restriction the user never asked for. pub fn validate_hash_chain_flags( hash_chain: bool, append_only: bool, @@ -117,6 +137,27 @@ pub fn validate_hash_chain_flags( Ok(()) } +/// `HASH_CHAIN` is refused on a CRDT collection. +pub fn validate_hash_chain_storage( + hash_chain: bool, + crdt: bool, +) -> Result<(), EnforcementDeclError> { + if hash_chain && crdt { + return Err(EnforcementDeclError::HashChainOnCrdt); + } + Ok(()) +} + +/// A `HASH_CHAIN` collection is refused as a materialized-sum target. +pub fn validate_sum_target(target: &StoredCollection) -> Result<(), EnforcementDeclError> { + if target.hash_chain { + return Err(EnforcementDeclError::HashChainAsSumTarget { + collection: target.name.clone(), + }); + } + Ok(()) +} + /// Parse `BALANCED ON (group_key = col, debit = 'DEBIT', /// credit = 'CREDIT', amount = col)` from the uppercase SQL /// string. Returns `None` if not present. @@ -478,6 +519,30 @@ mod tests { assert!(validate_hash_chain_flags(true, true).is_ok()); } + #[test] + fn a_hash_chained_collection_is_refused_as_a_sum_target() { + let mut target = StoredCollection::new(1, "ledger", "owner"); + assert!(validate_sum_target(&target).is_ok()); + target.hash_chain = true; + let error = validate_sum_target(&target).expect_err("must refuse"); + assert_eq!( + error, + EnforcementDeclError::HashChainAsSumTarget { + collection: "ledger".into() + } + ); + assert_eq!(error.sqlstate(), "42P16"); + } + + #[test] + fn hash_chain_on_crdt_storage_is_refused() { + let error = validate_hash_chain_storage(true, true).expect_err("must refuse"); + assert_eq!(error, EnforcementDeclError::HashChainOnCrdt); + assert_eq!(error.sqlstate(), "42P16"); + assert!(validate_hash_chain_storage(true, false).is_ok()); + assert!(validate_hash_chain_storage(false, true).is_ok()); + } + #[test] fn append_only_without_hash_chain_is_accepted() { assert!(validate_hash_chain_flags(false, true).is_ok()); diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/helpers.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/helpers.rs index 689b99fde..94af343a0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/helpers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/helpers.rs @@ -2,8 +2,7 @@ //! Shared parsing helpers used across collection DDL sub-modules. //! -//! Relocated verbatim from the pgwire `ddl::collection::helpers` module (now -//! deleted): its sole external caller was already `neutral::collection::alter::add_column`. +//! Its sole external caller is `neutral::collection::alter::add_column`. use nodedb_sql::parser::preprocess::lex::{ find_ascii_case_insensitive, find_ascii_case_insensitive_from, diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs index d350dc8f0..3eaaf647e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/build.rs @@ -154,7 +154,7 @@ async fn mark_ready(state: &SharedState, build: &SecondaryIndexBuild) -> Result< idx.state = IndexBuildState::Ready; } } - commit_collection_mutation(state, &ready_coll, build.database_id).await?; + commit_collection_mutation(state, &ready_coll).await?; } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/commit.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/commit.rs index ff7ebfa1f..1251535c0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/commit.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/commit.rs @@ -3,7 +3,6 @@ //! Commit a mutated collection record from an index DDL path. use crate::control::state::SharedState; -use crate::types::DatabaseId; use super::super::super::super::result::DdlError; @@ -11,33 +10,18 @@ pub(super) fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } -/// Commit a mutated [`StoredCollection`] through the replicated metadata -/// Raft group (cluster) or straight to the local `SystemCatalog` -/// (single-node fallback), then re-dispatch a `Register` to this node's -/// Data Plane so the new index vector lands in `doc_configs` immediately. +/// Commit a mutated [`StoredCollection`] through the metadata proposer. Its +/// apply re-dispatches a `Register` to this node's Data Plane, so the new +/// index vector lands in `doc_configs` before the propose returns. /// /// [`StoredCollection`]: crate::control::security::catalog::StoredCollection pub(super) async fn commit_collection_mutation( state: &SharedState, coll: &crate::control::security::catalog::StoredCollection, - database_id: DatabaseId, ) -> Result<(), DdlError> { let entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(coll.clone())); - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error(&e))?; - if outcome.needs_local_apply() { - { - let catalog = state.credentials.catalog(); - catalog - .put_collection(database_id, coll) - .map_err(|e| DdlError::from_error(&e))?; - } - // Single-node path bypasses the applier post-apply hook, so the - // Register refresh has to be fired here. In cluster mode the - // applier's `put_async` does it on every node. - super::super::dispatch_register_from_stored(state, coll) - .await - .map_err(|e| DdlError::from_error(&e))?; - } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs index 9c4f5c330..9085e59e8 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/create.rs @@ -144,7 +144,7 @@ pub async fn create_index( } // Reject a name already taken by an index of any kind in this database: - // the registry is keyed by name, so two kinds sharing one name would make + // the registry is keyed by name, so two kinds sharing one name will make // exactly one of them droppable. The registry read itself must still fail // loudly — only a genuine name collision is absorbed by `IF NOT EXISTS`. if let Some(existing) = catalog @@ -204,7 +204,7 @@ pub async fn create_index( owner: index_owner.clone(), }); - commit_collection_mutation(state, &coll, database_id).await?; + commit_collection_mutation(state, &coll).await?; // Phase 2 and 3: backfill on every node, then flip to Ready. Inside an // explicit transaction the build waits for COMMIT, after the Building @@ -235,7 +235,8 @@ pub async fn create_index( collection, fields: vec![canonical_field.clone()], }, - )?; + ) + .await?; // Ownership record backs authorization for later ALTER / DROP. crate::control::server::shared::ddl::owner::propose_owner( @@ -245,7 +246,8 @@ pub async fn create_index( tenant_id, &index_name, &index_owner, - )?; + ) + .await?; let kind = if is_unique { "unique index" } else { "index" }; let ci = if case_insensitive { diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs index c348f59a1..0db58acc3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/drop.rs @@ -136,7 +136,8 @@ pub async fn drop_index( tenant_id, index_name, &record.collection, - )?; + ) + .await?; crate::control::server::shared::ddl::owner::propose_delete_owner( state, @@ -144,7 +145,8 @@ pub async fn drop_index( database_id.as_u64(), tenant_id, index_name, - )?; + ) + .await?; if in_transaction { super::teardown::teardown(state, &record, database_id, tenant_id).await?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs index ce0ba9a2a..2d91e192c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index/teardown.rs @@ -14,7 +14,7 @@ //! | sparse | none beyond the registry + ownership rows | //! | sorted | an order-statistic tree on the core holding the collection's rows | //! -//! Every failure here propagates. A teardown that logs and continues would +//! Every failure here propagates. A teardown that logs and continues will //! report a successful drop over state that is still live — the same silent //! success that made a vector index undroppable in the first place. //! @@ -23,7 +23,7 @@ //! cannot propagate and files a `Capture` instead. use crate::control::security::catalog::{IndexKind, StoredIndexRecord}; -use crate::control::server::dispatch_utils::{MintedRecords, RecordOwner}; +use crate::control::server::dispatch_utils::MintedRecords; use crate::control::state::SharedState; use crate::types::{DatabaseId, TenantId, TraceId}; @@ -106,7 +106,7 @@ async fn secondary( .find(|i| i.name == record.name) .map(|i| i.field.clone()); coll.indexes.retain(|i| i.name != record.name); - commit_collection_mutation(state, &coll, database_id).await?; + commit_collection_mutation(state, &coll).await?; // Purge existing index entries from the sparse engine so stale rows // cannot leak into lookups on a re-created index of the same name. @@ -148,90 +148,19 @@ async fn secondary( /// /// The catalog row is replicated, and the WAL drop record plus the Data Plane /// drop are node-local physical state, so every node runs them from its own -/// post-apply lane. Only the single-node path has no applier to do that. +/// post-apply lane, this one included, before the propose returns. async fn vector( state: &SharedState, record: &StoredIndexRecord, database_id: DatabaseId, tenant_id: TenantId, ) -> Result<(), DdlError> { - let field_name = record.primary_field().to_string(); - let outcome = super::super::super::vector_replicate::propose_delete_params( + super::super::super::vector_replicate::propose_delete_params( state, database_id.as_u64(), tenant_id.as_u64(), &record.collection, - &field_name, - )?; - // Only the single-node path continues: everywhere else the post-apply - // lane has already appended, fsynced, and dropped on this node. - if !outcome.needs_local_apply() { - return Ok(()); - } - - let plan = crate::bridge::envelope::PhysicalPlan::Vector( - nodedb_physical::physical_plan::VectorOp::DropIndex { - collection: nodedb_types::QualifiedCollection::new(database_id, &record.collection), - field_name: field_name.clone(), - }, - ); - - // WAL first: the `VectorParams` record that created this index is still - // in the log, so without a durable drop record a restart rebuilds the - // index the user just dropped. - // - // The record's outcome-floor window opens before the append and closes - // from the drop's outcome. - let vshard = nodedb_types::CollectionKey::from_bare(database_id, &record.collection).vshard(); - let owner = RecordOwner { - tenant_id, - database_id, - vshard_id: vshard, - }; - let minted = MintedRecords::open(&state.outcome_floor); - let appended = match minted.append_plan( - &state.wal, - owner, - &plan, - // The drop is dispatched as a client statement. - crate::event::EventSource::User, - ) { - Ok(appended) => appended, - Err(e) => { - // Any record appended before the error never reaches a core. - minted.cancel(&state.wal, owner, 0).await.map_err(|c| { - DdlError::from_error_in_context("cancel vector index drop record", &c) - })?; - return Err(DdlError::from_error_in_context( - "persist vector index drop to WAL", - &e, - )); - } - }; - - // An append only buffers. The records this drop cancels were already - // fsynced by the writes that acked them, so a buffered-only drop is lost on - // restart while replay still rebuilds the index from those records. - let Some(lsn) = appended.lsn else { - minted.settle(); - return Err(DdlError::internal("vector index drop minted no WAL record")); - }; - if let Err(e) = state.wal.wait_durable(lsn).await { - // The record can still be on disk, so restart replay can reach it. - minted.hold(); - return Err(DdlError::from_error_in_context( - "fsync vector index drop", - &e, - )); - } - - dispatch( - state, - tenant_id, - database_id, - &record.collection, - plan, - Some(minted), + record.primary_field(), ) .await } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/index_fanout.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/index_fanout.rs index 6ae018608..c04d9516a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/index_fanout.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/index_fanout.rs @@ -7,7 +7,7 @@ //! collection's documents are spread across vShards hosted on //! multiple nodes, and the local dispatch only populates the //! coordinator's vShards. Non-coordinator nodes host rows the -//! coordinator never sees, so the Ready commit would pass with an +//! coordinator never sees, so the Ready commit will pass with an //! incomplete index — a silent-miss class. //! //! This module drives the fan-out: after the coordinator's local @@ -42,7 +42,7 @@ const PEER_BACKFILL_DEADLINE: Duration = Duration::from_secs(120); /// key is `23505` as on the single-node path. A transport fault is /// `XX000`. /// -/// Single-node clusters (no peers) return `Ok(())` immediately — the +/// A cluster with no peers returns `Ok(())` immediately — the /// coordinator's local dispatch already covered everything. /// Inputs to [`backfill_on_peers`]. Mirrors the fields of /// `DocumentOp::BackfillIndex` plus the owning tenant — grouped as a @@ -65,8 +65,8 @@ pub(super) async fn backfill_on_peers( args: PeerBackfill<'_>, ) -> Result<(), DdlError> { let Some(transport) = state.cluster_transport.as_ref() else { - // Non-cluster build / single-node without cluster transport: - // the local dispatch is the only required step. + // No transport is wired, so no peer is reachable: the local + // dispatch is the only step. return Ok(()); }; let Some(topology_lock) = state.cluster_topology.as_ref() else { @@ -125,6 +125,8 @@ pub(super) async fn backfill_on_peers( descriptor_versions: Vec::new(), // Index build fan-out is not session-transaction-scoped. txn_id: None, + vshard_id: None, + read_groups: Vec::new(), }); joins.push(tokio::spawn(async move { let outcome = transport.send_rpc(node_id, req).await; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/mod.rs index 9ea3af058..1bd52e958 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/mod.rs @@ -31,7 +31,7 @@ pub use dml::{insert_document, upsert_document}; pub use drop::{DropCollectionRequest, drop_collection}; pub use index::{CreateIndexRequest, DropIndexRequest, create_index, drop_index}; pub use register::{ - dispatch_register_by_name, dispatch_register_for_sum_sources, dispatch_register_from_stored, + dispatch_register_for_sum_sources, dispatch_register_from_stored, register_proposed_collection, }; pub use show_indexes::show_indexes; pub use undrop::undrop_collection; diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs index acf1db706..bd8222ede 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/purge/dispatch.rs @@ -9,7 +9,8 @@ use std::time::{Duration, Instant}; -use futures::future::join_all; +use futures::StreamExt; +use futures::stream::FuturesUnordered; use nodedb_physical::physical_plan::MetaOp; use crate::bridge::envelope::{PhysicalPlan, Priority, Request, Status}; @@ -20,7 +21,7 @@ use crate::types::{DatabaseId, ReadConsistency, TenantId, TraceId, VShardId}; /// `Ok` acknowledgement from each one. /// /// The catalog row has already been removed when this runs. Returning success -/// after only a subset of cores reclaimed would allow a same-name re-CREATE to +/// after only a subset of cores reclaimed will allow a same-name re-CREATE to /// observe predecessor state, so partial success is always an error. The caller /// records a durable pending-reclaim entry and the applied-index barrier fails /// closed. @@ -83,6 +84,7 @@ pub async fn dispatch_unregister_collection( txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: crate::bridge::envelope::Admission::Exempt( crate::bridge::envelope::ExemptReason::AlreadyOrdered, ), @@ -99,8 +101,11 @@ pub async fn dispatch_unregister_collection( } } - let responses = join_all(receivers.into_iter().map( - |(_request_id, core_id, mut receiver)| async move { + // Each core's acknowledgement counts as apply progress as it arrives, so + // a proposer waiting on this purge sees a slow reclaim move. + let mut pending: FuturesUnordered<_> = receivers + .into_iter() + .map(|(_request_id, core_id, mut receiver)| async move { let response = tokio::time::timeout(timeout, receiver.recv()) .await .map_err(|_| crate::Error::Dispatch { @@ -123,12 +128,21 @@ pub async fn dispatch_unregister_collection( }); } Ok(()) - }, - )) - .await; + }) + .collect(); - for response in responses { - response?; + let mut first_error = None; + while let Some(response) = pending.next().await { + match response { + Ok(()) => { + state + .metadata_apply_progress + .fetch_add(1, std::sync::atomic::Ordering::Release); + } + Err(error) => { + first_error.get_or_insert(error); + } + } } - Ok(()) + first_error.map_or(Ok(()), Err) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/register.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/register.rs index f0674723a..25915cd9f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/register.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/register.rs @@ -5,7 +5,7 @@ //! //! Two entry points: //! - [`dispatch_register_if_needed`] — leader-side, called from -//! the pgwire handler path. Parses the FIELDS clause from +//! the handler path. Parses the FIELDS clause from //! `parts` to derive index paths. //! - [`dispatch_register_from_stored`] — applier-side, called //! from the metadata applier's post-apply hook after a @@ -16,12 +16,9 @@ //! which builds the storage-mode + enforcement-options //! `EnforcementOptions` value and dispatches to the Data Plane. //! -//! Relocated verbatim from the pgwire -//! `pgwire::ddl::collection::create::register` module (now deleted) so the -//! neutral `continuous_agg` / `materialized_view` families, the -//! `catalog_entry::post_apply` hook, and the (still pgwire) `alter` -//! handlers can all depend on a protocol-neutral home instead of reaching -//! across the pgwire boundary. +//! The neutral `continuous_agg` / `materialized_view` families, the +//! `catalog_entry::post_apply` hook, and the pgwire `alter` handlers all +//! depend on this protocol-neutral home. use crate::bootstrap::constraint_reconcile::CollectionSource; use crate::control::security::catalog::StoredCollection; @@ -33,8 +30,8 @@ use nodedb_types::DatabaseId; use super::enforcement::{build_generated_column_specs, find_materialized_sum_bindings}; /// Dispatch a `DocumentOp::Register` to the Data Plane after -/// collection creation (leader-side pgwire path). Looks up the -/// just-created collection from catalog and parses the FIELDS +/// collection creation (leader-side handler path). Looks up the +/// created collection from catalog and parses the FIELDS /// clause from `parts` for index paths. /// /// Returns an error if any Data Plane core fails to acknowledge the @@ -73,31 +70,6 @@ pub async fn dispatch_register_if_needed( dispatch_register_from_stored_inner(state, tenant_id, &coll, indexes).await } -/// Typed leader-side entry point: dispatch `DocumentOp::Register` -/// after collection creation when the collection name is known but -/// no raw SQL parts are available (typed AST path). -/// -/// `database_id` must match the database the collection was created in so the -/// catalog lookup succeeds in non-default databases. -/// -/// Returns an error if any Data Plane core fails to acknowledge the -/// registration. -pub async fn dispatch_register_by_name( - state: &SharedState, - identity: &AuthenticatedIdentity, - name: &str, - database_id: DatabaseId, -) -> crate::Result<()> { - let tenant_id = identity.tenant_id; - let catalog = state.credentials.catalog(); - let Ok(Some(coll)) = catalog.get_collection(database_id, tenant_id.as_u64(), name) else { - return Ok(()); - }; - let mut indexes = derive_auto_indexes(coll.fields.iter().map(|(n, _)| n.as_str())); - extend_with_catalog_indexes(&mut indexes, &coll); - dispatch_register_from_stored_inner(state, tenant_id, &coll, indexes).await -} - /// Applier-side entry point: dispatch `DocumentOp::Register` using /// a fully-populated [`StoredCollection`]. Called from the /// production `MetadataCommitApplier` after it materializes a @@ -119,6 +91,27 @@ pub async fn dispatch_register_from_stored( dispatch_register_from_stored_inner(state, tenant_id, coll, indexes).await } +/// Register `coll` on this node's Data Plane after a handler proposed its +/// `PutCollection` with result `outcome`, when the proposal left that to the +/// handler. +/// +/// - A durable outcome ran the entry's awaited post-apply on this node. That +/// post-apply registered the collection on every core, so this call +/// registers nothing. +/// - A buffered outcome ran no post-apply. The open transaction's later +/// statements read the new shape, so this call registers it. ROLLBACK +/// registers the stored shape again. +pub async fn register_proposed_collection( + state: &SharedState, + outcome: crate::control::propose_outcome::ProposeOutcome, + coll: &StoredCollection, +) -> crate::Result<()> { + if outcome.is_durable() { + return Ok(()); + } + dispatch_register_from_stored(state, coll).await +} + /// Re-register every collection that DRIVES a materialized-sum binding `coll` /// declares, so those sources learn they now drive one. /// @@ -131,7 +124,7 @@ pub async fn dispatch_register_from_stored( /// asserting it drives nothing. Every write into it then folds nothing at all /// and the stored total silently stays where it was. /// -/// Only the co-resident path reads that config, which is why this could go +/// Only the co-resident path reads that config, which is why this can go /// unnoticed: a cross-shard binding is settled on the Control Plane at plan time /// from the catalog and travels on its own `ApplyBalanceDelta` task, which /// consults no Data-Plane config. @@ -242,7 +235,7 @@ fn build_timeseries_schema( })) } -/// Build the `CollectionConfig` a `DocumentOp::Register` would install in +/// Build the `CollectionConfig` a `DocumentOp::Register` will install in /// `doc_configs`, straight from the durable catalog — storage mode, /// enforcement options, generated columns, and secondary indexes. /// @@ -325,7 +318,7 @@ pub(crate) fn build_doc_config_from_stored( // Written as a struct literal with every field named — and deliberately // NOT `..Default::default()`. A field added to `CollectionConfig` that is // never derived from the catalog is invisible at runtime: the collection - // simply behaves as though the attribute was never declared. Naming every + // behaves as though the attribute was never declared. Naming every // field turns that into a compile error here, in the one function both the // live-DDL path and the boot seed go through. crate::engine::document::store::CollectionConfig { @@ -418,7 +411,7 @@ mod tests { /// This is the ONE builder both the live-DDL register broadcast and the /// boot-time `doc_configs` seed go through, so a marker that survives here /// survives a restart too. A marker that existed only after a live CREATE - /// would make a vector-primary collection readable until the first restart + /// will make a vector-primary collection readable until the first restart /// and unreadable after it — the same defect one layer down. #[test] fn a_vector_primary_collection_carries_its_marker_into_the_doc_config() { @@ -438,7 +431,7 @@ mod tests { } /// Every other engine must leave the marker unset, or every collection's - /// rows would be decoded as tagged sidecars. + /// rows will be decoded as tagged sidecars. #[test] fn a_plain_document_collection_carries_no_vector_primary_marker() { let coll = StoredCollection::new(1, "plain_docs", "owner"); diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/undrop.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/undrop.rs index feeac7c9b..b8ff2adfc 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/undrop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/undrop.rs @@ -11,13 +11,11 @@ //! //! Authorization matches `ALTER COLLECTION OWNER TO`: preserved owner, //! superuser, or tenant_admin. If the preserved-owner user no longer -//! exists, only superuser / tenant_admin may undrop; the restore is +//! exists, only superuser / tenant_admin can undrop; the restore is //! audit-logged with an `owner_user_missing` marker. //! -//! Ported from the pgwire `ddl::collection::undrop` handler. The catalog / -//! permission / audit reads and the metadata proposal are preserved verbatim; -//! only the result construction changed from pgwire `Response` / `Tag` to the -//! protocol-neutral `DdlResult` / `DdlError`. +//! The catalog / permission / audit reads and the metadata proposal run +//! here. The result is the protocol-neutral `DdlResult` / `DdlError`. use nodedb_types::DatabaseId; @@ -28,7 +26,7 @@ use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; -pub fn undrop_collection( +pub async fn undrop_collection( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -42,23 +40,6 @@ pub fn undrop_collection( let name = name_lower.as_str(); let tenant_id = identity.tenant_id; - // Metadata Raft serializes clustered lifecycle mutations. The local - // fallback must acquire the same exclusive name guard as CREATE/DROP - // before reading the preserved descriptor, otherwise it can restore an - // incarnation that a concurrent purge has already superseded. - let _local_lifecycle = if state.metadata_raft.get().is_none() { - Some( - state - .quiesce - .try_acquire_lifecycle(database_id.as_u64(), tenant_id.as_u64(), name) - .ok_or_else(|| { - DdlError::new("55006", format!("collection '{name}' lifecycle is busy")) - })?, - ) - } else { - None - }; - let catalog = state.credentials.catalog(); // Look up the soft-deleted record. Three distinct failures: @@ -101,7 +82,7 @@ pub fn undrop_collection( )); } - // If the preserved-owner user no longer exists, only admin may restore. + // If the preserved-owner user no longer exists, only admin can restore. let owner_user_missing = preserved_owner .as_deref() .is_some_and(|u| state.credentials.get_user(u).is_none()); @@ -130,19 +111,10 @@ pub fn undrop_collection( // group. Fresh entry carries `is_active = true` and the preserved // owner (already present on `stored`). stored.is_active = true; - let entry = - crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(stored.clone())); - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + let entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(stored)); + let outcome = crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error(&e))?; - if outcome.needs_local_apply() { - // Single-node fallback: run the same applier the replicated path runs - // on every node, so the restore carries every invariant of a - // `PutCollection` apply — the collection row, its owner row, and the - // visibility of the indexes the soft-delete hid. Writing the row - // directly here restored a collection whose indexes stayed hidden. - crate::control::catalog_entry::apply::collection::put(&stored, catalog) - .map_err(|e| DdlError::from_error_in_context("catalog restore failed", &e))?; - } let completion = UndropAuditDetail::new(name, UndropStage::Completed, owner_user_missing) .with_log_index(outcome.log_index()) diff --git a/nodedb/src/control/server/shared/ddl/neutral/collection/vector_metadata.rs b/nodedb/src/control/server/shared/ddl/neutral/collection/vector_metadata.rs index a2779e14e..e2642e305 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/collection/vector_metadata.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/collection/vector_metadata.rs @@ -6,14 +6,9 @@ //! - `SHOW VECTOR MODELS` — catalog view of all vector columns with model metadata //! - `SELECT VECTOR_METADATA('collection', 'column')` — inline JSON query //! -//! Ported verbatim from the pgwire `ddl::collection::vector_metadata` -//! handlers; only the result construction changed from pgwire `Response` / -//! `QueryResponse` (text fields) to the protocol-neutral [`DdlResult`] over -//! [`ShapedRows`] (all-text columns — every field the pgwire handlers encoded -//! via `text_field`, including `dimensions` / `strict_dimensions`, was a text -//! column). The parsing, catalog reads/writes, `chrono_format_utc` default, -//! SQLSTATE codes / messages, and the `ALTER COLLECTION` command tag are -//! unchanged. +//! The result is a protocol-neutral [`DdlResult`] over +//! [`ShapedRows`] (all-text columns, including `dimensions` / +//! `strict_dimensions`). use nodedb_sql::parser::preprocess::lex::find_ascii_case_insensitive; use nodedb_types::DatabaseId; @@ -32,7 +27,7 @@ fn err(sqlstate: &str, message: impl Into) -> DdlError { } /// Handle `ALTER COLLECTION x SET VECTOR METADATA ON column (model = '...', dimensions = N, ...)`. -pub fn handle_set_vector_metadata( +pub async fn handle_set_vector_metadata( state: &SharedState, identity: &AuthenticatedIdentity, sql: &str, @@ -148,7 +143,7 @@ pub fn handle_set_vector_metadata( }, }; - super::super::vector_replicate::propose_put_model(state, &entry)?; + super::super::vector_replicate::propose_put_model(state, &entry).await?; tracing::info!( %collection, diff --git a/nodedb/src/control/server/shared/ddl/neutral/conflict_policy.rs b/nodedb/src/control/server/shared/ddl/neutral/conflict_policy.rs index 20ce9e541..c0faecbac 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/conflict_policy.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/conflict_policy.rs @@ -28,11 +28,10 @@ use crate::control::server::response_shape::types::ShapedRows; use crate::control::state::SharedState; use crate::types::DatabaseId; -use super::super::catalog::propose_and_apply; +use super::super::catalog::propose_and_apply_async; use super::super::result::{DdlError, DdlResult}; -/// Construct a [`DdlError`], preserving the exact SQLSTATE codes and messages -/// the pgwire handlers produced. +/// Construct a [`DdlError`] from a SQLSTATE code and a message. fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } @@ -80,7 +79,7 @@ pub async fn alter_set_on_conflict( sonic_rs::to_string(&policy).map_err(|e| DdlError::internal(e.to_string()))?; coll.conflict_policy = Some(policy_json); let entry = CatalogEntry::PutCollection(Box::new(coll)); - propose_and_apply(state, &entry)?; + propose_and_apply_async(state, &entry).await?; state.schema_version.bump(); let mut row = Map::new(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/constraint/handlers.rs b/nodedb/src/control/server/shared/ddl/neutral/constraint/handlers.rs index 164697490..1a44098c9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/constraint/handlers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/constraint/handlers.rs @@ -2,11 +2,9 @@ //! Protocol-neutral DDL handlers for adding and dropping constraints. //! -//! Ported from the pgwire `ddl::constraint::handlers`. All non-return logic -//! (token parsing, catalog get/put, duplicate pre-checks, `schema_version.bump`) -//! is preserved verbatim; only the result construction changed from pgwire -//! `Response` / `PgWireError` to the protocol-neutral [`DdlResult`] / -//! [`DdlError`]. +//! The token parsing, catalog get/put, duplicate pre-checks, and +//! `schema_version.bump` run here. The result is the protocol-neutral +//! [`DdlResult`] / [`DdlError`]. use nodedb_types::DatabaseId; @@ -20,7 +18,7 @@ use crate::control::state::SharedState; use super::support::err; /// Handle `ALTER COLLECTION x ADD CONSTRAINT name ON COLUMN col TRANSITIONS (...)`. -pub fn add_state_constraint( +pub async fn add_state_constraint( state: &SharedState, identity: &AuthenticatedIdentity, sql: &str, @@ -78,7 +76,8 @@ pub fn add_state_constraint( } coll.state_constraints.push(def); - persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) + persist_collection_replicated(state, &coll) + .await .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -90,7 +89,7 @@ pub fn add_state_constraint( } /// Handle `ALTER COLLECTION x ADD TRANSITION CHECK name (predicate)`. -pub fn add_transition_check( +pub async fn add_transition_check( state: &SharedState, identity: &AuthenticatedIdentity, sql: &str, @@ -138,7 +137,8 @@ pub fn add_transition_check( } coll.transition_checks.push(def); - persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) + persist_collection_replicated(state, &coll) + .await .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -150,7 +150,7 @@ pub fn add_transition_check( } /// Handle `ALTER COLLECTION x ADD CONSTRAINT name CHECK (expr)`. -pub fn add_check_constraint( +pub async fn add_check_constraint( state: &SharedState, identity: &AuthenticatedIdentity, sql: &str, @@ -226,7 +226,8 @@ pub fn add_check_constraint( } coll.check_constraints.push(def); - persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) + persist_collection_replicated(state, &coll) + .await .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -238,7 +239,7 @@ pub fn add_check_constraint( } /// Handle `DROP CONSTRAINT name ON collection`. -pub fn drop_constraint( +pub async fn drop_constraint( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -284,7 +285,8 @@ pub fn drop_constraint( )); } - persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) + persist_collection_replicated(state, &coll) + .await .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/constraint/parse.rs b/nodedb/src/control/server/shared/ddl/neutral/constraint/parse.rs index 40890e6cc..afa80218d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/constraint/parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/constraint/parse.rs @@ -2,8 +2,7 @@ //! Parsing helpers for constraint DDL: transition rules, predicates, expressions. //! -//! Ported verbatim from the pgwire `ddl::constraint::parse`; only the error type -//! changed from pgwire `PgWireError` to the protocol-neutral [`DdlError`]. +//! Parse errors are the protocol-neutral [`DdlError`]. use nodedb_sql::parser::preprocess::lex::{ find_ascii_case_insensitive, find_ascii_case_insensitive_from, diff --git a/nodedb/src/control/server/shared/ddl/neutral/constraint/show.rs b/nodedb/src/control/server/shared/ddl/neutral/constraint/show.rs index 28e2e1cc7..32c684aad 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/constraint/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/constraint/show.rs @@ -2,10 +2,8 @@ //! `SHOW CONSTRAINTS ON ` — unified view of all constraint kinds. //! -//! Ported from the pgwire `ddl::constraint::show`. The per-row content is -//! preserved verbatim; only the result construction changed from a pgwire -//! `QueryResponse` (4 text columns via `DataRowEncoder`) to a protocol-neutral -//! [`DdlResult::Rows`] carrying the same four text columns. +//! The result is a protocol-neutral [`DdlResult::Rows`] with four text +//! columns. use nodedb_sql::parser::preprocess::lex::find_ascii_case_insensitive; use nodedb_types::DatabaseId; diff --git a/nodedb/src/control/server/shared/ddl/neutral/constraint/validate.rs b/nodedb/src/control/server/shared/ddl/neutral/constraint/validate.rs index 50df09d41..28fff6010 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/constraint/validate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/constraint/validate.rs @@ -2,8 +2,7 @@ //! DDL-time validation for CHECK constraint expressions. //! -//! Ported verbatim from the pgwire `ddl::constraint::validate`; only the error -//! type changed from pgwire `PgWireError` to the protocol-neutral [`DdlError`]. +//! Validation errors are the protocol-neutral [`DdlError`]. use crate::control::server::shared::ddl::result::DdlError; diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs index 33d1d705c..990cfc6a0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/commit.rs @@ -2,17 +2,18 @@ //! Protocol-neutral `COMMIT OFFSET` DDL handler. //! -//! Ported from the pgwire `ddl::consumer_group::commit` handler. The two-form -//! token parsing, the group-existence checks, the per-partition tail-tracker -//! batch-commit path (NOT a full buffer scan — preserved -//! verbatim), and the `OffsetRegression` error mapping are preserved verbatim; -//! only the result construction changed from pgwire `Response` / `PgWireError` -//! to the protocol-neutral [`DdlResult`] / [`DdlError`]. +//! The two-form token parsing, the group-existence checks, the +//! per-partition tail-tracker batch-commit path (NOT a full buffer scan), +//! and the `OffsetRegression` error mapping run here. The result is the +//! protocol-neutral [`DdlResult`] / [`DdlError`]. //! //! Syntax: -//! - `COMMIT OFFSET PARTITION

AT : ON CONSUMER GROUP ` -//! (a bare `` is legacy compatibility and acknowledges the whole LSN) +//! - `COMMIT OFFSET PARTITION

AT :: ON CONSUMER GROUP ` +//! (a bare `` acknowledges every event of that write) //! - `COMMIT OFFSETS ON CONSUMER GROUP ` (batch: commit all at latest) +//! +//! A commit is a replicated catalog entry: every node raises its offset +//! store on apply, so a consumer that moves to another node resumes from it. use crate::control::security::audit::ArcAuditEmitter; use crate::control::security::identity::{AuthenticatedIdentity, Permission}; @@ -20,11 +21,14 @@ use crate::control::server::shared::authorization::authorize_collection; use crate::control::server::shared::ddl::sql_parse::{parse_ident_token, parse_stream_ident_token}; use crate::control::state::SharedState; use crate::event::cdc::CdcOffset; +use crate::event::cdc::consumer_group::ConsumerGroupDef; +use crate::event::cdc::consumer_group::types::PartitionOffset; use crate::types::DatabaseId; use super::super::super::result::{DdlError, DdlResult}; use super::super::auth_support::status; use super::identity::canonical_stream_name; +use super::replicate::propose_commit_offsets; fn authorize_offset_commit( state: &SharedState, @@ -57,7 +61,7 @@ fn authorize_offset_commit( .map_err(|error| DdlError::new("42501", error.to_string())) } -fn migrate_legacy_group( +async fn migrate_legacy_group( state: &SharedState, database_id: DatabaseId, tenant_id: u64, @@ -71,10 +75,11 @@ fn migrate_legacy_group( stream_name, group_name, ) + .await .map(|_| ()) } -/// Handle `COMMIT OFFSET PARTITION

AT : ON CONSUMER GROUP `. +/// Handle `COMMIT OFFSET PARTITION

AT :: ON CONSUMER GROUP `. pub async fn commit_offset( state: &SharedState, identity: &AuthenticatedIdentity, @@ -83,7 +88,7 @@ pub async fn commit_offset( ) -> Result, DdlError> { let tenant_id = identity.tenant_id.as_u64(); - // Single partition: COMMIT OFFSET PARTITION

AT : ON CONSUMER GROUP + // Single partition: COMMIT OFFSET PARTITION

AT :: ON CONSUMER GROUP // parts: [COMMIT, OFFSET, PARTITION,

, AT, , ON, , CONSUMER, GROUP, ] // indices: 0 1 2 3 4 5 6 7 8 9 10 if parts.len() >= 11 @@ -131,35 +136,34 @@ pub async fn commit_offset( Some(lock) => Some(lock.lock_owned().await), None => None, }; - migrate_legacy_group(state, database_id, tenant_id, &stream_name, &group_name)?; - - // Verify group exists. - if state - .group_registry - .get(database_id, tenant_id, &stream_name, &group_name) - .is_none() - { - return Err(DdlError::new( - "42704", - format!("consumer group '{group_name}' does not exist on stream '{stream_name}'"), - )); - } + migrate_legacy_group(state, database_id, tenant_id, &stream_name, &group_name).await?; - state - .offset_store - .commit_offset( - database_id, - tenant_id, - &stream_name, - &group_name, + let def = registered_group(state, database_id, tenant_id, &stream_name, &group_name)?; + let current = state.offset_store.get_offset( + database_id, + tenant_id, + &stream_name, + &group_name, + partition_id, + ); + if offset < current { + let regression = crate::Error::OffsetRegression { + stream: stream_name, + group: group_name, partition_id, - offset, - ) - .map_err(|e| match e { - crate::Error::OffsetRegression { .. } => DdlError::new("22023", e.to_string()), - // Any other error keeps the class the SQLSTATE table gives it. - other => DdlError::from_error_in_context("offset commit", &other), - })?; + offsets: Box::new(crate::error::RegressedOffsets { + current, + attempted: offset, + }), + }; + return Err(DdlError::new("22023", regression.to_string())); + } + propose_commit_offsets( + state, + &def, + vec![PartitionOffset::new(partition_id, offset)], + ) + .await?; return Ok(status("COMMIT OFFSET")); } @@ -202,52 +206,35 @@ pub async fn commit_offset( Some(lock) => Some(lock.lock_owned().await), None => None, }; - migrate_legacy_group(state, database_id, tenant_id, &stream_name, &group_name)?; + migrate_legacy_group(state, database_id, tenant_id, &stream_name, &group_name).await?; - if state - .group_registry - .get(database_id, tenant_id, &stream_name, &group_name) - .is_none() - { - return Err(DdlError::new( - "42704", - format!("consumer group '{group_name}' does not exist on stream '{stream_name}'"), - )); - } + let def = registered_group(state, database_id, tenant_id, &stream_name, &group_name)?; // Use the buffer's per-partition tail tracker — NOT a full // buffer scan. A scan is O(N) and silently // misses partitions whose events have been evicted by retention. - if let Some(buffer) = state + // Every replica positions events alike, so this node's tails name + // the same events on every node. + let raised: Vec = state .cdc_router .get_buffer(database_id, tenant_id, &stream_name) - { - for (partition_id, offset) in buffer.partition_tails() { - // Skip partitions whose committed offset already meets - // or exceeds the current tail — commit_offset rejects - // regressions and we want idempotent auto-commit. - let current = state.offset_store.get_offset( - database_id, - tenant_id, - &stream_name, - &group_name, - partition_id, - ); - if offset <= current { - continue; - } - state - .offset_store - .commit_offset( + .map(|buffer| buffer.partition_tails()) + .unwrap_or_default() + .into_iter() + .filter(|(partition_id, offset)| { + *offset + > state.offset_store.get_offset( database_id, tenant_id, &stream_name, &group_name, - partition_id, - offset, + *partition_id, ) - .map_err(|e| DdlError::from_error_in_context("offset commit", &e))?; - } + }) + .map(|(partition_id, offset)| PartitionOffset::new(partition_id, offset)) + .collect(); + if !raised.is_empty() { + propose_commit_offsets(state, &def, raised).await?; } return Ok(status("COMMIT OFFSETS")); @@ -255,7 +242,41 @@ pub async fn commit_offset( Err(DdlError::new( "42601", - "expected COMMIT OFFSET PARTITION

AT : ON CONSUMER GROUP , \ - or COMMIT OFFSETS ON CONSUMER GROUP ; bare is legacy whole-LSN acknowledgement", + "expected COMMIT OFFSET PARTITION

AT :: ON CONSUMER GROUP , \ + or COMMIT OFFSETS ON CONSUMER GROUP ; a bare acknowledges every event of that write", )) } + +/// The registered definition of a consumer group, or `42704` when none is. +fn registered_group( + state: &SharedState, + database_id: DatabaseId, + tenant_id: u64, + stream_name: &str, + group_name: &str, +) -> Result { + state + .group_registry + .get(database_id, tenant_id, stream_name, group_name) + .ok_or_else(|| { + DdlError::new( + "42704", + format!("consumer group '{group_name}' does not exist on stream '{stream_name}'"), + ) + }) +} + +/// Commit `offsets` for a group through the replicated catalog, raising each +/// partition on every node. A partition already at or past its offset keeps +/// its position. Deferred `COMMIT OFFSET` inside a transaction flushes here. +pub async fn commit_group_offsets( + state: &SharedState, + database_id: DatabaseId, + tenant_id: u64, + stream_name: &str, + group_name: &str, + offsets: Vec, +) -> Result<(), DdlError> { + let def = registered_group(state, database_id, tenant_id, stream_name, group_name)?; + propose_commit_offsets(state, &def, offsets).await +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/create.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/create.rs index d88fd4a86..8fad31c86 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/create.rs @@ -84,7 +84,8 @@ pub async fn create_consumer_group( tenant_id, &stream_name, &group_name, - )?; + ) + .await?; if state .group_registry @@ -109,9 +110,10 @@ pub async fn create_consumer_group( stream_name: stream_name.clone(), owner: identity.username.clone(), created_at: now, + modification_hlc: nodedb_types::Hlc::ZERO, }; - super::replicate::propose_create(state, &def)?; + super::replicate::propose_create(state, &def).await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/drop.rs index 232f79921..fb48c4f46 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/drop.rs @@ -68,7 +68,8 @@ pub async fn drop_consumer_group( tenant_id, &stream_name, &group_name, - )?; + ) + .await?; if state .group_registry @@ -83,7 +84,8 @@ pub async fn drop_consumer_group( // The entry carries the registry teardown and the committed-offset delete // to every node; the offset store is node-local and no entry can hold it. - super::replicate::propose_delete(state, database_id, tenant_id, &stream_name, &group_name)?; + super::replicate::propose_delete(state, database_id, tenant_id, &stream_name, &group_name) + .await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/identity.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/identity.rs index 921b01586..91b27a0e3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/identity.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/identity.rs @@ -39,7 +39,7 @@ pub fn canonical_stream_name( /// /// The catalog re-key is replicated; the offsets move first, in their separate /// database, so a failure there leaves the legacy identity whole. -pub fn migrate_legacy_topic_group( +pub async fn migrate_legacy_topic_group( state: &SharedState, database_id: DatabaseId, tenant_id: u64, @@ -72,6 +72,6 @@ pub fn migrate_legacy_topic_group( group, ) .map_err(|error| DdlError::from_error_in_context("consumer-group migration", &error))?; - super::replicate::propose_migrate(state, &def, legacy_stream)?; + super::replicate::propose_migrate(state, &def, legacy_stream).await?; Ok(true) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/replicate.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/replicate.rs index 2e78a053d..3383dfc3e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/replicate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/replicate.rs @@ -6,75 +6,90 @@ //! each node writes the row and installs it in its own `GroupRegistry`. A group //! created on one node resolves on all. -use crate::control::catalog_entry::apply::consumer_group as apply; use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::catalog_entry::post_apply::consumer_group as post_apply; use crate::control::state::SharedState; -use crate::event::cdc::consumer_group::ConsumerGroupDef; +use crate::event::cdc::consumer_group::{ConsumerGroupDef, OffsetCommit, PartitionOffset}; use crate::types::DatabaseId; use super::super::super::result::DdlError; -use super::super::replicate::propose_and_apply; +use super::super::replicate::propose_and_apply_async; /// Propose the group definition. The leader reports the duplicate before /// proposing, so apply is a create-only write that never rejects. -pub(super) fn propose_create(state: &SharedState, def: &ConsumerGroupDef) -> Result<(), DdlError> { +pub(super) async fn propose_create( + state: &SharedState, + def: &ConsumerGroupDef, +) -> Result<(), DdlError> { let entry = CatalogEntry::PutConsumerGroupIfAbsent(Box::new(def.clone())); - propose_and_apply(state, &entry, || { - apply::put_if_absent(def, state.credentials.catalog()) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - post_apply::put_if_absent(def, state); - Ok(()) - }) + propose_and_apply_async(state, &entry).await } /// Propose removal of the group row, its registration, and its durable offsets /// on every node. -pub(super) fn propose_delete( +pub(super) async fn propose_delete( state: &SharedState, database_id: DatabaseId, tenant_id: u64, stream_name: &str, name: &str, ) -> Result<(), DdlError> { - let database_id = database_id.as_u64(); let entry = CatalogEntry::DeleteConsumerGroup { - database_id, + database_id: database_id.as_u64(), tenant_id, stream_name: stream_name.to_string(), name: name.to_string(), + target_hlc: nodedb_types::Hlc::ZERO, }; - propose_and_apply(state, &entry, || { - apply::delete( - database_id, - tenant_id, - stream_name, - name, - state.credentials.catalog(), - ) - .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; - post_apply::delete(database_id, tenant_id, stream_name, name, state); - Ok(()) - }) + propose_and_apply_async(state, &entry).await } /// Propose the re-key of a legacy bare-topic group onto its canonical stream. /// /// The caller moves the durable offsets first: they live in a separate /// database this entry cannot carry. -pub(super) fn propose_migrate( +pub(super) async fn propose_migrate( state: &SharedState, def: &ConsumerGroupDef, legacy_stream: &str, ) -> Result<(), DdlError> { + // The entry names the canonical row, which is what its incarnation fences. + let canonical = ConsumerGroupDef { + stream_name: format!("topic:{legacy_stream}"), + ..def.clone() + }; let entry = CatalogEntry::MigrateConsumerGroupStream { - def: Box::new(def.clone()), + def: Box::new(canonical), legacy_stream: legacy_stream.to_string(), }; - propose_and_apply(state, &entry, || { - apply::migrate_stream(def, legacy_stream, state.credentials.catalog()) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - post_apply::migrate_stream(def, legacy_stream, state); - Ok(()) - }) + propose_and_apply_async(state, &entry).await +} + +/// Propose raising the group's committed offsets. Every node raises its own +/// offset store on apply, so a consumer resumes from this commit on any node. +/// +/// Offsets apply as a monotonic max and stamp no descriptor version, so the +/// commit takes no DDL preparation lease. Inside a transaction block it is +/// buffered until COMMIT, like every other DDL the transaction issues. +pub(super) async fn propose_commit_offsets( + state: &SharedState, + def: &ConsumerGroupDef, + offsets: Vec, +) -> Result<(), DdlError> { + let commit = OffsetCommit { + database_id: def.database_id, + tenant_id: def.tenant_id, + stream_name: def.stream_name.clone(), + group_name: def.name.clone(), + group_hlc: def.modification_hlc, + offsets, + }; + if crate::control::server::shared::session::ddl_buffer::try_buffer( + CatalogEntry::CommitConsumerOffsets(Box::new(commit.clone())), + ) { + return Ok(()); + } + crate::control::metadata_proposer::propose_cursor_commit(state, commit) + .await + .map(|_| ()) + .map_err(|e| DdlError::from_error_in_context("consumer offset commit failed", &e)) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/show.rs b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/show.rs index 8b5f63cc5..e051406ec 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/consumer_group/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/consumer_group/show.rs @@ -3,11 +3,9 @@ //! Protocol-neutral `SHOW CONSUMER GROUPS ON ` and //! `SHOW PARTITIONS ON ` handlers. //! -//! Ported from the pgwire `ddl::consumer_group::show` handlers. The token-based -//! syntax checks, the tenant scoping, the per-group offset counting, and the -//! per-partition buffer-scan statistics are preserved verbatim; only the result -//! construction changed from a pgwire `QueryResponse` to the protocol-neutral -//! [`DdlResult::Rows`]. +//! The token-based syntax checks, the tenant scoping, the per-group offset +//! counting, and the per-partition buffer-scan statistics run here. The +//! result is the protocol-neutral [`DdlResult::Rows`]. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs index 059fd935e..b637ade2e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/create.rs @@ -2,28 +2,21 @@ //! Protocol-neutral `CREATE CONTINUOUS AGGREGATE` handler. //! -//! Ported from the pgwire `ddl::continuous_agg::create` handler. The catalog -//! path (`propose_and_apply` for the `PutContinuousAggregate` entry, then the -//! target-collection `propose_and_apply` + `dispatch_register_from_stored`, then -//! the `LocalOnly` single-node `RegisterContinuousAggregate` sync dispatch), -//! the source-existence / timeseries / duplicate checks, the def serialization, -//! and the target-collection descriptor are preserved verbatim; only the result -//! construction changed from pgwire `Response` / `PgWireError` to the -//! protocol-neutral [`DdlResult`] / [`DdlError`]. - -use std::time::Duration; +//! The catalog path (`propose_and_apply` for the `PutContinuousAggregate` +//! entry, then the target-collection `propose_and_apply` + +//! `register_proposed_collection`), the source-existence / timeseries / +//! duplicate checks, the def serialization, and the target-collection +//! descriptor run here. The result is the protocol-neutral [`DdlResult`] / +//! [`DdlError`]. use nodedb_types::DatabaseId; -use crate::bridge::envelope::PhysicalPlan; -use crate::control::security::catalog::{StoredCollection, StoredContinuousAggregate, StoredOwner}; +use crate::control::security::catalog::{StoredCollection, StoredContinuousAggregate}; use crate::control::security::identity::AuthenticatedIdentity; -use crate::control::server::shared::ddl::sync_dispatch; use crate::control::state::SharedState; use crate::engine::timeseries::continuous_agg::ContinuousAggregateDef; -use nodedb_physical::physical_plan::MetaOp; -use super::super::super::catalog::propose_and_apply; +use super::super::super::catalog::propose_and_apply_async; use super::super::super::result::{DdlError, DdlResult}; use super::super::collection; use super::parse::{extract_with_options, parse_create_sql}; @@ -159,10 +152,9 @@ pub async fn create_continuous_aggregate( modification_hlc: nodedb_types::Hlc::ZERO, }; - let entry = crate::control::catalog_entry::CatalogEntry::PutContinuousAggregate(Box::new( - stored.clone(), - )); - let outcome = propose_and_apply(state, &entry)?; + let entry = + crate::control::catalog_entry::CatalogEntry::PutContinuousAggregate(Box::new(stored)); + propose_and_apply_async(state, &entry).await?; // Create the target collection so `SELECT * FROM ` resolves // like any other relation. Schemaless document by parity with @@ -187,6 +179,7 @@ pub async fn create_continuous_aggregate( constraint_version: 0, crdt_signing_required: false, modification_hlc: nodedb_types::Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, fields: Vec::new(), field_defs: Vec::new(), event_defs: Vec::new(), @@ -223,41 +216,12 @@ pub async fn create_continuous_aggregate( }; let coll_entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(target.clone())); - propose_and_apply(state, &coll_entry)?; - collection::dispatch_register_from_stored(state, &target) + let outcome = propose_and_apply_async(state, &coll_entry).await?; + collection::register_proposed_collection(state, outcome, &target) .await .map_err(|e| DdlError::from_error(&e))?; } - // Single-node / no-applier path: the async post-apply dispatcher - // only fires on the raft-applier path. Mirror - // the dispatch here so the local `continuous_agg_mgr` registers - // immediately, matching the cluster behaviour. - if outcome.needs_local_apply() { - state.permissions.install_replicated_owner(&StoredOwner { - database_id: stored.database_id, - object_type: - crate::control::security::catalog::auth_types::object_type::CONTINUOUS_AGGREGATE - .to_string(), - object_name: stored.name.clone(), - tenant_id: stored.tenant_id, - owner_username: stored.owner.clone(), - }); - let plan = PhysicalPlan::Meta(MetaOp::RegisterContinuousAggregate { def: def.clone() }); - sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::CatalogMaintenance, - tenant_id, - nodedb_types::CollectionKey::from_bare(database_id, &def.source), - plan, - ), - Duration::from_secs(5), - ) - .await - .map_err(|e| DdlError::from_error_in_context("dispatch failed", &e))?; - } - tracing::info!( name = def.name, source = def.source, diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs index 54a6c77a6..fba96ce8e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/drop.rs @@ -2,26 +2,17 @@ //! Protocol-neutral `DROP CONTINUOUS AGGREGATE` handler. //! -//! Ported from the pgwire `ddl::continuous_agg::drop` handler. The catalog path -//! (`propose_and_apply` for the `DeleteContinuousAggregate` entry, then the -//! `LocalOnly` single-node `UnregisterContinuousAggregate` sync dispatch), -//! the `parts[3]` name extraction, and the arity check are preserved verbatim; -//! only the result construction changed from pgwire `Response` / `PgWireError` to -//! the protocol-neutral [`DdlResult`] / [`DdlError`]. The -//! [`continuous_aggregate_exists`] helper (moved from the pgwire router's -//! `ast::exists`) backs the router's IF EXISTS short-circuit guard. +//! The catalog path (`propose_and_apply` for the `DeleteContinuousAggregate` entry), +//! the `parts[3]` name extraction, and the arity check run here; +//! the result is the protocol-neutral [`DdlResult`] / [`DdlError`]. The +//! [`continuous_aggregate_exists`] helper backs the router's IF EXISTS short-circuit guard. -use std::time::Duration; - -use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::{AuthenticatedIdentity, Role}; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; -use crate::control::server::shared::ddl::sync_dispatch; use crate::control::state::SharedState; use crate::types::DatabaseId; -use nodedb_physical::physical_plan::MetaOp; -use super::super::super::catalog::propose_and_apply; +use super::super::super::catalog::propose_and_apply_async; use super::super::super::result::{DdlError, DdlResult}; fn err(sqlstate: &str, message: String) -> DdlError { @@ -95,33 +86,11 @@ pub async fn drop_continuous_aggregate( database_id: database_id.as_u64(), tenant_id: tenant_id.as_u64(), name: name.clone(), + // Frozen by the proposer's stamp. + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }; - let outcome = propose_and_apply(state, &entry)?; - - // Single-node / no-applier path: mirror the unregister dispatch the - // raft-applier path would have done so the local manager forgets the - // aggregate immediately. - if outcome.needs_local_apply() { - state.permissions.install_replicated_remove_owner( - crate::control::security::catalog::auth_types::object_type::CONTINUOUS_AGGREGATE, - database_id.as_u64(), - tenant_id.as_u64(), - &name, - ); - let plan = PhysicalPlan::Meta(MetaOp::UnregisterContinuousAggregate { name: name.clone() }); - sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::CatalogMaintenance, - tenant_id, - nodedb_types::CollectionKey::from_bare(database_id, &stored.source), - plan, - ), - Duration::from_secs(5), - ) - .await - .map_err(|e| DdlError::from_error_in_context("dispatch failed", &e))?; - } + propose_and_apply_async(state, &entry).await?; tracing::info!(name, "continuous aggregate dropped"); diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/parse.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/parse.rs index 5b3239792..962522744 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/parse.rs @@ -2,10 +2,8 @@ //! SQL parsing helpers for `CREATE CONTINUOUS AGGREGATE`. //! -//! Ported from the pgwire `ddl::continuous_agg::parse` helpers. The parsing -//! logic, keyword tables, auto-alias derivation, and SQLSTATE codes are -//! preserved verbatim; only the error type changed from pgwire -//! `PgWireError` (via `sqlstate_error`) to the protocol-neutral [`DdlError`]. +//! The parsing logic, keyword tables, auto-alias derivation, and SQLSTATE +//! codes live here. Errors are the protocol-neutral [`DdlError`]. use nodedb_sql::parser::preprocess::lex::{ find_ascii_case_insensitive, find_ascii_case_insensitive_from, rfind_ascii_case_insensitive, diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/register.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/register.rs index c93d283b6..98d319e74 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/register.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/register.rs @@ -1,71 +1,42 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Startup replay: re-register catalog-persisted continuous aggregates. -//! -//! Moved verbatim from the pgwire `ddl::continuous_agg::register` module. This is -//! a boot-time Data-Plane replay helper, not a DDL statement handler, so it -//! carries no pgwire response types and is unchanged by the neutral migration. +//! Boot re-registration of catalog-persisted continuous aggregates. -use std::time::Duration; - -use crate::bridge::envelope::PhysicalPlan; -use crate::control::server::shared::ddl::sync_dispatch; +use crate::control::catalog_entry::post_apply::{ + ContinuousAggregateRegisterFailure, register_continuous_aggregate_on_every_core, +}; use crate::control::state::SharedState; -use crate::engine::timeseries::continuous_agg::ContinuousAggregateDef; -use nodedb_physical::physical_plan::MetaOp; -/// Re-register every catalog-persisted continuous aggregate on the -/// local Data Plane. Called at startup on paths that don't trigger -/// the raft post-apply chain (single-node restart, the -/// `nodedb-test-support` harness): the `continuous_agg_mgr` is a -/// per-core in-memory registry, so without explicit replay every -/// aggregate becomes silently inactive after restart and the runtime -/// bucket-aggregation pipeline forgets the definitions. -pub async fn register_persisted_continuous_aggregates(state: &SharedState) { - let catalog = state.credentials.catalog(); - let stored = match catalog.load_all_continuous_aggregates() { - Ok(s) => s, - Err(e) => { - tracing::warn!(error = %e, "boot: failed to load continuous aggregates from catalog"); - return; - } - }; +/// Re-register every catalog-persisted continuous aggregate on every local +/// Data Plane core. +/// +/// The per-core `continuous_agg_mgr` is in-memory, and the metadata applier +/// skips entries it already applied. Boot therefore rebuilds the registry +/// from redb in single-node and cluster mode. Any aggregate that does not +/// reach every core fails the boot: serving with it inactive silently stops +/// its bucket aggregation. +pub async fn register_persisted_continuous_aggregates(state: &SharedState) -> crate::Result<()> { + let stored = state + .credentials + .catalog() + .load_all_continuous_aggregates()?; for s in stored { - let def: ContinuousAggregateDef = match zerompk::from_msgpack(&s.def_bytes) { - Ok(def) => def, - Err(e) => { - tracing::warn!( - cagg = %s.name, - tenant = s.tenant_id, - error = %e, - "boot: failed to decode continuous aggregate def — skipping replay" - ); - continue; - } - }; - let plan = PhysicalPlan::Meta(MetaOp::RegisterContinuousAggregate { def: def.clone() }); - let tenant_id = crate::types::TenantId::new(s.tenant_id); - if let Err(e) = sync_dispatch::dispatch_system( - state, - sync_dispatch::SystemTask::new( - sync_dispatch::SystemReason::CatalogMaintenance, - tenant_id, - nodedb_types::CollectionKey::from_bare( - crate::types::DatabaseId::new(def.database_id), - &def.source, - ), - plan, - ), - Duration::from_secs(5), - ) - .await - { - tracing::warn!( - cagg = %s.name, - tenant = s.tenant_id, - error = %e, - "boot: failed to re-register continuous aggregate on Data Plane" - ); - } + register_continuous_aggregate_on_every_core(state, s.tenant_id, &s.name, &s.def_bytes) + .await + .map_err( + |failure: ContinuousAggregateRegisterFailure| match failure.error { + crate::Error::Codec { .. } => failure.error, + error => crate::Error::Internal { + detail: format!( + "continuous aggregate '{}' (tenant {}, database {}): {}: {error}", + s.name, + s.tenant_id, + failure.database_id, + failure.stage.label() + ), + }, + }, + )?; } + Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs index b590371cc..f8a970ac1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/continuous_agg/show.rs @@ -2,17 +2,13 @@ //! Protocol-neutral `SHOW CONTINUOUS AGGREGATES [FOR ]` handler. //! -//! Ported from the pgwire `ddl::continuous_agg::show` handler. The catalog read -//! (durable source of truth), the best-effort runtime-stats merge from the local -//! Data Plane manager, the optional `FOR ` filter, the decode-failure -//! skip, and the exact column set are preserved verbatim; only the result -//! construction changed from pgwire `Response` / `QueryResponse` to the -//! protocol-neutral [`DdlResult::Rows`] over [`ShapedRows`]. The mixed -//! text/`int8` column OIDs (`watermark_ts`, `rows_aggregated`, -//! `materialized_buckets` are `int8`) are reproduced by building `column_types` -//! manually so the RowDescription stays byte-identical; the `int8` cells are -//! emitted as their decimal text form, the same bytes the pgwire -//! `DataRowEncoder::encode_field(&i64)` produced. +//! The catalog read (durable source of truth), the best-effort runtime-stats +//! merge from the local Data Plane manager, the optional `FOR ` +//! filter, the decode-failure skip, and the exact column set run here. The +//! result is the protocol-neutral [`DdlResult::Rows`] over [`ShapedRows`]. +//! The mixed text/`int8` column OIDs (`watermark_ts`, `rows_aggregated`, +//! `materialized_buckets` are `int8`) come from building `column_types` +//! manually; the `int8` cells are emitted as their decimal text form. use std::time::Duration; diff --git a/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs b/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs index a2fb189cb..59e6c98ac 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/convert/driver.rs @@ -43,6 +43,13 @@ pub async fn convert_collection( .get_collection(database_id, tenant_id.as_u64(), &collection) .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' does not exist")))?; + if coll.hash_chain { + let refusal = + crate::control::server::shared::ddl::neutral::collection::enforcement::EnforcementDeclError::HashChainConvert { + collection: collection.clone(), + }; + return Err(err(refusal.sqlstate(), refusal.to_string())); + } // Build columns before dispatch — needed for both Data Plane and catalog. let columns: Option> = match target_type.as_str() { @@ -72,7 +79,7 @@ pub async fn convert_collection( // Resolve the SOURCE storage mode from the catalog row read above, before // this DDL mutates `coll.collection_type`. Mirrors the exhaustive match in // `build_doc_config_from_stored`, so the Data Plane handler decodes the - // scanned rows the same way the collection's own register path would. + // scanned rows the same way the collection's own register path does. let source_storage_mode = match &coll.collection_type { nodedb_types::CollectionType::Document(nodedb_types::DocumentMode::Strict(schema)) => { nodedb_physical::physical_plan::StorageMode::Strict { @@ -135,9 +142,11 @@ pub async fn convert_collection( let new_type = match target_type.as_str() { "document_schemaless" => nodedb_types::CollectionType::document(), "document_strict" | "kv" => { - let columns = columns.expect( - "invariant: columns is Some for document_strict/kv targets, validated above", - ); + let Some(columns) = columns else { + return Err(DdlError::internal( + "columns missing for a document_strict/kv target", + )); + }; let schema = nodedb_types::columnar::StrictSchema { columns, version: 1, @@ -181,14 +190,16 @@ pub async fn convert_collection( coll.type_guards.clear(); } - persist_collection_replicated(state, database_id, &coll) + let outcome = persist_collection_replicated(state, &coll) + .await .map_err(|e| DdlError::from_error(&e))?; // Refresh this node's Data Plane `doc_configs` entry to the NEW storage // mode. Without this, every later read of the collection resolves its // body format from the pre-conversion entry until the process restarts. - crate::control::server::shared::ddl::neutral::collection::dispatch_register_from_stored( - state, &coll, + // A durable apply refreshed it in its post-apply. + crate::control::server::shared::ddl::neutral::collection::register_proposed_collection( + state, outcome, &coll, ) .await .map_err(|e| DdlError::from_error(&e))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs index 73f460d31..e310bf3e3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/crdt_ops.rs @@ -93,6 +93,9 @@ pub async fn crdt_state( // `AlreadyAdmitted` carries no address precisely so there is // nothing to invent here. admission: crate::control::server::shared::ddl::user_dispatch::RequestAdmission::AlreadyAdmitted, + // No session reaches this handler, so its read takes the strong + // default (see `DmlTxnCtx::linearizable_reads`). + linearizable: true, }, ) .await @@ -160,14 +163,15 @@ pub async fn crdt_apply( ) .map_err(|error| DdlError::new("42501", format!("permission denied: {}", error.resource())))?; - let surrogate = state - .surrogate_assigner - .assign( - nodedb_types::CollectionKey::from_bare(database_id, collection), - tenant_id, - document_id.as_bytes(), - ) - .map_err(|e| DdlError::from_error(&e))?; + let surrogate = crate::control::server::surrogate_exchange::assign_surrogate_routed( + state, + nodedb_types::CollectionKey::from_bare(database_id, collection), + tenant_id, + document_id.as_bytes(), + crate::types::TraceId::ZERO, + ) + .await + .map_err(|e| DdlError::from_error(&e))?; let plan = PhysicalPlan::Crdt(CrdtOp::Apply { collection: nodedb_types::QualifiedCollection::new(database_id, collection), @@ -203,8 +207,8 @@ pub async fn crdt_apply( .ok_or_else(|| DdlError::internal("authorization returned no capability"))?; // Route through the Raft proposer gate so the delta is quorum-durable under - // replication. A local-only dispatch would land the delta on the receiving - // node only — it would be lost to every follower and entirely on failover. + // replication. A local-only dispatch will land the delta on the receiving + // node only — it will be lost to every follower and entirely on failover. // // RLS write policies are stored keyed by `db_qualified(database_id, // collection)`, so the policy must be handed that same key or it silently diff --git a/nodedb/src/control/server/shared/ddl/neutral/custom_type.rs b/nodedb/src/control/server/shared/ddl/neutral/custom_type.rs index 837ccf027..62bc905f4 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/custom_type.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/custom_type.rs @@ -52,9 +52,7 @@ fn status(command: &str) -> Vec { /// Require that the identity is superuser or tenant_admin. /// -/// Folded in verbatim from the pgwire `require_tenant_admin` helper: it does -/// NOT emit an audit record on denial and returns SQLSTATE 42501 with the -/// identical message. +/// It does NOT emit an audit record on denial and returns SQLSTATE 42501. fn require_tenant_admin(identity: &AuthenticatedIdentity, action: &str) -> Result<(), DdlError> { if identity.is_superuser || identity.has_role(&Role::TenantAdmin) { Ok(()) @@ -67,7 +65,7 @@ fn require_tenant_admin(identity: &AuthenticatedIdentity, action: &str) -> Resul } /// Handle `CREATE TYPE AS ENUM ('label1', ...)`. -pub fn create_enum_type( +pub async fn create_enum_type( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -97,13 +95,13 @@ pub fn create_enum_type( created_at, }; - persist_and_register(state, stored)?; + persist_and_register(state, stored).await?; Ok(status("CREATE TYPE")) } /// Handle `CREATE TYPE AS ( , ...)`. -pub fn create_composite_type( +pub async fn create_composite_type( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -140,13 +138,13 @@ pub fn create_composite_type( created_at, }; - persist_and_register(state, stored)?; + persist_and_register(state, stored).await?; Ok(status("CREATE TYPE")) } /// Handle `DROP TYPE [IF EXISTS] `. -pub fn drop_type( +pub async fn drop_type( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -177,20 +175,14 @@ pub fn drop_type( )); } - let catalog = state.credentials.catalog(); - let entry = crate::control::catalog_entry::CatalogEntry::DeleteCustomType { database_id: database_id_u64, tenant_id, name: name.to_string(), }; - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - catalog - .delete_custom_type(database_id_u64, tenant_id, name) - .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; - } state .custom_type_registry @@ -200,7 +192,7 @@ pub fn drop_type( } /// Handle `ALTER TYPE ADD VALUE 'label'`. -pub fn alter_type_add_value( +pub async fn alter_type_add_value( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -234,7 +226,7 @@ pub fn alter_type_add_value( labels.push(label.to_string()); - persist_and_register(state, stored)?; + persist_and_register(state, stored).await?; Ok(status("ALTER TYPE")) } @@ -280,23 +272,17 @@ pub fn show_types( /// Persist the entry to catalog and register in the in-memory registry. /// /// The registry takes the record the catalog wrote, never `stored`: the -/// catalog assigns the OID and `stored` does not carry it. On the replicated -/// path the post-apply lane registers the written record on every node, this -/// one included, so the handler registers nothing. -fn persist_and_register(state: &SharedState, stored: StoredCustomType) -> Result<(), DdlError> { - let catalog = state.credentials.catalog(); - - let entry = - crate::control::catalog_entry::CatalogEntry::PutCustomType(Box::new(stored.clone())); - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) +/// catalog assigns the OID and `stored` does not carry it. The post-apply lane +/// registers the written record on every node, this one included, so the +/// handler registers nothing. +async fn persist_and_register( + state: &SharedState, + stored: StoredCustomType, +) -> Result<(), DdlError> { + let entry = crate::control::catalog_entry::CatalogEntry::PutCustomType(Box::new(stored)); + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - let written = catalog - .put_custom_type_assigning_oid(&stored) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - state.custom_type_registry.register(written); - } - Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/alter.rs b/nodedb/src/control/server/shared/ddl/neutral/database/alter.rs index 78232cc62..01c78d808 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/alter.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/alter.rs @@ -2,12 +2,10 @@ //! Handler for `ALTER DATABASE `. //! -//! Ported from the pgwire `ddl::database::alter` handler. Every per-operation -//! privilege gate, catalog read/write, Raft propose / single-node fallback, -//! live-cache / enforcement-component update, and audit record is preserved -//! verbatim; only the result construction changed from pgwire `Response` to the -//! protocol-neutral [`DdlResult`]. `MATERIALIZE` and `PROMOTE` delegate to the -//! `materialize` / `mirror::promote` handlers exactly as before. +//! Every per-operation privilege gate, catalog read, catalog propose, +//! live-cache / enforcement-component update, and audit record runs here. +//! The result is the protocol-neutral [`DdlResult`]. `MATERIALIZE` and +//! `PROMOTE` delegate to the `materialize` / `mirror::promote` handlers. use nodedb_sql::ddl_ast::AlterDatabaseOperation; use nodedb_types::QuotaRecord; @@ -17,14 +15,14 @@ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; -use super::super::replicate::propose_and_apply; +use super::super::replicate::propose_and_apply_async; use super::gate::{require_cluster_admin, require_database_owner}; use super::support::{ddl_err, status}; /// Handle `ALTER DATABASE `. /// /// Required role varies by operation (see per-arm gates below). -pub fn alter_database( +pub async fn alter_database( state: &SharedState, identity: &AuthenticatedIdentity, name: &str, @@ -67,15 +65,11 @@ pub fn alter_database( } } descriptor.name = new_name.clone(); - propose_and_apply( + propose_and_apply_async( state, &CatalogEntry::PutDatabase(Box::new(descriptor.clone())), - || { - catalog - .put_database(&descriptor) - .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e)) - }, - )?; + ) + .await?; state.audit_record_with_db( crate::control::security::audit::AuditEvent::DatabaseRenamed, @@ -113,22 +107,14 @@ pub fn alter_database( // Replicated: every node writes the row and installs the quota in // its live enforcement components via post-apply. - propose_and_apply( + propose_and_apply_async( state, &CatalogEntry::PutDatabaseQuota { db_id: db_id.as_u64(), record: Box::new(record.clone()), }, - || { - catalog - .write_database_quota(db_id, &record) - .map_err(|e| DdlError::from_error(&e))?; - crate::control::catalog_entry::post_apply::quota::put_database( - db_id, &record, state, - ); - Ok(()) - }, - )?; + ) + .await?; state.audit_record_with_db( crate::control::security::audit::AuditEvent::DatabaseQuotaChanged, @@ -171,15 +157,11 @@ pub fn alter_database( )?; // Update the descriptor's `audit_dml` field and persist it. descriptor.audit_dml = *mode; - propose_and_apply( + propose_and_apply_async( state, &CatalogEntry::PutDatabase(Box::new(descriptor.clone())), - || { - catalog - .put_database(&descriptor) - .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e)) - }, - )?; + ) + .await?; // Update live cache so the Event Plane consumer sees the new mode // without a restart. @@ -204,15 +186,11 @@ pub fn alter_database( )?; let before = descriptor.idle_session_timeout_secs; descriptor.idle_session_timeout_secs = *secs; - propose_and_apply( + propose_and_apply_async( state, &CatalogEntry::PutDatabase(Box::new(descriptor.clone())), - || { - catalog - .put_database(&descriptor) - .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e)) - }, - )?; + ) + .await?; // Update the live idle-timeout cache so the sweep loop sees the // new value immediately without a restart. @@ -228,11 +206,11 @@ pub fn alter_database( } AlterDatabaseOperation::Materialize => { - return super::materialize::alter_database_materialize(state, identity, name); + return super::materialize::alter_database_materialize(state, identity, name).await; } AlterDatabaseOperation::Promote => { - return super::mirror::promote::promote_database(state, identity, name); + return super::mirror::promote::promote_database(state, identity, name).await; } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/backup_restore.rs b/nodedb/src/control/server/shared/ddl/neutral/database/backup_restore.rs index 3a3950f5f..a28fe5cd0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/backup_restore.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/backup_restore.rs @@ -1,30 +1,64 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Handlers for `BACKUP DATABASE ` and `RESTORE DATABASE `. +//! Handlers for `BACKUP DATABASE TO ''` and +//! `RESTORE DATABASE FROM '' [FORCE] [DRY RUN]`. //! -//! Ported verbatim from the pgwire typed-AST database router (`database_ops`), -//! where both were inline placeholder arms. Both perform their privilege gate -//! (with the exact db-id resolution + audit-on-denial behaviour) BEFORE the -//! `0A000` (`feature_not_supported`) placeholder return, so the gate side -//! effects are preserved. Only the result type changed from pgwire -//! `PgWireResult` to the protocol-neutral [`DdlResult`] / [`DdlError`]. +//! Gate: superuser, or the owner of the named database. A restore into a +//! database this cluster lacks needs superuser, since no owner exists yet. +//! The gate runs before the URI is read, so an unauthorized caller learns +//! nothing about the store. A bad URI fails with SQLSTATE 22023 before any +//! store is touched. +use nodedb_types::error::sqlstate; + +use crate::control::backup::database; +use crate::control::backup::store::{BackupIoError, BackupObject, BackupUriError}; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::security::request_scope::RequestAuthScope; +use crate::control::server::shared::backup_metering::{ + admit_backup_restore_quota, admit_restore_write_quota, meter_backup_restore, + meter_restore_writes, +}; use crate::control::state::SharedState; +use crate::types::DatabaseId; use super::super::super::result::{DdlError, DdlResult}; -use super::gate::{require_database_owner_or_higher, require_superuser}; +use super::gate::{require_database_owner, require_superuser}; use super::support::ddl_err; -/// Handle `BACKUP DATABASE `. +fn uri_error(e: &BackupUriError) -> DdlError { + ddl_err(sqlstate::INVALID_PARAMETER_VALUE, e.to_string()) +} + +/// A path that leaves the local root at the open is 22023, as at resolve. +fn io_error(context: &str, e: BackupIoError) -> DdlError { + match e { + BackupIoError::Refused(refusal) => uri_error(&refusal), + BackupIoError::Failed(error) => DdlError::from_error_in_context(context, &error), + } +} + +/// A backup or restore writes outside the transaction, and a ROLLBACK cannot +/// undo it. +fn refuse_in_transaction(statement: &str) -> Result<(), DdlError> { + if crate::control::server::shared::session::ddl_buffer::is_active() { + return Err(DdlError::from_error(&crate::Error::NotInTransactionBlock { + statement: statement.into(), + })); + } + Ok(()) +} + +/// Handle `BACKUP DATABASE TO ''`. /// -/// Gate: `DatabaseOwner(db)` or higher before the placeholder return. Resolve -/// db_id first; unknown name returns 3D000, not 42501. -pub fn backup_database( +/// Resolves the database first: an unknown name returns 3D000, not 42501. +pub async fn backup_database( state: &SharedState, identity: &AuthenticatedIdentity, name: &str, + uri: &str, ) -> Result, DdlError> { + refuse_in_transaction("BACKUP DATABASE")?; let catalog = state.credentials.catalog(); let db_id = match catalog.get_database_id_by_name(name) { Ok(Some(id)) => id, @@ -38,30 +72,143 @@ pub fn backup_database( return Err(DdlError::from_error_in_context("catalog lookup failed", &e)); } }; - require_database_owner_or_higher(state, identity, db_id, &format!("BACKUP DATABASE {name}"))?; - Err(ddl_err("0A000", "BACKUP DATABASE is not yet implemented")) + require_database_owner(state, identity, db_id, &format!("BACKUP DATABASE {name}"))?; + let object = + BackupObject::resolve(uri, state.backup_storage.as_deref()).map_err(|e| uri_error(&e))?; + let failed = |e: &crate::Error| DdlError::from_error_in_context("BACKUP DATABASE", e); + + // A spent hard quota of any tenant refuses the backup before it reads a + // byte, as the tenant COPY backup does. + let scope = RequestAuthScope::for_database(identity, state.auth_stores(), db_id); + let tenants = database::database_tenants(state, db_id).map_err(|e| failed(&e))?; + for &tenant_id in &tenants { + admit_backup_restore_quota(state, &scope, tenant_id).map_err(|e| failed(&e))?; + } + + let shared = state.self_arc().map_err(|e| failed(&e))?; + let bytes = database::backup_database(&shared, db_id, name, &tenants) + .await + .map_err(|e| failed(&e))?; + let size = bytes.len() as u64; + object + .put(bytes) + .await + .map_err(|e| io_error("BACKUP DATABASE", e))?; + for &tenant_id in &tenants { + meter_backup_restore(state, &scope, tenant_id, None); + } + state.audit_record( + crate::control::security::audit::AuditEvent::AdminAction, + None, + &identity.username, + &format!( + "BACKUP DATABASE {name} TO '{}' wrote {size} bytes", + object.uri() + ), + ); + Ok(vec![DdlResult::Status { + command: "BACKUP DATABASE".into(), + rows_affected: Some(size), + }]) } -/// Handle `RESTORE DATABASE `. +/// Handle `RESTORE DATABASE FROM '' [FORCE] [DRY RUN]`. /// -/// Gate: `Superuser` required before the placeholder return. The target -/// database may not exist yet; if it doesn't, pass db_id=None. -pub fn restore_database( +/// The affected-row count is the number of rows the restore verified on the +/// destination, `0` for a DRY RUN. +pub async fn restore_database( state: &SharedState, identity: &AuthenticatedIdentity, name: &str, + uri: &str, + force: bool, + dry_run: bool, ) -> Result, DdlError> { - let db_id_opt = state + refuse_in_transaction("RESTORE DATABASE")?; + let action = format!("RESTORE DATABASE {name}"); + let db_id = state .credentials .catalog() .get_database_id_by_name(name) - .ok() - .flatten(); - require_superuser( - state, + .map_err(|e| DdlError::from_error_in_context("catalog lookup failed", &e))?; + match db_id { + Some(db_id) => require_database_owner(state, identity, db_id, &action)?, + None => require_superuser(state, identity, None, &action)?, + } + let object = + BackupObject::resolve(uri, state.backup_storage.as_deref()).map_err(|e| uri_error(&e))?; + + let failed = |e: &crate::Error| DdlError::from_error_in_context("RESTORE DATABASE", e); + + let bytes = object + .get() + .await + .map_err(|e| io_error("RESTORE DATABASE", e))?; + let backup = database::open_database_backup(state, name, &bytes).map_err(|e| failed(&e))?; + drop(bytes); + + // A spent hard backup quota of any restored tenant refuses the restore + // before it reads further, as the tenant COPY restore admits. + let scope = RequestAuthScope::for_database( identity, - db_id_opt, - &format!("RESTORE DATABASE {name}"), - )?; - Err(ddl_err("0A000", "RESTORE DATABASE is not yet implemented")) + state.auth_stores(), + db_id + .or(identity.default_database) + .unwrap_or(DatabaseId::DEFAULT), + ); + for &tenant_id in backup.tenant_ids() { + admit_backup_restore_quota(state, &scope, tenant_id).map_err(|e| failed(&e))?; + } + + // Every tenant envelope passes its checks before any tenant writes. The + // check lists each collection's rows, and a spent hard write quota on any + // of them, on its tenant's marker, or on `*`, refuses the restore before + // its first write. A DRY RUN writes nothing and admits no write quota. + let shared = state.self_arc().map_err(|e| failed(&e))?; + let checked = database::check_database(&shared, &backup, force) + .await + .map_err(|e| failed(&e))?; + let restored = if dry_run { + checked + } else { + for (tenant_id, stats) in &checked.tenants { + admit_restore_write_quota(state, &scope, *tenant_id, &stats.collection_rows) + .map_err(|e| failed(&e))?; + } + database::apply_database(&shared, &backup, force) + .await + .map_err(|e| failed(&e))? + }; + // Charged on the success path, so the charge never refuses anything: the + // backup quota with the row count the tenant COPY restore reports, and + // the write quota with the rows the restore verified per collection. + for (tenant_id, stats) in &restored.tenants { + let rows = + stats.documents + stats.kv_tables + stats.vectors + stats.timeseries + stats.edges; + meter_backup_restore(state, &scope, *tenant_id, Some(rows as u64)); + if !dry_run { + meter_restore_writes(state, &scope, *tenant_id, &stats.collection_rows); + } + } + let stats = restored.total; + if !dry_run { + state.audit_record( + crate::control::security::audit::AuditEvent::AdminAction, + None, + &identity.username, + &format!( + "RESTORE DATABASE {name} FROM '{}' verified {} rows", + object.uri(), + stats.verified_rows + ), + ); + } + Ok(vec![DdlResult::Status { + command: if dry_run { + "RESTORE DATABASE DRY RUN".into() + } else { + "RESTORE DATABASE".into() + }, + rows_affected: Some(stats.verified_rows), + }]) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/clone.rs b/nodedb/src/control/server/shared/ddl/neutral/database/clone.rs index 8803683b9..2184b83bb 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/clone.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/clone.rs @@ -2,32 +2,27 @@ //! Handler for `CLONE DATABASE FROM [AS OF SYSTEM TIME | LATEST]`. //! -//! Ported from the pgwire `ddl::database::clone` handler. Source resolution, -//! the superuser gate (after source resolution so the audit carries the source -//! db), mirror rejection, `MAX_CLONE_DEPTH` enforcement, duplicate-name check, -//! as-of LSN resolution, descriptor build, Raft propose / single-node -//! lineage-then-descriptor write with compensating rollback, shadow-collection -//! stamping, allocator-hwm flush, and `DatabaseCloned` audit are preserved -//! verbatim; only the result construction changed from pgwire `Response` to the -//! protocol-neutral [`DdlResult`]. +//! Source resolution, the superuser gate (after source resolution so the +//! audit carries the source db), mirror rejection, `MAX_CLONE_DEPTH` +//! enforcement, duplicate-name check, as-of LSN resolution, descriptor build, +//! catalog propose (whose apply stamps the shadow collections), and +//! `DatabaseCloned` audit run here. The result is the protocol-neutral +//! [`DdlResult`]. use nodedb_sql::ddl_ast::CloneAsOf; use nodedb_types::{DatabaseId, MAX_CLONE_DEPTH}; use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::catalog_entry::post_apply::custom_type::register_written; -use crate::control::clone::catalog_copy::copy_database_metadata; use crate::control::clone::lsn_resolve::wall_ms_to_lsn; -use crate::control::metadata_proposer::propose_catalog_entry; -use crate::control::security::catalog::auth_types::object_type; +use crate::control::metadata_proposer::propose_catalog_entry_async; +use crate::control::security::catalog::UNASSIGNED_OID; use crate::control::security::catalog::database_types::{ DatabaseDescriptor, DatabaseStatus, ParentCloneRef, }; -use crate::control::security::catalog::{StoredOwner, UNASSIGNED_OID}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; -use super::super::super::catalog::propose_and_apply; +use super::super::super::catalog::propose_and_apply_async; use super::super::super::result::{DdlError, DdlResult}; use super::gate::require_superuser; use super::support::{ddl_err, status}; @@ -123,19 +118,36 @@ pub async fn clone_database( // // For `Latest` we use the current WAL frontier as the clone point. // - // For `SystemTimeMs(t)` we resolve ms → LSN via the `LsnMsAnchor` map - // held on `SharedState`. When the map is populated (WAL anchors have been - // replayed or emitted) this is a precise interpolation. When the map is - // empty the WAL frontier is used as the best available approximation, - // which is correct for recent timestamps (within the same server session). + // For `SystemTimeMs(t)` the clone point is the highest LSN committed by + // the end of millisecond `t`, from the WAL's time anchors. A `t` before + // the oldest retained anchor is refused. So is a `t` before the source + // database existed: it predates the commit of the WAL state the database + // was created on. let now_ms = current_wall_ms().map_err(|e| DdlError::from_error_in_context("clock read failed", &e))?; let (as_of_lsn, as_of_ms) = match params.as_of { CloneAsOf::Latest => (state.wal.next_lsn(), now_ms), CloneAsOf::SystemTimeMs(ms) => { - // wall_ms_to_lsn resolves via the LsnMsAnchor map; falls back to - // wal.next_lsn() when the map is empty (correct for recent clones). - let lsn = wall_ms_to_lsn(state, *ms); + let lsn = wall_ms_to_lsn(state, *ms).map_err(|e| { + ddl_err( + "22000", + format!("CLONE DATABASE AS OF SYSTEM TIME {ms}: {e}"), + ) + })?; + let created = nodedb_types::Lsn::new(source_descriptor.created_at_lsn); + if let Some(created_ms) = state.ms_to_lsn_inverse(created) + && *ms < created_ms + { + return Err(ddl_err( + "22000", + format!( + "CLONE DATABASE AS OF SYSTEM TIME {ms}: the time predates database \ + '{}', created on the WAL state committed at {created_ms} ms; \ + clone it AS OF a later time", + params.source_name + ), + )); + } (lsn, *ms) } }; @@ -143,7 +155,9 @@ pub async fn clone_database( let clone_created_at = state.wal.next_lsn(); // ── Allocate target database id ─────────────────────────────────────────── - let target_db_id = state.database_registry.alloc_one(); + let target_db_id = crate::control::database::allocate_database_id(state) + .await + .map_err(|e| DdlError::from_error_in_context("database id allocation failed", &e))?; // ── Build descriptor ────────────────────────────────────────────────────── let target_descriptor = DatabaseDescriptor { @@ -169,133 +183,27 @@ pub async fn clone_database( }; // ── Propose via Raft ────────────────────────────────────────────────────── + // The proposer stamps the incarnation the shadow collections take. let entry = CatalogEntry::CloneDatabase { - target_descriptor: Box::new(target_descriptor.clone()), + target_descriptor: Box::new(target_descriptor), source_db_id: source_db_id.as_u64(), + incarnation: nodedb_types::Hlc::ZERO, }; - let outcome = propose_catalog_entry(state, &entry) + // The apply writes the descriptor, stamps a shadow descriptor and an + // owner row for every active source collection, copies the source's + // database-scoped catalog rows, and writes the lineage edge last. + propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; - // Single-node fast path (`LocalOnly` means "no Raft, apply directly"). - // - // Order matters for partial-failure safety: write the lineage edge first, - // then the descriptor. If lineage succeeds and descriptor fails we roll the - // lineage entry back — leaving no partial state. If we reversed the order, - // a descriptor-then-lineage failure would create a clone that DROP DATABASE - // on the source would not see as a dependent, allowing unsafe drops. - if outcome.needs_local_apply() { - catalog - .add_clone_child(source_db_id, target_db_id) - .map_err(|e| DdlError::from_error_in_context("lineage write failed", &e))?; - - if let Err(put_err) = catalog.put_database(&target_descriptor) { - // Compensate: remove the lineage edge we just wrote. A failure here - // is fatal — surface both errors so on-call can repair the catalog. - if let Err(rb_err) = catalog.remove_clone_child(source_db_id, target_db_id) { - return Err(DdlError::from_error_in_context( - &format!( - "lineage rollback ALSO failed: {rb_err} — \ - catalog left with orphan lineage edge \ - (source={source_db_id}, target={target_db_id}); catalog write failed", - ), - &put_err, - )); - } - return Err(DdlError::from_error_in_context( - "catalog write failed", - &put_err, - )); - } - - // Stamp every active source collection into the target database with - // `cloned_from` set. This lets the SQL planner resolve collection - // names against the clone without knowing about clone indirection; - // CoW delegation happens at dispatch time. - let source_colls = catalog.load_all_collections(source_db_id).map_err(|e| { - DdlError::from_error_in_context("clone: enumerate source collections", &e) - })?; - let kv_surrogate_ceiling = Some(state.surrogate_assigner.current_hwm()); - for mut coll in source_colls.into_iter().filter(|c| c.is_active) { - coll.database_id = target_db_id; - coll.cloned_from = Some(nodedb_types::CloneOrigin { - source_database: source_db_id, - source_collection: coll.name.clone(), - as_of_lsn, - clone_created_at, - kv_surrogate_ceiling, - }); - coll.clone_status = nodedb_types::CloneStatus::Shadowed; - coll.descriptor_version = 0; - // Fatal, not a warning: an unstamped descriptor means the clone is - // reported as created while one of the source's collections simply - // does not resolve in it, and nothing later re-stamps it. Surfacing - // the failure is the only way the caller learns the clone is - // incomplete. - catalog.put_collection(target_db_id, &coll).map_err(|e| { - DdlError::from_error_in_context( - &format!( - "clone: stamping shadow descriptor for collection '{}' failed", - coll.name - ), - &e, - ) - })?; - - // The owner row is keyed by database, so the clone needs its own. - // Without it the collection resolves but every ownership check - // against it reports no owner. - let owner = StoredOwner { - database_id: target_db_id.as_u64(), - object_type: object_type::COLLECTION.to_string(), - object_name: coll.name.clone(), - tenant_id: coll.tenant_id, - owner_username: coll.owner.clone(), - }; - catalog.put_owner(&owner).map_err(|e| { - DdlError::from_error_in_context( - &format!( - "clone: stamping owner for collection '{}' failed", - coll.name - ), - &e, - ) - })?; - } - - // Copy the source's database-scoped catalog rows: vector index - // params, vector models, column statistics, index records, RLS and - // redaction policies, triggers, retention policies, alert rules, - // continuous aggregates, and streaming materialized views. Schedules - // are not copied — see `catalog_copy::copy_scoped_objects`. - // - // A missing row is fatal, for the same reason an unstamped - // descriptor is fatal. The clone reports itself created while it - // answers queries the source answers differently, and nothing later - // re-copies the row. - copy_database_metadata(catalog, source_db_id, target_db_id) - .map_err(|e| DdlError::from_error_in_context("clone: copying catalog metadata", &e))?; - } - // Synonym groups and custom types travel as proposed entries, not as a // catalog copy. Each needs two more effects than a redb write: the // in-memory registry SHOW reads, and for a group the FTS backend on every // node. A propose runs the applier and both post-apply lanes everywhere, // which is the only path that delivers all three. - // - // Outside the `needs_local_apply` block above on purpose: a propose is the - // thing that reaches every node, and on a cluster proposer that branch is - // false. copy_synonym_groups(state, source_db_id, target_db_id).await?; - copy_custom_types(state, source_db_id, target_db_id)?; - - // Flush the allocator hwm so restarts pick up the correct next-id boundary. - if state.database_registry.should_flush() { - let hwm = state.database_registry.current_hwm(); - if let Err(e) = catalog.put_database_hwm(hwm) { - tracing::warn!("database hwm flush failed after clone: {e}"); - } - } + copy_custom_types(state, source_db_id, target_db_id).await?; state.audit_record_with_db( crate::control::security::audit::AuditEvent::DatabaseCloned, @@ -316,8 +224,7 @@ pub async fn clone_database( /// Propose one `PutSynonymGroup` per source group, rewritten to the target. /// /// A group also lives in each node's FTS backend, which only the post-apply -/// lane reaches. On the `LocalOnly` path no applier runs, so this installs the -/// group and registers it here instead. +/// lane reaches. /// /// A failed propose is fatal, for the same reason an unstamped descriptor is: /// the clone reports itself created while a text query against it expands @@ -337,12 +244,9 @@ async fn copy_synonym_groups( for mut group in groups { group.database_id = target.as_u64(); let entry = CatalogEntry::PutSynonymGroup(Box::new(group.clone())); - let outcome = propose_and_apply(state, &entry) + propose_and_apply_async(state, &entry) + .await .map_err(|e| e.in_context(&format!("clone: copying synonym group '{}'", group.name)))?; - if outcome.needs_local_apply() { - state.synonym_registry.register(group.clone()); - crate::control::catalog_entry::post_apply::install_synonym_group(group, state).await; - } } Ok(()) } @@ -354,9 +258,9 @@ async fn copy_synonym_groups( /// two definitions under one identity. The catalog assigns each copy a fresh /// OID when the entry applies, identically on every node. /// -/// A failed propose is fatal: the clone would resolve neither a copied +/// A failed propose is fatal: the clone will resolve neither a copied /// descriptor's typed column nor the OID a pgwire client reads back. -fn copy_custom_types( +async fn copy_custom_types( state: &SharedState, source: DatabaseId, target: DatabaseId, @@ -370,20 +274,12 @@ fn copy_custom_types( custom_type.database_id = target.as_u64(); custom_type.oid = UNASSIGNED_OID; let entry = CatalogEntry::PutCustomType(Box::new(custom_type.clone())); - let outcome = propose_and_apply(state, &entry).map_err(|e| { + propose_and_apply_async(state, &entry).await.map_err(|e| { e.in_context(&format!( "clone: copying custom type '{}'", custom_type.name )) })?; - if outcome.needs_local_apply() { - register_written( - custom_type.database_id, - custom_type.tenant_id, - &custom_type.name, - state, - ); - } } Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/create.rs b/nodedb/src/control/server/shared/ddl/neutral/database/create.rs index add19995c..9296864ce 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/create.rs @@ -2,14 +2,12 @@ //! Handler for `CREATE [IF NOT EXISTS] DATABASE [WITH (...)]`. //! -//! Ported from the pgwire `ddl::database::create` handler. The catalog -//! allocation, Raft propose / single-node fallback, allocator-hwm flush, -//! per-database metric registration, and the `DatabaseCreated` audit record are -//! preserved verbatim; only the result construction changed from pgwire -//! `Response` to the protocol-neutral [`DdlResult`]. +//! The database id comes from `allocate_database_id`, which replicates it +//! through the metadata log. The descriptor is proposed through metadata +//! Raft. use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::catalog::database_types::{DatabaseDescriptor, DatabaseStatus}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; @@ -53,7 +51,7 @@ fn parse_create_options(options: &[(String, String)]) -> Result [WITH (...)]`. /// /// Required role: `ClusterAdmin` or `Superuser`. -pub fn create_database( +pub async fn create_database( state: &SharedState, identity: &AuthenticatedIdentity, name: &str, @@ -83,12 +81,12 @@ pub fn create_database( } } - // Allocate a new DatabaseId from the registry (local atomic counter; - // authoritative proposal via Raft metadata group 0 is wired separately). - let db_id = state.database_registry.alloc_one(); + let db_id = crate::control::database::allocate_database_id(state) + .await + .map_err(|e| DdlError::from_error_in_context("database id allocation failed", &e))?; // Stamp the descriptor with the next WAL LSN. This is the LSN the very - // next WAL append on this server would receive; it is monotonically + // next WAL append on this server will receive; it is monotonically // greater than any record observed before this DDL ran and gives the // descriptor a well-ordered creation point relative to the WAL. let created_at_lsn = state.wal.next_lsn().as_u64(); @@ -105,32 +103,15 @@ pub fn create_database( idle_session_timeout_secs: 0, }; - // Propose through metadata Raft group 0 so all replicas apply the - // descriptor atomically. In single-node mode `propose_catalog_entry` - // reports `LocalOnly` and falls through to the direct write below. - let outcome = propose_catalog_entry( + // Propose through the metadata proposer so every replica applies the + // descriptor atomically. + propose_catalog_entry_async( state, &CatalogEntry::PutDatabase(Box::new(descriptor.clone())), ) + .await .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; - // Direct write for single-node mode (`LocalOnly`) or as a fallback - // when the cluster is in mixed-version compat mode. - if outcome.needs_local_apply() { - catalog - .put_database(&descriptor) - .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e))?; - } - - // Flush the allocator hwm on the periodic threshold so restarts - // pick up the correct next-id boundary. - if state.database_registry.should_flush() { - let hwm = state.database_registry.current_hwm(); - if let Err(e) = catalog.put_database_hwm(hwm) { - tracing::warn!("database hwm flush failed: {e}"); - } - } - // Register per-database metric series so the names appear in Prometheus // output immediately after creation. Tenants, memory, and storage start // at zero and are updated by their respective subsystems. @@ -154,26 +135,12 @@ pub fn create_database( #[cfg(test)] mod tests { - use std::sync::Arc; - use nodedb_types::error::ErrorCode; use super::*; - use crate::bridge::dispatch::Dispatcher; + use crate::control::cluster::test_one_node; use crate::control::security::identity::{DatabaseSet, Role}; use crate::types::TenantId; - use crate::wal::WalManager; - - fn test_state() -> (tempfile::TempDir, Arc) { - let dir = tempfile::tempdir().expect("create test directory"); - let wal = Arc::new( - WalManager::open_for_testing(&dir.path().join("create-database.wal")) - .expect("open test WAL"), - ); - let (dispatcher, _data_sides) = Dispatcher::new(1, 64); - let state = SharedState::new(dispatcher, wal).expect("construct shared state"); - (dir, state) - } fn admin() -> AuthenticatedIdentity { AuthenticatedIdentity::new_internal_service( @@ -189,19 +156,46 @@ mod tests { /// CREATE DATABASE of a name already taken is `duplicate_database` /// (`42P04`) with the already-exists code, never an internal error. - #[test] - fn creating_an_existing_database_is_a_duplicate_database() { - let (_dir, state) = test_state(); + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn creating_an_existing_database_is_a_duplicate_database() { + let cluster = test_one_node::boot().await; + let state = &cluster.state; let identity = admin(); - create_database(&state, &identity, "orders", false, &[]).expect("first create succeeds"); + create_database(state, &identity, "orders", false, &[]) + .await + .expect("first create succeeds"); - let err = create_database(&state, &identity, "orders", false, &[]) + let err = create_database(state, &identity, "orders", false, &[]) + .await .expect_err("a second create of the same name is refused"); assert_eq!(err.sqlstate, "42P04", "{err:?}"); assert_eq!(err.code, ErrorCode::ALREADY_EXISTS); - let existing = create_database(&state, &identity, "orders", true, &[]) + let existing = create_database(state, &identity, "orders", true, &[]) + .await .expect("IF NOT EXISTS on an existing name succeeds"); assert_eq!(existing.len(), 1); + cluster.shutdown().await; + } + + /// Every CREATE persists the hwm before it returns, so the id survives a + /// restart that happens right after it. + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] + async fn create_persists_the_hwm_of_every_issued_id() { + let cluster = test_one_node::boot().await; + let state = &cluster.state; + let identity = admin(); + let catalog = state.credentials.catalog(); + for name in ["a", "b"] { + create_database(state, &identity, name, false, &[]) + .await + .expect("create"); + let id = catalog + .get_database_id_by_name(name) + .expect("lookup") + .expect("created"); + assert_eq!(catalog.get_database_hwm().expect("hwm"), id.as_u64()); + } + cluster.shutdown().await; } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs index 0931d9ae3..8d97dc8f3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/drop.rs @@ -2,34 +2,28 @@ //! Handler for `DROP [IF EXISTS] DATABASE [CASCADE | FORCE]`. //! -//! Ported from the pgwire `ddl::database::drop` handler. The default-database -//! guard, superuser gate, mirror-unsubscribe teardown, clone orphan-protection -//! (reject / force-materialize), cascade collection drop, `DatabaseDropped` -//! audit (emitted before the catalog mutation), Raft propose / single-node -//! fallback, and per-database metrics cleanup are preserved verbatim; only the -//! result construction changed from pgwire `Response` to the protocol-neutral -//! [`DdlResult`]. +//! Every catalog removal of the drop, the objects inside the database and the +//! descriptor itself, travels in one metadata commit built by +//! [`plan_database_teardown`]. Each node applies it, reclaims the dropped +//! collections' storage, and replays it as one unit after a restart. use nodedb_types::DatabaseId; -use nodedb_types::MirrorStatus; -use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::maintenance::clone_materializer::{ - CloneMaterializerHandle, force_materialize_blocking, -}; -use crate::control::metadata_proposer::propose_catalog_entry; -use crate::control::security::catalog::{StoredCollection, SystemCatalog}; +use crate::control::maintenance::clone_materializer::{CloneMaterializerHandle, force_materialize}; +use crate::control::metadata_proposer::propose_catalog_batch_async; +use crate::control::security::catalog::SystemCatalog; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; use super::super::super::result::{DdlError, DdlResult}; use super::gate::require_superuser; use super::support::{ddl_err, status}; +use super::teardown::plan_database_teardown; /// Handle `DROP [IF EXISTS] DATABASE [CASCADE | FORCE]`. /// /// Required role: `Superuser`. -pub fn drop_database( +pub async fn drop_database( state: &SharedState, identity: &AuthenticatedIdentity, name: &str, @@ -78,51 +72,6 @@ pub fn drop_database( )); } - // ── Mirror unsubscribe ──────────────────────────────────────────────────── - // - // If this database is an active mirror, tear down the cross-cluster observer - // link before removing local state. On the source side the observer simply - // stops receiving entries; the source cluster does not need to be notified - // (this is consistent with the design where promotion is the mirror's local - // decision, e.g. in a DR scenario where the source is unreachable). - // - // The link teardown is best-effort: if the link is already disconnected - // (e.g. source was unreachable), the drop proceeds anyway. The mirror's - // catalog state is the authoritative record of whether a subscription exists. - { - let descriptor_for_mirror = catalog - .get_database(db_id) - .map_err(|e| DdlError::from_error_in_context("catalog read failed", &e))?; - if let Some(descriptor) = descriptor_for_mirror - && let Some(origin) = descriptor.mirror_origin.as_ref() - // Promoted mirrors are now standalone writable databases — the - // observer link was torn down and the mirror_collection_map / - // mirror_lag rows were cleared at promotion time. Skip the - // teardown branch so we don't re-delete already-removed rows - // and don't emit a misleading "subscription teardown" log line. - && !matches!(origin.status, MirrorStatus::Promoted) - { - // Remove the mirror collection map and lag records. - // These are best-effort; we proceed even if they fail because - // the descriptor delete below is the authoritative removal. - if let Err(e) = catalog.delete_mirror_collection_map(db_id) { - tracing::warn!( - db = ?db_id, "DROP DATABASE mirror: failed to remove collection map: {e}" - ); - } - if let Err(e) = catalog.delete_mirror_lag(db_id) { - tracing::warn!( - db = ?db_id, "DROP DATABASE mirror: failed to remove lag record: {e}" - ); - } - tracing::info!( - db = ?db_id, - source_cluster = %origin.source_cluster, - "DROP DATABASE mirror: observer subscription teardown complete" - ); - } - } - // ── Orphan protection ───────────────────────────────────────────────────── // // Check whether any live clones depend on this database as their source. @@ -130,7 +79,7 @@ pub fn drop_database( // If dependents exist and `cascade` is true, block-materialize each one // before proceeding. let dependent_ids = catalog - .get_clone_children(db_id) + .get_live_clone_children(db_id) .map_err(|e| DdlError::from_error_in_context("lineage check failed", &e))?; if !dependent_ids.is_empty() { @@ -148,7 +97,7 @@ pub fn drop_database( ))); } - // FORCE path: block-materialize each dependent clone so it is no longer + // FORCE path: materialize each dependent clone so it is no longer // backed by this source, then proceed with the drop. // // Crash safety: if the server dies mid-force-drop, the dependents @@ -157,10 +106,10 @@ pub fn drop_database( // succeed once the dependents are fully materialized. for dep_id in &dependent_ids { let handle = CloneMaterializerHandle::new(*dep_id); - // Blocking materialization on this thread (pgwire DDL handlers - // execute on a blocking thread pool). - force_materialize_blocking(*dep_id, state, catalog, Some(&handle)).map_err( - |e| match e { + // Awaited on this handler's runtime. + force_materialize(*dep_id, state, catalog, Some(&handle)) + .await + .map_err(|e| match e { // Gated until per-engine row copy lands — surface `0A000` // (`feature_not_supported`) so clients know not to retry. crate::Error::BadRequest { detail } => ddl_err("0A000", detail), @@ -173,12 +122,11 @@ pub fn drop_database( ), &other, ), - }, - )?; + })?; } } - // ── Cascade: drop all collections ──────────────────────────────────────── + // ── Teardown ────────────────────────────────────────────────────────────── let collections = catalog .load_all_collections(db_id) .map_err(|e| DdlError::from_error_in_context("catalog scan failed", &e))?; @@ -193,9 +141,22 @@ pub fn drop_database( ), )); } - - if cascade { - drop_all_collections_in_database(catalog, db_id, &collections)?; + // Arrays register per node, so this node's array catalog is the one to + // read. Every node drops its own arrays when it applies the drop. + let arrays = catalog + .load_all_arrays() + .map_err(|e| DdlError::from_error_in_context("array catalog scan failed", &e))? + .into_iter() + .filter(|entry| entry.array_id.database_id == db_id) + .count(); + if !cascade && arrays > 0 { + return Err(ddl_err( + "2BP01", + format!( + "database '{name}' has {arrays} array(s); \ + use CASCADE to drop them automatically" + ), + )); } // Emit audit BEFORE the catalog mutation so the record is durable even @@ -209,26 +170,19 @@ pub fn drop_database( &format!("DROP DATABASE {name}"), ); - // Propose the delete through Raft; fall back to direct write in single-node mode. - let outcome = propose_catalog_entry( - state, - &CatalogEntry::DeleteDatabase { - db_id: db_id.as_u64(), - }, - ) + // One metadata commit removes every object of the database and then the + // database, on every node. The plan is rebuilt under the DDL preparation + // lease, so a collection created after the check above is still refused + // without CASCADE. + propose_catalog_batch_async(state, |catalog| { + if !cascade { + refuse_if_collections(catalog, db_id, name)?; + } + plan_database_teardown(catalog, db_id) + }) + .await .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; - if outcome.needs_local_apply() { - catalog - .delete_database(db_id) - .map_err(|e| DdlError::from_error_in_context("catalog delete failed", &e))?; - // Single-node path: no applier runs, so this branch owns both the - // quota row deletion and the live cap release. - crate::control::catalog_entry::apply::quota::purge_database_scope(db_id.as_u64(), catalog) - .map_err(|e| DdlError::from_error_in_context("quota purge failed", &e))?; - crate::control::catalog_entry::post_apply::quota::release_database_scope(db_id, state); - } - // Remove per-database metrics entries on drop. if let Some(m) = &state.system_metrics { if let Ok(mut map) = m.database_collections_by_name.write() { @@ -245,30 +199,24 @@ pub fn drop_database( Ok(status("DROP DATABASE")) } -/// Drop every collection in `collections` from the catalog under `db_id`. -/// -/// On the first failure the cascade aborts and the error is returned to the -/// caller. The descriptor is left intact so retrying the DROP picks up the -/// remaining collections; this is the only way to avoid a half-dropped -/// database where the descriptor is gone but the collection rows persist. -fn drop_all_collections_in_database( +/// Refuse a drop without CASCADE of a database that holds collections. +fn refuse_if_collections( catalog: &SystemCatalog, db_id: DatabaseId, - collections: &[StoredCollection], -) -> Result<(), DdlError> { - for coll in collections { - catalog - .delete_collection(db_id, coll.tenant_id, &coll.name) - .map_err(|e| { - DdlError::from_error_in_context( - &format!( - "CASCADE DROP DATABASE {}: failed to delete collection '{}'", - db_id.as_u64(), - coll.name - ), - &e, - ) - })?; - } - Ok(()) + name: &str, +) -> crate::Result<()> { + let collections = catalog.load_all_collections(db_id)?; + let Some(first) = collections.first() else { + return Ok(()); + }; + Err(crate::Error::DependentObjectsExist { + tenant_id: first.tenant_id, + root_kind: "database", + root_name: name.to_string(), + dependent_count: collections.len(), + dependents: collections + .iter() + .map(|c| ("collection".to_string(), c.name.clone())) + .collect(), + }) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/gate.rs b/nodedb/src/control/server/shared/ddl/neutral/database/gate.rs index b2c2c3a3c..e210ef2bd 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/gate.rs @@ -2,11 +2,9 @@ //! Privilege-gate helpers for the protocol-neutral database DDL handlers. //! -//! These mirror the pgwire `types::privilege` gates verbatim — same allowed -//! roles, same `AuditEvent::PermissionDenied`-on-denial behaviour, same -//! SQLSTATE (42501, `INSUFFICIENT_PRIVILEGE`), and byte-identical messages — -//! but return [`DdlError`] instead of `PgWireError` so they carry no pgwire -//! types. +//! Each gate emits `AuditEvent::PermissionDenied` on denial and returns +//! SQLSTATE 42501 (`INSUFFICIENT_PRIVILEGE`) as a [`DdlError`], so the gates +//! carry no pgwire types. use nodedb_types::error::sqlstate; use nodedb_types::id::DatabaseId; @@ -22,10 +20,9 @@ use super::support::ddl_err; /// /// Emits `AuditEvent::PermissionDenied` and returns SQLSTATE 42501 on failure. /// -/// Visibility is widened to the whole `neutral` tree (not just `database`) so -/// the tenant family's `MOVE TENANT` handler — which uses this exact gate -/// verbatim from the pgwire `types::privilege::require_superuser` — can reuse -/// it instead of duplicating the audit-on-denial logic. +/// Visibility spans the whole `neutral` tree (not just `database`) so the +/// tenant family's `MOVE TENANT` handler reuses it instead of duplicating the +/// audit-on-denial logic. pub(in crate::control::server::shared::ddl::neutral) fn require_superuser( state: &SharedState, identity: &AuthenticatedIdentity, @@ -114,8 +111,7 @@ pub(super) fn require_database_owner_or_higher( /// record on denial. Used by the read-only database SHOW handlers, and reused /// (visibility widened to the `neutral` tree) by the tenant family's /// `SHOW TENANT QUOTA|USAGE FOR ... IN DATABASE ...` and -/// `ALTER TENANT ... IN DATABASE ... SET QUOTA` handlers, which used the -/// identical pgwire gate. +/// `ALTER TENANT ... IN DATABASE ... SET QUOTA` handlers. pub(in crate::control::server::shared::ddl::neutral) fn require_tenant_admin( identity: &AuthenticatedIdentity, action: &str, diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs b/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs index ba30a415d..502cc390e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/materialize.rs @@ -2,15 +2,12 @@ //! Handler for `ALTER DATABASE MATERIALIZE`. //! -//! Ported from the pgwire `ddl::database::materialize` handler. The catalog -//! lookup, `DatabaseOwner`-or-higher gate, blocking force-materialization (with -//! `BadRequest` → `0A000` mapping), and `DatabaseMaterialized` audit record are -//! preserved verbatim; only the result construction changed from pgwire -//! `Response` to the protocol-neutral [`DdlResult`]. +//! The catalog lookup, `DatabaseOwner`-or-higher gate, awaited +//! force-materialization (with `BadRequest` → `0A000` mapping), and +//! `DatabaseMaterialized` audit record run here. The result is the +//! protocol-neutral [`DdlResult`]. -use crate::control::maintenance::clone_materializer::{ - CloneMaterializerHandle, force_materialize_blocking, -}; +use crate::control::maintenance::clone_materializer::{CloneMaterializerHandle, force_materialize}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; @@ -22,9 +19,9 @@ use super::support::{ddl_err, status}; /// /// Required role: `DatabaseOwner(db)`, `ClusterAdmin`, or `Superuser`. /// -/// Forces synchronous full materialization of all clone collections in the -/// named database. Returns once all collections are in `Materialized` state. -pub fn alter_database_materialize( +/// Forces full materialization of all clone collections in the named +/// database. Returns once all collections are in `Materialized` state. +pub async fn alter_database_materialize( state: &SharedState, identity: &AuthenticatedIdentity, name: &str, @@ -46,21 +43,22 @@ pub fn alter_database_materialize( // Build a completion handle so callers can observe progress if needed. let handle = CloneMaterializerHandle::new(db_id); - // Run blocking materialization. The pgwire handler executes on a - // dedicated blocking thread pool, so this will not starve the Tokio runtime. + // The materialization is awaited on this handler's runtime. // // `BadRequest` from the gating walker is surfaced as SQLSTATE `0A000` // (`feature_not_supported`) so clients can distinguish it from generic // failures and retry strategy is unambiguous (don't retry — wait for the // per-engine bulk-copy implementation to land). - force_materialize_blocking(db_id, state, catalog, Some(&handle)).map_err(|e| match e { - crate::Error::BadRequest { detail } => ddl_err("0A000", detail), - // Any other error keeps the class the SQLSTATE table gives it. - other => DdlError::from_error_in_context( - &format!("clone materialization of '{name}' failed"), - &other, - ), - })?; + force_materialize(db_id, state, catalog, Some(&handle)) + .await + .map_err(|e| match e { + crate::Error::BadRequest { detail } => ddl_err("0A000", detail), + // Any other error keeps the class the SQLSTATE table gives it. + other => DdlError::from_error_in_context( + &format!("clone materialization of '{name}' failed"), + &other, + ), + })?; state.audit_record_with_db( crate::control::security::audit::AuditEvent::DatabaseMaterialized, diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/create.rs b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/create.rs index ebcd19e18..54b2b1533 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/create.rs @@ -2,17 +2,15 @@ //! Handler for `MIRROR DATABASE FROM . [MODE = sync | async]`. //! -//! Ported from the pgwire `ddl::database::mirror::create` handler. The superuser -//! gate, duplicate-name rejection, self-mirror pre-flight, descriptor build with -//! `MirrorStatus::Bootstrapping`, Raft propose / single-node fallback, -//! allocator-hwm flush, and `DatabaseMirrored` audit are preserved verbatim; -//! only the result construction changed from pgwire `Response` to the -//! protocol-neutral [`DdlResult`]. +//! The superuser gate, duplicate-name rejection, self-mirror pre-flight, +//! descriptor build with `MirrorStatus::Bootstrapping`, Raft propose / +//! single-node fallback, and `DatabaseMirrored` audit run here. The result is +//! the protocol-neutral [`DdlResult`]. use nodedb_types::{DatabaseId, Lsn, MirrorMode, MirrorOrigin, MirrorStatus}; use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::catalog::database_types::{DatabaseDescriptor, DatabaseStatus}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; @@ -24,7 +22,7 @@ use super::super::support::{ddl_err, status}; /// Handle `MIRROR DATABASE FROM . [MODE = ...]`. /// /// Required role: `Superuser`. -pub fn mirror_database( +pub async fn mirror_database( state: &SharedState, identity: &AuthenticatedIdentity, local_name: &str, @@ -71,8 +69,9 @@ pub fn mirror_database( )); } - // Allocate a DatabaseId for the new mirror. - let db_id = state.database_registry.alloc_one(); + let db_id = crate::control::database::allocate_database_id(state) + .await + .map_err(|e| DdlError::from_error_in_context("database id allocation failed", &e))?; let created_at_lsn = state.wal.next_lsn().as_u64(); // The source database numeric id on the source cluster is not known until @@ -105,27 +104,14 @@ pub fn mirror_database( idle_session_timeout_secs: 0, }; - // Propose through Raft; fall back to direct write in single-node mode. - let outcome = propose_catalog_entry( + // Propose through the metadata proposer. + propose_catalog_entry_async( state, &CatalogEntry::PutDatabase(Box::new(descriptor.clone())), ) + .await .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; - if outcome.needs_local_apply() { - catalog - .put_database(&descriptor) - .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e))?; - } - - // Flush allocator hwm on threshold. - if state.database_registry.should_flush() { - let hwm = state.database_registry.current_hwm(); - if let Err(e) = catalog.put_database_hwm(hwm) { - tracing::warn!("database hwm flush failed after MIRROR DATABASE: {e}"); - } - } - state.audit_record_with_db( crate::control::security::audit::AuditEvent::DatabaseMirrored, None, diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/promote.rs b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/promote.rs index 6e120b02f..a8be98d01 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/promote.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/promote.rs @@ -2,18 +2,16 @@ //! Handler for `ALTER DATABASE PROMOTE`. //! -//! Ported from the pgwire `ddl::database::mirror::promote` handler. The catalog -//! lookup, superuser gate (after db-id resolution), idempotent already-promoted -//! short-circuit, not-a-mirror rejection, observer-link teardown BEFORE the -//! descriptor mutation, status flip, Raft propose / single-node fallback, -//! mirror-only catalog cleanup, and `DatabasePromoted` audit are preserved -//! verbatim; only the result construction changed from pgwire `Response` to the -//! protocol-neutral [`DdlResult`]. +//! The catalog lookup, superuser gate (after db-id resolution), idempotent +//! already-promoted short-circuit, not-a-mirror rejection, observer-link +//! teardown BEFORE the descriptor mutation, status flip, Raft propose / +//! single-node fallback, mirror-only catalog cleanup, and `DatabasePromoted` +//! audit run here. The result is the protocol-neutral [`DdlResult`]. use nodedb_types::MirrorStatus; use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::catalog::database_types::DatabaseStatus; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; @@ -25,7 +23,7 @@ use super::super::support::{ddl_err, status}; /// Handle `ALTER DATABASE PROMOTE`. /// /// Required role: `Superuser`. -pub fn promote_database( +pub async fn promote_database( state: &SharedState, identity: &AuthenticatedIdentity, name: &str, @@ -93,19 +91,14 @@ pub fn promote_database( // Persist atomically through Raft. On restart the descriptor is reloaded // with status=Active + origin.status=Promoted, so the database remains // writable without any further intervention. - let outcome = propose_catalog_entry( + propose_catalog_entry_async( state, &CatalogEntry::PutDatabase(Box::new(descriptor.clone())), ) + .await .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; - if outcome.needs_local_apply() { - catalog - .put_database(&descriptor) - .map_err(|e| DdlError::from_error_in_context("catalog write failed", &e))?; - } - - // The database is now writable. Clear the mirror-only catalog state so + // The database is writable. Clear the mirror-only catalog state so // it does not linger as stale data: // - mirror_collection_map: source→local collection name routing used // by the observer-side DDL applier; meaningless once writes are local. diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/show.rs b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/show.rs index 8ccc5f73f..f28c50b29 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/mirror/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/mirror/show.rs @@ -2,13 +2,10 @@ //! Handler for `SHOW DATABASE MIRROR STATUS [FOR ]`. //! -//! Ported from the pgwire `ddl::database::mirror::show` handler. The tenant-admin -//! gate, catalog list, mirror-only filtering, `FOR ` filter, status / -//! mode / lag rendering, `mirror_lag` fallback reads, and the not-found error -//! for a specific name are preserved verbatim; only the result construction -//! changed from pgwire `QueryResponse` to the protocol-neutral [`DdlResult`] -//! over `ShapedRows`. Every column is a `text_field` in the original, so all -//! columns stay `Text`. +//! The tenant-admin gate, catalog list, mirror-only filtering, `FOR ` +//! filter, status / mode / lag rendering, `mirror_lag` fallback reads, and +//! the not-found error for a specific name run here. The result is the +//! protocol-neutral [`DdlResult`] over `ShapedRows`. Every column is `Text`. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/database/mod.rs index c6e84957c..8b25d63e5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/mod.rs @@ -2,9 +2,8 @@ //! Protocol-neutral database DDL family handlers (CREATE / DROP / ALTER //! DATABASE, SHOW DATABASES / QUOTA / USAGE / LINEAGE, CLONE / MIRROR / PROMOTE, -//! BACKUP / RESTORE). Ported from the pgwire `ddl::database` handlers; every -//! catalog / data-plane / audit / privilege-gate side effect is preserved -//! verbatim. +//! BACKUP / RESTORE). Every catalog / data-plane / audit / privilege-gate +//! side effect runs in these handlers. //! //! `USE DATABASE` is intentionally NOT here — it is session-coupled (mutates the //! per-connection current database) and stays on the pgwire side. @@ -22,3 +21,4 @@ pub mod show_lineage; pub mod show_quota; pub mod show_usage; pub mod support; +pub mod teardown; diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/show.rs b/nodedb/src/control/server/shared/ddl/neutral/database/show.rs index a8d9acb9c..3d765cd57 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/show.rs @@ -2,12 +2,9 @@ //! Handler for `SHOW DATABASES`. //! -//! Ported from the pgwire `ddl::database::show` handler. The tenant-admin gate, -//! catalog list, per-database collection count, status mapping, and parent -//! clone rendering are preserved verbatim; only the result construction changed -//! from pgwire `QueryResponse` to the protocol-neutral [`DdlResult`] over -//! `ShapedRows`. Every column is a `text_field` in the original, so all columns -//! stay `Text`. +//! The tenant-admin gate, catalog list, per-database collection count, status +//! mapping, and parent clone rendering run here. The result is the +//! protocol-neutral [`DdlResult`] over `ShapedRows`. Every column is `Text`. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/show_lineage.rs b/nodedb/src/control/server/shared/ddl/neutral/database/show_lineage.rs index d72067aba..88a08412b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/show_lineage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/show_lineage.rs @@ -2,11 +2,9 @@ //! Handler for `SHOW DATABASE LINEAGE FOR `. //! -//! Ported from the pgwire `ddl::database::show_lineage` handler. The tenant-admin -//! gate, bounded `parent_clone` chain walk, and per-ancestor row rendering are -//! preserved verbatim; only the result construction changed from pgwire -//! `QueryResponse` to the protocol-neutral [`DdlResult`] over `ShapedRows`. -//! Every column is a `text_field` in the original, so all columns stay `Text`. +//! The tenant-admin gate, bounded `parent_clone` chain walk, and per-ancestor +//! row rendering run here. The result is the protocol-neutral [`DdlResult`] +//! over `ShapedRows`. Every column is `Text`. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/show_quota.rs b/nodedb/src/control/server/shared/ddl/neutral/database/show_quota.rs index 14ae8ca99..8f515e0f6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/show_quota.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/show_quota.rs @@ -2,12 +2,10 @@ //! Handler for `SHOW DATABASE QUOTA FOR `. //! -//! Ported from the pgwire `ddl::database::show_quota` handler. The tenant-admin -//! gate, catalog lookup, quota-record fallback to `QuotaRecord::DEFAULT`, and -//! per-dimension row rendering (including the `unlimited` special-case) are -//! preserved verbatim; only the result construction changed from pgwire -//! `QueryResponse` to the protocol-neutral [`DdlResult`] over `ShapedRows`. -//! Every column is a `text_field` in the original, so all columns stay `Text`. +//! The tenant-admin gate, catalog lookup, quota-record fallback to +//! `QuotaRecord::DEFAULT`, and per-dimension row rendering (including the +//! `unlimited` special-case) run here. The result is the protocol-neutral +//! [`DdlResult`] over `ShapedRows`. Every column is `Text`. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/show_usage.rs b/nodedb/src/control/server/shared/ddl/neutral/database/show_usage.rs index 9e8ffbb01..46d8a10ff 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/show_usage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/show_usage.rs @@ -2,12 +2,10 @@ //! Handler for `SHOW DATABASE USAGE FOR `. //! -//! Ported from the pgwire `ddl::database::show_usage` handler. The tenant-admin -//! gate, catalog lookup, live-gauge reads from `SystemMetrics`, and per-dimension -//! row rendering (`unlimited` limit + `percent_used`) are preserved verbatim; -//! only the result construction changed from pgwire `QueryResponse` to the -//! protocol-neutral [`DdlResult`] over `ShapedRows`. Every column is a -//! `text_field` in the original, so all columns stay `Text`. +//! The tenant-admin gate, catalog lookup, live-gauge reads from +//! `SystemMetrics`, and per-dimension row rendering (`unlimited` limit + +//! `percent_used`) run here. The result is the protocol-neutral +//! [`DdlResult`] over `ShapedRows`. Every column is `Text`. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/support.rs b/nodedb/src/control/server/shared/ddl/neutral/database/support.rs index 04269fc67..a2acafb90 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/database/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/database/support.rs @@ -9,9 +9,7 @@ use super::super::super::result::{DdlError, DdlResult}; /// Build a [`DdlError`] from an ANSI SQLSTATE code and a message. /// -/// Preserves the exact SQLSTATE / message the pgwire database handlers -/// produced (via `sqlstate_error`), so error parity stays byte-identical after -/// the migration off the pgwire router. +/// The SQLSTATE and message reach the client unchanged. pub(super) fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } @@ -25,9 +23,8 @@ pub(super) fn status(command: &str) -> Vec { }] } -/// Build an all-text [`ShapedRows`] result. Every column produced by the pgwire -/// database SHOW handlers was a `text_field`, so `column_types` is uniformly -/// `Text`, matching the pgwire schema exactly. +/// Build an all-text [`ShapedRows`] result. Every database SHOW column is +/// text, so `column_types` is uniformly `Text`. pub(super) fn text_rows( columns: Vec, rows: Vec>, diff --git a/nodedb/src/control/server/shared/ddl/neutral/database/teardown.rs b/nodedb/src/control/server/shared/ddl/neutral/database/teardown.rs new file mode 100644 index 000000000..ce6179e0f --- /dev/null +++ b/nodedb/src/control/server/shared/ddl/neutral/database/teardown.rs @@ -0,0 +1,399 @@ +// SPDX-License-Identifier: BUSL-1.1 + +//! The catalog entries that drop one database and every object in it. +//! +//! `DROP DATABASE` proposes the whole plan as one metadata commit. Each entry +//! is the replicated delete the object's own `DROP` uses, so every node runs +//! the same apply and post-apply for it: registry eviction, owner and grant +//! cleanup, and Data Plane reclaim for a collection or an array. A restart +//! replays the commit as one unit. + +use std::collections::HashSet; + +use nodedb_types::{DatabaseId, Hlc}; + +use crate::control::catalog_entry::CatalogEntry; +use crate::control::security::catalog::SystemCatalog; +use crate::control::security::permission::parse_scoped_target; + +/// Build the teardown of `db_id` from the catalog as it stands. +/// +/// Dependents come before the objects they name, collections come after +/// every object defined over them, and `DeleteDatabase` comes last. Fenced +/// deletes carry an unstamped target: the proposer freezes it. +pub fn plan_database_teardown( + catalog: &SystemCatalog, + db_id: DatabaseId, +) -> crate::Result> { + let db = db_id.as_u64(); + let mut plan = Vec::new(); + + for group in catalog + .load_all_consumer_groups()? + .into_iter() + .filter(|g| g.database_id == db_id) + { + plan.push(CatalogEntry::DeleteConsumerGroup { + database_id: db, + tenant_id: group.tenant_id, + stream_name: group.stream_name, + name: group.name, + target_hlc: Hlc::ZERO, + }); + } + for stream in catalog + .load_all_change_streams()? + .into_iter() + .filter(|s| s.database_id == db_id) + { + plan.push(CatalogEntry::DeleteChangeStream { + database_id: db, + tenant_id: stream.tenant_id, + name: stream.name, + target_hlc: Hlc::ZERO, + }); + } + for topic in catalog + .load_all_ep_topics()? + .into_iter() + .filter(|t| t.database_id == db_id) + { + plan.push(CatalogEntry::DeleteTopicWithConsumerGroups { + database_id: db, + tenant_id: topic.tenant_id, + name: topic.name, + target_hlc: Hlc::ZERO, + }); + } + for trigger in catalog.load_triggers_for_database(db_id)? { + plan.push(CatalogEntry::DeleteTrigger { + database_id: db_id, + tenant_id: trigger.tenant_id, + name: trigger.name, + target_descriptor_version: 0, + target_hlc: Hlc::ZERO, + }); + } + for schedule in catalog + .load_all_schedules()? + .into_iter() + .filter(|s| s.database_id == db) + { + plan.push(CatalogEntry::DeleteSchedule { + database_id: db_id, + tenant_id: schedule.tenant_id, + name: schedule.name, + }); + } + for view in catalog.load_streaming_mvs_for_database(db_id)? { + plan.push(CatalogEntry::DeleteStreamingMaterializedView { + database_id: db, + tenant_id: view.tenant_id, + name: view.name, + }); + } + for aggregate in catalog.list_continuous_aggregates_in_database(db)? { + plan.push(CatalogEntry::DeleteContinuousAggregate { + database_id: db, + tenant_id: aggregate.tenant_id, + name: aggregate.name, + target_descriptor_version: 0, + target_hlc: Hlc::ZERO, + }); + } + // A materialized view's delete also purges its same-name target + // collection, so the collection pass below skips those targets. + let mut view_targets = HashSet::new(); + for view in catalog.list_materialized_views_in_database(db)? { + view_targets.insert((view.tenant_id, view.name.clone())); + plan.push(CatalogEntry::DeleteMaterializedView { + database_id: db, + tenant_id: view.tenant_id, + name: view.name, + target_descriptor_version: 0, + target_hlc: Hlc::ZERO, + }); + } + for rule in catalog.load_alert_rules_in_database(db)? { + plan.push(CatalogEntry::DeleteAlertRule { + database_id: db, + tenant_id: rule.tenant_id, + name: rule.name, + }); + } + for policy in catalog.load_retention_policies_in_database(db)? { + plan.push(CatalogEntry::DeleteRetentionPolicy { + database_id: db, + tenant_id: policy.tenant_id, + name: policy.name, + collection: policy.collection, + }); + } + for policy in catalog + .load_all_rls_policies()? + .into_iter() + .filter(|p| p.database_id == db) + { + plan.push(CatalogEntry::DeleteRlsPolicy { + tenant_id: policy.tenant_id, + collection: policy.collection, + name: policy.name, + }); + } + for function in catalog + .load_all_functions()? + .into_iter() + .filter(|f| f.database_id == db_id) + { + plan.push(CatalogEntry::DeleteFunction { + database_id: db_id, + tenant_id: function.tenant_id, + name: function.name, + target_descriptor_version: 0, + target_hlc: Hlc::ZERO, + }); + } + for procedure in catalog + .load_all_procedures()? + .into_iter() + .filter(|p| p.database_id == db_id) + { + plan.push(CatalogEntry::DeleteProcedure { + database_id: db_id, + tenant_id: procedure.tenant_id, + name: procedure.name, + target_descriptor_version: 0, + target_hlc: Hlc::ZERO, + }); + } + for sequence in catalog.load_sequences_in_database(db)? { + plan.push(CatalogEntry::DeleteSequence { + database_id: db, + tenant_id: sequence.tenant_id, + name: sequence.name, + target_descriptor_version: 0, + target_hlc: Hlc::ZERO, + }); + } + // The purge removes each collection's indexes, vector parameters and + // models, surrogates, redaction policies, owner row, and engine storage. + for collection in catalog + .load_all_collections(db_id)? + .into_iter() + .filter(|c| !view_targets.contains(&(c.tenant_id, c.name.clone()))) + { + plan.push(CatalogEntry::PurgeCollection { + database_id: db, + tenant_id: collection.tenant_id, + name: collection.name, + target_descriptor_version: 0, + target_hlc: Hlc::ZERO, + }); + } + // Each array delete drops the cell store on every core of every node. + for array in catalog + .load_all_arrays()? + .into_iter() + .filter(|a| a.array_id.database_id == db_id) + { + plan.push(CatalogEntry::DeleteArray { + database_id: db, + tenant_id: array.array_id.tenant_id.as_u64(), + name: array.name, + target_hlc: Hlc::ZERO, + moved_to: None, + }); + } + for group in catalog.load_synonym_groups_in_database(db)? { + plan.push(CatalogEntry::DeleteSynonymGroup { + database_id: db, + tenant_id: group.tenant_id, + name: group.name, + target_hlc: Hlc::ZERO, + }); + } + for custom_type in catalog.load_custom_types_in_database(db)? { + plan.push(CatalogEntry::DeleteCustomType { + database_id: db, + tenant_id: custom_type.tenant_id, + name: custom_type.name, + }); + } + // Grants on the database's collections, functions, and procedures. + for grant in catalog.load_all_permissions()?.into_iter().filter(|grant| { + parse_scoped_target(&grant.target).is_some_and(|target| target.database_id == db) + }) { + plan.push(CatalogEntry::DeletePermission { + target: grant.target, + grantee: grant.grantee, + permission: grant.permission, + }); + } + for grant in catalog.list_database_grants(db_id)? { + plan.push(CatalogEntry::DeleteDatabaseGrant { + db_id: db, + user_id: grant.user_id, + privilege: grant.privilege, + }); + } + plan.push(CatalogEntry::DeleteDatabase { db_id: db }); + Ok(plan) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::control::security::catalog::StoredCollection; + + fn open_catalog() -> (tempfile::TempDir, SystemCatalog) { + let dir = tempfile::tempdir().unwrap(); + let catalog = SystemCatalog::open(&dir.path().join("system.redb")).unwrap(); + (dir, catalog) + } + + fn collection(db: DatabaseId, tenant: u64, name: &str) -> StoredCollection { + let mut stored = StoredCollection::stamped_for_test(tenant, name, "admin"); + stored.database_id = db; + stored + } + + /// Every collection of the database, active or soft-deleted, is purged + /// through the plan, the database goes last, and other databases are + /// untouched. + #[test] + fn plan_purges_every_collection_then_drops_the_database() { + let (_dir, catalog) = open_catalog(); + let dropped = DatabaseId::new(1024); + let kept = DatabaseId::new(1025); + catalog + .put_collection(dropped, &collection(dropped, 1, "orders")) + .unwrap(); + let mut inactive = collection(dropped, 2, "archive"); + inactive.is_active = false; + catalog.put_collection(dropped, &inactive).unwrap(); + catalog + .put_collection(kept, &collection(kept, 1, "orders")) + .unwrap(); + + let plan = plan_database_teardown(&catalog, dropped).unwrap(); + + let mut purged: Vec<(u64, u64, String)> = plan + .iter() + .filter_map(|entry| match entry { + CatalogEntry::PurgeCollection { + database_id, + tenant_id, + name, + .. + } => Some((*database_id, *tenant_id, name.clone())), + _ => None, + }) + .collect(); + purged.sort(); + assert_eq!( + purged, + vec![ + (1024, 1, "orders".to_string()), + (1024, 2, "archive".to_string()) + ] + ); + assert!(matches!( + plan.last(), + Some(CatalogEntry::DeleteDatabase { db_id: 1024 }) + )); + } + + /// Grants on the dropped database's objects are revoked. A grant on the + /// same-name collection of another database stays. + #[test] + fn plan_revokes_grants_of_the_database_only() { + use crate::control::security::catalog::StoredPermission; + use crate::control::security::permission::collection_target; + use crate::types::TenantId; + + let (_dir, catalog) = open_catalog(); + let dropped = DatabaseId::new(1024); + let kept = DatabaseId::new(1025); + for db in [dropped, kept] { + catalog + .put_permission(&StoredPermission { + target: collection_target(db, TenantId::new(1), "orders"), + grantee: "user:bob".into(), + permission: "read".into(), + granted_by: "admin".into(), + granted_at: 0, + }) + .unwrap(); + } + + let plan = plan_database_teardown(&catalog, dropped).unwrap(); + let revoked: Vec<&str> = plan + .iter() + .filter_map(|entry| match entry { + CatalogEntry::DeletePermission { target, .. } => Some(target.as_str()), + _ => None, + }) + .collect(); + let expected = collection_target(dropped, TenantId::new(1), "orders"); + assert_eq!(revoked, vec![expected.as_str()]); + } + + /// Every array of the database is dropped through the plan, ahead of the + /// database. An array of another database stays. + #[test] + fn plan_drops_every_array_of_the_database() { + use crate::control::array_catalog::ArrayCatalogEntry; + use nodedb_array::types::ArrayId; + use nodedb_types::TenantId; + + let (_dir, catalog) = open_catalog(); + let dropped = DatabaseId::new(1024); + let kept = DatabaseId::new(1025); + for (db, tenant) in [(dropped, 1), (dropped, 2), (kept, 1)] { + catalog + .put_array(&ArrayCatalogEntry { + array_id: ArrayId::in_database(TenantId::new(tenant), db, "grid"), + name: "grid".to_string(), + schema_msgpack: vec![0x90], + schema_hash: 7, + created_at_ms: 0, + prefix_bits: 8, + audit_retain_ms: None, + minimum_audit_retain_ms: None, + modification_hlc: Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, + }) + .unwrap(); + } + + let plan = plan_database_teardown(&catalog, dropped).unwrap(); + + let mut arrays: Vec<(u64, u64, &str, bool)> = plan + .iter() + .filter_map(|entry| match entry { + CatalogEntry::DeleteArray { + database_id, + tenant_id, + name, + moved_to, + .. + } => Some((*database_id, *tenant_id, name.as_str(), moved_to.is_some())), + _ => None, + }) + .collect(); + arrays.sort(); + assert_eq!( + arrays, + vec![(1024, 1, "grid", false), (1024, 2, "grid", false)] + ); + let last_array = plan + .iter() + .rposition(|entry| matches!(entry, CatalogEntry::DeleteArray { .. })) + .unwrap(); + assert!(matches!( + plan.last(), + Some(CatalogEntry::DeleteDatabase { db_id: 1024 }) + )); + assert!(last_array < plan.len() - 1); + } +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs index 40e7d84d7..3929ad00a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/crdt_merge.rs @@ -84,6 +84,9 @@ pub async fn crdt_merge( // Admitted at this request's own transport entry; this handler // has no session or peer information of its own. admission: crate::control::server::shared::ddl::user_dispatch::RequestAdmission::AlreadyAdmitted, + // No session reaches this handler, so its read takes the strong + // default (see `DmlTxnCtx::linearizable_reads`). + linearizable: true, }, ) .await @@ -95,14 +98,15 @@ pub async fn crdt_merge( )); } - let target_surrogate = state - .surrogate_assigner - .assign( - nodedb_types::CollectionKey::from_bare(database_id, collection), - tenant_id, - target_id.as_bytes(), - ) - .map_err(|e| DdlError::from_error(&e))?; + let target_surrogate = crate::control::server::surrogate_exchange::assign_surrogate_routed( + state, + nodedb_types::CollectionKey::from_bare(database_id, collection), + tenant_id, + target_id.as_bytes(), + crate::types::TraceId::ZERO, + ) + .await + .map_err(|e| DdlError::from_error(&e))?; let apply_plan = PhysicalPlan::Crdt(CrdtOp::Apply { collection: nodedb_types::QualifiedCollection::new(database_id, collection), diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/search_fusion.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/search_fusion.rs index 334767513..0f5a9d5b0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/search_fusion.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/search_fusion.rs @@ -24,6 +24,7 @@ pub async fn search_fusion( identity: &AuthenticatedIdentity, database_id: DatabaseId, sql: &str, + linearizable: bool, ) -> Result, DdlError> { let (collection, params) = parse_search_using_fusion(sql).ok_or_else(|| { ddl_err( @@ -31,5 +32,13 @@ pub async fn search_fusion( "syntax: SEARCH USING FUSION(ARRAY[...] ...)", ) })?; - rag_fusion(state, identity, database_id, collection, params).await + rag_fusion( + state, + identity, + database_id, + collection, + params, + linearizable, + ) + .await } diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/sparse_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/sparse_index.rs index c998b4c0e..44d19fe31 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/sparse_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/sparse_index.rs @@ -34,14 +34,14 @@ const HEADER: HeaderSpec = HeaderSpec { const DEFAULT_FIELD: &str = "_sparse"; /// `CREATE SPARSE INDEX [IF NOT EXISTS] [] ON [()]` -pub fn create_sparse_index( +pub async fn create_sparse_index( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, sql: &str, ) -> Result, DdlError> { // This surface carries no options, so any trailing token is a statement - // the handler does not implement rather than one it may ignore. + // the handler does not implement rather than one it can ignore. let stmt = parse_index_statement(sql, LEADING, &HEADER, &[], CONTEXT)?; let index_name = &stmt.header.name; @@ -53,7 +53,7 @@ pub fn create_sparse_index( let tenant_id = identity.tenant_id; // The parser substitutes a placeholder when the name is omitted; a - // tenant-global placeholder would collide across collections and leave + // tenant-global placeholder will collide across collections and leave // only one of them droppable, so it resolves per collection and field. let index_name = if index_name == PLACEHOLDER_NAME { format!("{collection}_{field}_sparse_idx") @@ -94,7 +94,8 @@ pub fn create_sparse_index( collection, fields: vec![field.to_string()], }, - )?; + ) + .await?; crate::control::server::shared::ddl::owner::propose_owner( state, IndexKind::Sparse.owner_object_type(), @@ -102,7 +103,8 @@ pub fn create_sparse_index( tenant_id, &index_name, &identity.username, - )?; + ) + .await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/support.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/support.rs index 835c3511d..ead7e6b3b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/support.rs @@ -6,9 +6,7 @@ use super::super::super::result::DdlError; /// Build a [`DdlError`] from an ANSI SQLSTATE code and a message. /// -/// Preserves the exact SQLSTATE / message the pgwire DSL handlers produced -/// (via `sqlstate_error`), so error parity stays byte-identical after the -/// migration off the pgwire router. +/// The SQLSTATE and message reach the client unchanged. pub(super) fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs index 84b2622c6..85458db9d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/text_index.rs @@ -3,10 +3,8 @@ //! `CREATE SEARCH INDEX` / `CREATE FULLTEXT INDEX` DSL handler. //! //! The two keywords are documented as equivalents, so they share one -//! implementation rather than two parsers that drift: they previously -//! disagreed on whether the column list is written `(a, b)` or `FIELDS a, b`, -//! on whether a list may name more than one column, and on whether `ANALYZER` -//! exists at all. Both spellings of the column list are accepted here, and the +//! implementation rather than two parsers that drift on the column list +//! syntax, the column count, and the `ANALYZER` clause. Both spellings of the column list are accepted here, and the //! statement is rejected if any token goes unread. use crate::bridge::envelope::PhysicalPlan; @@ -125,7 +123,7 @@ async fn create_text_index( let fuzzy_default = stmt.options.boolean("FUZZY"); // One index, under the name the statement declared. Synthesizing a - // per-column name and discarding the declared one would leave + // per-column name and discarding the declared one will leave // `DROP INDEX ` unable to match. let index_name = resolve_index_name(&stmt, &collection); if let Some(taken) = state @@ -162,7 +160,8 @@ async fn create_text_index( collection: &collection, fields: stmt.header.columns.clone(), }, - )?; + ) + .await?; crate::control::server::shared::ddl::owner::propose_owner( state, IndexKind::FullText.owner_object_type(), @@ -170,7 +169,8 @@ async fn create_text_index( tenant_id, &index_name, &identity.username, - )?; + ) + .await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, Some(tenant_id), @@ -232,7 +232,7 @@ async fn create_text_index( /// declared, or a per-collection default when the name was omitted. /// /// The parser substitutes a fixed placeholder for an omitted name; a -/// placeholder would collide across collections, so it is replaced by a name +/// placeholder will collide across collections, so it is replaced by a name /// derived from the collection. fn resolve_index_name(stmt: &IndexStatement, collection: &str) -> String { const PLACEHOLDERS: [&str; 2] = ["_auto_search", "_auto_fulltext"]; diff --git a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs index c68bc8501..aa74d3a30 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/dsl/vector_index.rs @@ -135,7 +135,7 @@ pub async fn create_vector_index( )); } - // A name already taken by an index of any kind would leave exactly one of + // A name already taken by an index of any kind will leave exactly one of // the two droppable, since the registry is keyed by name. if let Some(taken) = state .credentials @@ -163,7 +163,8 @@ pub async fn create_vector_index( tenant_id, index_name, &identity.username, - )?; + ) + .await?; let set_params_plan = PhysicalPlan::Vector(VectorOp::SetParams { collection: nodedb_types::QualifiedCollection::new(database_id, collection), @@ -227,20 +228,10 @@ pub async fn create_vector_index( pq_m: params.pq_m, ivf_cells: params.ivf_cells, ivf_nprobe: params.ivf_nprobe, + // Frozen by the proposer's stamp. + modification_hlc: nodedb_types::Hlc::ZERO, }; - let outcome = super::super::vector_replicate::propose_put_params(state, &stored)?; - - // Single node: no applier runs, so post-apply never fires. Run the - // per-node install the post-apply lane runs everywhere else — the redo - // record plus the fan-out that reaches every core, not just the one the - // pre-flight dispatched to. - if outcome.needs_local_apply() { - let shared = state - .self_arc() - .map_err(|e| DdlError::from_error_in_context("install vector index params", &e))?; - crate::control::catalog_entry::post_apply::install_vector_index_params(stored, shared) - .await; - } + super::super::vector_replicate::propose_put_params(state, &stored).await?; propose_index_record( state, @@ -252,7 +243,8 @@ pub async fn create_vector_index( collection, fields: vec![field_name.clone()], }, - )?; + ) + .await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/emergency_ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/emergency_ddl.rs index 49bd47882..42fc0c948 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/emergency_ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/emergency_ddl.rs @@ -8,19 +8,17 @@ //! BLACKLIST AUTH USERS WHERE email LIKE '%@compromised.com' WITH KILL SESSIONS //! ``` //! -//! Ported from the pgwire `ddl::emergency_ddl` handlers; the superuser gates, -//! two-party approval check, emergency-state mutation, blacklist / session -//! side effects, and audit records are preserved verbatim. Only the result -//! construction changed from pgwire `Response` / `Tag` to the protocol-neutral -//! [`DdlResult`]; the SQLSTATE codes, messages, and command tags are unchanged. +//! The superuser gates, two-party approval check, emergency-state mutation, +//! blacklist / session side effects, and audit records run here. The result +//! is the protocol-neutral [`DdlResult`] with its SQLSTATE codes, messages, +//! and command tags. use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; use super::super::result::{DdlError, DdlResult}; -/// Construct a [`DdlError`], preserving the exact SQLSTATE codes and messages -/// the pgwire handlers produced. +/// Construct a [`DdlError`] from a SQLSTATE code and a message. fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/explain_ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/explain_ddl.rs index ca4f88bbd..916f9dfd5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/explain_ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/explain_ddl.rs @@ -7,10 +7,8 @@ //! EXPLAIN SCOPE FOR AUTH USER 'user_123' //! ``` //! -//! Ported from the pgwire `ddl::explain_ddl` handlers. The permission / -//! scope evaluation reads and the synthetic-identity construction are -//! preserved verbatim; only the result construction changed from pgwire -//! `Response` / `QueryResponse` to the protocol-neutral `DdlResult` over +//! The permission / scope evaluation reads and the synthetic-identity +//! construction run here. The result is the protocol-neutral `DdlResult` over //! `ShapedRows`. use serde_json::{Map, Value as JsonValue}; @@ -24,9 +22,7 @@ use super::super::result::{DdlError, DdlResult}; /// Build a [`DdlError`] from an ANSI SQLSTATE code and a message. /// -/// Preserves the exact SQLSTATE / message the pgwire explain handlers -/// produced (via `sqlstate_error`), so error parity stays byte-identical -/// after the migration off the pgwire router. +/// The SQLSTATE and message reach the client unchanged. fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/field_def.rs b/nodedb/src/control/server/shared/ddl/neutral/field_def.rs index df47ab0e5..d07482d75 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/field_def.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/field_def.rs @@ -33,7 +33,7 @@ fn err(sqlstate: &str, message: &str) -> DdlError { } /// Parse and store a DEFINE FIELD statement. -pub fn define_field( +pub async fn define_field( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -121,11 +121,9 @@ pub fn define_field( // replicated catalog state, and an in-place local mutation // both diverges peers and breaks the version/bytes invariant // the metadata applier enforces on replay after a restart. - if let Err(e) = crate::control::catalog_entry::persist_collection_replicated( - state, - database_id, - &coll, - ) { + if let Err(e) = + crate::control::catalog_entry::persist_collection_replicated(state, &coll).await + { return Err(DdlError::from_error_in_context("save collection", &e)); } } @@ -154,7 +152,7 @@ pub fn define_field( /// Parse and store a DEFINE EVENT statement. /// /// Syntax: DEFINE EVENT ON WHEN THEN -pub fn define_event( +pub async fn define_event( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -224,11 +222,9 @@ pub fn define_event( // replicated catalog state, and an in-place local mutation // both diverges peers and breaks the version/bytes invariant // the metadata applier enforces on replay after a restart. - if let Err(e) = crate::control::catalog_entry::persist_collection_replicated( - state, - database_id, - &coll, - ) { + if let Err(e) = + crate::control::catalog_entry::persist_collection_replicated(state, &coll).await + { return Err(DdlError::from_error_in_context("save collection", &e)); } } @@ -262,7 +258,7 @@ pub fn define_event( /// way DEFINE EVENT replicates it with one. Inside a transaction the change /// is held for COMMIT, as DEFINE EVENT's is. An undefined name is an error /// with SQLSTATE 42704. -pub fn remove_event( +pub async fn remove_event( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -311,7 +307,8 @@ pub fn remove_event( &format!("event '{event_name}' on '{collection}' does not exist"), )); } - crate::control::catalog_entry::persist_collection_replicated(state, database_id, &coll) + crate::control::catalog_entry::persist_collection_replicated(state, &coll) + .await .map_err(|e| DdlError::from_error_in_context("save collection", &e))?; state.audit_record( diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/alter.rs b/nodedb/src/control/server/shared/ddl/neutral/function/alter.rs index 2c234659e..9d555af6a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/alter.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/alter.rs @@ -2,17 +2,15 @@ //! `ALTER FUNCTION ... OWNER TO` DDL handler. //! -//! Ported from the pgwire `ddl::function::alter` handler. The catalog path -//! (`propose_and_apply` for both OWNER TO and SET (FUEL/MEMORY), plus the -//! `audit_record` calls) is preserved verbatim; only the result construction -//! changed from pgwire `Response` / `PgWireError` to the protocol-neutral -//! [`DdlResult`] / [`DdlError`]. +//! The catalog path (`propose_and_apply_async` for both OWNER TO and SET +//! (FUEL/MEMORY), plus the `audit_record` calls) runs here. The result is the +//! protocol-neutral [`DdlResult`] / [`DdlError`]. use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; use crate::control::state::SharedState; -use super::super::super::catalog::propose_and_apply; +use super::super::super::catalog::propose_and_apply_async; use super::super::super::result::{DdlError, DdlResult}; use super::super::auth_support::{require_tenant_admin, status}; @@ -47,7 +45,7 @@ fn attach_wasm_payload_for_reproposal( } /// Handle `ALTER FUNCTION OWNER TO ` -pub fn alter_function( +pub async fn alter_function( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -66,7 +64,7 @@ pub fn alter_function( // ALTER FUNCTION SET (FUEL = N, MEMORY = N) if action == "SET" { - return alter_function_limits(state, identity, &name, parts); + return alter_function_limits(state, identity, &name, parts).await; } // ALTER FUNCTION OWNER TO @@ -102,7 +100,7 @@ pub fn alter_function( // function to the previous owner, silently breaking permission // transfer. let entry = crate::control::catalog_entry::CatalogEntry::PutFunction(Box::new(func.clone())); - propose_and_apply(state, &entry)?; + propose_and_apply_async(state, &entry).await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, @@ -115,7 +113,7 @@ pub fn alter_function( } /// Handle `ALTER FUNCTION SET (FUEL = N, MEMORY = N)` -fn alter_function_limits( +async fn alter_function_limits( state: &SharedState, identity: &AuthenticatedIdentity, name: &str, @@ -161,7 +159,7 @@ fn alter_function_limits( attach_wasm_payload_for_reproposal(&mut func, catalog)?; let entry = crate::control::catalog_entry::CatalogEntry::PutFunction(Box::new(func.clone())); - propose_and_apply(state, &entry)?; + propose_and_apply_async(state, &entry).await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/create/handler.rs b/nodedb/src/control/server/shared/ddl/neutral/function/create/handler.rs index c44bb069c..e7b81d6da 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/create/handler.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/create/handler.rs @@ -2,17 +2,15 @@ //! The protocol-neutral `create_function` handler. //! -//! Ported from the pgwire `ddl::function::create` handler. All non-return logic -//! (privilege gate, parsing, body compilation/validation, StoredFunction build, -//! catalog propose-and-apply, dependency extraction into the replicated -//! definition, Lite definition-sync broadcast, and the `audit_record` call) -//! is preserved verbatim; only the -//! result construction changed from pgwire `Response` / `PgWireError` to the -//! protocol-neutral [`DdlResult`] / [`DdlError`]. +//! The privilege gate, parsing, body compilation/validation, StoredFunction +//! build, catalog propose-and-apply, dependency extraction into the +//! replicated definition, Lite definition-sync broadcast, and the +//! `audit_record` call run here. The result is the protocol-neutral +//! [`DdlResult`] / [`DdlError`]. use crate::control::security::catalog::StoredFunction; use crate::control::security::identity::AuthenticatedIdentity; -use crate::control::server::shared::ddl::catalog::propose_and_apply; +use crate::control::server::shared::ddl::catalog::propose_and_apply_async; use crate::control::server::shared::ddl::neutral::auth_support::{require_tenant_admin, status}; use crate::control::server::shared::ddl::result::{DdlError, DdlResult}; use crate::control::state::SharedState; @@ -27,7 +25,7 @@ use super::parse::{ParsedCreateFunction, parse_create_function}; /// Requires superuser or tenant_admin — function bodies are SQL /// expressions that can reference any collection, so creation is /// a privileged operation. -pub fn create_function( +pub async fn create_function( state: &SharedState, identity: &AuthenticatedIdentity, sql: &str, @@ -121,18 +119,10 @@ pub fn create_function( // only until a future batch adds replicated WASM distribution.) // Ownership replicates through the parent `PutFunction` // post_apply on every node — `stored.owner` carries the creator - // and `apply::function::put` installs the owner record. On the - // single-node / rolling-upgrade / DDL-buffer fallback path - // `propose_and_apply` runs the same applier locally so the - // OWNERS row lands too. + // and `apply::function::put` installs the owner record, this node + // included. let entry = crate::control::catalog_entry::CatalogEntry::PutFunction(Box::new(stored.clone())); - let outcome = propose_and_apply(state, &entry)?; - if outcome.needs_local_apply() { - // The no-Raft fallback still uses the CatalogEntry applier for the - // durable row. Run the matching post-apply hook so its owner and - // function-cache effects match a replicated apply. - crate::control::catalog_entry::post_apply::function::put(stored.clone(), state); - } + propose_and_apply_async(state, &entry).await?; // Broadcast to connected Lite sessions after the catalog commit is durable. emit_function_put(state, &stored); diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/function/drop.rs index a5327e759..7df5724f5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/drop.rs @@ -2,11 +2,10 @@ //! `DROP FUNCTION [IF EXISTS]` DDL handler. //! -//! Ported from the pgwire `ddl::function::drop` handler. The catalog path -//! (`propose_catalog_entry` + local applier fallback, dependency-block check, -//! replicated dependency deletion, Lite definition-sync broadcast, and the `audit_record` -//! call) is preserved verbatim; only the result construction changed from -//! pgwire `Response` / `PgWireError` to protocol-neutral result types. +//! The catalog path (`propose_catalog_entry_async` + local applier fallback, +//! dependency-block check, replicated dependency deletion, Lite +//! definition-sync broadcast, and the `audit_record` call) runs here. The +//! result types are protocol-neutral. use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; @@ -18,7 +17,7 @@ use super::parse::validate_identifier; /// Handle `DROP FUNCTION [IF EXISTS] ` /// /// Requires superuser or tenant_admin — same privilege level as CREATE FUNCTION. -pub fn drop_function( +pub async fn drop_function( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -77,10 +76,13 @@ pub fn drop_function( database_id, tenant_id, name: name.clone(), + // Frozen by the proposer's stamp. + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }; - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - crate::control::catalog_entry::apply::local::apply_locally_if_needed(state, &entry, outcome); // Broadcast deletion to connected Lite sessions. { diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/parse.rs b/nodedb/src/control/server/shared/ddl/neutral/function/parse.rs index a66c5dbce..5ecabd878 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/parse.rs @@ -5,11 +5,8 @@ //! SQL type mapping, identifier validation, parameter parsing, and //! utility functions used by both CREATE and DROP handlers. //! -//! Ported verbatim from the pgwire `ddl::function::parse` helpers; only the -//! error type changed from pgwire `PgWireError` to the protocol-neutral -//! [`DdlError`]. `find_matching_paren` is inlined here (the pgwire helper -//! delegated to the pgwire-private `ddl::parse_utils`) to keep this family -//! self-contained. +//! Parse errors are the protocol-neutral [`DdlError`]. `find_matching_paren` +//! is inlined here to keep this family self-contained. use arrow::datatypes::DataType; use nodedb_sql::parser::preprocess::lex::find_ascii_case_insensitive; diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/show.rs b/nodedb/src/control/server/shared/ddl/neutral/function/show.rs index edb512dff..79da5a619 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/show.rs @@ -4,11 +4,9 @@ //! //! Lists all user-defined functions for the current tenant, plus system functions. //! -//! Ported from the pgwire `ddl::function::show` handler. The catalog reads and -//! row ordering (user-defined functions first, then system functions) are -//! preserved verbatim; only the result construction changed from a pgwire -//! `QueryResponse` (6 text columns) to a protocol-neutral [`DdlResult::Rows`] -//! carrying the same columns and per-row values. +//! The catalog reads and row ordering (user-defined functions first, then +//! system functions) run here. The result is a protocol-neutral +//! [`DdlResult::Rows`] with six text columns. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/wasm_aggregate.rs b/nodedb/src/control/server/shared/ddl/neutral/function/wasm_aggregate.rs index a7675db7f..07c771fcd 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/wasm_aggregate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/wasm_aggregate.rs @@ -13,7 +13,7 @@ use nodedb_sql::parser::preprocess::lex::find_ascii_case_insensitive; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; -use super::super::super::catalog::propose_and_apply; +use super::super::super::catalog::propose_and_apply_async; use super::super::super::result::{DdlError, DdlResult}; use super::super::auth_support::{require_tenant_admin, status}; use super::create::emit_function_put; @@ -21,7 +21,7 @@ use super::parse::{find_matching_paren, parse_parameters, validate_identifier}; /// Handle `CREATE [OR REPLACE] AGGREGATE FUNCTION () /// RETURNS LANGUAGE WASM AS ''` -pub fn create_wasm_aggregate( +pub async fn create_wasm_aggregate( state: &SharedState, identity: &AuthenticatedIdentity, sql: &str, @@ -95,10 +95,7 @@ pub fn create_wasm_aggregate( }; let entry = crate::control::catalog_entry::CatalogEntry::PutFunction(Box::new(stored.clone())); - let outcome = propose_and_apply(state, &entry)?; - if outcome.needs_local_apply() { - crate::control::catalog_entry::post_apply::function::put(stored.clone(), state); - } + propose_and_apply_async(state, &entry).await?; emit_function_put(state, &stored); state.audit_record( diff --git a/nodedb/src/control/server/shared/ddl/neutral/function/wasm_create.rs b/nodedb/src/control/server/shared/ddl/neutral/function/wasm_create.rs index 493e14ff1..9dc4fc6aa 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/function/wasm_create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/function/wasm_create.rs @@ -13,14 +13,14 @@ use crate::control::security::catalog::{ use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; -use super::super::super::catalog::propose_and_apply; +use super::super::super::catalog::propose_and_apply_async; use super::super::super::result::{DdlError, DdlResult}; use super::super::auth_support::{require_tenant_admin, status}; use super::create::emit_function_put; use super::parse::parse_function_header; /// Handle `CREATE [OR REPLACE] FUNCTION ... LANGUAGE WASM AS ''` -pub fn create_wasm_function( +pub async fn create_wasm_function( state: &SharedState, identity: &AuthenticatedIdentity, sql: &str, @@ -83,10 +83,7 @@ pub fn create_wasm_function( }; let entry = crate::control::catalog_entry::CatalogEntry::PutFunction(Box::new(stored.clone())); - let outcome = propose_and_apply(state, &entry)?; - if outcome.needs_local_apply() { - crate::control::catalog_entry::post_apply::function::put(stored.clone(), state); - } + propose_and_apply_async(state, &entry).await?; emit_function_put(state, &stored); state.audit_record( diff --git a/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs b/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs index df64a28aa..3c3b2a15a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/grant/database_permission.rs @@ -9,19 +9,17 @@ //! REVOKE ALL ON DATABASE FROM ; //! ``` //! -//! Ported from the pgwire `ddl::grant::database_permission` handlers. All -//! non-return logic (tenant-admin gate, catalog resolution of the database id -//! and grantee user record, `ALL` privilege expansion, catalog propose + -//! single-node fallback, and `audit_record`) is preserved verbatim; only the -//! result construction changed from pgwire `Response` / `PgWireError` to the -//! protocol-neutral [`DdlResult`] / [`DdlError`]. +//! The tenant-admin gate, catalog resolution of the database id and grantee +//! user record, `ALL` privilege expansion, catalog propose + single-node +//! fallback, and `audit_record` run here. The result is the protocol-neutral +//! [`DdlResult`] / [`DdlError`]. //! //! Grants are stored in `_system.database_grants`. They are also reflected //! into the user's `accessible_databases` set — new grants add the database //! to the set; all privileges revoked removes it. use crate::control::catalog_entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::audit::AuditEvent; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; @@ -32,7 +30,7 @@ use super::support::{require_tenant_admin, status}; /// Handle `GRANT ON DATABASE TO `. /// /// Accepted privileges: `ALL`, `CREATE COLLECTION`, `SELECT`. -pub fn grant_database( +pub async fn grant_database( state: &SharedState, identity: &AuthenticatedIdentity, privilege: &str, @@ -60,7 +58,7 @@ pub fn grant_database( }; for priv_name in &privileges { - let outcome = propose_catalog_entry( + propose_catalog_entry_async( state, &CatalogEntry::PutDatabaseGrant { db_id: db_id.as_u64(), @@ -68,13 +66,8 @@ pub fn grant_database( privilege: priv_name.to_string(), }, ) + .await .map_err(|e| DdlError::from_error_in_context("catalog propose", &e))?; - - if outcome.needs_local_apply() { - catalog - .put_database_grant(db_id, user_record.user_id, priv_name) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } } state.audit_record( @@ -88,7 +81,7 @@ pub fn grant_database( } /// Handle `REVOKE ON DATABASE FROM `. -pub fn revoke_database( +pub async fn revoke_database( state: &SharedState, identity: &AuthenticatedIdentity, privilege: &str, @@ -114,7 +107,7 @@ pub fn revoke_database( }; for priv_name in &privileges { - let outcome = propose_catalog_entry( + propose_catalog_entry_async( state, &CatalogEntry::DeleteDatabaseGrant { db_id: db_id.as_u64(), @@ -122,13 +115,8 @@ pub fn revoke_database( privilege: priv_name.to_string(), }, ) + .await .map_err(|e| DdlError::from_error_in_context("catalog propose", &e))?; - - if outcome.needs_local_apply() { - catalog - .delete_database_grant(db_id, user_record.user_id, priv_name) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } } state.audit_record( diff --git a/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs b/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs index 192cc7875..471d9629e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/grant/permission.rs @@ -3,27 +3,26 @@ //! Protocol-neutral `GRANT/REVOKE ON TO/FROM ` //! handlers. //! -//! Ported from the pgwire `ddl::grant::permission` handlers. All non-return -//! logic (grantee canonicalization, target resolution incl. the cross-tenant +//! The grantee canonicalization, target resolution incl. the cross-tenant //! superuser gate, `ALL` expansion, tenant-admin gate, `prepare_permission`, //! catalog propose + single-node fallback, `install_replicated_permission` / -//! `install_replicated_revoke`, and `audit_record`) is preserved verbatim; -//! only the result construction changed from pgwire `Response` / `PgWireError` -//! to the protocol-neutral [`DdlResult`] / [`DdlError`]. +//! `install_replicated_revoke`, and `audit_record` run here. The result is the +//! protocol-neutral [`DdlResult`] / [`DdlError`]. //! //! Proposes `CatalogEntry::{PutPermission, DeletePermission}` so every //! follower's `PermissionStore` and `OWNERS` redb stay in sync. use crate::control::catalog_entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::audit::AuditEvent; use crate::control::security::identity::{AuthenticatedIdentity, Permission, Role}; use crate::control::security::permission::{ - format_permission, function_target, parse_permission, procedure_target, tenant_target, + collection_target, format_permission, function_target, parse_permission, procedure_target, + tenant_target, }; use crate::control::server::shared::ddl::sql_parse::{parse_ident_token, parse_relation_token}; use crate::control::state::SharedState; -use crate::types::TenantId; +use crate::types::{DatabaseId, TenantId}; use super::super::super::result::{DdlError, DdlResult}; use super::support::{require_tenant_admin, status}; @@ -55,7 +54,7 @@ fn canonicalize_grantee(state: &SharedState, raw: &str) -> Result Result<(), DdlError> { - let perm_str = format_permission(perm); let entry = CatalogEntry::DeletePermission { target: target.to_string(), grantee: grantee.to_string(), - permission: perm_str.clone(), + permission: format_permission(perm), }; - let outcome = propose_catalog_entry(state, &entry) + propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - { - let catalog = state.credentials.catalog(); - catalog - .delete_permission(target, grantee, &perm_str) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } - state - .permissions - .install_replicated_revoke(target, grantee, &perm_str); - } Ok(()) } @@ -135,24 +115,25 @@ fn resolve_tenant_id(state: &SharedState, name: &str) -> Result Result<(String, String), DdlError> { if target_type.eq_ignore_ascii_case("FUNCTION") { let name = parse_ident_token(target_name)?; Ok(( - function_target(identity.tenant_id, &name), + function_target(database_id, identity.tenant_id, &name), format!("function '{name}'"), )) } else if target_type.eq_ignore_ascii_case("PROCEDURE") { let name = parse_ident_token(target_name)?; Ok(( - procedure_target(identity.tenant_id, &name), + procedure_target(database_id, identity.tenant_id, &name), format!("procedure '{name}'"), )) } else if target_type.eq_ignore_ascii_case("TENANT") { let tenant_id = resolve_tenant_id(state, target_name)?; - // A tenant admin may only manage grants within their own tenant; + // A tenant admin can only manage grants within their own tenant; // granting across tenant boundaries requires superuser. if tenant_id != identity.tenant_id && !identity.is_superuser { return Err(DdlError::new( @@ -167,7 +148,7 @@ fn resolve_target( // resolves them. let name = parse_relation_token(target_name)?; Ok(( - format!("collection:{}:{name}", identity.tenant_id.as_u64()), + collection_target(database_id, identity.tenant_id, &name), format!("collection '{name}'"), )) } @@ -198,15 +179,17 @@ fn resolve_permissions(permissions: &[String]) -> Result, DdlErr /// `GRANT [, ...] ON TO ` /// /// Called with typed fields from the AST router. -pub fn grant_permission( +pub async fn grant_permission( state: &SharedState, identity: &AuthenticatedIdentity, + database_id: DatabaseId, permissions: &[String], target_type: &str, target_name: &str, grantee: &str, ) -> Result, DdlError> { - let (target, object_desc) = resolve_target(state, identity, target_type, target_name)?; + let (target, object_desc) = + resolve_target(state, identity, database_id, target_type, target_name)?; require_tenant_admin(identity, "grant permissions")?; @@ -214,7 +197,7 @@ pub fn grant_permission( let canonical = canonicalize_grantee(state, grantee)?; for perm in &perms { - propose_grant(state, &target, &canonical, *perm, &identity.username)?; + propose_grant(state, &target, &canonical, *perm, &identity.username).await?; } state.audit_record( @@ -233,15 +216,17 @@ pub fn grant_permission( /// `REVOKE [, ...] ON FROM ` /// /// Called with typed fields from the AST router. -pub fn revoke_permission( +pub async fn revoke_permission( state: &SharedState, identity: &AuthenticatedIdentity, + database_id: DatabaseId, permissions: &[String], target_type: &str, target_name: &str, grantee: &str, ) -> Result, DdlError> { - let (target, object_desc) = resolve_target(state, identity, target_type, target_name)?; + let (target, object_desc) = + resolve_target(state, identity, database_id, target_type, target_name)?; require_tenant_admin(identity, "revoke permissions")?; @@ -249,7 +234,7 @@ pub fn revoke_permission( let canonical = canonicalize_grantee(state, grantee)?; for perm in &perms { - propose_revoke(state, &target, &canonical, *perm)?; + propose_revoke(state, &target, &canonical, *perm).await?; } state.audit_record( diff --git a/nodedb/src/control/server/shared/ddl/neutral/grant/role.rs b/nodedb/src/control/server/shared/ddl/neutral/grant/role.rs index e065c4bc0..527a3b4e6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/grant/role.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/grant/role.rs @@ -2,13 +2,11 @@ //! Protocol-neutral `GRANT/REVOKE ROLE x TO/FROM user` handlers. //! -//! Ported from the pgwire `ddl::grant::role` handlers. All non-return logic -//! (tenant-admin gate, superuser-grant guard, self-superuser-revoke guard, +//! The tenant-admin gate, superuser-grant guard, self-superuser-revoke guard, //! grantee resolution, role-list mutation, `prepare_user_update`, catalog //! propose + single-node fallback, `install_replicated_user`, the role-to-role -//! `set_role_parent` delegation, and `audit_record`) is preserved verbatim; -//! only the result construction changed from pgwire `Response` / `PgWireError` -//! to the protocol-neutral [`DdlResult`] / [`DdlError`]. +//! `set_role_parent` delegation, and `audit_record` run here. The result is the +//! protocol-neutral [`DdlResult`] / [`DdlError`]. //! //! Reuses the existing `CatalogEntry::PutUser` variant. The mutated role list //! is built locally from the user's current record, then @@ -18,7 +16,7 @@ //! separate `Add/RemoveRole` variant needed. use crate::control::catalog_entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::audit::AuditEvent; use crate::control::security::identity::{AuthenticatedIdentity, Role}; use crate::control::state::SharedState; @@ -37,32 +35,22 @@ fn current_roles( Ok((roles, crate::types::TenantId::new(user.tenant_id))) } -fn propose_user_with_roles( +async fn propose_user_with_roles( state: &SharedState, username: &str, tenant_id: crate::types::TenantId, new_roles: Vec, - invalidation: crate::control::security::buses::SessionInvalidationReason, ) -> Result<(), DdlError> { let base = super::super::role_checks::visible_user_or_missing(state, username)?; let stored = state .credentials .prepare_user_update_from(base, None, Some(new_roles.clone())) .map_err(|e| DdlError::new("42704", e.to_string()))?; - let entry = CatalogEntry::PutUser(Box::new(stored.clone())); - let outcome = propose_catalog_entry(state, &entry) + let entry = CatalogEntry::PutUser(Box::new(stored)); + let outcome = propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - { - let catalog = state.credentials.catalog(); - catalog - .put_user(&stored) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } - state - .credentials - .install_replicated_user(&stored, Some(invalidation)); - } else if outcome.is_replicated() { + if outcome.is_durable() { super::super::role_checks::confirm_user_roles(state, username, &new_roles, tenant_id)?; } Ok(()) @@ -75,7 +63,7 @@ fn propose_user_with_roles( /// listed role becomes the grantee's inheritance parent (role-to-role /// membership) — the role hierarchy permits one parent, so granting more /// than one role to a role is rejected. -pub fn grant_role( +pub async fn grant_role( state: &SharedState, identity: &AuthenticatedIdentity, roles: &[String], @@ -87,9 +75,9 @@ pub fn grant_role( } if super::super::role_checks::visible_user(state, grantee).is_some() { - grant_roles_to_user(state, identity, roles, grantee) + grant_roles_to_user(state, identity, roles, grantee).await } else if super::super::role_checks::visible_roles(state).contains_key(grantee) { - grant_role_to_role(state, identity, roles, grantee) + grant_role_to_role(state, identity, roles, grantee).await } else { Err(DdlError::new( "42704", @@ -98,7 +86,7 @@ pub fn grant_role( } } -fn grant_roles_to_user( +async fn grant_roles_to_user( state: &SharedState, identity: &AuthenticatedIdentity, role_names: &[String], @@ -106,7 +94,7 @@ fn grant_roles_to_user( ) -> Result, DdlError> { let (mut roles, tenant_id) = current_roles(state, username)?; let granted: Vec = role_names.iter().map(|name| parse_role(name)).collect(); - // A role that is neither built in nor defined in the user's tenant would + // A role that is neither built in nor defined in the user's tenant will // grant nothing: refuse it by name. super::super::role_checks::check_user_roles(state, &granted, tenant_id)?; for role in granted { @@ -120,13 +108,7 @@ fn grant_roles_to_user( roles.push(role); } } - propose_user_with_roles( - state, - username, - tenant_id, - roles, - crate::control::security::buses::SessionInvalidationReason::RoleGranted, - )?; + propose_user_with_roles(state, username, tenant_id, roles).await?; state.audit_record( AuditEvent::PrivilegeChange, @@ -141,7 +123,7 @@ fn grant_roles_to_user( Ok(status("GRANT")) } -fn grant_role_to_role( +async fn grant_role_to_role( state: &SharedState, identity: &AuthenticatedIdentity, role_names: &[String], @@ -154,7 +136,7 @@ fn grant_role_to_role( )); } let parent = &role_names[0]; - super::super::role::set_role_parent(state, child, Some(parent))?; + super::super::role::set_role_parent(state, child, Some(parent)).await?; state.audit_record( AuditEvent::PrivilegeChange, @@ -167,7 +149,7 @@ fn grant_role_to_role( } /// `REVOKE [, ...] FROM `. -pub fn revoke_role( +pub async fn revoke_role( state: &SharedState, identity: &AuthenticatedIdentity, roles: &[String], @@ -193,9 +175,9 @@ pub fn revoke_role( } if super::super::role_checks::visible_user(state, grantee).is_some() { - revoke_roles_from_user(state, identity, roles, grantee) + revoke_roles_from_user(state, identity, roles, grantee).await } else if super::super::role_checks::visible_roles(state).contains_key(grantee) { - revoke_role_from_role(state, identity, roles, grantee) + revoke_role_from_role(state, identity, roles, grantee).await } else { Err(DdlError::new( "42704", @@ -204,7 +186,7 @@ pub fn revoke_role( } } -fn revoke_roles_from_user( +async fn revoke_roles_from_user( state: &SharedState, identity: &AuthenticatedIdentity, role_names: &[String], @@ -221,13 +203,7 @@ fn revoke_roles_from_user( } } roles.retain(|r| !revoked.contains(r)); - propose_user_with_roles( - state, - username, - tenant_id, - roles, - crate::control::security::buses::SessionInvalidationReason::RoleRevoked, - )?; + propose_user_with_roles(state, username, tenant_id, roles).await?; state.audit_record( AuditEvent::PrivilegeChange, @@ -242,7 +218,7 @@ fn revoke_roles_from_user( Ok(status("REVOKE")) } -fn revoke_role_from_role( +async fn revoke_role_from_role( state: &SharedState, identity: &AuthenticatedIdentity, role_names: &[String], @@ -262,7 +238,7 @@ fn revoke_role_from_role( format!("role '{child}' does not inherit from '{parent}'"), )); } - super::super::role::set_role_parent(state, child, None)?; + super::super::role::set_role_parent(state, child, None).await?; state.audit_record( AuditEvent::PrivilegeChange, diff --git a/nodedb/src/control/server/shared/ddl/neutral/grant/support.rs b/nodedb/src/control/server/shared/ddl/neutral/grant/support.rs index 4cf4a2e6f..cba71afb9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/grant/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/grant/support.rs @@ -17,9 +17,7 @@ pub(super) fn status(command: &str) -> Vec { /// Require that the identity is superuser or tenant_admin. /// -/// Folded in verbatim from the pgwire `require_tenant_admin` helper: it does -/// NOT emit an audit record on denial and returns SQLSTATE 42501 with the -/// identical message. +/// It does NOT emit an audit record on denial and returns SQLSTATE 42501. pub(super) fn require_tenant_admin( identity: &AuthenticatedIdentity, action: &str, diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/algo.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/algo.rs index 06bcb95f2..40056c1f1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/algo.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/algo.rs @@ -4,15 +4,11 @@ use serde_json::{Map, Value as JsonValue}; -use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; -use crate::control::server::broadcast; use crate::control::server::response_shape::types::ShapedRows; use crate::control::state::SharedState; use crate::data::executor::response_codec; use crate::engine::graph::algo::GraphAlgorithm; -use crate::types::TraceId; -use nodedb_physical::physical_plan::GraphOp; use nodedb_types::DatabaseId; use super::super::super::result::{DdlError, DdlResult}; @@ -43,6 +39,9 @@ pub struct AlgoRequest<'a> { pub direction: Option, pub mode: Option, pub personalization: Option, + /// The session reads linearizably: each node the algorithm reads on + /// confirms its groups first. + pub linearizable: bool, } pub async fn algo( @@ -64,10 +63,11 @@ pub async fn algo( direction, mode, personalization, + linearizable, } = request; let algorithm = resolve_algorithm(algorithm_name)?; - // Every dispatch shape below — the single-node broadcast and both cluster + // Every dispatch shape below — the gathered run and both cluster BSP // coordinators — reaches the Data Plane without a single plan for the // planner's authorization and RLS passes to inspect, so both are resolved // here. The result is a rank / component id / count derived from every edge @@ -96,57 +96,20 @@ pub async fn algo( let tenant_id = identity.tenant_id; - // Cluster PageRank / WCC route through their distributed coordinators: graph - // edges are Raft-homed on `from_key(src)` and each core's CSR is partitioned, - // so a single-node `broadcast_to_all_cores` would only see the coordinator's - // local partitions. Each coordinator runs its per-shard primitive - // (`GraphOp::BspSuperstep` for PageRank, `GraphOp::WccSuperstep` for WCC) - // across every shard and assembles the result into the SAME `AlgoResultBatch` - // payload the single-node path produces, so `algo_payload_to_rows` - // renders identical output. - // - // Single-node (`cluster_routing.is_none()`) and every other algorithm keep - // the existing `broadcast_to_all_cores` path byte-identical — only - // cluster-mode PageRank and WCC diverge here. - if state.cluster_routing.is_some() - && matches!(algorithm, GraphAlgorithm::PageRank | GraphAlgorithm::Wcc) + // The algorithm reads every partition of the collection's graph, on every + // node that leads a data group (`graph_dispatch::whole_graph`). + match crate::control::server::graph_dispatch::run_graph_algo( + state, + tenant_id, + database_id, + algorithm, + params, + None, + linearizable, + ) + .await { - let deadline_ms = state.tuning.network.default_deadline_secs * 1_000; - let result = match algorithm { - GraphAlgorithm::PageRank => { - crate::control::server::graph_dispatch::run_bsp_pagerank( - state, - tenant_id, - database_id, - params, - deadline_ms, - ) - .await - } - _ => { - // Wcc — the outer guard guarantees this. - crate::control::server::graph_dispatch::run_bsp_wcc( - state, - tenant_id, - database_id, - params, - deadline_ms, - ) - .await - } - }; - return match result { - Ok(payload) => Ok(algo_payload_to_rows(&payload, algorithm)?), - Err(e) => Err(DdlError::from_error(&e)), - }; - } - - let plan = PhysicalPlan::Graph(GraphOp::Algo { algorithm, params }); - - match broadcast::broadcast_to_all_cores(state, tenant_id, database_id, plan, TraceId::ZERO) - .await - { - Ok(resp) => Ok(algo_payload_to_rows(&resp.payload, algorithm)?), + Ok(payload) => Ok(algo_payload_to_rows(&payload, algorithm)?), Err(e) => Err(DdlError::from_error(&e)), } } @@ -219,11 +182,10 @@ fn clamp_opt( /// Render an algorithm result payload into a protocol-neutral row set. /// /// Every column is emitted as `Text` with its cell pre-rendered to the exact -/// string the pgwire handler wrote (all algorithm result columns used -/// `text_field`): `Text` → the raw string, `Float64` → `format!("{v}")` or the +/// string (all algorithm result columns are text): `Text` → the raw string, `Float64` → `format!("{v}")` or the /// literal `Infinity` for a non-representable/non-finite score, `Int64` → /// decimal or `0`. Pre-rendering keeps the wire bytes byte-identical (a native -/// float path would change both the column OID and the `Infinity` fallback). +/// float path will change both the column OID and the `Infinity` fallback). fn algo_payload_to_rows( payload: &crate::bridge::envelope::Payload, algorithm: GraphAlgorithm, diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/dispatch.rs index ef582b406..5368e5e41 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/dispatch.rs @@ -25,6 +25,9 @@ pub async fn dispatch_graph( stmt: NodedbStatement, txn_ctx: &DmlTxnCtx<'_>, ) -> Option, DdlError>> { + // Graph reads honour the session's read consistency, like every other + // read (see `DmlTxnCtx::linearizable_reads` for the default). + let linearizable = txn_ctx.linearizable_reads(); match stmt { NodedbStatement::Graph(GraphStmt::GraphInsertEdge { collection, @@ -90,6 +93,7 @@ pub async fn dispatch_graph( depth, edge_label, direction, + linearizable, }, ) .await, @@ -116,6 +120,7 @@ pub async fn dispatch_graph( edge_label, direction, txn_id, + linearizable, }, ) .await, @@ -138,6 +143,7 @@ pub async fn dispatch_graph( dst, max_depth, edge_label, + linearizable, }, ) .await, @@ -173,19 +179,37 @@ pub async fn dispatch_graph( direction, mode, personalization, + linearizable, }, ) .await, ), - NodedbStatement::Graph(GraphStmt::GraphRagFusion { collection, params }) => { - Some(rag_fusion::rag_fusion(state, identity, database_id, collection, params).await) - } + NodedbStatement::Graph(GraphStmt::GraphRagFusion { collection, params }) => Some( + rag_fusion::rag_fusion( + state, + identity, + database_id, + collection, + params, + linearizable, + ) + .await, + ), NodedbStatement::Graph(GraphStmt::ShowGraphStats { collection, verbose, as_of, }) => Some( - stats::show_graph_stats(state, identity, database_id, collection, verbose, as_of).await, + stats::show_graph_stats( + state, + identity, + database_id, + collection, + verbose, + as_of, + linearizable, + ) + .await, ), // `MatchQuery` (handled by the router's typed arm → neutral `match_ops`) // and every non-graph-overlay variant return None so the caller can route diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs index fcf346e98..36820e2b6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge.rs @@ -9,13 +9,13 @@ use nodedb_sql::ddl_ast::GraphProperties; use crate::bridge::envelope::PhysicalPlan; -use crate::control::planner::calvin::{build_static_tx_class, submit_calvin_routed}; +use crate::control::planner::calvin::{build_static_tx_class, submit_calvin_routed_write}; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::session::{DmlTxnCtx, TransactionState}; use crate::control::server::shared::sql::staging_predicates::require_affected_count; use crate::control::server::surrogate_exchange::assign_surrogate_routed; use crate::control::state::SharedState; -use crate::types::{DatabaseId, TraceId, VShardId}; +use crate::types::{DatabaseId, RecordHomes, TraceId, VShardId}; use nodedb_physical::physical_plan::GraphOp; use nodedb_physical::physical_task::{PhysicalTask, PostSetOp}; @@ -74,16 +74,16 @@ pub async fn insert_edge( // Dual-home: a cross-shard edge must be written on the home vShard of both src // and dst, or reverse/IN traversal never finds it. - let vsrc = VShardId::from_key(src.as_bytes()); - let vdst = VShardId::from_key(dst.as_bytes()); + let homes = RecordHomes::edge(&src, &dst); + let (vsrc, vdst) = (homes.owner(), homes.second()); let key = nodedb_types::CollectionKey::from_bare(database_id, &collection); let src_surrogate = - assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), TraceId::ZERO) + assign_surrogate_routed(state, key, tenant_id, src.as_bytes(), TraceId::ZERO) .await .map_err(|e| DdlError::from_error(&e))?; let dst_surrogate = - assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), TraceId::ZERO) + assign_surrogate_routed(state, key, tenant_id, dst.as_bytes(), TraceId::ZERO) .await .map_err(|e| DdlError::from_error(&e))?; @@ -104,11 +104,8 @@ pub async fn insert_edge( }, )?; - // Calvin cross-shard atomicity needs cluster mode with a wired sequencer. In - // single-node, one write already lands both EDGES and REVERSE_EDGES locally. - let calvin_available = - state.cluster_transport.is_some() && state.sequencer_inbox.get().is_some(); - let single_home = vsrc == vdst || !calvin_available; + // A cross-shard edge commits on both homes atomically through Calvin. + let single_home = homes.is_single(); // In a transaction, an insert stages into `GraphTxnOverlay` instead of applying now // (COMMIT replays, ROLLBACK discards); a cross-shard edge stages into both endpoints. @@ -133,7 +130,7 @@ pub async fn insert_edge( } let affected = if single_home { - // F1a fast path: single-home write to `vsrc` covers both forward and reverse. + // Both endpoints share one home, so one write to `vsrc` stores the edge. let plan = PhysicalPlan::Graph(edge_put); let response = crate::control::server::sync::raft_dispatch::dispatch_trusted_internal_sync_response( @@ -142,7 +139,6 @@ pub async fn insert_edge( database_id, vsrc, plan, - TraceId::ZERO, crate::event::EventSource::User, ) .await @@ -162,26 +158,13 @@ pub async fn insert_edge( }; let tx_class = build_static_tx_class(&[task], tenant_id, &[]).map_err(|e| DdlError::from_error(&e))?; - let response = submit_calvin_routed(state, tx_class) + // Every participant of this graph-only transaction reports its + // answer in its completion ack, so the count arrives on any node. + let response = submit_calvin_routed_write(state, tx_class) .await .map_err(|e| DdlError::from_error(&e))?; - match response { - Some(response) => { - data_plane_verdict(&response)?; - response_affected(&response)? - } - // `plans_have_primary_write` treats a Graph-only tx_class (no - // accompanying primary DML, exactly this DSL's own tx_class) as - // primary, so `commit_apply_tail` always deposits this - // participant's applied response. A missing deposit here is a - // scheduler invariant violation, never a value to guess. - None => { - return Err(DdlError::internal( - "cross-shard edge insert completed with no applied response to read \ - its affected count from", - )); - } - } + data_plane_verdict(&response)?; + response_affected(&response)? }; Ok(vec![DdlResult::Status { @@ -200,8 +183,7 @@ pub struct EdgeRef { } /// The home vShard(s) an edge resolves to: `vsrc` holds the forward row, `vdst` -/// the reverse row. `single_home` is true when both share one vShard or Calvin -/// is unavailable. Bundled for [`stage_edge_dual_home`](super::edge_stage::stage_edge_dual_home). +/// the reverse row. `single_home` is true when both share one vShard. Bundled for [`stage_edge_dual_home`](super::edge_stage::stage_edge_dual_home). pub struct EdgeHomes { pub vsrc: VShardId, pub vdst: VShardId, @@ -236,16 +218,16 @@ pub async fn delete_edge( // Dual-home: stored forward on `from_key(src)` and reverse on `from_key(dst)`, // so delete must tombstone both homes. - let vsrc = VShardId::from_key(src.as_bytes()); - let vdst = VShardId::from_key(dst.as_bytes()); + let homes = RecordHomes::edge(&src, &dst); + let (vsrc, vdst) = (homes.owner(), homes.second()); let key = nodedb_types::CollectionKey::from_bare(database_id, &collection); let src_surrogate = - assign_surrogate_routed(state, vsrc, key, tenant_id, src.as_bytes(), TraceId::ZERO) + assign_surrogate_routed(state, key, tenant_id, src.as_bytes(), TraceId::ZERO) .await .map_err(|e| DdlError::from_error(&e))?; let dst_surrogate = - assign_surrogate_routed(state, vdst, key, tenant_id, dst.as_bytes(), TraceId::ZERO) + assign_surrogate_routed(state, key, tenant_id, dst.as_bytes(), TraceId::ZERO) .await .map_err(|e| DdlError::from_error(&e))?; @@ -266,11 +248,8 @@ pub async fn delete_edge( }, )?; - // Calvin needs cluster mode with a wired sequencer; single-node already - // tombstones both EDGES and REVERSE_EDGES in one write. - let calvin_available = - state.cluster_transport.is_some() && state.sequencer_inbox.get().is_some(); - let single_home = vsrc == vdst || !calvin_available; + // A cross-shard edge delete commits on both homes atomically through Calvin. + let single_home = homes.is_single(); // Inside a transaction, an edge delete stages into `GraphTxnOverlay` instead of // applying now, so RYOW sees it removed; COMMIT replays it, ROLLBACK discards it. @@ -298,7 +277,6 @@ pub async fn delete_edge( // writing identity to decide it. Resolve against stored properties while it's live. if let Some(resolver) = crate::control::write_resolve::resolver_for_plan(&PhysicalPlan::Graph(edge_delete.clone())) - && state.async_raft_proposer().is_some() { let ctx = crate::control::write_resolve::WriteResolveContext { tenant_id, @@ -314,7 +292,7 @@ pub async fn delete_edge( } let affected = if single_home { - // F1a fast path: single-home write to `vsrc` tombstones both rows together. + // Both endpoints share one home, so one delete on `vsrc` removes the edge. let plan = PhysicalPlan::Graph(edge_delete); let response = crate::control::server::sync::raft_dispatch::dispatch_trusted_internal_sync_response( @@ -323,7 +301,6 @@ pub async fn delete_edge( database_id, vsrc, plan, - TraceId::ZERO, crate::event::EventSource::User, ) .await @@ -343,28 +320,13 @@ pub async fn delete_edge( }; let tx_class = build_static_tx_class(&[task], tenant_id, &[]).map_err(|e| DdlError::from_error(&e))?; - let response = submit_calvin_routed(state, tx_class) + // Every participant of this graph-only transaction reports its + // answer in its completion ack, so the count arrives on any node. + let response = submit_calvin_routed_write(state, tx_class) .await .map_err(|e| DdlError::from_error(&e))?; - match response { - Some(response) => { - data_plane_verdict(&response)?; - response_affected(&response)? - } - // `plans_have_primary_write` treats a Graph-only tx_class (no - // accompanying primary DML, exactly this DSL's own tx_class) as - // primary, so `commit_apply_tail` always deposits this - // participant's applied response. A delete's count is not - // deterministic from the plan alone (the edge may already be - // absent), so a missing deposit here is a scheduler invariant - // violation, never a value to guess. - None => { - return Err(DdlError::internal( - "cross-shard edge delete completed with no applied response to read \ - its affected count from", - )); - } - } + data_plane_verdict(&response)?; + response_affected(&response)? }; Ok(vec![DdlResult::Status { @@ -402,37 +364,16 @@ pub async fn set_node_labels( }; // Single-keyed on `node_id`, so single-home: route to `from_key(node_id)`. - // No redb durability — a WAL record is the bitset's only backing. The - // record's outcome-floor window opens before the append and closes from - // the dispatch's outcome. - let owner = crate::control::server::dispatch_utils::RecordOwner { - tenant_id, - database_id: DatabaseId::DEFAULT, - vshard_id, - }; - let minted = crate::control::server::dispatch_utils::MintedRecords::open(&state.outcome_floor); - if let Err(e) = minted.append_plan( - &state.wal, - owner, - &plan, - // The same source the edge write is dispatched with below. - crate::event::EventSource::User, - ) { - // Any record appended before the error never reaches a core. - minted - .cancel(&state.wal, owner, 0) - .await - .map_err(|c| DdlError::from_error(&c))?; - return Err(DdlError::from_error(&e)); - } - + // No redb durability: the WAL record the Raft entry's apply appends on + // every replica is the bitset's only backing. let response = - crate::control::server::sync::raft_dispatch::dispatch_trusted_internal_minted_sync_response( + crate::control::server::sync::raft_dispatch::dispatch_trusted_internal_sync_response( state, - owner, + tenant_id, + DatabaseId::DEFAULT, + vshard_id, plan, crate::event::EventSource::User, - minted, ) .await .map_err(|e| DdlError::from_error(&e))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs index 5022b7d0c..0d87597dc 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/edge_stage.rs @@ -5,10 +5,10 @@ //! //! The `GRAPH INSERT EDGE` / `GRAPH DELETE EDGE` handlers dispatch a single //! `GraphOp::EdgePut` / `EdgeDelete` directly to the Data Plane in autocommit. -//! Inside an explicit `BEGIN..COMMIT` block that direct dispatch would apply +//! Inside an explicit `BEGIN..COMMIT` block that direct dispatch will apply //! the write DURABLY at statement time, so an in-transaction `MATCH` / `GRAPH -//! NEIGHBORS` would not observe it as staged (breaking read-your-own-writes) -//! and a ROLLBACK could not undo it. These helpers instead route the write +//! NEIGHBORS` will not observe it as staged (breaking read-your-own-writes) +//! and a ROLLBACK cannot undo it. These helpers instead route the write //! through the protocol-neutral staging gate //! ([`route_in_tx_write`](crate::control::server::shared::session::staging_gate::route_in_tx_write)), //! exactly like every other in-transaction point write: the Data Plane stages @@ -16,7 +16,7 @@ //! Hop for RYOW), the plan is buffered for COMMIT's durable replay, and //! ROLLBACK drops the overlay. //! -//! A SINGLE-HOME edge (both endpoints on one vShard, or single-node) stages +//! A SINGLE-HOME edge (both endpoints on one vShard) stages //! once into `vsrc`. A cross-shard (dual-home) edge is reachable from BOTH //! endpoints, and each core merges only its OWN overlay on a read, so //! [`stage_edge_dual_home`] stages the same edge into both the `vsrc` and `vdst` @@ -41,10 +41,10 @@ use super::support::ddl_err; /// Stage a graph-edge write into the active transaction's overlay on EVERY /// vShard the edge homes to. /// -/// A single-home edge (both endpoints on one vShard, or single-node) stages +/// A single-home edge (both endpoints on one vShard) stages /// once, into `vsrc`. A cross-shard (dual-home) edge is reachable from BOTH /// endpoints, and each Data-Plane core merges only its OWN transaction overlay -/// on a read — so a reverse/IN traversal that scatters to `from_key(dst)` would +/// on a read — so a reverse/IN traversal that scatters to `from_key(dst)` will /// never observe an edge staged only on `vsrc`. The dual-home case therefore /// stages the SAME `GraphOp` into both the `vsrc` and `vdst` overlays, giving /// read-your-own-writes from either endpoint. Each stage buffers its task, so diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/rag_fusion.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/rag_fusion.rs index df6d66902..dad1e6fa2 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/rag_fusion.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/rag_fusion.rs @@ -37,6 +37,7 @@ pub async fn rag_fusion( database_id: DatabaseId, collection: String, params: FusionParams, + linearizable: bool, ) -> Result, DdlError> { // Gate on catalog `is_active` (see `support::ensure_collection_active`): // a plain `DROP COLLECTION` only flips `is_active=false` without @@ -97,10 +98,7 @@ pub async fn rag_fusion( }; let options = match params.max_visited { - Some(mv) => GraphTraversalOptions { - max_visited: mv, - ..Default::default() - }, + Some(mv) => GraphTraversalOptions { max_visited: mv }, None => GraphTraversalOptions::default(), }; @@ -118,6 +116,7 @@ pub async fn rag_fusion( options, bm25_query: params.bm25_query, bm25_field: params.bm25_field, + stage: nodedb_physical::physical_plan::RagStage::Local, }); // Only reached through `shared::ddl::dispatch`, which native/pgwire/HTTP @@ -132,6 +131,7 @@ pub async fn rag_fusion( // Admitted at this request's own transport entry; this handler has no // session or peer information of its own. admission: user_dispatch::RequestAdmission::AlreadyAdmitted, + linearizable, }) .await .map_err(|e| DdlError::from_error(&e))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/stats.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/stats.rs index 51ed521be..75bcaf184 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/stats.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/stats.rs @@ -2,48 +2,31 @@ //! `SHOW GRAPH STATS` handler. //! -//! Reads persistent graph-stats counters from every Data-Plane core via -//! `broadcast_to_all_cores`, aggregates the per-core +//! Reads graph-stats counters from every core of one node per data group +//! (`graph_dispatch::scatter_to_graph_owners`), aggregates the per-core //! [`CollectionStats`](crate::engine::graph::edge_store::stats::CollectionStats) //! payloads, and emits a protocol-neutral result row set. //! -//! Aggregation rules: -//! - `edge_count`: summed across cores (each core holds a disjoint partition). -//! - `distinct_node_count`: summed across cores. Per-core CSR partitions are -//! hash-disjoint by node id, so the cross-core sum equals the global distinct -//! count — no double-count. -//! - `distinct_label_count`: re-derived from the merged `labels` vec rather than -//! summed (labels are NOT partition-disjoint — the same label name can appear -//! in multiple cores). -//! - `labels`: merged by name; counts summed; output is sorted ascending by name. +//! Every read asks for the exact logical edges, so aggregation is an identity +//! union. An edge that two cores or two nodes hold (both endpoint homes, or a +//! replica) counts once: +//! - `edge_count`, `distinct_node_count` and `labels` are re-derived from the +//! union of `(src, label, dst)` edges. +//! - `distinct_label_count` is the number of merged labels. +//! - `labels` is sorted ascending by name. use std::collections::BTreeMap; -use std::sync::atomic::{AtomicU64, Ordering}; use nodedb_types::DatabaseId; use nodedb_types::diagnostic::DiagnosticLayer; use serde_json::{Map, Value as JsonValue}; use tracing::info_span; -/// Total number of `SHOW GRAPH STATS` calls served since process start. -/// Read by the metrics endpoint via [`graph_stats_calls_total`]. -static GRAPH_STATS_CALLS: AtomicU64 = AtomicU64::new(0); - -/// Counter for observability. Mirrors the `broadcast_call_count()` style -/// used elsewhere in the Control Plane. Exposed for metrics endpoints -/// and test harnesses to assert call counts. -#[allow(dead_code)] -pub fn graph_stats_calls_total() -> u64 { - GRAPH_STATS_CALLS.load(Ordering::Relaxed) -} - use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; -use crate::control::server::broadcast::broadcast_to_all_cores; use crate::control::server::response_shape::types::{DdlColType, ShapedRows}; use crate::control::state::SharedState; use crate::engine::graph::edge_store::stats::CollectionStats; -use crate::types::TraceId; use nodedb_physical::physical_plan::GraphOp; use super::super::super::result::{DdlError, DdlResult}; @@ -64,8 +47,8 @@ pub async fn show_graph_stats( collection: Option, verbose: bool, as_of: Option, + linearizable: bool, ) -> Result, DdlError> { - GRAPH_STATS_CALLS.fetch_add(1, Ordering::Relaxed); let scope = if collection.is_some() { "collection" } else { @@ -81,13 +64,13 @@ pub async fn show_graph_stats( as_of = ?as_of, ); - // The counters reach the Data Plane through `broadcast_to_all_cores`, which + // The counters reach the Data Plane through `scatter_to_graph_owners`, which // never runs the planner's authorization or RLS passes, so both are // resolved here. A counter carries no row for a filter to apply to, and it // counts the edges of rows a policy hides, so a read policy refuses. The // tenant-wide form names no collection to ask the narrow question about, so // it asks the tenant-wide one — and narrows its rows to the collections the - // caller may actually read, below. + // caller can actually read, below. let gate = RefusingReadGate::for_request(state, identity, database_id); match collection.as_deref() { Some(name) => gate.gate_collection(name, STATS_WHAT)?, @@ -108,7 +91,7 @@ pub async fn show_graph_stats( // Exact logical-edge scans are required even for current-time reads: // source-owned summaries cannot deduplicate a destination shared by edges - // from different source vShards, and mixed legacy/current collections may + // from different source vShards, and mixed legacy/current collections can // have no summary row at all. Identity union below is the correctness path. let plan = PhysicalPlan::Graph(GraphOp::Stats { collection: collection @@ -117,12 +100,32 @@ pub async fn show_graph_stats( as_of: as_of.or(Some(i64::MAX)), }); - let resp = broadcast_to_all_cores(state, identity.tenant_id, database_id, plan, TraceId::ZERO) - .await - .map_err(|e| DdlError::from_error_in_context("graph stats dispatch failed", &e))?; - - let merged: Vec = decode_merged_stats(resp.payload.as_bytes()) - .map_err(|e| DdlError::from_error_in_context("graph stats decode failed", &e))?; + // A collection's edges spread over every data group. The stats read goes + // to one node per group, and every one of its cores answers + // (`graph_dispatch::whole_graph`). A linearizable read is confirmed on each + // node that serves it. + let payloads = crate::control::server::graph_dispatch::scatter_to_graph_owners( + state, + identity.tenant_id, + database_id, + plan, + linearizable, + collection.as_deref().map(|name| { + nodedb_types::QualifiedCollection::new(database_id, name) + .as_str() + .to_owned() + }), + ) + .await + .map_err(|e| DdlError::from_error_in_context("graph stats dispatch failed", &e))?; + + let mut merged: Vec = Vec::new(); + for payload in &payloads { + merged.extend( + decode_merged_stats(payload.as_bytes()) + .map_err(|e| DdlError::from_error_in_context("graph stats decode failed", &e))?, + ); + } let aggregated = aggregate_by_collection(merged); @@ -149,7 +152,7 @@ pub async fn show_graph_stats( } } -/// Decode the merged msgpack array produced by `broadcast_to_all_cores`. +/// Decode one node's merged msgpack array of per-core stats. fn decode_merged_stats(payload: &[u8]) -> crate::Result> { if payload.is_empty() { return Ok(Vec::new()); diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/support.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/support.rs index 2cf23600c..b15dd4498 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/support.rs @@ -10,9 +10,7 @@ use super::super::super::result::DdlError; /// Build a [`DdlError`] from an ANSI SQLSTATE code and a message. /// -/// Preserves the exact SQLSTATE / message the pgwire graph-ops handlers -/// produced (via `sqlstate_error`), so error parity stays byte-identical after -/// the migration off the pgwire router. +/// The SQLSTATE and message reach the client unchanged. pub(super) fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/traverse.rs b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/traverse.rs index 60a63b52b..9200f09e0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/graph_ops/traverse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/graph_ops/traverse.rs @@ -10,7 +10,6 @@ use crate::control::state::SharedState; use crate::engine::graph::edge_store::Direction; use crate::engine::graph::traversal_options::GraphTraversalOptions; use crate::engine::graph::traversal_options::MAX_GRAPH_TRAVERSAL_DEPTH; -use crate::types::TraceId; use nodedb_physical::physical_plan::GraphOp; use nodedb_types::DatabaseId; @@ -82,7 +81,7 @@ fn check_tenant_graph_depth( /// depth-1 `GraphOp::Hop` dispatch -- merging staged edges into an N-hop /// cross-core BFS is out of scope for this single-hop read-your-own-writes /// unit (see `graph_txn_merge`'s doc comment). -/// Fail closed unless `identity` may read the traversal's collection. +/// Fail closed unless `identity` can read the traversal's collection. /// /// A traversal discloses which nodes exist in a collection and how they are /// connected, so it carries the same read grant the collection's rows do. @@ -104,7 +103,7 @@ fn authorize_traversal( // seam: the traversal returns topology, so there are no stored columns in // its result for the redaction hook to mask. // `collection` is passed bare: the callee qualifies it against - // `database_id` itself, so qualifying it here too would double-qualify. + // `database_id` itself, so qualifying it here too will double-qualify. crate::control::planner::redaction_refusal::refuse_unredactable_graph_collection( collection, database_id, @@ -124,6 +123,8 @@ pub struct TraverseRequest { pub depth: usize, pub edge_label: Option, pub direction: GraphDirection, + /// The session reads linearizably. + pub linearizable: bool, } pub async fn traverse( @@ -138,6 +139,7 @@ pub async fn traverse( depth, edge_label, direction, + linearizable, } = req; if start.is_empty() { return Err(ddl_err("42601", "missing FROM ''")); @@ -157,12 +159,18 @@ pub async fn traverse( crate::control::server::graph_dispatch::CrossCoreTraverseSubgraphParams { tenant_id, database_id, - collection: Some(collection), + // The walk plans read edges by the stored, database-qualified name. + collection: Some( + nodedb_types::QualifiedCollection::new(database_id, &collection) + .as_str() + .to_owned(), + ), start, edge_label, direction: dir, max_depth: depth, options: &GraphTraversalOptions::default(), + linearizable, }, ) .await @@ -185,6 +193,8 @@ pub struct NeighborsRequest { pub direction: GraphDirection, /// The session's active transaction, for read-your-own-writes overlay merge. pub txn_id: Option, + /// The session reads linearizably. + pub linearizable: bool, } pub async fn neighbors( @@ -199,6 +209,7 @@ pub async fn neighbors( edge_label, direction, txn_id, + linearizable, } = req; if node.is_empty() { return Err(ddl_err("42601", "missing OF ''")); @@ -207,6 +218,7 @@ pub async fn neighbors( let dir = to_engine_direction(direction); let tenant_id = identity.tenant_id; + let node_key = node.clone(); let plan = PhysicalPlan::Graph(GraphOp::Neighbors { collection: Some(nodedb_types::QualifiedCollection::new( database_id, @@ -218,17 +230,20 @@ pub async fn neighbors( rls_filters: Vec::new(), }); - match crate::control::server::broadcast::broadcast_to_all_cores_txn( + // The node's edges live on its key vShard: the read runs on that vShard's + // leader, confirmed there when linearizable. + match crate::control::server::graph_dispatch::read_on_key_owner( state, tenant_id, database_id, + &node_key, plan, - TraceId::ZERO, txn_id, + linearizable, ) .await { - Ok(resp) => Ok(payload_to_rows(&resp.payload)), + Ok(payload) => Ok(payload_to_rows(&payload)), Err(e) => Err(DdlError::from_error(&e)), } } @@ -247,6 +262,8 @@ pub struct ShortestPathRequest { pub dst: String, pub max_depth: usize, pub edge_label: Option, + /// The session reads linearizably. + pub linearizable: bool, } pub async fn shortest_path( @@ -261,6 +278,7 @@ pub async fn shortest_path( dst, max_depth, edge_label, + linearizable, } = req; if src.is_empty() || dst.is_empty() { return Err(ddl_err( @@ -277,11 +295,18 @@ pub async fn shortest_path( crate::control::server::graph_dispatch::CrossCoreShortestPathParams { tenant_id, database_id, - collection, + // The walk plans read edges by the stored, database-qualified name. + collection: Some( + nodedb_types::QualifiedCollection::new(database_id, &collection) + .as_str() + .to_owned(), + ), src, dst, edge_label, max_depth, + options: GraphTraversalOptions::default(), + linearizable, }, ) .await diff --git a/nodedb/src/control/server/shared/ddl/neutral/impersonation.rs b/nodedb/src/control/server/shared/ddl/neutral/impersonation.rs index 03a57094a..c50513fee 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/impersonation.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/impersonation.rs @@ -10,13 +10,12 @@ //! SHOW DELEGATIONS //! ``` //! -//! Ported from the pgwire `ddl::impersonation_ddl` handlers. All five mutate +//! All five handlers mutate //! or read the GLOBAL `state.impersonation` registry (keyed by user_id, not //! by connection) plus the audit log — not the current connection's identity //! — so they carry no per-connection state. The superuser / delegator gates, //! the token parsing (`AS` / `SCOPES` / `EXPIRES` / `REASON` extraction), the -//! registry calls, and the audit records are preserved verbatim; only the -//! result construction changed from pgwire `Response` / `PgWireError` to the +//! registry calls, and the audit records run here. The result is the //! protocol-neutral [`DdlResult`] / [`DdlError`]. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/inspect/grants.rs b/nodedb/src/control/server/shared/ddl/neutral/inspect/grants.rs index 621438822..bcb74d644 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/inspect/grants.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/inspect/grants.rs @@ -3,10 +3,8 @@ //! Protocol-neutral grant / permission introspection: SHOW GRANTS, //! SHOW PERMISSIONS. //! -//! Ported from the pgwire `ddl::inspect` handlers. The credential / -//! permission-store reads and the target-key rendering are preserved -//! verbatim; only the result construction changed from pgwire `Response` / -//! `QueryResponse` to the protocol-neutral `DdlResult` over `ShapedRows`. +//! The credential / permission-store reads and the target-key rendering run +//! here. The result is the protocol-neutral `DdlResult` over `ShapedRows`. use serde_json::{Map, Value as JsonValue}; @@ -81,7 +79,7 @@ pub fn show_permissions( on_collection: Option<&str>, for_grantee: Option<&str>, ) -> Result, DdlError> { - // Non-admins may only view their own grants. + // Non-admins can only view their own grants. if let Some(grantee) = for_grantee && grantee != identity.username && !identity.is_superuser @@ -102,7 +100,11 @@ pub fn show_permissions( let mut rows = Vec::new(); if let Some(collection) = on_collection { - let target = format!("collection:{}:{collection}", identity.tenant_id.as_u64()); + let target = crate::control::security::permission::collection_target( + database_id, + identity.tenant_id, + collection, + ); // Show owner row (only when collection is specified). if for_grantee.is_none() @@ -158,14 +160,12 @@ pub fn show_permissions( // All grants for a specific grantee (direct grants only, no inheritance walk). let grants = state.permissions.grants_for(grantee); for grant in &grants { - // Extract a human-readable target from the internal target key - // (e.g. "collection:1:users" → "users"). - let display_target = grant - .target - .rsplit(':') - .next() - .unwrap_or(&grant.target) - .to_string(); + // Show the object name of a scoped target ("collection:0:1:users" + // shows "users"), and any other target as stored. + let display_target = + crate::control::security::permission::parse_scoped_target(&grant.target) + .map_or(grant.target.as_str(), |t| t.name) + .to_string(); let mut row = Map::new(); row.insert( "grantee".to_string(), @@ -190,12 +190,10 @@ pub fn show_permissions( state.permissions.grants_for(&identity.username) }; for grant in &all_grants { - let display_target = grant - .target - .rsplit(':') - .next() - .unwrap_or(&grant.target) - .to_string(); + let display_target = + crate::control::security::permission::parse_scoped_target(&grant.target) + .map_or(grant.target.as_str(), |t| t.name) + .to_string(); let mut row = Map::new(); row.insert( "grantee".to_string(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/inspect/support.rs b/nodedb/src/control/server/shared/ddl/neutral/inspect/support.rs index 2d2b324e6..c6aba3458 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/inspect/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/inspect/support.rs @@ -6,9 +6,7 @@ use super::super::super::result::DdlError; /// Build a [`DdlError`] from an ANSI SQLSTATE code and a message. /// -/// Preserves the exact SQLSTATE / message the pgwire inspect handlers -/// produced (via `sqlstate_error`), so error parity stays byte-identical -/// after the migration off the pgwire router. +/// The SQLSTATE and message reach the client unchanged. pub(super) fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/inspect/tenants.rs b/nodedb/src/control/server/shared/ddl/neutral/inspect/tenants.rs index 8dad4a30e..d56a698da 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/inspect/tenants.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/inspect/tenants.rs @@ -3,10 +3,8 @@ //! Protocol-neutral tenant introspection: SHOW TENANTS, SHOW TENANT //! , SHOW TENANTS WITH NAME . //! -//! Ported from the pgwire `ddl::inspect` handlers. The tenant-set union -//! (catalog-registered tenants + tenants owning at least one user) and the -//! per-tenant usage reads are preserved verbatim; only the result -//! construction changed from pgwire `Response` / `QueryResponse` to the +//! The tenant-set union (catalog-registered tenants + tenants owning at least +//! one user) and the per-tenant usage reads run here. The result is the //! protocol-neutral `DdlResult` over `ShapedRows`. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/inspect/trigger_dlq.rs b/nodedb/src/control/server/shared/ddl/neutral/inspect/trigger_dlq.rs index 0a964091b..e698687b9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/inspect/trigger_dlq.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/inspect/trigger_dlq.rs @@ -157,5 +157,7 @@ fn requeue_take_sqlstate(error: &crate::event::trigger::RequeueTakeError) -> &'s E::NotFound { .. } => "42704", // The object exists but is not in a state this action accepts. E::AlreadyResolved { .. } => "55000", + // redb refused the write. The entry stays unresolved in the DLQ. + E::Persist { .. } => "58030", } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/inspect/users.rs b/nodedb/src/control/server/shared/ddl/neutral/inspect/users.rs index bfe7e0f56..534a7abf7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/inspect/users.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/inspect/users.rs @@ -3,10 +3,8 @@ //! Protocol-neutral user / role / session introspection: SHOW USERS, //! SHOW ROLES, SHOW SESSION. //! -//! Ported from the pgwire `ddl::inspect` handlers. The credential / role / -//! identity reads are preserved verbatim; only the result construction -//! changed from pgwire `Response` / `QueryResponse` to the protocol-neutral -//! `DdlResult` over `ShapedRows`. +//! The credential / role / identity reads run here. The result is the +//! protocol-neutral `DdlResult` over `ShapedRows`. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/inspect_audit.rs b/nodedb/src/control/server/shared/ddl/neutral/inspect_audit.rs index 7ebd69c64..37f6f7f5e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/inspect_audit.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/inspect_audit.rs @@ -3,11 +3,9 @@ //! Protocol-neutral audit-log SHOW commands: `SHOW AUDIT LOG`, //! `SHOW AUDIT WHERE`, `SHOW AUDIT IN DATABASE`, and `EXPORT AUDIT`. //! -//! Ported from the pgwire `ddl::inspect_audit` handlers. The catalog / -//! in-memory audit-log reads, ordering (most-recent-first), and the -//! catalog fall-through scan are preserved verbatim; only the result -//! construction changed from pgwire `Response` / `QueryResponse` to the -//! protocol-neutral `DdlResult` over `ShapedRows`. +//! The catalog / in-memory audit-log reads, ordering (most-recent-first), and +//! the catalog fall-through scan run here. The result is the protocol-neutral +//! `DdlResult` over `ShapedRows`. use serde_json::{Map, Value as JsonValue}; @@ -19,9 +17,7 @@ use super::super::result::{DdlError, DdlResult}; /// Build a [`DdlError`] from an ANSI SQLSTATE code and a message. /// -/// Preserves the exact SQLSTATE / message the pgwire audit handlers -/// produced (via `sqlstate_error`), so error parity stays byte-identical -/// after the migration off the pgwire router. +/// The SQLSTATE and message reach the client unchanged. fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs index af622d164..cf35f294f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_atomic/handlers.rs @@ -53,14 +53,15 @@ pub async fn kv_incr( let ttl_ms = parse_optional_ttl(&args[3..])?; let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); - let surrogate = state - .surrogate_assigner - .assign( - nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), - identity.tenant_id, - key.as_bytes(), - ) - .map_err(|e| DdlError::from_error(&e))?; + let surrogate = crate::control::server::surrogate_exchange::assign_surrogate_routed( + state, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), + identity.tenant_id, + key.as_bytes(), + crate::types::TraceId::ZERO, + ) + .await + .map_err(|e| DdlError::from_error(&e))?; let shape = counter_shape(state, identity, &collection, &key, KvCounterKind::Integer)?; let plan = PhysicalPlan::Kv(KvOp::Incr { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), @@ -120,14 +121,15 @@ pub async fn kv_incr_float( } let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); - let surrogate = state - .surrogate_assigner - .assign( - nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), - identity.tenant_id, - key.as_bytes(), - ) - .map_err(|e| DdlError::from_error(&e))?; + let surrogate = crate::control::server::surrogate_exchange::assign_surrogate_routed( + state, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), + identity.tenant_id, + key.as_bytes(), + crate::types::TraceId::ZERO, + ) + .await + .map_err(|e| DdlError::from_error(&e))?; let shape = counter_shape(state, identity, &collection, &key, KvCounterKind::Float)?; let plan = PhysicalPlan::Kv(KvOp::IncrFloat { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), @@ -212,14 +214,15 @@ pub async fn kv_cas( let new_value = unquote(&args[3]); let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); - let surrogate = state - .surrogate_assigner - .assign( - nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), - identity.tenant_id, - key.as_bytes(), - ) - .map_err(|e| DdlError::from_error(&e))?; + let surrogate = crate::control::server::surrogate_exchange::assign_surrogate_routed( + state, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), + identity.tenant_id, + key.as_bytes(), + crate::types::TraceId::ZERO, + ) + .await + .map_err(|e| DdlError::from_error(&e))?; let plan = PhysicalPlan::Kv(KvOp::Cas { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), key: key.as_bytes().to_vec(), @@ -265,14 +268,15 @@ pub async fn kv_getset( let new_value = unquote(&args[2]); let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection).vshard(); - let surrogate = state - .surrogate_assigner - .assign( - nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), - identity.tenant_id, - key.as_bytes(), - ) - .map_err(|e| DdlError::from_error(&e))?; + let surrogate = crate::control::server::surrogate_exchange::assign_surrogate_routed( + state, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, &collection), + identity.tenant_id, + key.as_bytes(), + crate::types::TraceId::ZERO, + ) + .await + .map_err(|e| DdlError::from_error(&e))?; let plan = PhysicalPlan::Kv(KvOp::GetSet { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, &collection), key: key.as_bytes().to_vec(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs index c039056b2..323c81820 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/kv_sorted_index/ddl.rs @@ -32,7 +32,7 @@ use super::parse::{ddl_err, parse_key_column, parse_sort_columns, parse_window_c /// Building the index reads every row of the collection into the order-stat /// tree, so the caller must be allowed to read it, and a read policy makes the /// index underivable: it is shared by every reader, so one principal's -/// filtered view would answer other principals' `TOPK` from rows that were +/// filtered view will answer other principals' `TOPK` from rows that were /// never indexed. const CREATE_WHAT: &str = "CREATE SORTED INDEX, which backfills the index from every row of the collection"; @@ -150,7 +150,8 @@ pub async fn create_sorted_index( collection: &collection, fields, }, - )?; + ) + .await?; // Ownership record backs authorization for a later DROP. crate::control::server::shared::ddl::owner::propose_owner( @@ -160,7 +161,8 @@ pub async fn create_sorted_index( tenant_id, &index_name, &identity.username, - )?; + ) + .await?; if in_transaction { let deferred = ddl_buffer::defer_effect(DeferredDdlEffect::SortedIndexRegister { @@ -248,14 +250,15 @@ pub async fn drop_sorted_index( drop_in_engine(state, &target, &index_name).await?; } - propose_delete_index_record(state, database_id, tenant_id, &index_name, &collection)?; + propose_delete_index_record(state, database_id, tenant_id, &index_name, &collection).await?; crate::control::server::shared::ddl::owner::propose_delete_owner( state, IndexKind::Sorted.owner_object_type(), database_id.as_u64(), tenant_id, &index_name, - )?; + ) + .await?; if in_transaction && !ddl_buffer::defer_effect(DeferredDdlEffect::SortedIndexDrop { diff --git a/nodedb/src/control/server/shared/ddl/neutral/last_value.rs b/nodedb/src/control/server/shared/ddl/neutral/last_value.rs index 4d48c6e8e..4583e3c94 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/last_value.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/last_value.rs @@ -48,6 +48,9 @@ pub async fn query_last_values( // Admitted at this request's own transport entry; this handler // has no session or peer information of its own. admission: crate::control::server::shared::ddl::user_dispatch::RequestAdmission::AlreadyAdmitted, + // No session reaches this handler, so its read takes the strong + // default (see `DmlTxnCtx::linearizable_reads`). + linearizable: true, }, ) .await @@ -117,6 +120,9 @@ pub async fn query_last_value( // Admitted at this request's own transport entry; this handler // has no session or peer information of its own. admission: crate::control::server::shared::ddl::user_dispatch::RequestAdmission::AlreadyAdmitted, + // No session reaches this handler, so its read takes the strong + // default (see `DmlTxnCtx::linearizable_reads`). + linearizable: true, }, ) .await @@ -124,7 +130,7 @@ pub async fn query_last_value( // MessagePack, as in `query_last_values` above — an absent series is // encoded as a null (decoding to `None`), which is a different fact from a - // payload that could not be read at all. + // payload that cannot be read at all. let entry: Option<(i64, f64)> = crate::data::executor::response_codec::decode_payload(&payload) .map_err(|e| DdlError::from_error_in_context("LAST_VALUE reply", &e))?; diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/analyze.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/analyze.rs index a0270d05a..c6f0f97db 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/analyze.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/analyze.rs @@ -136,13 +136,8 @@ pub async fn handle_analyze( .collect() }; - let local_rows = stats_rows.clone(); let entry = CatalogEntry::PutColumnStats(Box::new(stats_rows)); - super::super::replicate::propose_and_apply(state, &entry, || { - catalog - .put_column_stats_batch(&local_rows) - .map_err(|e| DdlError::from_error_in_context("failed to store column stats", &e)) - })?; + super::super::replicate::propose_and_apply_async(state, &entry).await?; state .dml_counter diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/auto_analyze.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/auto_analyze.rs index f4e40b619..041b51aac 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/auto_analyze.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/auto_analyze.rs @@ -18,7 +18,8 @@ use std::sync::{Arc, Mutex}; use nodedb_types::DatabaseId; -use crate::control::maintenance::{MaintenanceOutcome, with_budget}; +use crate::control::maintenance::MaintenanceOutcome; +use crate::control::maintenance::wrapper::with_budget_async; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; @@ -199,8 +200,8 @@ pub fn record_and_maybe_analyze( }; let identity = identity.clone(); let collection = collection.to_string(); - tokio::task::spawn_blocking(move || { - run_budgeted_analyze(&guard.owner, &identity, database_id, &collection); + tokio::spawn(async move { + run_budgeted_analyze(&guard.owner, &identity, database_id, &collection).await; // `guard` drops here and releases the slot. }); } @@ -234,21 +235,20 @@ fn last_analyzed_row_count( } /// Run ANALYZE under the database's maintenance budget and log the outcome. -/// -/// Runs on a `spawn_blocking` thread, so the budget lease stays inside one -/// synchronous scope and never crosses an await. -fn run_budgeted_analyze( +/// The budget lease covers the whole awaited ANALYZE. +async fn run_budgeted_analyze( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, collection: &str, ) { - let outcome = with_budget( + let outcome = with_budget_async( &state.maintenance_budget, database_id, ANALYZE_ESTIMATED_SECS, - || blocking_analyze(state, identity, database_id, collection), - ); + analyze(state, identity, database_id, collection), + ) + .await; match outcome { MaintenanceOutcome::Deferred => tracing::debug!( %collection, @@ -264,25 +264,17 @@ fn run_budgeted_analyze( } } -/// Drive the async `handle_analyze` to completion from a blocking thread. -fn blocking_analyze( +/// Run ANALYZE on `collection`. +async fn analyze( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, collection: &str, ) -> Result<(), DdlError> { - let handle = tokio::runtime::Handle::try_current().map_err(|error| { - DdlError::internal(format!("auto-ANALYZE needs a Tokio runtime: {error}")) - })?; // `handle_analyze` reads the collection name off the second whitespace // token and lowercases it, so the bare name is what it expects. let sql = format!("ANALYZE {collection}"); - handle.block_on(super::analyze::handle_analyze( - state, - identity, - &sql, - database_id, - ))?; + super::analyze::handle_analyze(state, identity, &sql, database_id).await?; Ok(()) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/distributed.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/distributed.rs index da220e665..20f4d0084 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/distributed.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/distributed.rs @@ -2,9 +2,9 @@ //! Distributed maintenance operations: ANALYZE/COMPACT/REINDEX across shards. //! -//! In cluster mode, these operations are dispatched to each shard leader -//! independently. Results are merged on the coordinator node. -//! In single-node mode, they execute directly on the local Data Plane. +//! These operations are dispatched to each shard leader independently. +//! Results are merged on the coordinator node. On a one-node cluster every +//! leader is this node, so they run on the local Data Plane. use crate::bridge::envelope::{PhysicalPlan, Priority, Request}; use crate::control::state::SharedState; @@ -16,14 +16,14 @@ use nodedb_physical::physical_plan::MetaOp; /// /// Deliberately not `tuning.network.default_deadline_secs`: that bounds a /// client statement, and a maintenance pass rewrites whole segments at -/// `Priority::Background`. Bounding it by a query deadline would abandon the +/// `Priority::Background`. Bounding it by a query deadline will abandon the /// rewrite partway on every collection large enough to need it. const MAINTENANCE_DEADLINE: std::time::Duration = std::time::Duration::from_secs(300); /// Dispatch a maintenance operation (COMPACT/REINDEX) to all Data Plane cores. /// /// In single-node: dispatches to core 0 (the only core in most test configs). -/// In cluster: would dispatch to each shard leader. Currently dispatches locally. +/// In cluster: will dispatch to each shard leader. Currently dispatches locally. pub fn dispatch_maintenance_to_all_cores( state: &SharedState, tenant_id: TenantId, @@ -48,6 +48,7 @@ pub fn dispatch_maintenance_to_all_cores( txn_id: None, wal_lsn: None, resolved_now_ms: None, + commit_hlc: None, admission: crate::bridge::envelope::Admission::Exempt( crate::bridge::envelope::ExemptReason::AlreadyOrdered, ), @@ -69,7 +70,7 @@ pub fn dispatch_maintenance_to_all_cores( /// The coordinator calls this to merge them: /// - row_count: sum across shards /// - null_count: sum across shards -/// - distinct_count: max across shards (HLL merge would be more accurate) +/// - distinct_count: max across shards (HLL merge will be more accurate) /// - min_value: min of all shard mins /// - max_value: max of all shard maxes pub fn merge_column_stats( @@ -85,7 +86,7 @@ pub fn merge_column_stats( for shard in &shards[1..] { merged.row_count += shard.row_count; merged.null_count += shard.null_count; - // Approximate: take max distinct count (HLL merge would be better). + // Approximate: take max distinct count (HLL merge will be better). merged.distinct_count = merged.distinct_count.max(shard.distinct_count); // Merge min/max. diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/storage_info.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/storage_info.rs index 1eb210070..22c45f9fa 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/storage_info.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/storage_info.rs @@ -2,11 +2,9 @@ //! `SHOW STORAGE FOR collection` and `SHOW COMPACTION STATUS` handlers. //! -//! Ported from the pgwire maintenance handlers. Both result sets are all-text -//! columns (`text_field`), so the protocol-neutral [`ShapedRows`] carries -//! `DdlColType::Text` per column and each cell as its `String` form — the same -//! bytes `DataRowEncoder::encode_field(&str)` produced, keeping the -//! RowDescription and DataRow output byte-identical. +//! Both result sets are all-text columns, so the protocol-neutral +//! [`ShapedRows`] carries `DdlColType::Text` per column and each cell as its +//! `String` form. use nodedb_types::DatabaseId; diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/support.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/support.rs index b28a57569..c9dfd561e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/support.rs @@ -6,9 +6,7 @@ use super::super::super::result::DdlError; /// Build a [`DdlError`] from an ANSI SQLSTATE code and a message. /// -/// Preserves the exact SQLSTATE / message the pgwire maintenance handlers -/// produced (via `ErrorInfo::new` / `sqlstate_error`), so error parity stays -/// byte-identical after the migration off the pgwire router. +/// The SQLSTATE and message reach the client unchanged. pub(super) fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs index e63e2e2f7..d94cf23e8 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index.rs @@ -8,12 +8,10 @@ //! //! `ALTER VECTOR INDEX ... SET (...)` lives in [`super::vector_index_set`]. //! -//! Ported from the pgwire maintenance handlers. The `SHOW` result set is -//! all-text columns (`text_field`), so the protocol-neutral [`ShapedRows`] -//! carries `DdlColType::Text` per column and each cell as its `String` form — -//! the same bytes `DataRowEncoder::encode_field(&str)` produced. The Data Plane -//! dispatch paths (`dispatch_to_data_plane`, plan construction, ordering) are -//! preserved verbatim. +//! The `SHOW` result set is all-text columns, so the protocol-neutral +//! [`ShapedRows`] carries `DdlColType::Text` per column and each cell as its +//! `String` form. The Data Plane dispatch paths (`dispatch_to_data_plane`, +//! plan construction, ordering) run here. use nodedb_sql::parser::preprocess::lex::find_ascii_case_insensitive; use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs index b743cf1d7..18b6df554 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/maintenance/vector_index_set.rs @@ -74,17 +74,7 @@ pub async fn handle_alter_vector_index_set( let merged = merge(current, &overlay); validate(&merged, &overlay)?; - let outcome = super::super::vector_replicate::propose_put_params(state, &merged)?; - - // Single node: no applier runs, so post-apply never fires. Run the - // per-node install the post-apply lane runs everywhere else. - if outcome.needs_local_apply() { - let shared = state - .self_arc() - .map_err(|e| DdlError::from_error_in_context("install vector index params", &e))?; - crate::control::catalog_entry::post_apply::install_vector_index_params(merged, shared) - .await; - } + super::super::vector_replicate::propose_put_params(state, &merged).await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, @@ -151,7 +141,7 @@ fn merge(stored: StoredVectorIndexParams, overlay: &ParamOverlay) -> StoredVecto /// Refuse a merged row the engine cannot build, before it replicates. /// /// Apply writes the row on every node and cannot reject, so a row that fails -/// these rules would reach every node and break each one's build. CREATE +/// these rules will reach every node and break each one's build. CREATE /// enforces the same three rules on its own row. /// /// The first rule reads the overlay, not the merged row: an index moving from @@ -226,7 +216,7 @@ fn parse_set_clause(sql: &str) -> Result { for pair in inner.split(',') { let pair = pair.trim(); // A list item with no `=` must not be skipped — silently dropping a - // typo'd item would report success for the ones around it. + // typo'd item will report success for the ones around it. let Some((key, val)) = pair.split_once('=') else { return Err(ddl_err( "42601", @@ -253,7 +243,7 @@ fn parse_set_clause(sql: &str) -> Result { "ivf_nprobe" => overlay.ivf_nprobe = uint(val, "ivf_nprobe")?, // `m0` is derived as `2 * m` by every path that installs an index — // CREATE, the boot seed, and WAL replay alike. A row cannot carry a - // different ratio, so honouring one here would hold only until the + // different ratio, so honouring one here will hold only until the // next restart, on the one node that ran the statement. "m0" => { return Err(ddl_err( @@ -312,6 +302,7 @@ mod tests { pq_m: 0, ivf_cells: 0, ivf_nprobe: 0, + modification_hlc: nodedb_types::Hlc::ZERO, } } @@ -345,7 +336,7 @@ mod tests { assert_eq!(merged("index_type = 'ivf_pq', ivf_cells = 64").dim, 384); } - /// The row keys the catalog write, so a merge that moved it would land the + /// The row keys the catalog write, so a merge that moved it will land the /// altered parameters on a different index. #[test] fn the_identity_fields_survive_every_clause() { @@ -418,7 +409,7 @@ mod tests { assert!(parse("m = 0").is_err()); } - /// `m0` cannot be made durable, so accepting it would apply the ratio on + /// `m0` cannot be made durable, so accepting it will apply the ratio on /// one node until its next restart. #[test] fn m0_is_rejected_rather_than_dropped() { diff --git a/nodedb/src/control/server/shared/ddl/neutral/match_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/match_ops.rs index e8229b71d..ead880675 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/match_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/match_ops.rs @@ -6,7 +6,7 @@ //! The handler builds [`DdlResult`](super::super::result::DdlResult) directly //! and carries no pgwire types. It is dispatched from the neutral router's typed //! `GraphStmt::MatchQuery` arm, but re-parses the raw `sql` with the graph -//! pattern compiler (as the pgwire handler did) to build the physical plan. +//! pattern compiler to build the physical plan. use serde_json::{Map, Value as JsonValue}; @@ -15,7 +15,7 @@ use crate::control::server::graph_dispatch; use crate::control::server::response_shape::types::ShapedRows; use crate::control::state::SharedState; use crate::data::executor::response_codec; -use crate::types::{DatabaseId, TraceId, TxnId}; +use crate::types::{DatabaseId, TraceId}; use nodedb_physical::physical_plan::GraphOp; use super::super::result::{DdlError, DdlResult}; @@ -27,14 +27,14 @@ use super::refuse_gate::RefusingReadGate; /// time. const MATCH_WHAT: &str = "a pattern match, which returns bindings over graph topology"; -/// Returned when a MATCH could not be fully resolved within its expansion +/// Returned when a MATCH cannot be fully resolved within its expansion /// budget — either the cross-shard hop rounds or the variable-length paging /// rounds were exhausted with work still pending, or a single-node /// variable-length expansion hit its hard cap with no coordinator to drain it. -/// The result set would be INCOMPLETE, so it is surfaced as a fail-closed error +/// The result set will be INCOMPLETE, so it is surfaced as a fail-closed error /// (SQLSTATE 54001, `program_limit_exceeded`) rather than silently returning a /// truncated result the client cannot distinguish from a complete one. -const MATCH_INCOMPLETE_MESSAGE: &str = "MATCH result incomplete: the pattern exceeded the expansion budget; \ +pub(crate) const MATCH_INCOMPLETE_MESSAGE: &str = "MATCH result incomplete: the pattern exceeded the expansion budget; \ narrow the pattern or its variable-length `*min..max` bound"; /// Handle a MATCH query. @@ -46,7 +46,7 @@ pub async fn match_query( identity: &AuthenticatedIdentity, database_id: DatabaseId, sql: &str, - txn_id: Option, + read: graph_dispatch::GraphRead, ) -> Result, DdlError> { // Parse the MATCH query. let query = crate::engine::graph::pattern::compiler::parse(sql) @@ -57,13 +57,13 @@ pub async fn match_query( // resolved here, on the pattern's own scope. // // A pattern scoped with `IN ''` asks the narrow question about - // that collection. An unscoped pattern may walk any collection the tenant - // holds, so the set it could touch is the set it must be granted: every + // that collection. An unscoped pattern can walk any collection the tenant + // holds, so the set it can touch is the set it must be granted: every // active collection of the database, failing closed on the first denial. - // Requiring an explicit `IN` instead would refuse the unscoped form for + // Requiring an explicit `IN` instead will refuse the unscoped form for // every caller, including one already granted everything the pattern can - // reach; this keeps that caller's behavior exactly as it was and refuses - // only the caller who would otherwise walk a collection it cannot read. + // reach; this keeps that caller's behavior and refuses + // only the caller who will otherwise walk a collection it cannot read. // The RLS half mirrors it: the narrow question when the pattern names a // collection, the tenant-wide one when it names none. let gate = RefusingReadGate::for_request(state, identity, database_id); @@ -144,7 +144,7 @@ pub async fn match_query( // Both dispatch shapes below send the same query bytes, so the refusal is // applied here, once, before either runs. `query.collection` is already // parsed, so the scoped check is used directly rather than re-decoding - // `query_bytes` back into a `MatchQuery` just to read it again. The gate's + // `query_bytes` back into a `MatchQuery` only to read it again. The gate's // own scope is reused so the redaction refusal, the RBAC check, and the RLS // refusal all resolve against the same principal. crate::control::planner::redaction_refusal::refuse_unredactable_graph_match_scoped( @@ -173,7 +173,7 @@ pub async fn match_query( database_id, plan, TraceId::ZERO, - txn_id, + read, ) .await { @@ -207,7 +207,7 @@ pub async fn match_query( database_id, query_bytes, deadline_ms, - txn_id, + read, ) .await { diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/create.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/create.rs index b72f90b88..ed8fc3478 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/create.rs @@ -3,12 +3,11 @@ //! Protocol-neutral `CREATE MATERIALIZED VIEW` handler — replicates through the //! metadata raft group via `CatalogEntry::PutMaterializedView`. //! -//! Ported from the pgwire `ddl::materialized_view::create` handler. The catalog -//! path (`propose_and_apply` for the view definition, then `propose_and_apply` -//! for the target collection, then `dispatch_register_from_stored`), the -//! duplicate / source-existence checks, and the target-collection descriptor are -//! preserved verbatim; only the result construction changed from pgwire -//! `Response` / `PgWireError` to the protocol-neutral [`DdlResult`] / [`DdlError`]. +//! The catalog path (`propose_and_apply` for the view definition, then +//! `propose_and_apply` for the target collection, then +//! `register_proposed_collection`), the duplicate / source-existence checks, +//! and the target-collection descriptor run here. The result is the +//! protocol-neutral [`DdlResult`] / [`DdlError`]. use nodedb_types::DatabaseId; @@ -16,7 +15,7 @@ use crate::control::security::catalog::{StoredCollection, StoredMaterializedView use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; -use super::super::super::catalog::propose_and_apply; +use super::super::super::catalog::propose_and_apply_async; use super::super::super::result::{DdlError, DdlResult}; fn err(sqlstate: &str, message: String) -> DdlError { @@ -49,20 +48,6 @@ pub async fn create_materialized_view( return create_streaming_mv(state, identity, database_id, &name, &query_sql).await; } - // Metadata Raft serializes clustered DDL. Without it, hold an exclusive - // name lifecycle guard through definition+target creation and Data Plane - // registration so DROP or another CREATE cannot interleave. - let _local_lifecycle = if state.metadata_raft.get().is_none() { - Some( - state - .quiesce - .acquire_lifecycle(database_id.as_u64(), tenant_id.as_u64(), &name) - .await, - ) - } else { - None - }; - // Validate source collection exists. { let catalog = state.credentials.catalog(); @@ -76,7 +61,7 @@ pub async fn create_materialized_view( } } - // A catalog-read fault must abort the CREATE: proceeding could adopt a + // A catalog-read fault must abort the CREATE: proceeding can adopt a // same-name target over an object whose existence check transiently // failed. if catalog @@ -123,11 +108,11 @@ pub async fn create_materialized_view( // it up on its next tick. let entry = crate::control::catalog_entry::CatalogEntry::PutMaterializedView(Box::new(view.clone())); - propose_and_apply(state, &entry)?; + propose_and_apply_async(state, &entry).await?; // Create the implementation-owned target collection so REFRESH can insert // into it and clients can SELECT from it. The pre-check rejects every - // same-name collection; DROP may therefore purge this target without ever + // same-name collection; DROP can therefore purge this target without ever // deleting a user-owned collection. let target = StoredCollection { tenant_id: tenant_id.as_u64(), @@ -139,6 +124,7 @@ pub async fn create_materialized_view( constraint_version: 0, crdt_signing_required: false, modification_hlc: nodedb_types::Hlc::ZERO, + incarnation: nodedb_types::Hlc::ZERO, fields: Vec::new(), field_defs: Vec::new(), event_defs: Vec::new(), @@ -175,8 +161,8 @@ pub async fn create_materialized_view( }; let coll_entry = crate::control::catalog_entry::CatalogEntry::PutCollection(Box::new(target.clone())); - propose_and_apply(state, &coll_entry)?; - super::super::collection::dispatch_register_from_stored(state, &target) + let outcome = propose_and_apply_async(state, &coll_entry).await?; + super::super::collection::register_proposed_collection(state, outcome, &target) .await .map_err(|e| DdlError::from_error(&e))?; @@ -195,10 +181,9 @@ pub async fn create_materialized_view( /// `CREATE MATERIALIZED VIEW [ON ] STREAMING AS SELECT ... FROM ...` /// -/// Ported from the deleted pgwire `ddl::streaming_mv::create` handler: the -/// tenant-admin gate, source-stream existence check, duplicate guard, catalog -/// persist, in-memory registration, and buffer backfill are preserved; only the -/// result / error types changed to the protocol-neutral [`DdlResult`] / +/// The handler runs the tenant-admin gate, source-stream existence check, +/// duplicate guard, catalog persist, in-memory registration, and buffer +/// backfill. The result / error types are the protocol-neutral [`DdlResult`] / /// [`DdlError`]. The source is the change stream named in the query's FROM /// clause, not the `ON` lineage collection. async fn create_streaming_mv( @@ -273,25 +258,11 @@ async fn create_streaming_mv( created_at: now, }; - let entry = crate::control::catalog_entry::CatalogEntry::PutStreamingMaterializedView( - Box::new(def.clone()), - ); - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + let entry = + crate::control::catalog_entry::CatalogEntry::PutStreamingMaterializedView(Box::new(def)); + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|error| DdlError::from_error_in_context("metadata propose", &error))?; - crate::control::catalog_entry::apply::local::apply_locally_if_needed(state, &entry, outcome); - if outcome.needs_local_apply() { - state.permissions.install_replicated_owner( - &crate::control::security::catalog::StoredOwner { - database_id: database_id.as_u64(), - object_type: crate::control::security::catalog::auth_types::object_type::STREAMING_MATERIALIZED_VIEW - .to_string(), - object_name: name.to_string(), - tenant_id, - owner_username: identity.username.clone(), - }, - ); - state.mv_registry.register(def); - } // Backfill: replay events already in the source stream's buffer so the MV // bootstraps with historical data instead of only future events. diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/drop.rs index 6b3daf6df..77a0b7fe6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/drop.rs @@ -2,10 +2,8 @@ //! Protocol-neutral `DROP MATERIALIZED VIEW [IF EXISTS]` handler. //! -//! Ported from the pgwire `ddl::materialized_view::drop` handler. The DIRECT -//! catalog path (`propose_catalog_entry` for the compound -//! `DeleteMaterializedView` definition+target deletion, with synchronous local -//! apply/reclaim when metadata Raft is absent), the token-based name / IF EXISTS +//! The DIRECT catalog path (`propose_catalog_entry` for the compound +//! `DeleteMaterializedView` definition+target deletion), the token-based name / IF EXISTS //! extraction, and the pre-check existence gate are shared by every protocol. use crate::control::security::identity::AuthenticatedIdentity; @@ -31,7 +29,7 @@ pub fn materialized_view_exists( state.mv_registry.get_def(database_id, tid, name).is_some() } -pub fn drop_materialized_view( +pub async fn drop_materialized_view( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -69,22 +67,9 @@ pub fn drop_materialized_view( tenant_id: tenant_id.as_u64(), name: name.clone(), }; - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|error| DdlError::from_error_in_context("metadata propose", &error))?; - crate::control::catalog_entry::apply::local::apply_locally_if_needed( - state, &entry, outcome, - ); - if outcome.needs_local_apply() { - state - .mv_registry - .unregister(database_id, tenant_id.as_u64(), &name); - state.permissions.install_replicated_remove_owner( - crate::control::security::catalog::auth_types::object_type::STREAMING_MATERIALIZED_VIEW, - database_id.as_u64(), - tenant_id.as_u64(), - &name, - ); - } tracing::info!(view = name, "streaming materialized view dropped"); return Ok(vec![DdlResult::Status { command: "DROP MATERIALIZED VIEW".to_string(), @@ -119,59 +104,15 @@ pub fn drop_materialized_view( database_id: database_id.as_u64(), tenant_id: tenant_id.as_u64(), name: name.clone(), + // Frozen by the proposer's stamp. + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }; - let mut local_lifecycle = if state.metadata_raft.get().is_none() { - Some( - state - .quiesce - .try_acquire_lifecycle(database_id.as_u64(), tenant_id.as_u64(), &name) - .ok_or_else(|| { - err( - "55006", - format!("materialized view '{name}' lifecycle is busy"), - ) - })?, - ) - } else { - None - }; - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + // The apply deletes the definition and its target, and reclaims the + // target's storage in post-apply on every node, this one included. + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|error| DdlError::from_error_in_context("metadata propose", &error))?; - if outcome.needs_local_apply() { - // No metadata Raft is active, so apply the same compound catalog - // deletion locally and synchronously reclaim the implementation-owned - // target collection. A reclaim failure after catalog deletion is - // fatal: continuing would permit a same-name CREATE over stale rows. - crate::control::catalog_entry::apply::apply_to(&entry, state.credentials.catalog()) - .map_err(|e| DdlError::from_error_in_context("catalog apply", &e))?; - let purge_lsn = state.wal.next_lsn().as_u64(); - let purge_result = tokio::task::block_in_place(|| { - tokio::runtime::Handle::current().block_on(async { - crate::control::server::shared::ddl::neutral::collection::purge::hard_purge_collection( - state, - database_id.as_u64(), - tenant_id.as_u64(), - &name, - purge_lsn, - local_lifecycle.is_some(), - ) - .await - }) - }); - if let Err(failure) = purge_result { - // Disarm only when a durable retry record owns the drain; a - // no-retry failure releases the hold via the guard's unwind Drop. - if failure.retry_queued - && let Some(guard) = local_lifecycle.take() - { - guard.disarm(); - } - panic!( - "local materialized-view target reclaim failed: {}", - failure.error - ); - } - } tracing::info!(view = name, "materialized view dropped"); diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs index 1dfc83369..e64332319 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/refresh.rs @@ -17,11 +17,9 @@ //! specialised refresh opcode; every engine feature reachable by a normal //! `SELECT` is reachable by refresh. //! -//! Ported from the pgwire `ddl::materialized_view::refresh` handler. The plan / -//! Data-Plane dispatch path (`plan_sql`, `wal_append_if_write`, +//! The plan / Data-Plane dispatch path (`plan_sql`, `wal_append_if_write`, //! `dispatch_to_data_plane`), the scan-row normalisation, and the INSERT -//! synthesis are preserved verbatim; only the result construction changed from -//! pgwire `Response` / `PgWireError` to the protocol-neutral [`DdlResult`] / +//! synthesis run here. The result is the protocol-neutral [`DdlResult`] / //! [`DdlError`]. use nodedb_types::DatabaseId; @@ -116,7 +114,7 @@ pub async fn refresh_materialized_view( } /// Plan and execute a `SELECT` via the standard SQL pipeline, collect -/// the result rows as `serde_json::Map` objects. Response payloads may +/// the result rows as `serde_json::Map` objects. Response payloads can /// come back as wrapped scan rows (`{id, data: {...}}`) or as flat /// aggregate/join rows — both are normalised to the logical row map. async fn execute_select( @@ -306,38 +304,15 @@ async fn dispatch_sql( checked } }; - // The record's outcome-floor window opens before the append and - // closes from the task's outcome inside the funnel. - let owner = crate::control::server::dispatch_utils::RecordOwner { - tenant_id: identity.tenant_id, - database_id: checked.database_id(), - vshard_id: checked.vshard_id(), - }; - let minted = - crate::control::server::dispatch_utils::MintedRecords::open(&state.outcome_floor); - if let Err(e) = minted.append_plan( - &state.wal, - owner, - checked.plan(), - // The refresh is dispatched as a client write. - crate::event::EventSource::User, - ) { - // Any record appended before the error never reaches a core. - minted - .cancel(&state.wal, owner, 0) - .await - .map_err(|c| err(sqlstate::IO_ERROR, format!("cancel refresh record: {c}")))?; - return Err(err(sqlstate::IO_ERROR, format!("wal append: {e}"))); - } - let response = - crate::control::server::dispatch_utils::dispatch_authorized_minted_to_data_plane( - state, - checked, - TraceId::ZERO, - minted, - ) - .await - .map_err(|e| err(sqlstate::CONNECTION_FAILURE, format!("dispatch: {e}")))?; + // The refresh write applies through its replicated entry, as every + // client write does. + let response = crate::control::server::dispatch_utils::dispatch_authorized_durable_write( + state, + checked, + TraceId::ZERO, + ) + .await + .map_err(|e| err(sqlstate::CONNECTION_FAILURE, format!("dispatch: {e}")))?; require_ok_response(&response)?; } Ok(()) diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/show.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/show.rs index efbb38f3c..db37c204e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/show.rs @@ -2,12 +2,10 @@ //! Protocol-neutral `SHOW MATERIALIZED VIEWS [FOR ]` handler. //! -//! Ported from the pgwire `ddl::materialized_view::show` handler. The catalog -//! read, the optional `FOR ` filter, and the exact column set (all five -//! columns `text`) are preserved verbatim; only the result construction changed -//! from pgwire `Response` / `QueryResponse` to the protocol-neutral +//! The catalog read, the optional `FOR ` filter, and the exact column +//! set (all five columns `text`) run here. The result is the protocol-neutral //! [`DdlResult::Rows`] over [`ShapedRows`]. All columns are `text`, so -//! `ShapedRows::text_types(5)` reproduces the RowDescription byte-identically. +//! `ShapedRows::text_types(5)` gives the column types. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/streaming_parse.rs b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/streaming_parse.rs index f50433464..d2632d453 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/materialized_view/streaming_parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/materialized_view/streaming_parse.rs @@ -2,13 +2,11 @@ //! Parser for the streaming variant of `CREATE MATERIALIZED VIEW ... STREAMING`. //! -//! Ported from the deleted pgwire `ddl::streaming_mv::create` parser. The -//! extraction logic (source stream from the `FROM` clause, `GROUP BY` columns, -//! `WHERE` filter, and the `COUNT/SUM/MIN/MAX/AVG` aggregate list) is preserved -//! verbatim; the only behavioural change is that it operates on the already-split -//! query body (`query_sql`, the text after ` AS `) plus the view `name` that the -//! DDL parser extracted, rather than re-parsing the full statement string. Parse -//! failures surface as protocol-neutral [`DdlError`] with SQLSTATE `42601`. +//! The extraction logic (source stream from the `FROM` clause, `GROUP BY` +//! columns, `WHERE` filter, and the `COUNT/SUM/MIN/MAX/AVG` aggregate list) +//! operates on the already-split query body (`query_sql`, the text after +//! ` AS `) plus the view `name` that the DDL parser extracted. Parse failures +//! surface as protocol-neutral [`DdlError`] with SQLSTATE `42601`. use crate::control::server::shared::ddl::sql_parse::parse_ident_token; use crate::event::streaming_mv::types::{AggDef, AggFunction}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/metering_ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/metering_ddl.rs index fcd4e7d0c..63d5d5121 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/metering_ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/metering_ddl.rs @@ -9,11 +9,9 @@ //! SHOW QUOTA FOR AUTH USER 'user_42' //! ``` //! -//! Ported from the pgwire `ddl::metering_ddl` handlers. The usage-store / -//! quota-manager / tenant-usage reads, ordering, and the superuser gates are -//! preserved verbatim; only the result construction changed from pgwire -//! `Response` / `QueryResponse` to the protocol-neutral `DdlResult` over -//! `ShapedRows`. +//! The usage-store / quota-manager / tenant-usage reads, ordering, and the +//! superuser gates run here. The result is the protocol-neutral `DdlResult` +//! over `ShapedRows`. use serde_json::{Map, Value as JsonValue}; @@ -25,9 +23,7 @@ use super::super::result::{DdlError, DdlResult}; /// Build a [`DdlError`] from an ANSI SQLSTATE code and a message. /// -/// Preserves the exact SQLSTATE / message the pgwire metering handlers -/// produced (via `sqlstate_error`), so error parity stays byte-identical -/// after the migration off the pgwire router. +/// The SQLSTATE and message reach the client unchanged. fn ddl_err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/observability.rs b/nodedb/src/control/server/shared/ddl/neutral/observability.rs index 94cfa2a4f..e5e9e47f2 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/observability.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/observability.rs @@ -14,10 +14,8 @@ //! `SHOW MEMORY` reports per-engine memory budgets and current //! utilisation from `nodedb_mem::MemoryGovernor`. //! -//! Ported from the pgwire `ddl::observability` handlers. The metric source -//! reads, ordering, and the tenant-admin gate are preserved verbatim; only the -//! result construction changed from pgwire `Response` / `QueryResponse` to the -//! protocol-neutral `DdlResult` over `ShapedRows`. +//! The metric source reads, ordering, and the tenant-admin gate run here. The +//! result is the protocol-neutral `DdlResult` over `ShapedRows`. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/oidc.rs b/nodedb/src/control/server/shared/ddl/neutral/oidc.rs index 192188157..e5fc10f46 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/oidc.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/oidc.rs @@ -2,11 +2,9 @@ //! Protocol-neutral OIDC provider DDL — CREATE / ALTER / DROP / SHOW. //! -//! Ported from the pgwire `ddl::oidc` handlers. All non-return logic -//! (superuser gate + its denial audit, empty-field validation, duplicate +//! The superuser gate + its denial audit, empty-field validation, duplicate //! name / issuer pre-checks, `StoredClaimMappingRule` build, catalog proposes, -//! local single-node fallbacks, `audit_record`) is preserved verbatim; only the -//! result construction changed from pgwire `Response` / `PgWireError` to the +//! local single-node fallbacks, and `audit_record` run here. The result is the //! protocol-neutral [`DdlResult`] / [`DdlError`]. //! //! OIDC providers are system-scoped (superuser-only) and backed by the @@ -15,7 +13,7 @@ use serde_json::{Map, Value as JsonValue}; use crate::control::catalog_entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::security::audit::AuditEvent; use crate::control::security::catalog::StoredOidcProvider; use crate::control::security::catalog::oidc_providers::StoredClaimMappingRule; @@ -34,10 +32,8 @@ fn status(command: &str) -> Vec { }] } -/// Superuser gate, folded in verbatim from the pgwire `require_superuser` -/// helper: on denial it emits `AuditEvent::PermissionDenied` (database-less -/// scope, matching the `None` `db_id` the pgwire handlers passed) and returns -/// SQLSTATE 42501, preserving both the side effect and the wire error. +/// Superuser gate: on denial it emits `AuditEvent::PermissionDenied` +/// (database-less scope, a `None` `db_id`) and returns SQLSTATE 42501. fn require_superuser( state: &SharedState, identity: &AuthenticatedIdentity, @@ -60,7 +56,7 @@ fn require_superuser( } } -/// Whether two providers would make the issuer route ambiguous. +/// Whether two providers will make the issuer route ambiguous. fn has_ambiguous_issuer_route(existing_audience: Option<&str>, audience: Option<&str>) -> bool { let existing_audience = existing_audience.filter(|value| !value.is_empty()); let audience = audience.filter(|value| !value.is_empty()); @@ -71,7 +67,7 @@ fn has_ambiguous_issuer_route(existing_audience: Option<&str>, audience: Option< } /// Refuse a claim mapping that grants superuser, or a role that is neither -/// built in nor defined in the provider's tenant: a login mapped to it would +/// built in nor defined in the provider's tenant: a login mapped to it will /// hold nothing. A role dropped after this check refuses the login instead. fn validate_claim_mapping_roles( state: &SharedState, @@ -115,7 +111,7 @@ pub struct CreateOidcProviderParams<'a> { } /// Handle `CREATE OIDC PROVIDER ISSUER '' JWKS_URI '' ...`. -pub fn create_oidc_provider( +pub async fn create_oidc_provider( state: &SharedState, identity: &AuthenticatedIdentity, params: CreateOidcProviderParams<'_>, @@ -167,7 +163,7 @@ pub fn create_oidc_provider( } // A route is `(issuer, audience)`. An absent or empty audience makes an - // issuer route ambiguous, while distinct non-empty audiences may share it. + // issuer route ambiguous, while distinct non-empty audiences can share it. match catalog.list_oidc_providers() { Ok(providers) => { if providers.iter().any(|p| { @@ -207,14 +203,10 @@ pub fn create_oidc_provider( created_at_lsn: 0, }; - let entry = CatalogEntry::PutOidcProvider(Box::new(provider.clone())); - let outcome = propose_catalog_entry(state, &entry) + let entry = CatalogEntry::PutOidcProvider(Box::new(provider)); + propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - catalog - .put_oidc_provider(&provider) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } state.audit_record( AuditEvent::OidcProviderChanged, @@ -229,7 +221,7 @@ pub fn create_oidc_provider( /// Handle `ALTER OIDC PROVIDER SET CLAIM MAPPING WHEN ...`. /// /// Replaces the entire claim-mapping list for the named provider. -pub fn alter_oidc_provider_claim_mapping( +pub async fn alter_oidc_provider_claim_mapping( state: &SharedState, identity: &AuthenticatedIdentity, name: &str, @@ -259,13 +251,9 @@ pub fn alter_oidc_provider_claim_mapping( provider.claim_mapping = stored_mappings; let entry = CatalogEntry::PutOidcProvider(Box::new(provider.clone())); - let outcome = propose_catalog_entry(state, &entry) + propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - catalog - .put_oidc_provider(&provider) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } state.audit_record( AuditEvent::OidcProviderChanged, @@ -281,7 +269,7 @@ pub fn alter_oidc_provider_claim_mapping( } /// Handle `DROP OIDC PROVIDER [IF EXISTS] `. -pub fn drop_oidc_provider( +pub async fn drop_oidc_provider( state: &SharedState, identity: &AuthenticatedIdentity, name: &str, @@ -308,13 +296,9 @@ pub fn drop_oidc_provider( let entry = CatalogEntry::DeleteOidcProvider { name: name.to_string(), }; - let outcome = propose_catalog_entry(state, &entry) + propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - catalog - .delete_oidc_provider(name) - .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; - } state.audit_record( AuditEvent::OidcProviderChanged, diff --git a/nodedb/src/control/server/shared/ddl/neutral/org_ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/org_ddl.rs index 85b9b9b7f..ea2970370 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/org_ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/org_ddl.rs @@ -10,10 +10,9 @@ //! SHOW MEMBERS OF ORG 'acme' //! ``` //! -//! Ported from the pgwire `ddl::org_ddl` handlers. The superuser gate, org-store -//! mutations, and `audit_record` side effects are preserved verbatim; only the -//! result construction changed from pgwire `Response` / `QueryResponse` / `Tag` -//! to the protocol-neutral [`DdlResult`] over [`ShapedRows`]. +//! The superuser gate, org-store mutations, and `audit_record` side effects +//! run here. The result is the protocol-neutral [`DdlResult`] over +//! [`ShapedRows`]. use serde_json::{Map, Value as JsonValue}; @@ -23,8 +22,7 @@ use crate::control::state::SharedState; use super::super::result::{DdlError, DdlResult}; -/// Construct a [`DdlError`], preserving the exact SQLSTATE codes and messages -/// the pgwire handlers produced. +/// Construct a [`DdlError`] from a SQLSTATE code and a message. fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/period_lock.rs b/nodedb/src/control/server/shared/ddl/neutral/period_lock.rs index eb62a6e27..50c374e1c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/period_lock.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/period_lock.rs @@ -13,11 +13,9 @@ //! ALTER COLLECTION journal_entries DROP PERIOD LOCK; //! ``` //! -//! Ported from the pgwire `ddl::period_lock` handlers; the token parsing, -//! catalog get/put, schema-version bump, and audit side effects are preserved -//! verbatim. Only the result construction changed from pgwire `Response` / -//! `Tag` to the protocol-neutral [`DdlResult`]; the SQLSTATE codes, messages, -//! and command tags are unchanged. +//! The token parsing, catalog get/put, schema-version bump, and audit side +//! effects run here. The result is the protocol-neutral [`DdlResult`] with its +//! SQLSTATE codes, messages, and command tags. use nodedb_types::DatabaseId; @@ -29,14 +27,13 @@ use crate::control::state::SharedState; use super::super::result::{DdlError, DdlResult}; -/// Construct a [`DdlError`], preserving the exact SQLSTATE codes and messages -/// the pgwire handlers produced. +/// Construct a [`DdlError`] from a SQLSTATE code and a message. fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } /// Handle `ALTER COLLECTION x ADD PERIOD LOCK ON col REFERENCES table(pk) ...` -pub fn add_period_lock( +pub async fn add_period_lock( state: &SharedState, identity: &AuthenticatedIdentity, sql: &str, @@ -105,7 +102,8 @@ pub fn add_period_lock( coll.period_lock = Some(def); - persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) + persist_collection_replicated(state, &coll) + .await .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); @@ -125,7 +123,7 @@ pub fn add_period_lock( } /// Handle `ALTER COLLECTION x DROP PERIOD LOCK`. -pub fn drop_period_lock( +pub async fn drop_period_lock( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -146,7 +144,8 @@ pub fn drop_period_lock( coll.period_lock = None; - persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) + persist_collection_replicated(state, &coll) + .await .map_err(|e| DdlError::from_error(&e))?; state.schema_version.bump(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs b/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs index e101f7f0a..c967830fd 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/permission_tree.rs @@ -10,28 +10,24 @@ //! }'; //! //! ALTER COLLECTION documents DROP PERMISSION_TREE; -//! -//! SELECT RESOLVE_PERMISSION('user-42', 'doc-123', 'documents'); //! ``` //! -//! Ported from the pgwire `ddl::permission_tree` handlers. The JSON parse / -//! validate, catalog get/put, in-memory permission-cache update, and audit -//! side effects are preserved verbatim; only the result construction changed -//! from pgwire `Response` / `Tag` to the protocol-neutral [`DdlResult`]. +//! Both statements act on the named collection of the session's database. +//! The permission table resolves in that same database. use nodedb_sql::parser::preprocess::lex::find_ascii_case_insensitive; use nodedb_types::DatabaseId; use crate::control::catalog_entry::persist_collection_replicated; use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::security::permission_tree::TreeKey; use crate::control::security::permission_tree::types::PermissionTreeDef; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; use crate::control::state::SharedState; use super::super::result::{DdlError, DdlResult}; -/// Construct a [`DdlError`], preserving the exact SQLSTATE codes and messages -/// the pgwire handlers produced. +/// Construct a [`DdlError`] from a SQLSTATE code and a message. fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } @@ -48,6 +44,7 @@ fn status(command: impl Into) -> Vec { pub async fn set_permission_tree( state: &SharedState, identity: &AuthenticatedIdentity, + database_id: DatabaseId, sql: &str, ) -> Result, DdlError> { // Extract collection name: between "ALTER COLLECTION " and " SET PERMISSION_TREE". @@ -78,7 +75,7 @@ pub async fn set_permission_tree( // Verify collection exists. let catalog = state.credentials.catalog(); let mut coll = catalog - .get_collection(DatabaseId::DEFAULT, tenant_id.as_u64(), &collection) + .get_collection(database_id, tenant_id.as_u64(), &collection) .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' does not exist")))?; @@ -93,17 +90,22 @@ pub async fn set_permission_tree( let def_json = sonic_rs::to_string(&def) .map_err(|e| DdlError::internal(format!("serialize PERMISSION_TREE: {e}")))?; coll.permission_tree_def = Some(def_json); - persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) + persist_collection_replicated(state, &coll) + .await .map_err(|e| DdlError::from_error(&e))?; - let sources = [collection.clone(), def.permission_table.clone()]; + let key = TreeKey::new(database_id, tenant_id.as_u64(), collection.clone()); + let sources = [ + key.clone(), + key.scope.collection(def.permission_table.clone()), + ]; // Update in-memory cache. state .permission_cache .write() .await - .register_tree_def(tenant_id.as_u64(), &collection, def); + .register_tree_def(key, def); // Rows already in the sources are grants and edges too. Hold the // acknowledgement until every lease holder covers each source group @@ -128,16 +130,15 @@ pub async fn set_permission_tree( } /// Barrier on every Raft group homing one of `sources`, at a read index -/// taken now. A single node has no groups; its planning reloads the cache. -async fn source_group_barrier(state: &SharedState, sources: &[String]) -> crate::Result<()> { - let Some(timing) = state.authorization_fence.timing() else { - return Ok(()); - }; +/// taken now. +async fn source_group_barrier(state: &SharedState, sources: &[TreeKey]) -> crate::Result<()> { + let timing = crate::control::security::auth_lease::barrier::lease_timing(state)?; let mut targets: Vec = Vec::new(); for source in sources { - let vshard = nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, source).vshard(); - let group_id = - crate::control::security::auth_fence::cluster::group_of_vshard(state, vshard.as_u32())?; + let group_id = crate::control::security::auth_fence::cluster::group_of_vshard( + state, + source.vshard().as_u32(), + )?; if targets.iter().any(|target| target.group_id == group_id) { continue; } @@ -156,6 +157,7 @@ async fn source_group_barrier(state: &SharedState, sources: &[String]) -> crate: pub async fn drop_permission_tree( state: &SharedState, identity: &AuthenticatedIdentity, + database_id: DatabaseId, sql: &str, ) -> Result, DdlError> { let start = "ALTER COLLECTION ".len(); @@ -167,12 +169,13 @@ pub async fn drop_permission_tree( let catalog = state.credentials.catalog(); let mut coll = catalog - .get_collection(DatabaseId::DEFAULT, tenant_id.as_u64(), &collection) + .get_collection(database_id, tenant_id.as_u64(), &collection) .map_err(|e| DdlError::from_error(&e))? .ok_or_else(|| err("42P01", format!("collection '{collection}' does not exist")))?; coll.permission_tree_def = None; - persist_collection_replicated(state, DatabaseId::DEFAULT, &coll) + persist_collection_replicated(state, &coll) + .await .map_err(|e| DdlError::from_error(&e))?; // Update in-memory cache. @@ -180,7 +183,11 @@ pub async fn drop_permission_tree( .permission_cache .write() .await - .unregister_tree_def(tenant_id.as_u64(), &collection); + .unregister_tree_def(&TreeKey::new( + database_id, + tenant_id.as_u64(), + collection.clone(), + )); state .audit diff --git a/nodedb/src/control/server/shared/ddl/neutral/planning.rs b/nodedb/src/control/server/shared/ddl/neutral/planning.rs index 039a7aeee..50957c3a1 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/planning.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/planning.rs @@ -36,10 +36,9 @@ pub async fn plan_authorized_sql( ) -> Result<(Vec, OutputSchema, QueryLeaseScope), DdlError> { // Internal DDL scans still plan in the caller-selected database context. let scope = RequestAuthScope::for_database(identity, state.auth_stores(), database_id); - let permission_cache = - crate::control::security::auth_fence::permission_view(state, identity.tenant_id) - .await - .map_err(|error| DdlError::from_error(&error))?; + crate::control::security::auth_fence::admit_permission_view(state, identity.tenant_id) + .await + .map_err(|error| DdlError::from_error(&error))?; let sec = PlanSecurityContext { identity, auth: scope.auth(), @@ -47,7 +46,9 @@ pub async fn plan_authorized_sql( redaction_store: &state.redaction, permissions: &state.permissions, roles: &state.roles, - permission_cache: Some(&*permission_cache), + permission_tree: crate::control::planner::context::PermissionTreeSource::Live( + &state.permission_cache, + ), }; let query_ctx = QueryContext::for_state(state); let (tasks, output_schema, versions, _) = query_ctx @@ -72,11 +73,14 @@ pub async fn plan_authorized_sql( // Admission follows authorization so denied requests never consume // descriptor leases. Callers retain this scope through dispatch and // response consumption. - let lease_scope = state.acquire_plan_lease_scope(&versions).map_err(|error| { - let (_, sqlstate, message) = - crate::control::server::pgwire::types::error_to_sqlstate(&error); - DdlError::new(sqlstate.to_string(), message) - })?; + let lease_scope = state + .acquire_plan_lease_scope(&versions) + .await + .map_err(|error| { + let (_, sqlstate, message) = + crate::control::server::pgwire::types::error_to_sqlstate(&error); + DdlError::new(sqlstate.to_string(), message) + })?; Ok((tasks, output_schema, lease_scope)) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/procedure/call.rs b/nodedb/src/control/server/shared/ddl/neutral/procedure/call.rs index 7296fa088..33fa21177 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/procedure/call.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/procedure/call.rs @@ -2,12 +2,10 @@ //! `CALL (args)` execution handler. //! -//! Ported from the pgwire `ddl::procedure::call` handler. The CALL parsing, -//! catalog resolution, argument binding, budgeted body execution, and OUT -//! parameter extraction are preserved verbatim; only the result construction -//! changed from pgwire `Response` / `PgWireError` to the protocol-neutral -//! [`DdlResult`] / [`DdlError`]. The OUT-parameter result set carries the same -//! text columns and per-row values as the pgwire `QueryResponse` it replaces. +//! The CALL parsing, catalog resolution, argument binding, budgeted body +//! execution, and OUT parameter extraction run here. The result is the +//! protocol-neutral [`DdlResult`] / [`DdlError`]. The OUT-parameter result set +//! carries text columns. use crate::control::planner::procedural::executor::bindings::RowBindings; use crate::control::planner::procedural::executor::core::StatementExecutor; @@ -67,7 +65,8 @@ pub async fn call_procedure( let block = crate::control::planner::procedural::parse_block(&proc.body_sql) .map_err(|e| DdlError::new("42601", format!("procedure body parse error: {e}")))?; - // Execute with fuel metering, timeout, and transaction context. + // Execute with fuel metering and timeout. Every statement stages into a + // transaction that COMMIT, ROLLBACK and the block's end resolve. let mut budget = ExecutionBudget::new(proc.max_iterations, proc.timeout_secs); let executor = StatementExecutor::with_source_in_database( state, @@ -76,13 +75,14 @@ pub async fn call_procedure( database_id, 0, crate::event::EventSource::User, - ) - .with_transaction_context(); + ); executor .execute_block_with_budget(&block, &bindings, &mut budget) .await .map_err(|e| DdlError::new("P0001", e.to_string()))?; + // A procedure runs with no cross-shard origin, so it holds no cross-node + // write, and its `PUBLISH TO` messages committed in its redo record. // Check for OUT parameter values. let out_params: Vec<_> = proc diff --git a/nodedb/src/control/server/shared/ddl/neutral/procedure/create/handler.rs b/nodedb/src/control/server/shared/ddl/neutral/procedure/create/handler.rs index 0a3bbff7f..e64e627f3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/procedure/create/handler.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/procedure/create/handler.rs @@ -2,16 +2,14 @@ //! The protocol-neutral `create_procedure` handler. //! -//! Ported from the pgwire `ddl::procedure::create::handler`. All non-return -//! logic (privilege gate, parsing, or-replace pre-check, body parse validation, +//! The privilege gate, parsing, or-replace pre-check, body parse validation, //! StoredProcedure build, catalog propose-and-apply, routability extraction, -//! Lite definition-sync broadcast, and the `audit_record` call) is preserved -//! verbatim; only the result construction changed from pgwire `Response` / -//! `PgWireError` to the protocol-neutral [`DdlResult`] / [`DdlError`]. +//! Lite definition-sync broadcast, and the `audit_record` call run here. The +//! result is the protocol-neutral [`DdlResult`] / [`DdlError`]. use crate::control::security::catalog::procedure_types::StoredProcedure; use crate::control::security::identity::AuthenticatedIdentity; -use crate::control::server::shared::ddl::catalog::propose_and_apply; +use crate::control::server::shared::ddl::catalog::propose_and_apply_async; use crate::control::server::shared::ddl::neutral::auth_support::{require_tenant_admin, status}; use crate::control::server::shared::ddl::result::{DdlError, DdlResult}; use crate::control::state::SharedState; @@ -20,7 +18,7 @@ use super::parse::parse_create_procedure; use super::routability::extract_routability; /// Handle `CREATE [OR REPLACE] PROCEDURE ...` -pub fn create_procedure( +pub async fn create_procedure( state: &SharedState, identity: &AuthenticatedIdentity, sql: &str, @@ -74,7 +72,7 @@ pub fn create_procedure( // applier writes the record to local redb and clears the // parsed block cache so the next CALL re-parses the new body. let entry = crate::control::catalog_entry::CatalogEntry::PutProcedure(Box::new(stored.clone())); - propose_and_apply(state, &entry)?; + propose_and_apply_async(state, &entry).await?; // Broadcast to connected Lite sessions after the catalog commit is durable. emit_procedure_put(state, &stored); diff --git a/nodedb/src/control/server/shared/ddl/neutral/procedure/create/parse.rs b/nodedb/src/control/server/shared/ddl/neutral/procedure/create/parse.rs index 3d83c39aa..d4c0ade65 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/procedure/create/parse.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/procedure/create/parse.rs @@ -8,9 +8,7 @@ //! parser here lifts the body verbatim and lets the procedural-SQL //! parser handle it downstream. //! -//! Ported from the pgwire `ddl::procedure::create::parse` module; only the -//! error type changed from pgwire `PgWireError` to the protocol-neutral -//! [`DdlError`]. +//! Parse errors are the protocol-neutral [`DdlError`]. use crate::control::security::catalog::procedure_types::{ParamDirection, ProcedureParam}; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; diff --git a/nodedb/src/control/server/shared/ddl/neutral/procedure/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/procedure/drop.rs index fa75f4579..abec09907 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/procedure/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/procedure/drop.rs @@ -2,12 +2,10 @@ //! `DROP PROCEDURE [IF EXISTS]` DDL handler. //! -//! Ported from the pgwire `ddl::procedure::drop` handler. The catalog path -//! (existence pre-check so `IF EXISTS` on a missing procedure never touches -//! raft, `propose_catalog_entry` + `LocalOnly` local-delete fallback, Lite -//! definition-sync broadcast, and the `audit_record` call) is preserved -//! verbatim; only the result construction changed from pgwire `Response` / -//! `PgWireError` to the protocol-neutral [`DdlResult`] / [`DdlError`]. +//! The catalog path (existence pre-check so `IF EXISTS` on a missing +//! procedure never touches raft, `propose_catalog_entry_async`, Lite +//! definition-sync broadcast, and the `audit_record` call) runs here. The +//! result is the protocol-neutral [`DdlResult`] / [`DdlError`]. use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; @@ -17,7 +15,7 @@ use super::super::super::result::{DdlError, DdlResult}; use super::super::auth_support::{require_tenant_admin, status}; /// Handle `DROP PROCEDURE [IF EXISTS] ` -pub fn drop_procedure( +pub async fn drop_procedure( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -73,14 +71,13 @@ pub fn drop_procedure( database_id, tenant_id, name: name.clone(), + // Frozen by the proposer's stamp. + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, }; - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - let _ = catalog - .delete_procedure_in_database(database_id, tenant_id, &name) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } // Broadcast deletion to connected Lite sessions. { diff --git a/nodedb/src/control/server/shared/ddl/neutral/procedure/parens.rs b/nodedb/src/control/server/shared/ddl/neutral/procedure/parens.rs index b130c48ad..f8954854d 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/procedure/parens.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/procedure/parens.rs @@ -3,9 +3,7 @@ //! Paren-matching helper shared by the procedure CREATE parser and the CALL //! parser. //! -//! Inlined here (the pgwire `parse_utils` helper carried a pgwire-tinted module -//! path) so the neutral procedure family carries no pgwire dependency; the -//! matching logic is preserved verbatim. +//! Defined here so the neutral procedure family carries no pgwire dependency. /// Find the matching closing paren for the open paren at `start`. /// diff --git a/nodedb/src/control/server/shared/ddl/neutral/procedure/show.rs b/nodedb/src/control/server/shared/ddl/neutral/procedure/show.rs index 169fed8ff..f1fe4988c 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/procedure/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/procedure/show.rs @@ -2,11 +2,8 @@ //! `SHOW PROCEDURES` DDL handler. //! -//! Ported from the pgwire `ddl::procedure::show` handler. The catalog read and -//! per-row parameter formatting are preserved verbatim; only the result -//! construction changed from a pgwire `QueryResponse` (5 text columns) to a -//! protocol-neutral [`DdlResult::Rows`] carrying the same columns and per-row -//! values. +//! The catalog read and per-row parameter formatting run here. The result is +//! a protocol-neutral [`DdlResult::Rows`] with five text columns. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs index 81e5261c6..d974aa681 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/balance_as_of.rs @@ -44,7 +44,7 @@ pub async fn balance_as_of( // `collection` is a caller argument, so the read it names is authorized and // row-filtered here. The returned balance is arithmetic over `column`, so a - // redaction rule on that column has no honest answer — masking it would + // redaction rule on that column has no honest answer — masking it will // report a number no row holds. let gate = CollectionReadGate::open(state, identity, database_id, &collection)?; gate.require_document_engine(&collection, "BALANCE_AS_OF")?; @@ -53,45 +53,51 @@ pub async fn balance_as_of( // Read current balance from the target document. let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); let pk_bytes = key.as_bytes().to_vec(); - let surrogate = state - .surrogate_assigner - .lookup( - nodedb_types::CollectionKey::from_bare(database_id, &collection), - tenant_id, - &pk_bytes, - ) - .map_err(|e| DdlError::from_error_in_context("surrogate lookup failed", &e))? - .unwrap_or(nodedb_types::Surrogate::ZERO); - let mut get_plan = - PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::PointGet { - collection: nodedb_types::QualifiedCollection::new(database_id, &collection), - document_id: key.clone(), - surrogate, - pk_bytes, - rls_filters: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - }); - gate.inject_rls(&mut get_plan)?; - - let get_resp = dispatch_utils::dispatch_to_data_plane( + let surrogate = crate::control::server::surrogate_exchange::lookup_surrogate_routed( state, + nodedb_types::CollectionKey::from_bare(database_id, &collection), tenant_id, - database_id, - vshard, - get_plan, - TraceId::ZERO, + &pk_bytes, + crate::types::TraceId::ZERO, ) .await - .map_err(|e| DdlError::from_error_in_context("point get failed", &e))?; - - let doc_json = crate::data::executor::response_codec::decode_payload_to_json(&get_resp.payload); - let doc: serde_json::Value = sonic_rs::from_str(&doc_json).unwrap_or(serde_json::Value::Null); - - let current_balance = doc - .get(&column) - .and_then(json_to_decimal) - .unwrap_or(rust_decimal::Decimal::ZERO); + .map_err(|e| DdlError::from_error_in_context("surrogate lookup failed", &e))?; + // A key the home never bound names no row: its current balance is zero. + let current_balance = match surrogate { + None => rust_decimal::Decimal::ZERO, + Some(surrogate) => { + let mut get_plan = + PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::PointGet { + collection: nodedb_types::QualifiedCollection::new(database_id, &collection), + document_id: key.clone(), + surrogate: Some(surrogate), + pk_bytes, + rls_filters: Vec::new(), + system_time: nodedb_types::SystemTimeScope::Current, + valid_at_ms: None, + }); + gate.inject_rls(&mut get_plan)?; + + let get_resp = dispatch_utils::dispatch_to_data_plane( + state, + tenant_id, + database_id, + vshard, + get_plan, + TraceId::ZERO, + ) + .await + .map_err(|e| DdlError::from_error_in_context("point get failed", &e))?; + + let doc_json = + crate::data::executor::response_codec::decode_payload_to_json(&get_resp.payload); + let doc: serde_json::Value = + sonic_rs::from_str(&doc_json).unwrap_or(serde_json::Value::Null); + doc.get(&column) + .and_then(json_to_decimal) + .unwrap_or(rust_decimal::Decimal::ZERO) + } + }; // Find materialized sum definitions to know the source collection and value_expr. let catalog = state.credentials.catalog(); @@ -109,7 +115,7 @@ pub async fn balance_as_of( }; // The source collection is a second read, resolved from the catalog rather - // than the argument list, and it needs its own grant: a caller who may read + // than the argument list, and it needs its own grant: a caller who can read // the balance is not thereby entitled to the ledger it was summed from. // `value_expr` can name any of its columns, so a redaction rule anywhere on // it is refused rather than silently summed over hidden values. diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs index dc3141fdd..5f1ef98c9 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/helpers.rs @@ -2,18 +2,16 @@ //! Shared helpers for the protocol-neutral query-function handlers. //! -//! Ported verbatim from the pgwire `ddl::query_functions::helpers` module; the -//! only change is that fallible helpers now yield a protocol-neutral -//! [`DdlError`] (SQLSTATE + message) instead of a pgwire `PgWireError`, and the -//! single-column `result` output builds a [`DdlResult::Rows`] directly instead -//! of a pgwire `QueryResponse`. SQLSTATE codes and messages are unchanged. +//! Fallible helpers yield a protocol-neutral [`DdlError`] (SQLSTATE + +//! message), and the single-column `result` output builds a +//! [`DdlResult::Rows`] directly. use nodedb_sql::parser::preprocess::lex::find_ascii_case_insensitive; use nodedb_types::Value; use serde_json::{Map, Value as JsonValue}; use crate::control::server::response_shape::cell::row_to_wire_json; -use crate::control::server::response_shape::project::{is_scan_wrapper_json, push_flat_rows}; +use crate::control::server::response_shape::project::push_flat_rows; use crate::control::server::response_shape::types::ShapedRows; use super::super::super::result::{DdlError, DdlResult}; @@ -70,8 +68,8 @@ pub fn json_to_decimal(v: &serde_json::Value) -> Option { /// Build the single-row `result` output carrying `value`. /// -/// Mirrors the pgwire `return_single_value` helper: one text column named -/// `result` with one row holding `value`. +/// The output has one text column named `result` with one row holding +/// `value`. pub fn single_result(value: &str) -> Vec { let mut row = Map::new(); row.insert("result".to_string(), JsonValue::String(value.to_string())); @@ -99,40 +97,9 @@ pub fn unwrap_scan_docs(docs: Vec) -> Result (String, Map) { - let JsonValue::Object(mut map) = doc else { - return (String::new(), Map::new()); - }; - if is_scan_wrapper_json(&map) { - let id = map - .get("id") - .and_then(|v| v.as_str()) - .unwrap_or_default() - .to_string(); - let data = match map.remove("data") { - Some(JsonValue::Object(inner)) => inner, - _ => Map::new(), - }; - return (id, data); - } - (String::new(), map) -} - /// Build a zero-row `result` output. /// -/// Mirrors the pgwire empty-`QueryResponse` case (one text column named -/// `result`, no rows). +/// The output has one text column named `result` and no rows. pub fn empty_result() -> Vec { vec![DdlResult::Rows(ShapedRows::text_rows( vec!["result".to_string()], diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/router.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/router.rs index 6a5297c61..6cc8336db 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/router.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/router.rs @@ -3,13 +3,12 @@ //! Protocol-neutral routing for the temporal / audit query functions. //! //! These are `SELECT (...)` calls that never parse into a typed DDL AST -//! statement — the pgwire router recognized them by substring (`upper.contains`) -//! in its `router::function::dispatch`, after the typed-AST parse gate and the -//! auth family. Replicate that exactly: this router is invoked only from the -//! `None` (non-DDL-parse) branch of the parent neutral router, so any typed DDL -//! statement (or parse error) whose body happens to contain one of these -//! substrings is handled by the typed path first, byte-identically to before. -//! The substring recognition order is preserved verbatim. +//! statement — the router recognizes them by substring (`upper.contains`) +//! after the typed-AST parse gate and the auth family. This router is invoked +//! only from the `None` (non-DDL-parse) branch of the parent neutral router, +//! so any typed DDL statement (or parse error) whose body contains one of +//! these substrings is handled by the typed path first. The substring +//! recognition order is fixed. use crate::control::security::identity::AuthenticatedIdentity; use crate::control::state::SharedState; diff --git a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs index e1bfea4a1..e435331ed 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/query_functions/verify_hash_chain.rs @@ -2,20 +2,21 @@ //! `SELECT VERIFY_HASH_CHAIN('collection')` //! -//! Scans documents in the collection, verifies each hash chain link. -//! Returns `{valid: true/false, entries: N, broken_at: index, last_hash: ...}`. - -use sonic_rs; +//! Dispatches `MetaOp::VerifyHashChain` to the core that owns the collection. +//! The Data Plane walks the chain over the raw stored rows. This handler only +//! authorizes the call and shapes the verdict as +//! `{valid, entries, last_hash, broken_at, document_id, expected, found}`. use crate::bridge::envelope::PhysicalPlan; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::dispatch_utils; use crate::control::state::SharedState; +use crate::types::hash_chain::ChainVerdict; use crate::types::{DatabaseId, TraceId}; use super::super::super::result::{DdlError, DdlResult}; use super::super::read_gate::CollectionReadGate; -use super::helpers::{err, extract_function_args, single_result, unwrap_scan_doc_with_id}; +use super::helpers::{clean_arg, err, extract_function_args, single_result}; pub async fn verify_hash_chain( state: &SharedState, @@ -25,121 +26,87 @@ pub async fn verify_hash_chain( ) -> Result, DdlError> { let tenant_id = identity.tenant_id; let args = extract_function_args(sql, "VERIFY_HASH_CHAIN")?; - if args.is_empty() { + let collection = args.first().map(|arg| clean_arg(arg).to_lowercase()); + let Some(collection) = collection.filter(|name| !name.is_empty()) else { return Err(err("42601", "VERIFY_HASH_CHAIN requires (collection)")); - } + }; - let collection = args[0] - .trim() - .trim_matches('\'') - .trim_matches('"') - .to_lowercase(); - - // `collection` is a caller argument, so the scan it names is authorized and - // row-filtered here. Each link is recomputed over the whole document body, - // so any redaction rule on the collection is refused: hashing a masked row - // would report an intact chain as broken. + // Each link covers its whole stored row, so a caller that cannot read + // every row of the collection is refused rather than handed a verdict + // over rows it cannot see. let gate = CollectionReadGate::open(state, identity, database_id, &collection)?; gate.require_document_engine(&collection, "VERIFY_HASH_CHAIN")?; gate.refuse_if_any_redaction(&collection, "the hash chain")?; - // Scan all documents. let vshard = nodedb_types::CollectionKey::from_bare(database_id, &collection).vshard(); - let mut scan_plan = PhysicalPlan::Document(nodedb_physical::physical_plan::DocumentOp::Scan { + let mut plan = PhysicalPlan::Meta(nodedb_physical::physical_plan::MetaOp::VerifyHashChain { collection: nodedb_types::QualifiedCollection::new(database_id, &collection), - limit: usize::MAX, - offset: 0, - sort_keys: Vec::new(), - filters: Vec::new(), - distinct: false, - projection: Vec::new(), - computed_columns: Vec::new(), - window_functions: Vec::new(), - system_time: nodedb_types::SystemTimeScope::Current, - valid_at_ms: None, - prefilter: None, }); - gate.inject_rls(&mut scan_plan)?; + gate.inject_rls(&mut plan)?; - let scan_resp = dispatch_utils::dispatch_to_data_plane( + let response = dispatch_utils::dispatch_to_data_plane( state, tenant_id, database_id, vshard, - scan_plan, + plan, TraceId::ZERO, ) .await - .map_err(|e| DdlError::from_error_in_context("scan failed", &e))?; - - let payload_json = - crate::data::executor::response_codec::decode_payload_to_json(&scan_resp.payload); - let docs: Vec = sonic_rs::from_str(&payload_json) - .map_err(|e| err("22P02", &format!("invalid JSON in scan response: {e}")))?; - - // Walk the chain: each doc should have `_chain_hash` field. - let mut prev_hash = crate::data::executor::enforcement::hash_chain::GENESIS_HASH.to_string(); - let mut entries = 0usize; - let mut valid = true; - let mut broken_at: Option = None; - - for (i, doc) in docs.into_iter().enumerate() { - // The raw document-scan codec wraps each row as `{"id": , - // "data": {..fields incl. _chain_hash..}}`. `doc_id` must be the - // *wrapper's* id — the same `document_id` the original INSERT fed - // into `compute_chain_hash` — not a same-named field inside the - // document body, which may not exist. - let (wrapper_id, obj) = unwrap_scan_doc_with_id(doc); - let doc_id = if !wrapper_id.is_empty() { - wrapper_id - } else { - obj.get("id") - .or_else(|| obj.get("_id")) - .and_then(|v| v.as_str()) - .unwrap_or("") - .to_string() - }; - - let stored_hash = obj - .get("_chain_hash") - .and_then(|v| v.as_str()) - .unwrap_or("") - .to_string(); + .map_err(|e| DdlError::from_error_in_context("hash-chain verification failed", &e))?; - if stored_hash.is_empty() { - valid = false; - broken_at = Some(i); - break; - } - - // Recompute the hash from the document contents (without _chain_hash). - let mut doc_for_hash = serde_json::Value::Object(obj); - if let Some(obj) = doc_for_hash.as_object_mut() { - obj.remove("_chain_hash"); - } - let doc_bytes = sonic_rs::to_vec(&doc_for_hash) - .map_err(|e| DdlError::internal(format!("failed to serialize document: {e}")))?; - - let expected = crate::data::executor::enforcement::hash_chain::compute_chain_hash( - &prev_hash, &doc_id, &doc_bytes, - ); + let verdict: ChainVerdict = zerompk::from_msgpack(response.payload.as_ref()) + .map_err(|e| DdlError::internal(format!("hash-chain verdict does not decode: {e}")))?; + Ok(single_result(&verdict_json(&verdict).to_string())) +} - if expected != stored_hash { - valid = false; - broken_at = Some(i); - break; - } +/// The verdict as the JSON the API boundary returns. +fn verdict_json(verdict: &ChainVerdict) -> serde_json::Value { + let brk = verdict.broken.as_ref(); + serde_json::json!({ + "valid": brk.is_none(), + "entries": verdict.entries, + "last_hash": verdict.last_hash, + "broken_at": brk.map(|b| b.index), + "document_id": brk.and_then(|b| b.document_id.clone()), + "expected": brk.and_then(|b| b.expected.clone()), + "found": brk.and_then(|b| b.found.clone()), + }) +} - prev_hash = stored_hash; - entries += 1; +#[cfg(test)] +mod tests { + use super::*; + use crate::types::hash_chain::ChainBreak; + + #[test] + fn an_intact_verdict_is_valid_with_no_break_fields() { + let json = verdict_json(&ChainVerdict { + entries: 3, + last_hash: "ab".into(), + broken: None, + }); + assert_eq!(json["valid"], true); + assert_eq!(json["entries"], 3); + assert!(json["broken_at"].is_null()); } - let result = serde_json::json!({ - "valid": valid, - "entries": entries, - "broken_at": broken_at, - "last_hash": prev_hash, - }); - - Ok(single_result(&result.to_string())) + #[test] + fn a_break_names_its_position_row_and_links() { + let json = verdict_json(&ChainVerdict { + entries: 1, + last_hash: "h1".into(), + broken: Some(ChainBreak { + index: 1, + document_id: Some("00000002".into()), + expected: Some("e".into()), + found: Some("f".into()), + }), + }); + assert_eq!(json["valid"], false); + assert_eq!(json["broken_at"], 1); + assert_eq!(json["document_id"], "00000002"); + assert_eq!(json["expected"], "e"); + assert_eq!(json["found"], "f"); + } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/quota_ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/quota_ddl.rs index cb5cd6e95..132cf0b61 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/quota_ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/quota_ddl.rs @@ -16,14 +16,13 @@ use serde_json::{Map, Value as JsonValue}; use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::catalog_entry::post_apply::scope_quota as scope_quota_post_apply; use crate::control::security::identity::AuthenticatedIdentity; use crate::control::security::metering::quota::{QuotaDefinition, QuotaEnforcement}; use crate::control::server::response_shape::types::ShapedRows; use crate::control::state::SharedState; use super::super::result::{DdlError, DdlResult}; -use super::replicate::propose_and_apply; +use super::replicate::propose_and_apply_async; /// Default warning threshold when `WARN AT` is omitted, matching the /// documented default on `QuotaDefinition::warning_threshold`. @@ -57,7 +56,7 @@ fn required_u64(parts: &[&str], keyword: &str) -> Result { /// DEFINE QUOTA ON SCOPE '' MAX TOKENS PER SECONDS /// ENFORCEMENT [WARN AT ] -pub fn define_quota( +pub async fn define_quota( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -121,19 +120,7 @@ pub fn define_quota( // Replicated: every node writes the row and installs the definition in // its `QuotaManager` via post-apply. - propose_and_apply( - state, - &CatalogEntry::PutScopeQuota(Box::new(stored.clone())), - || { - state - .credentials - .catalog() - .put_scope_quota(&stored) - .map_err(|e| DdlError::from_error(&e))?; - scope_quota_post_apply::put(&stored, state); - Ok(()) - }, - )?; + propose_and_apply_async(state, &CatalogEntry::PutScopeQuota(Box::new(stored))).await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, @@ -152,7 +139,7 @@ pub fn define_quota( } /// DROP QUOTA ON SCOPE '' -pub fn drop_quota( +pub async fn drop_quota( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -167,7 +154,7 @@ pub fn drop_quota( .to_string(); // Absence is decided on the leader, before the propose: a rejection at - // apply time would leave the followers disagreeing with the leader. + // apply time will leave the followers disagreeing with the leader. if !state.quota_manager.has_quota(&scope_name) { return Err(err( "42704", @@ -175,21 +162,13 @@ pub fn drop_quota( )); } - propose_and_apply( + propose_and_apply_async( state, &CatalogEntry::DeleteScopeQuota { scope_name: scope_name.clone(), }, - || { - state - .credentials - .catalog() - .delete_scope_quota(&scope_name) - .map_err(|e| DdlError::from_error(&e))?; - scope_quota_post_apply::delete(&scope_name, state); - Ok(()) - }, - )?; + ) + .await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs index b9ec9ee32..7574d56e6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rate_gate.rs @@ -85,14 +85,15 @@ pub async fn rate_check( let actual_ttl = if key_exists { 0 } else { ttl_ms }; - let surrogate = state - .surrogate_assigner - .assign( - nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, RATE_COLLECTION), - tenant_id, - rate_key.as_bytes(), - ) - .map_err(|e| DdlError::from_error_in_context("RATE_CHECK", &e))?; + let surrogate = crate::control::server::surrogate_exchange::assign_surrogate_routed( + state, + nodedb_types::CollectionKey::from_bare(DatabaseId::DEFAULT, RATE_COLLECTION), + tenant_id, + rate_key.as_bytes(), + crate::types::TraceId::ZERO, + ) + .await + .map_err(|e| DdlError::from_error_in_context("RATE_CHECK", &e))?; let plan = PhysicalPlan::Kv(KvOp::Incr { collection: nodedb_types::QualifiedCollection::new(DatabaseId::DEFAULT, RATE_COLLECTION), key: rate_key.as_bytes().to_vec(), diff --git a/nodedb/src/control/server/shared/ddl/neutral/redaction/create.rs b/nodedb/src/control/server/shared/ddl/neutral/redaction/create.rs index 791b56250..d71f655a6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/redaction/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/redaction/create.rs @@ -4,14 +4,12 @@ //! set for one `(tenant, collection, for_role)` triple. //! //! Shaped after the sibling `neutral::rls` create handler: authorize, validate, -//! pre-check the duplicate, propose `CatalogEntry::PutRedactionPolicy`, and fall -//! back to an inline catalog write plus in-memory install when no metadata raft -//! group is configured (`LocalOnly`). +//! pre-check the duplicate, and propose `CatalogEntry::PutRedactionPolicy`. use nodedb_sql::ddl_ast::statement::RedactionRuleSpec; use crate::control::catalog_entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::planner::sql_plan_convert::convert::db_qualified; use crate::control::security::audit::AuditEvent; use crate::control::security::catalog::StoredRedactionPolicy; @@ -80,7 +78,7 @@ fn compile_rules(specs: &[RedactionRuleSpec]) -> Result, DdlE /// /// All fields are pre-parsed by the `nodedb-sql` AST layer; this handler only /// validates the modes, refuses array targets, and mutates the catalog. -pub fn create_redaction_policy( +pub async fn create_redaction_policy( state: &SharedState, identity: &AuthenticatedIdentity, req: &CreateRedactionPolicyRequest<'_>, @@ -133,17 +131,9 @@ pub fn create_redaction_policy( .map_err(|e| DdlError::from_error_in_context("redaction serialize", &e))?; let entry = CatalogEntry::PutRedactionPolicy(Box::new(stored.clone())); - let outcome = propose_catalog_entry(state, &entry) + propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - { - let catalog = state.credentials.catalog(); - catalog - .put_redaction_policy(&stored) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } - state.redaction.install_replicated_policy(policy); - } state.audit_record( AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/redaction/drop_show.rs b/nodedb/src/control/server/shared/ddl/neutral/redaction/drop_show.rs index 0baa11a0b..9106da1d7 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/redaction/drop_show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/redaction/drop_show.rs @@ -9,7 +9,7 @@ use serde_json::{Map, Value as JsonValue}; use crate::control::catalog_entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::planner::sql_plan_convert::convert::db_qualified; use crate::control::security::audit::AuditEvent; use crate::control::security::identity::AuthenticatedIdentity; @@ -23,7 +23,7 @@ use super::scope::{authorize_redaction_scope, status}; /// `DROP REDACTION POLICY [IF EXISTS] ON FOR ROLE /// [TENANT ]` -pub fn drop_redaction_policy( +pub async fn drop_redaction_policy( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -53,19 +53,9 @@ pub fn drop_redaction_policy( collection: qualified_collection.clone(), for_role: for_role.to_string(), }; - let outcome = propose_catalog_entry(state, &entry) + propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - { - let catalog = state.credentials.catalog(); - catalog - .delete_redaction_policy(tenant_id, &qualified_collection, for_role) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } - state - .redaction - .install_replicated_drop_policy(tenant_id, &qualified_collection, for_role); - } state.audit_record( AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/replicate.rs b/nodedb/src/control/server/shared/ddl/neutral/replicate.rs index ddbefd261..1e601ff11 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/replicate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/replicate.rs @@ -1,63 +1,64 @@ // SPDX-License-Identifier: BUSL-1.1 -//! Shared propose-and-apply branch for replicated neutral DDL writes. +//! Shared propose helper for replicated neutral DDL writes. //! -//! A replicated catalog mutation proposes a `CatalogEntry` through the -//! metadata raft group. Only the node that owns the write runs the typed -//! catalog call and its post-apply hook. Each handler supplies both in -//! `local`, so this module never names a catalog method or a post-apply arm. +//! A replicated catalog mutation proposes a `CatalogEntry`. The proposer +//! applies it on this node through the metadata group, with its post-apply +//! hooks. A handler never writes the catalog itself. use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::propose_outcome::ProposeOutcome; use crate::control::state::SharedState; use super::super::result::DdlError; -/// Propose `entry`, then run `local` when this node owns the catalog write. +/// Propose `entry`, on any runtime flavor. It awaits every wait of the +/// propose, the entry's post-apply Data Plane work included. /// /// A propose failure keeps the SQLSTATE class of its typed error. -pub(crate) fn propose_and_apply( +pub(crate) async fn propose_and_apply_async( state: &SharedState, entry: &CatalogEntry, - local: impl FnOnce() -> Result<(), DdlError>, ) -> Result<(), DdlError> { - propose_and_apply_outcome(state, entry, local).map(|_| ()) + propose_and_apply_outcome_async(state, entry) + .await + .map(|_| ()) } -/// [`propose_and_apply`], returning the outcome the proposer reported. -/// -/// A handler whose per-node side effects live in the post-apply lane needs it: -/// only `LocalOnly` means no applier will run them for this node. -pub(crate) fn propose_and_apply_outcome( +/// [`propose_and_apply_async`], returning the outcome the proposer reported. +pub(crate) async fn propose_and_apply_outcome_async( state: &SharedState, entry: &CatalogEntry, - local: impl FnOnce() -> Result<(), DdlError>, ) -> Result { - let outcome = propose_catalog_entry(state, entry) - .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e))?; - if outcome.needs_local_apply() { - local()?; - } - Ok(outcome) + propose_catalog_entry_async(state, entry) + .await + .map_err(|e| DdlError::from_error_in_context("catalog propose failed", &e)) } #[cfg(test)] mod tests { use std::sync::Arc; - use std::sync::atomic::{AtomicBool, Ordering}; use super::*; use crate::bridge::dispatch::Dispatcher; + use crate::control::security::credential::CredentialStore; use crate::control::server::shared::session::{conn_scope, ddl_buffer}; use crate::wal::WalManager; + /// A state no boot wired to a cluster. A buffered entry never reaches the + /// metadata group, so buffering needs none. fn test_state(name: &str) -> (tempfile::TempDir, Arc) { let dir = tempfile::tempdir().expect("create test directory"); let wal = Arc::new(WalManager::open_for_testing(&dir.path().join(name)).expect("open test WAL")); + let credentials = Arc::new( + CredentialStore::open(&dir.path().join("system.redb")).expect("open credential store"), + ); let (dispatcher, _data_sides) = Dispatcher::new(1, 64); - let state = SharedState::new(dispatcher, wal).expect("construct shared state"); + let state = SharedState::new_with_credentials(dispatcher, wal, credentials, false) + .expect("construct shared state"); + crate::bootstrap::state_wiring::install_gateway(&state).expect("install gateway"); (dir, state) } @@ -66,39 +67,40 @@ mod tests { database_id: 0, tenant_id: 1, name: "replicate-helper".to_string(), + target_descriptor_version: 0, + target_hlc: nodedb_types::Hlc::ZERO, } } - #[tokio::test] - async fn local_runs_without_a_metadata_group() { - let (_dir, state) = test_state("replicate-local.wal"); - let ran = AtomicBool::new(false); - - propose_and_apply(&state, &sample_entry(), || { - ran.store(true, Ordering::SeqCst); - Ok(()) - }) - .expect("single-node propose succeeds"); + #[tokio::test(flavor = "multi_thread")] + async fn applies_through_the_one_node_metadata_group() { + let cluster = crate::control::cluster::test_one_node::boot().await; + let outcome = propose_and_apply_outcome_async(&cluster.state, &sample_entry()) + .await + .expect("one-node propose succeeds"); + assert!(outcome.is_replicated(), "{outcome:?}"); + cluster.shutdown().await; + } - assert!(ran.load(Ordering::SeqCst)); + #[tokio::test] + async fn a_state_with_no_metadata_group_refuses_the_propose() { + let (_dir, state) = test_state("replicate-unbooted.wal"); + let refused = propose_and_apply_outcome_async(&state, &sample_entry()).await; + assert!(refused.is_err(), "no metadata group, no propose"); } #[tokio::test] - async fn local_is_skipped_while_ddl_is_buffered() { + async fn nothing_applies_while_ddl_is_buffered() { let (_dir, state) = test_state("replicate-buffered.wal"); - let ran = AtomicBool::new(false); - - conn_scope::scoped(async { + let outcome = conn_scope::scoped(async { ddl_buffer::activate(); - propose_and_apply(&state, &sample_entry(), || { - ran.store(true, Ordering::SeqCst); - Ok(()) - }) - .expect("buffering an entry succeeds"); + let outcome = propose_and_apply_outcome_async(&state, &sample_entry()) + .await + .expect("buffering an entry succeeds"); ddl_buffer::discard(); + outcome }) .await; - - assert!(!ran.load(Ordering::SeqCst)); + assert_eq!(outcome, ProposeOutcome::Buffered); } } diff --git a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/alter.rs b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/alter.rs index beeeaa90c..59d27d8a6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/alter.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/alter.rs @@ -40,7 +40,7 @@ fn parse_auto_tier(value: &str) -> Result { /// `name`, `action`, `set_key`, and `set_value` come from the typed /// [`PolicyStmt::AlterRetentionPolicy`] variant. `database_id` scopes /// the in-memory registry lookup to the session's database. -pub fn alter_retention_policy( +pub async fn alter_retention_policy( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -88,7 +88,7 @@ pub fn alter_retention_policy( } // Replicated: every node writes the row and refreshes its registry. - propose_put(state, &def)?; + propose_put(state, &def).await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/create.rs b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/create.rs index 5cb97b7c1..7c6208f5b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/create.rs @@ -143,7 +143,7 @@ pub async fn create_retention_policy( // Replicated: every node writes the row and installs the definition in // its own `RetentionPolicyRegistry` via post-apply. - propose_put(state, &def)?; + propose_put(state, &def).await?; // Emit CRDT sync delta for Lite visibility. { @@ -169,7 +169,7 @@ pub async fn create_retention_policy( { // Roll back through the same replicated path that created it, so the // policy disappears on every node, not only on this one. - if let Err(rollback) = propose_delete(state, &def) { + if let Err(rollback) = propose_delete(state, &def).await { return Err(DdlError::from_error_in_context( &format!( "rollback left the policy in place: {}; failed to auto-wire aggregates", diff --git a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/drop.rs index 237a2315b..8187c131b 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/drop.rs @@ -42,7 +42,7 @@ pub async fn drop_retention_policy( .ok_or_else(|| err("42704", format!("retention policy '{name}' does not exist")))?; // Replicated: every node deletes the row and drops its registry entry. - propose_delete(state, &policy_def)?; + propose_delete(state, &policy_def).await?; // Emit CRDT tombstone delta. { diff --git a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/replicate.rs b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/replicate.rs index 4b5d42aba..1ee9e2277 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/replicate.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/replicate.rs @@ -7,31 +7,25 @@ //! `RetentionPolicyRegistry`. A policy created on one node enforces on all. use crate::control::catalog_entry::entry::CatalogEntry; -use crate::control::catalog_entry::post_apply::retention_policy as post_apply; use crate::control::state::SharedState; use crate::engine::timeseries::retention_policy::RetentionPolicyDef; use super::super::super::result::DdlError; -use super::super::replicate::propose_and_apply; +use super::super::replicate::propose_and_apply_async; /// Propose the full policy record. CREATE and ALTER both re-put the row. /// /// The leader validates before proposing, so apply never rejects. -pub(super) fn propose_put(state: &SharedState, def: &RetentionPolicyDef) -> Result<(), DdlError> { +pub(super) async fn propose_put( + state: &SharedState, + def: &RetentionPolicyDef, +) -> Result<(), DdlError> { let entry = CatalogEntry::PutRetentionPolicy(Box::new(def.clone())); - propose_and_apply(state, &entry, || { - state - .credentials - .catalog() - .put_retention_policy(def) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - post_apply::put(def, state); - Ok(()) - }) + propose_and_apply_async(state, &entry).await } /// Propose removal of the policy row and the registry entry on every node. -pub(super) fn propose_delete( +pub(super) async fn propose_delete( state: &SharedState, def: &RetentionPolicyDef, ) -> Result<(), DdlError> { @@ -41,13 +35,5 @@ pub(super) fn propose_delete( name: def.name.clone(), collection: def.collection.clone(), }; - propose_and_apply(state, &entry, || { - state - .credentials - .catalog() - .delete_retention_policy(def.database_id, def.tenant_id, &def.name) - .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; - post_apply::delete(def.database_id, def.tenant_id, &def.name, state); - Ok(()) - }) + propose_and_apply_async(state, &entry).await } diff --git a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/show.rs b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/show.rs index b67dfd866..412192ffb 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/retention_policy/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/retention_policy/show.rs @@ -2,16 +2,12 @@ //! Protocol-neutral `SHOW RETENTION POLICY` DDL handler. //! -//! Ported from the pgwire `ddl::retention_policy::show` handler. The registry -//! listing, the optional `ON ` filter, the tier / duration -//! formatting, and the exact column set are preserved verbatim; only the result -//! construction changed from pgwire `Response` / `QueryResponse` / -//! `DataRowEncoder` to the protocol-neutral [`DdlResult::Rows`] over -//! [`ShapedRows`]. The mixed text/`int8` column OIDs (`tier_count` and -//! `created_at` are `int8`, every other column is text) are reproduced by -//! building `column_types` manually so the RowDescription stays byte-identical; -//! the `int8` cells are emitted as their decimal text form, the same bytes the -//! pgwire `DataRowEncoder::encode_field(&i64)` produced. +//! The registry listing, the optional `ON ` filter, the tier / +//! duration formatting, and the exact column set run here. The result is the +//! protocol-neutral [`DdlResult::Rows`] over [`ShapedRows`]. The mixed +//! text/`int8` column OIDs (`tier_count` and `created_at` are `int8`, every +//! other column is text) come from building `column_types` manually; the +//! `int8` cells are emitted as their decimal text form. //! //! Syntax: //! ```sql diff --git a/nodedb/src/control/server/shared/ddl/neutral/rls.rs b/nodedb/src/control/server/shared/ddl/neutral/rls.rs index eace83d3d..dd3415a68 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/rls.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/rls.rs @@ -2,17 +2,15 @@ //! Protocol-neutral RLS policy DDL — CREATE / DROP / SHOW. //! -//! Ported from the pgwire `ddl::rls` handlers. All non-return logic -//! (permission checks, predicate compilation, duplicate pre-checks, catalog -//! proposes, in-memory `RlsPolicyStore` install/drop, `audit_record`, tenant -//! scoping, and the token-based `parts` parsing for DROP / SHOW) is preserved -//! verbatim; only the result construction changed from pgwire `Response` / -//! `PgWireError` to the protocol-neutral [`DdlResult`] / [`DdlError`]. +//! The permission checks, predicate compilation, duplicate pre-checks, +//! catalog proposes, in-memory `RlsPolicyStore` install/drop, `audit_record`, +//! tenant scoping, and the token-based `parts` parsing for DROP / SHOW run +//! here. The result is the protocol-neutral [`DdlResult`] / [`DdlError`]. use serde_json::{Map, Value as JsonValue}; use crate::control::catalog_entry::CatalogEntry; -use crate::control::metadata_proposer::propose_catalog_entry; +use crate::control::metadata_proposer::propose_catalog_entry_async; use crate::control::planner::sql_plan_convert::convert::db_qualified; use crate::control::security::audit::AuditEvent; use crate::control::security::catalog::StoredRlsPolicy; @@ -119,7 +117,7 @@ fn status(command: &str) -> Vec { /// Resolve and authorize the tenant scope for RLS administration. /// -/// Tenant administrators may manage only their authenticated tenant. An +/// Tenant administrators can manage only their authenticated tenant. An /// explicit cross-tenant target is reserved for superusers. fn authorize_rls_scope( identity: &AuthenticatedIdentity, @@ -148,7 +146,7 @@ fn authorize_rls_scope( /// /// All fields are pre-parsed by the `nodedb-sql` AST layer; this handler /// only performs predicate compilation and catalog mutation. -pub fn create_rls_policy( +pub async fn create_rls_policy( state: &SharedState, identity: &AuthenticatedIdentity, req: &CreateRlsPolicyRequest<'_>, @@ -231,18 +229,10 @@ pub fn create_rls_policy( let stored = StoredRlsPolicy::from_runtime(&policy, database_id, predicate_raw) .map_err(|e| DdlError::from_error_in_context("rls serialize", &e))?; - let entry = CatalogEntry::PutRlsPolicy(Box::new(stored.clone())); - let outcome = propose_catalog_entry(state, &entry) + let entry = CatalogEntry::PutRlsPolicy(Box::new(stored)); + propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - { - let catalog = state.credentials.catalog(); - catalog - .put_rls_policy(&stored) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } - state.rls.install_replicated_policy(policy); - } let mode_str = if is_restrictive { " RESTRICTIVE" } else { "" }; state.audit_record( @@ -259,7 +249,7 @@ pub fn create_rls_policy( } /// `DROP RLS POLICY ON [TENANT ]`. -pub fn drop_rls_policy( +pub async fn drop_rls_policy( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -289,19 +279,9 @@ pub fn drop_rls_policy( collection: qualified_collection.clone(), name: name.to_string(), }; - let outcome = propose_catalog_entry(state, &entry) + propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - { - let catalog = state.credentials.catalog(); - catalog - .delete_rls_policy(tenant_id, &qualified_collection, name) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - } - state - .rls - .install_replicated_drop_policy(tenant_id, &qualified_collection, name); - } state.audit_record( AuditEvent::AdminAction, diff --git a/nodedb/src/control/server/shared/ddl/neutral/role.rs b/nodedb/src/control/server/shared/ddl/neutral/role.rs index 2e7a0daaf..a16ef034f 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/role.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/role.rs @@ -3,12 +3,9 @@ //! Protocol-neutral `role` DDL — CREATE / DROP ROLE, ALTER ROLE (GRANT / //! REVOKE / SET INHERIT), and the shared `set_role_parent` inheritance mutator. //! -//! Ported from the pgwire `ddl::role` handlers. All non-return logic -//! (tenant-admin gate, IF [NOT] EXISTS short-circuits, `prepare_role`, -//! parent-existence + inheritance-cycle validation, catalog propose + -//! single-node `LocalOnly` fallback, `install_replicated_role`, -//! `drop_role`, and `audit_record`) is preserved verbatim; only the result -//! construction changed from pgwire `Response` / `PgWireError` to the +//! The tenant-admin gate, IF [NOT] EXISTS short-circuits, `prepare_role`, +//! parent-existence + inheritance-cycle validation, catalog propose, the +//! post-apply role check, and `audit_record` run here. The result is the //! protocol-neutral [`DdlResult`] / [`DdlError`]. use nodedb_sql::ddl_ast::AlterRoleOp; @@ -22,7 +19,7 @@ use super::auth_support::{require_tenant_admin, status, strip_if_exists, strip_i use super::grant; /// CREATE ROLE [IF NOT EXISTS] [INHERIT ] -pub fn create_role( +pub async fn create_role( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -66,16 +63,11 @@ pub fn create_role( ) .map_err(|e| DdlError::new("42710", e.to_string()))?; - let entry = crate::control::catalog_entry::CatalogEntry::PutRole(Box::new(stored.clone())); - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + let entry = crate::control::catalog_entry::CatalogEntry::PutRole(Box::new(stored)); + let outcome = crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - let catalog = state.credentials.catalog(); - catalog - .put_role(&stored) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - state.roles.install_replicated_role(&stored); - } else if outcome.is_replicated() { + if outcome.is_durable() { super::role_checks::confirm_role(state, name, parent)?; } @@ -93,7 +85,7 @@ pub fn create_role( } /// DROP ROLE [IF EXISTS] -pub fn drop_role( +pub async fn drop_role( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -123,53 +115,35 @@ pub fn drop_role( } // As PostgreSQL does, a role that users hold or roles inherit from is - // not dropped: dropping it would leave them naming a role that grants + // not dropped: dropping it will leave them naming a role that grants // nothing. super::role_checks::check_role_droppable(state, name)?; let entry = crate::control::catalog_entry::CatalogEntry::DeleteRole { name: name.to_string(), }; - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + let outcome = crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - let dropped = if outcome.needs_local_apply() { - let catalog = state.credentials.catalog(); - state - .roles - .drop_role(name, Some(catalog)) - .map_err(|e| DdlError::from_error(&e))? - } else if outcome.is_replicated() { - // The synchronous post-apply removed the role from this node's - // cache before the applied index advanced. A role still present - // means the applier skipped the entry: a user or role came to - // depend on it before the drop committed. - if state.roles.get_role(name).is_some() { - super::role_checks::check_role_droppable(state, name)?; - return Err(DdlError::new( - "40001", - format!("transient: the drop of role '{name}' was superseded, retry"), - )); - } - true - } else { - // Buffered in an open transaction: COMMIT applies it. - true - }; - - if dropped { - state.audit_record( - AuditEvent::PrivilegeChange, - Some(identity.tenant_id), - &identity.username, - &format!("dropped role '{name}'"), - ); - Ok(status("DROP ROLE")) - } else { - Err(DdlError::new( - "42704", - format!("role '{name}' does not exist"), - )) + // The synchronous post-apply removed the role from this node's cache + // before the propose returned. A role still present means the apply + // skipped the entry: a user or role came to depend on it before the drop + // committed. A buffered drop applies at COMMIT. + if outcome.is_durable() && state.roles.get_role(name).is_some() { + super::role_checks::check_role_droppable(state, name)?; + return Err(DdlError::new( + "40001", + format!("transient: the drop of role '{name}' was superseded, retry"), + )); } + + state.audit_record( + AuditEvent::PrivilegeChange, + Some(identity.tenant_id), + &identity.username, + &format!("dropped role '{name}'"), + ); + Ok(status("DROP ROLE")) } /// Typed dispatch for `ALTER ROLE` — covers GRANT, REVOKE, and SET INHERIT forms. @@ -177,9 +151,10 @@ pub fn drop_role( /// Reuses the protocol-neutral `grant_permission` / `revoke_permission` for the /// permission forms so all permission mutations go through the same /// catalog-propose path and emit `AuditEvent::PrivilegeChange`. -pub fn alter_role_typed( +pub async fn alter_role_typed( state: &SharedState, identity: &AuthenticatedIdentity, + database_id: crate::types::DatabaseId, role_name: &str, sub_op: &AlterRoleOp, ) -> Result, DdlError> { @@ -199,30 +174,38 @@ pub fn alter_role_typed( permission, target_type, target_name, - } => grant::permission::grant_permission( - state, - identity, - std::slice::from_ref(permission), - target_type, - target_name, - role_name, - ), + } => { + grant::permission::grant_permission( + state, + identity, + database_id, + std::slice::from_ref(permission), + target_type, + target_name, + role_name, + ) + .await + } AlterRoleOp::Revoke { permission, target_type, target_name, - } => grant::permission::revoke_permission( - state, - identity, - std::slice::from_ref(permission), - target_type, - target_name, - role_name, - ), + } => { + grant::permission::revoke_permission( + state, + identity, + database_id, + std::slice::from_ref(permission), + target_type, + target_name, + role_name, + ) + .await + } AlterRoleOp::SetInherit { parent } => { - set_role_parent(state, role_name, Some(parent))?; + set_role_parent(state, role_name, Some(parent)).await?; state.audit_record( AuditEvent::PrivilegeChange, @@ -244,7 +227,7 @@ pub fn alter_role_typed( /// inheritance mutation goes through one catalog-propose path. The caller /// is responsible for the `require_tenant_admin` privilege check and for /// emitting the audit record. -pub fn set_role_parent( +pub async fn set_role_parent( state: &SharedState, role_name: &str, parent: Option<&str>, @@ -285,16 +268,11 @@ pub fn set_role_parent( created_at: now, }; - let entry = crate::control::catalog_entry::CatalogEntry::PutRole(Box::new(stored.clone())); - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + let entry = crate::control::catalog_entry::CatalogEntry::PutRole(Box::new(stored)); + let outcome = crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - let catalog = state.credentials.catalog(); - catalog - .put_role(&stored) - .map_err(|e| DdlError::from_error_in_context("catalog write", &e))?; - state.roles.install_replicated_role(&stored); - } else if outcome.is_replicated() { + if outcome.is_durable() { super::role_checks::confirm_role(state, role_name, parent)?; } Ok(()) diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/dispatch.rs b/nodedb/src/control/server/shared/ddl/neutral/router/dispatch.rs index 059873586..8eb184338 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/dispatch.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/dispatch.rs @@ -50,7 +50,9 @@ pub async fn try_dispatch( if let Some(r) = string_schema::try_string(state, identity, sql, &upper, database_id).await { return Some(r); } - if let Some(r) = string_streaming::try_string(state, identity, sql, &upper, database_id).await { + if let Some(r) = + string_streaming::try_string(state, identity, sql, &upper, database_id, txn_ctx).await + { return Some(r); } if let Some(r) = @@ -73,14 +75,13 @@ pub async fn try_dispatch( // planner path gives the same `SqlError` (`UnsupportedConstraint` is // `0A000`, a parse error `42601`) and the parser's own `Display` text as // the message. This is the sole parse-error gate for the DDL router; the - // GRAPH / MATCH / SHOW GRAPH STATS prefixed inputs that previously carried - // their own parse-error reproduction are subsumed by this arm. + // GRAPH / MATCH / SHOW GRAPH STATS prefixed inputs reach their parse + // error through this arm. // // Non-DDL statements (`None`) include the temporal / audit query functions — - // `SELECT (...)` calls that never parse into a typed DDL AST. In the - // pgwire router these were recognized by substring after the typed-AST parse - // gate and the auth family; recognizing them here, in the `None` branch, - // preserves that ordering exactly (any typed DDL whose body contains one of + // `SELECT (...)` calls that never parse into a typed DDL AST. They are + // recognized by substring after the typed-AST parse gate and the auth + // family, in the `None` branch (any typed DDL whose body contains one of // the substrings is handled by the typed match above first). A non-match // returns `None` so the caller falls through to the SQL planner. let stmt = match nodedb_sql::ddl_ast::parse(sql) { @@ -109,10 +110,8 @@ pub async fn try_dispatch( } // INSERT INTO x { } — object literal syntax; intercept for - // trigger/sequence handling. Ported from the pgwire `dsl` - // string router, which ran after the typed-AST parse gate — - // recognizing it here in the `None` branch preserves that - // ordering exactly. + // trigger/sequence handling. It runs after the typed-AST parse + // gate, in the `None` branch. if sql .get(.."INSERT INTO ".len()) .is_some_and(|prefix| prefix.eq_ignore_ascii_case("INSERT INTO ")) @@ -179,15 +178,17 @@ pub async fn try_dispatch( // it to the MATCH handler for the Data-Plane overlay merge — mirroring // the single-hop `GRAPH NEIGHBORS` path. let (txn_id, _) = txn_ctx.sessions.txn_identity(txn_ctx.session_id); - return Some(match_ops::match_query(state, identity, database_id, sql, txn_id).await); + let read = crate::control::server::graph_dispatch::GraphRead { + txn_id, + linearizable: txn_ctx.linearizable_reads(), + }; + return Some(match_ops::match_query(state, identity, database_id, sql, read).await); } // Graph-overlay statements (GRAPH INSERT/DELETE EDGE, GRAPH LABEL/UNLABEL, // GRAPH TRAVERSE/NEIGHBORS/PATH, GRAPH ALGO, GRAPH RAG FUSION, SHOW GRAPH - // STATS) parse into typed `GraphStmt` variants. In the pgwire router these - // were dispatched from the typed AST by the `dsl` string router (last). - // Recognizing them here on the typed path preserves that: `dispatch_graph` - // returns `Some` for the graph-overlay variants and `None` otherwise. + // STATS) parse into typed `GraphStmt` variants. `dispatch_graph` returns `Some` for the + // graph-overlay variants and `None` otherwise. if let NodedbStatement::Graph(_) = &stmt { // The graph-overlay handlers thread the session's transaction context // through `txn_ctx`: single-hop reads (Neighbors/Hop) resolve the diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/helpers.rs b/nodedb/src/control/server/shared/ddl/neutral/router/helpers.rs index b57b4dded..10fe682a6 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/helpers.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/helpers.rs @@ -10,10 +10,6 @@ use crate::types::DatabaseId; /// Existence check backing the `CreateCollection` `if_not_exists: true` /// short-circuit above. -/// -/// Relocated verbatim from the pgwire `router::ast::exists::collection_exists` -/// helper (now deleted, along with the pgwire guard arms that were its only -/// callers). pub(super) fn collection_exists( state: &SharedState, identity: &AuthenticatedIdentity, @@ -26,9 +22,6 @@ pub(super) fn collection_exists( } /// Extract the single-quoted collection argument from `SELECT LAST_VALUES('coll')`. -/// -/// Mirrors the pgwire router's `extract_quoted_arg(sql, "LAST_VALUES(")` exactly -/// so the parse behaviour stays byte-identical. pub(super) fn extract_last_values_arg(sql: &str) -> Option { let prefix = "LAST_VALUES("; let pos = find_ascii_case_insensitive(sql, prefix)?; @@ -39,8 +32,6 @@ pub(super) fn extract_last_values_arg(sql: &str) -> Option { } /// Extract `('collection', series_id)` from a `SELECT LAST_VALUE(...)` call. -/// -/// Mirrors the pgwire router's `extract_lv_args` exactly. pub(super) fn extract_last_value_args(sql: &str) -> Option<(String, u64)> { let pos = find_ascii_case_insensitive(sql, "LAST_VALUE(")?; let after = &sql[pos + 11..]; diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/string_admin.rs b/nodedb/src/control/server/shared/ddl/neutral/router/string_admin.rs index 688e94664..dc1e6205e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/string_admin.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/string_admin.rs @@ -30,62 +30,54 @@ pub(super) async fn try_string( // String-recognized user/role families. `DROP USER` parses into a typed // `AuthStmt::DropUser` that carries no `if_exists` flag (so it mishandles // `DROP USER IF EXISTS`), and `CREATE ROLE` / `DROP ROLE` do not parse into - // any typed variant at all — the pgwire router dispatched all three from the - // raw token slice. Replicate that exactly here, before the parse gate, so - // the token-based `strip_if_exists` / `strip_if_not_exists` handling and the - // syntax messages stay byte-identical. + // any typed variant at all — the router dispatches all three from the + // raw token slice, before the parse gate, with token-based + // `strip_if_exists` / `strip_if_not_exists` handling. if upper.starts_with("DROP USER ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(user::drop_user(state, identity, &parts)); + return Some(user::drop_user(state, identity, &parts).await); } if upper.starts_with("CREATE ROLE ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(role::create_role(state, identity, &parts)); + return Some(role::create_role(state, identity, &parts).await); } if upper.starts_with("DROP ROLE ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(role::drop_role(state, identity, &parts)); + return Some(role::drop_role(state, identity, &parts).await); } // Service accounts. These statements do not parse into any typed AST - // variant — the pgwire router dispatched all three from the raw token - // slice by string prefix. Replicate that exactly here, before the parse - // gate, so the token-based `IF [NOT] EXISTS` stripping and syntax messages - // stay byte-identical. + // variant — the router dispatches all three from the raw token slice by + // string prefix, before the parse gate, with token-based + // `IF [NOT] EXISTS` stripping. if upper.starts_with("CREATE SERVICE ACCOUNT ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(service_account::create_service_account( - state, identity, &parts, - )); + return Some(service_account::create_service_account(state, identity, &parts).await); } if upper.starts_with("DROP SERVICE ACCOUNT ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(service_account::drop_service_account( - state, identity, &parts, - )); + return Some(service_account::drop_service_account(state, identity, &parts).await); } if upper.starts_with("ALTER SERVICE ACCOUNT ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(service_account::alter_service_account_set_databases( - state, identity, &parts, - )); + return Some( + service_account::alter_service_account_set_databases(state, identity, &parts).await, + ); } // Auth-admin DDL families (API keys, auth-scoped API keys, auth user // management, blacklist). None of these parse into any typed AST variant — - // the pgwire admin router dispatched all of them by string prefix from the - // raw token slice. Replicate that exactly here, before the parse gate, so - // the prefix recognition and syntax messages stay byte-identical. The - // `BLACKLIST ` prefix intentionally precedes the (non-migrated) emergency - // `BLACKLIST AUTH USERS WHERE` handler exactly as it did in the pgwire admin - // router, so the shadowing behavior is unchanged. + // the router dispatches all of them by string prefix from the raw token + // slice, before the parse gate. The `BLACKLIST ` prefix intentionally + // precedes the emergency `BLACKLIST AUTH USERS WHERE` handler, which it + // shadows. if upper.starts_with("CREATE API KEY ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(apikey::create_api_key(state, identity, &parts)); + return Some(apikey::create_api_key(state, identity, &parts).await); } if upper.starts_with("REVOKE API KEY ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(apikey::revoke_api_key(state, identity, &parts)); + return Some(apikey::revoke_api_key(state, identity, &parts).await); } if upper.starts_with("LIST API KEYS") { let parts: Vec<&str> = sql.split_whitespace().collect(); @@ -133,11 +125,10 @@ pub(super) async fn try_string( } // Tenant management. `CREATE TENANT`, `DROP TENANT`, and `PURGE TENANT` - // parse into no typed AST variant — the pgwire auth router dispatched all - // three by string prefix from the raw token slice. Replicate that exactly - // here, before the parse gate, so the `IF [NOT] EXISTS` stripping and - // syntax messages stay byte-identical. `PURGE TENANT` dispatches an async - // Data Plane meta op. + // parse into no typed AST variant — the router dispatches all three by + // string prefix from the raw token slice, before the parse gate, with + // `IF [NOT] EXISTS` stripping. `PURGE TENANT` dispatches an async Data + // Plane meta op. // // `ALTER TENANT ` is ambiguous: `ALTER TENANT SET QUOTA ...` // (this string form) and `ALTER TENANT IN DATABASE SET QUOTA @@ -152,13 +143,11 @@ pub(super) async fn try_string( // `SHOW TENANT USAGE` / `SHOW TENANT QUOTA` (bare, no `IN DATABASE`) are // NOT recognized here: the typed `ddl_ast` tenant parser never returns // `None` for `SHOW TENANT USAGE|QUOTA...` — every such input resolves to - // either the typed `IN DATABASE` variant or a `42601` parse error. Their - // pgwire string handlers were therefore confirmed dead code and deleted, - // not migrated; adding a neutral string prefix for them would make that - // dead code reachable and break parity. + // either the typed `IN DATABASE` variant or a `42601` parse error. A + // neutral string prefix for them will be unreachable and break parity. if upper.starts_with("CREATE TENANT ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(tenant::create_tenant(state, identity, &parts)); + return Some(tenant::create_tenant(state, identity, &parts).await); } if upper.starts_with("ALTER TENANT ") { let parts: Vec<&str> = sql.split_whitespace().collect(); @@ -166,12 +155,12 @@ pub(super) async fn try_string( && parts[3].eq_ignore_ascii_case("IN") && parts[4].eq_ignore_ascii_case("DATABASE"); if !is_in_database_form { - return Some(tenant::alter_tenant(state, identity, database_id, &parts)); + return Some(tenant::alter_tenant(state, identity, database_id, &parts).await); } } if upper.starts_with("DROP TENANT ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(tenant::drop_tenant(state, identity, &parts)); + return Some(tenant::drop_tenant(state, identity, &parts).await); } if upper.starts_with("PURGE TENANT ") { let parts: Vec<&str> = sql.split_whitespace().collect(); @@ -179,15 +168,12 @@ pub(super) async fn try_string( } // Emergency & incident response DDL. `EMERGENCY LOCKDOWN` / `EMERGENCY - // UNLOCK` parse into no typed AST variant — the pgwire admin router - // dispatched both by string prefix from the raw token slice. Replicate that - // exactly here, before the parse gate, so the prefix recognition and syntax - // messages stay byte-identical. `BLACKLIST AUTH USERS WHERE …` is likewise - // string-recognized, but the `BLACKLIST ` prefix above already claims it - // (exactly as it shadowed the pgwire emergency handler, which ran only after - // this neutral router). This guard is therefore intentionally kept after the - // `BLACKLIST ` guard so `bulk_blacklist` remains unreachable — preserving the - // dead-but-present state verbatim. + // UNLOCK` parse into no typed AST variant — the router dispatches both by + // string prefix from the raw token slice, before the parse gate. + // `BLACKLIST AUTH USERS WHERE …` is likewise string-recognized, but the + // `BLACKLIST ` prefix above already claims it. This guard is therefore + // intentionally kept after the `BLACKLIST ` guard, so `bulk_blacklist` + // stays unreachable. if upper.starts_with("EMERGENCY LOCKDOWN") { let parts: Vec<&str> = sql.split_whitespace().collect(); return Some(emergency_ddl::emergency_lockdown(state, identity, &parts)); @@ -202,10 +188,9 @@ pub(super) async fn try_string( } // System-level settings: `ALTER SYSTEM SET = `. Parses into - // no typed AST variant — the pgwire auth router dispatched it by string - // prefix from the raw token slice. Replicate that exactly here, before the - // parse gate, so the prefix recognition and the `parts`-based field / value - // extraction stay byte-identical. + // no typed AST variant — the router dispatches it by string prefix from + // the raw token slice, before the parse gate, with `parts`-based field / + // value extraction. if upper.starts_with("ALTER SYSTEM ") { let parts: Vec<&str> = sql.split_whitespace().collect(); return Some(system_ddl::alter_system(state, identity, &parts)); diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs b/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs index ec80f408f..52653aa95 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/string_engine_ops.rs @@ -42,20 +42,17 @@ pub(super) async fn try_string( txn_ctx: &DmlTxnCtx<'_>, ) -> Option, DdlError>> { // Engine-ops SQL functions and DDL. None of these are dispatched from a - // typed AST arm — the pgwire engine_ops router recognized all of them by - // string prefix from the raw SQL (these keywords do not appear in the DDL - // AST grammar, so `ddl_ast::parse` returns `None` for them). Replicate that - // exactly here, before the parse gate, so the prefix recognition, guard - // ordering, and syntax messages stay byte-identical. The three vector + // typed AST arm — the router recognizes all of them by string prefix from + // the raw SQL, before the parse gate (these keywords do not appear in the + // DDL AST grammar, so `ddl_ast::parse` returns `None` for them). The three vector // model / metadata forms (`ALTER COLLECTION … SET VECTOR METADATA ON`, // `SHOW VECTOR MODELS`, `SELECT VECTOR_METADATA(…)`) are routed by the // string-prefix arms above (alongside the vector-index lifecycle forms). // // `CREATE TIMESERIES` / `ALTER TIMESERIES` / `REWRITE PARTITIONS` are // routed here, but `SHOW PARTITIONS ` is intentionally NOT — it is already - // claimed by the consumer-group handler above (which ran before engine_ops - // on the pgwire path too), so the timeseries `show_partitions` handler stays - // shadowed exactly as it was. + // claimed by the consumer-group handler above, so the timeseries + // `show_partitions` handler stays shadowed. // Weighted random selection — anchored to the statement prefix so a // doc-object UPSERT body carrying the token never reaches this arm. @@ -144,16 +141,11 @@ pub(super) async fn try_string( // (SHOW PARTITIONS is shadowed by consumer_group above, as noted.) if upper.starts_with("CREATE TIMESERIES ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(timeseries::create_timeseries( - state, - identity, - &parts, - database_id, - )); + return Some(timeseries::create_timeseries(state, identity, &parts, database_id).await); } if upper.starts_with("ALTER TIMESERIES ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(timeseries::alter_timeseries(state, identity, &parts)); + return Some(timeseries::alter_timeseries(state, identity, &parts).await); } if upper.starts_with("REWRITE PARTITIONS ") { let parts: Vec<&str> = sql.split_whitespace().collect(); @@ -181,12 +173,10 @@ pub(super) async fn try_string( // Materialized views (HTAP). `REFRESH MATERIALIZED VIEW` parses into no typed // AST variant, and `SHOW MATERIALIZED VIEWS` parses into a typed - // `StreamViewStmt::ShowMaterializedViews` but the pgwire admin router - // dispatched it from the raw token slice by string prefix (the `SHOW + // `StreamViewStmt::ShowMaterializedViews` but the router dispatches it from + // the raw token slice by string prefix, before the parse gate (the `SHOW // MATERIALIZED VIEW` prefix, trailing-space-less, captures both the plural - // `SHOW MATERIALIZED VIEWS` and the bare-singular input). Replicate both here, - // before the parse gate, so the prefix recognition and the `parts`-based name - // extraction stay byte-identical. `CREATE` / `DROP MATERIALIZED VIEW` are + // `SHOW MATERIALIZED VIEWS` and the bare-singular input). `CREATE` / `DROP MATERIALIZED VIEW` are // handled in the typed match below (they parse into typed StreamView variants). if upper.starts_with("REFRESH MATERIALIZED VIEW") { return Some( @@ -205,12 +195,10 @@ pub(super) async fn try_string( // Continuous aggregates (timeseries). `SHOW CONTINUOUS AGGREGATES [FOR // ]` parses into a typed `StreamViewStmt::ShowContinuousAggregates` - // but the pgwire admin router dispatched it from the raw token slice by - // string prefix (the `SHOW CONTINUOUS AGGREGATE` prefix, trailing-space-less, - // captures both the plural `SHOW CONTINUOUS AGGREGATES` and the bare-singular - // input). Replicate that here, before the parse gate, so the prefix - // recognition and the `parts`-based `FOR ` extraction stay - // byte-identical. `CREATE` / `DROP CONTINUOUS AGGREGATE` are handled in the + // but the router dispatches it from the raw token slice by string prefix, + // before the parse gate (the `SHOW CONTINUOUS AGGREGATE` prefix, + // trailing-space-less, captures both the plural `SHOW CONTINUOUS + // AGGREGATES` and the bare-singular input). `CREATE` / `DROP CONTINUOUS AGGREGATE` are handled in the // typed match below (they parse into typed StreamView variants). if upper.starts_with("SHOW CONTINUOUS AGGREGATE") { let parts: Vec<&str> = sql.split_whitespace().collect(); @@ -220,12 +208,10 @@ pub(super) async fn try_string( } // CONVERT COLLECTION between storage modes. `CONVERT COLLECTION TO - // ` parses into no typed AST variant — the pgwire admin router - // dispatched it by string prefix from the raw SQL. Replicate that exactly - // here, before the parse gate, so the prefix recognition (the + // ` parses into no typed AST variant — the router dispatches it by + // string prefix from the raw SQL, before the parse gate (the // `CONVERT COLLECTION ` form plus the broader `CONVERT ... TO ...` form, in - // that `||`/`&&` precedence) and the parse / syntax messages stay - // byte-identical. + // that `||`/`&&` precedence). if upper.starts_with("CONVERT COLLECTION ") || upper.starts_with("CONVERT ") && upper.contains(" TO ") { @@ -233,13 +219,11 @@ pub(super) async fn try_string( } // Retention policies (timeseries). `SHOW RETENTION POLICIES` parses into a - // typed `PolicyStmt::ShowRetentionPolicies`, but the pgwire admin router - // dispatched it from the raw token slice by the `SHOW RETENTION POLIC` - // prefix (trailing-space-less, captures both the plural `SHOW RETENTION + // typed `PolicyStmt::ShowRetentionPolicies`, but the router dispatches it + // from the raw token slice by the `SHOW RETENTION POLIC` prefix, before the + // parse gate (trailing-space-less, captures both the plural `SHOW RETENTION // POLICIES` and the singular `SHOW RETENTION POLICY ON `). - // Replicate that exactly here, before the parse gate, so the prefix - // recognition and the `parts`-based `ON ` filter stay - // byte-identical. `CREATE` / `ALTER` / `DROP RETENTION POLICY` are handled in + // `CREATE` / `ALTER` / `DROP RETENTION POLICY` are handled in // the typed match below (they parse into typed Policy variants). if upper.starts_with("SHOW RETENTION POLIC") { let parts: Vec<&str> = sql.split_whitespace().collect(); @@ -252,15 +236,23 @@ pub(super) async fn try_string( } // DSL extensions (custom SQL-like surfaces). None of these are dispatched - // from a typed AST arm — the pgwire dsl router recognized all six by string - // prefix from the raw SQL. Replicate that exactly here, before the parse - // gate, so the prefix recognition and syntax messages stay byte-identical. - // `SEARCH ... USING FUSION` must precede the parse gate because it would + // from a typed AST arm — the router recognizes all six by string + // prefix from the raw SQL, before the parse gate. + // `SEARCH ... USING FUSION` must precede the parse gate because it will // otherwise parse into a typed graph statement and be captured by the graph // dispatch below. `SEARCH ... USING VECTOR(...)` never reaches here — it is // preprocessor-rewritten to a canonical `SELECT ... vector_distance(...)`. if upper.starts_with("SEARCH ") && upper.contains("USING FUSION") { - return Some(dsl::search_fusion(state, identity, database_id, sql).await); + return Some( + dsl::search_fusion( + state, + identity, + database_id, + sql, + txn_ctx.linearizable_reads(), + ) + .await, + ); } if upper.starts_with("CREATE VECTOR INDEX ") { return Some(dsl::create_vector_index(state, identity, database_id, sql).await); @@ -272,18 +264,12 @@ pub(super) async fn try_string( return Some(dsl::create_search_index(state, identity, database_id, sql).await); } if upper.starts_with("CREATE SPARSE INDEX ") { - return Some(dsl::create_sparse_index(state, identity, database_id, sql)); + return Some(dsl::create_sparse_index(state, identity, database_id, sql).await); } - // CREATE SPATIAL INDEX — string-recognized (no typed AST variant); the pgwire - // schema string router dispatched it from the raw SQL. Replicate that - // exactly here, before the parse gate. + // CREATE SPATIAL INDEX — string-recognized (no typed AST variant); the router + // dispatches it from the raw SQL, before the parse gate. if upper.starts_with("CREATE SPATIAL INDEX ") { - return Some(spatial::create_spatial_index( - state, - identity, - database_id, - sql, - )); + return Some(spatial::create_spatial_index(state, identity, database_id, sql).await); } if upper.starts_with("CRDT MERGE ") { if crdt_apply_forbidden_in_transaction(txn_ctx) { @@ -293,9 +279,8 @@ pub(super) async fn try_string( return Some(dsl::crdt_merge(state, identity, database_id, &parts).await); } // `SELECT crdt_state(...)` / `SELECT crdt_apply(...)` CRDT DSL functions — - // string-recognized (they parse into no typed DDL variant). The pgwire dsl - // string router recognized both by prefix from the raw SQL; replicate that - // exactly here, before the parse gate. + // string-recognized (they parse into no typed DDL variant). The router + // recognizes both by prefix from the raw SQL, before the parse gate. if upper.starts_with("SELECT CRDT_STATE(") || upper.starts_with("SELECT CRDT_STATE (") { return Some(crdt_ops::crdt_state(state, identity, database_id, sql).await); } @@ -333,18 +318,18 @@ pub(super) async fn try_string( // `DEFINE FIELD …` / `DEFINE EVENT …` / `REMOVE EVENT …` — // string-recognized (no typed DDL variant), before the parse gate. if upper.starts_with("DEFINE FIELD ") { - return Some(field_def::define_field(state, identity, database_id, sql)); + return Some(field_def::define_field(state, identity, database_id, sql).await); } if upper.starts_with("DEFINE EVENT ") { - return Some(field_def::define_event(state, identity, database_id, sql)); + return Some(field_def::define_event(state, identity, database_id, sql).await); } if upper.starts_with("REMOVE EVENT ") { - return Some(field_def::remove_event(state, identity, database_id, sql)); + return Some(field_def::remove_event(state, identity, database_id, sql).await); } // `EXPLAIN TIERS ON [RANGE …]` — string-recognized (no typed - // DDL variant); the pgwire admin string router dispatched it from the raw - // token slice. Replicate that exactly here, before the parse gate. + // DDL variant); the router dispatches it from the raw token slice, before + // the parse gate. if upper.starts_with("EXPLAIN TIERS ") { let parts: Vec<&str> = sql.split_whitespace().collect(); return Some(explain_tiers::explain_tiers( diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/string_introspection.rs b/nodedb/src/control/server/shared/ddl/neutral/router/string_introspection.rs index f613a552a..8dd590199 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/string_introspection.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/string_introspection.rs @@ -37,18 +37,15 @@ pub(super) async fn try_string( // SHOW GRANTS. `SHOW USERS`, `SHOW GRANTS`, and `SHOW AUDIT…` parse into // typed AST variants (`AuthStmt::ShowUsers` / `ShowGrants`, // `MiscStmt::ShowAuditLog`) and bare `SHOW TENANTS` into - // `DatabaseStmt::ShowTenants`, but the pgwire typed-AST path has no arm for - // any of them — they fell through to the admin/observability string router, - // which dispatched them by prefix from the raw token slice. `SHOW ROLES`, + // `DatabaseStmt::ShowTenants`, but the router dispatches all of them by + // prefix from the raw token slice, before the parse gate. `SHOW ROLES`, // `SHOW SESSION`, and `EXPORT AUDIT` parse into no typed DDL variant. - // Replicate the string dispatch exactly here, before the parse gate, so the - // prefix recognition and the `parts`-based extraction stay byte-identical. // `SHOW SESSIONS` is excluded here (see the `session_admin::show_sessions` - // arm above, which is now checked first) so the two never race. + // arm above, which is checked first) so the two never race. // // Audit-log truncation is rejected: entries are pruned only by the retention // policy. Recognized by string prefix before the parse gate so the message - // stays byte-identical regardless of how the tail parses. + // is the same regardless of how the tail parses. if upper.starts_with("TRUNCATE AUDIT") || upper.starts_with("DELETE AUDIT") || upper.starts_with("CLEAR AUDIT") @@ -63,7 +60,7 @@ pub(super) async fn try_string( } // Exact-match only. Filtered forms (`SHOW TENANTS WITH NAME `, // `SHOW TENANT `) are parsed into typed variants and routed through - // the typed match below; a prefix match here would silently drop the filter + // the typed match below; a prefix match here will silently drop the filter // and list every tenant. if upper == "SHOW TENANTS" { return Some(inspect::show_tenants(state, identity)); @@ -144,10 +141,8 @@ pub(super) async fn try_string( // Impersonation & delegation: IMPERSONATE AUTH USER, STOP IMPERSONATION, // DELEGATE AUTH USER, REVOKE DELEGATION, SHOW DELEGATIONS. None of these - // parse into any typed AST variant — the pgwire admin router dispatched - // all five by string prefix from the raw token slice. Replicate that - // exactly here, before the parse gate, so the prefix recognition and the - // `parts`-based extraction / syntax messages stay byte-identical. + // parse into any typed AST variant — the router dispatches all five by + // string prefix from the raw token slice, before the parse gate. if upper.starts_with("IMPERSONATE AUTH USER ") { let parts: Vec<&str> = sql.split_whitespace().collect(); return Some(impersonation::impersonate(state, identity, &parts)); @@ -171,12 +166,9 @@ pub(super) async fn try_string( // Session management: SHOW SESSIONS, KILL SESSION, KILL USER SESSIONS, // VERIFY AUDIT CHAIN. None of these parse into any typed AST variant — - // the pgwire admin router dispatched all four by string prefix from the - // raw token slice. Replicate that exactly here, before the parse gate, so - // the prefix recognition and the `parts`-based extraction / syntax - // messages stay byte-identical. `SHOW SESSIONS` is matched here (before - // the observability `SHOW SESSION` prefix below), mirroring the pgwire - // admin router's precedence over the pgwire observability router; the + // the router dispatches all four by string prefix from the raw token + // slice, before the parse gate. `SHOW SESSIONS` is matched here (before + // the observability `SHOW SESSION` prefix below); the // `SHOW SESSION` guard below already excludes `SHOW SESSIONS` explicitly, // so the two never race regardless of which is checked first. if upper.starts_with("SHOW SESSIONS") { @@ -198,11 +190,9 @@ pub(super) async fn try_string( // Administrative observability: SHOW SERVER STATS / SHOW STATS / SHOW // METRICS / SHOW MEMORY. None of these parse into a typed DDL AST variant — - // the pgwire admin observability router recognized all four by the - // exact-or-trailing-space prefix from the raw SQL. Replicate that exactly - // here, before the parse gate, so the recognition (and the `SHOW SERVER - // STATS` / `SHOW STATS` shared handler) stays byte-identical. `SHOW SERVER - // STATS` is checked before `SHOW STATS` exactly as the pgwire router did. + // the router recognizes all four by the exact-or-trailing-space prefix + // from the raw SQL, before the parse gate. `SHOW SERVER STATS` and + // `SHOW STATS` share a handler, and `SHOW SERVER STATS` is checked first. if upper == "SHOW SERVER STATS" || upper.starts_with("SHOW SERVER STATS ") { return Some(observability::show_server_stats(state, identity)); } @@ -217,14 +207,11 @@ pub(super) async fn try_string( } // Permission / scope introspection: EXPLAIN PERMISSION / EXPLAIN SCOPE. - // Neither parses into a typed DDL AST variant — the pgwire admin router - // recognized both by string prefix from the raw token slice. Replicate that - // exactly here, before the parse gate, so the prefix recognition and the - // `parts`-based extraction / syntax messages stay byte-identical. The - // pgwire wire path reaches these full-`EXPLAIN …` statements through the + // Neither parses into a typed DDL AST variant — the router recognizes + // both by string prefix from the raw token slice, before the parse gate. + // The wire path reaches these full-`EXPLAIN …` statements through the // DDL dispatch (native / http always; pgwire only for the non-`EXPLAIN ` - // full-SQL dispatch), so recognizing them here preserves behavior; the - // `EXPLAIN ` handler strips the leading `EXPLAIN ` and never yields a + // full-SQL dispatch). The `EXPLAIN ` handler strips the leading `EXPLAIN ` and never yields a // `PERMISSION …` / `SCOPE …` prefix, so it is unaffected. if upper.starts_with("EXPLAIN PERMISSION ") { let parts: Vec<&str> = sql.split_whitespace().collect(); @@ -237,11 +224,9 @@ pub(super) async fn try_string( // Usage metering: DEFINE METERING DIMENSION, SHOW USAGE FOR TENANT, EXPORT // USAGE, SHOW USAGE, SHOW QUOTA. None of these parse into a typed DDL AST - // variant — the pgwire admin router recognized all five by string prefix - // from the raw token slice. Replicate that exactly here, before the parse - // gate, so the prefix recognition and the `parts`-based extraction / syntax - // messages stay byte-identical. Guard ordering (SHOW USAGE FOR TENANT and - // EXPORT USAGE before the broader SHOW USAGE) mirrors the pgwire router. + // variant — the router recognizes all five by string prefix from the raw + // token slice, before the parse gate. Guard ordering: SHOW USAGE FOR + // TENANT and EXPORT USAGE come before the broader SHOW USAGE. if upper.starts_with("DEFINE METERING DIMENSION ") { let parts: Vec<&str> = sql.split_whitespace().collect(); return Some(metering_ddl::define_dimension(state, identity, &parts)); @@ -267,11 +252,11 @@ pub(super) async fn try_string( // has an `S`. if upper.starts_with("DEFINE QUOTA ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(quota_ddl::define_quota(state, identity, &parts)); + return Some(quota_ddl::define_quota(state, identity, &parts).await); } if upper.starts_with("DROP QUOTA ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(quota_ddl::drop_quota(state, identity, &parts)); + return Some(quota_ddl::drop_quota(state, identity, &parts).await); } if upper.starts_with("SHOW QUOTAS") { let parts: Vec<&str> = sql.split_whitespace().collect(); @@ -280,10 +265,8 @@ pub(super) async fn try_string( // Organization management. None of `CREATE ORG`, `ALTER ORG`, `DROP ORG`, // `SHOW ORGS`, or `SHOW MEMBERS OF ORG` parse into any typed AST variant — - // the pgwire admin router dispatched all of them by string prefix from the - // raw token slice. Replicate that exactly here, before the parse gate, so - // the prefix recognition and the `parts`-based extraction / syntax messages - // stay byte-identical. + // the router dispatches all of them by string prefix from the raw token + // slice, before the parse gate. if upper.starts_with("CREATE ORG ") || upper.starts_with("ALTER ORG ") || upper.starts_with("DROP ORG ") @@ -304,15 +287,11 @@ pub(super) async fn try_string( // SHOW MY SCOPES, SHOW SCOPES FOR, SHOW SCOPE GRANTS, SHOW SCOPE(S). None of // these parse into any typed AST variant — `GRANT SCOPE` / `REVOKE SCOPE` // are explicitly excluded from the typed grant parser (returning `None`), - // and the rest have no grammar at all — so the pgwire admin router - // dispatched all of them by string prefix from the raw token slice. - // Replicate that exactly here, before the parse gate, so the prefix - // recognition and the `parts`-based extraction / syntax messages stay - // byte-identical. Guard ordering mirrors the pgwire admin router: `SHOW MY - // SCOPES` and `SHOW SCOPES FOR ` are matched before the broader `SHOW SCOPE - // GRANTS` / `SHOW SCOPE` pair (nothing between them in the pgwire router - // claimed a scope input, so grouping them here is behavior-preserving), and - // `SHOW SCOPE GRANTS` is checked before the `SHOW SCOPE` catch-all. + // and the rest have no grammar at all — so the router + // dispatches all of them by string prefix from the raw token slice, before + // the parse gate. Guard ordering: `SHOW MY SCOPES` and `SHOW SCOPES FOR ` + // are matched before the broader `SHOW SCOPE GRANTS` / `SHOW SCOPE` pair, + // and `SHOW SCOPE GRANTS` is checked before the `SHOW SCOPE` catch-all. if upper.starts_with("DEFINE SCOPE ") { let parts: Vec<&str> = sql.split_whitespace().collect(); return Some(scope_ddl::define_scope(state, identity, &parts)); @@ -323,11 +302,11 @@ pub(super) async fn try_string( } if upper.starts_with("GRANT SCOPE ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(scope_ddl::grant_scope(state, identity, &parts)); + return Some(scope_ddl::grant_scope(state, identity, &parts).await); } if upper.starts_with("REVOKE SCOPE ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(scope_ddl::revoke_scope(state, identity, &parts)); + return Some(scope_ddl::revoke_scope(state, identity, &parts).await); } if upper.starts_with("ALTER SCOPE ") { let parts: Vec<&str> = sql.split_whitespace().collect(); @@ -343,7 +322,7 @@ pub(super) async fn try_string( } if upper.starts_with("RENEW SCOPE ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(scope_ddl::renew_scope(state, identity, &parts)); + return Some(scope_ddl::renew_scope(state, identity, &parts).await); } if upper.starts_with("SHOW SCOPE GRANTS") { let parts: Vec<&str> = sql.split_whitespace().collect(); @@ -356,17 +335,15 @@ pub(super) async fn try_string( // Collection introspection: DESCRIBE / `\D `, // UNDROP COLLECTION|TABLE, SHOW COLLECTIONS, SHOW INDEXES|INDEX. All four - // parse into typed `CollectionStmt` variants, but the pgwire schema string - // router dispatched them by string prefix from the raw token slice, using + // parse into typed `CollectionStmt` variants, but the router + // dispatches them by string prefix from the raw token slice, using // `parts`-based name / filter extraction and the `\D` alias that the typed // parser does not reproduce (`\D ` never parses into // `DescribeCollection`; bare `\D` parses into `ShowCollections`; the // `SHOW INDEXES` typed `collection` field is `parts[2]`, not the handler's - // `parts[3]` filter). Replicate the string dispatch exactly here, before the - // parse gate, so the prefix recognition, `parts` extraction, and syntax - // messages stay byte-identical. `DESCRIBE SEQUENCE` is excluded so it falls - // through to the typed `DescribeSequence` arm (claimed by the sequence - // family), exactly as it was before this block existed. + // `parts[3]` filter). The string dispatch runs before the parse gate. + // `DESCRIBE SEQUENCE` is excluded so it falls through to the typed + // `DescribeSequence` arm (claimed by the sequence family). if (upper.starts_with("DESCRIBE ") && !upper.starts_with("DESCRIBE SEQUENCE")) || upper.starts_with("\\D ") { @@ -380,12 +357,7 @@ pub(super) async fn try_string( } if upper.starts_with("UNDROP COLLECTION ") || upper.starts_with("UNDROP TABLE ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(collection::undrop_collection( - state, - identity, - &parts, - database_id, - )); + return Some(collection::undrop_collection(state, identity, &parts, database_id).await); } if upper == "SHOW COLLECTIONS" || upper.starts_with("SHOW COLLECTIONS") { return Some(collection::show_collections(state, identity, database_id)); diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/string_schema.rs b/nodedb/src/control/server/shared/ddl/neutral/router/string_schema.rs index 53d24cb8b..cd635fe5e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/string_schema.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/string_schema.rs @@ -24,16 +24,14 @@ pub(super) async fn try_string( ) -> Option, DdlError>> { // Stored procedures. None of `CREATE [OR REPLACE] PROCEDURE`, `DROP // PROCEDURE`, `SHOW PROCEDURES`, or `CALL (...)` parse into any - // typed AST variant — the pgwire router dispatched all of them by string - // prefix from the raw SQL / token slice. Replicate that exactly here, before - // the parse gate, so the prefix recognition and syntax messages stay - // byte-identical. + // typed AST variant — the router dispatches all of them by string + // prefix from the raw SQL / token slice, before the parse gate. if upper.starts_with("CREATE OR REPLACE PROCEDURE ") || upper.starts_with("CREATE PROCEDURE ") { - return Some(procedure::create_procedure(state, identity, sql)); + return Some(procedure::create_procedure(state, identity, sql).await); } if upper.starts_with("DROP PROCEDURE ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(procedure::drop_procedure(state, identity, &parts)); + return Some(procedure::drop_procedure(state, identity, &parts).await); } if upper == "SHOW PROCEDURES" || upper.starts_with("SHOW PROCEDURES") { return Some(procedure::show_procedures(state, identity, database_id)); @@ -44,28 +42,27 @@ pub(super) async fn try_string( // User-defined functions. None of `CREATE [OR REPLACE] [AGGREGATE] FUNCTION`, // `DROP FUNCTION`, `ALTER FUNCTION`, or `SHOW FUNCTIONS` parse into any typed - // AST variant — the pgwire router dispatched all of them by string prefix - // from the raw SQL / token slice. Replicate that exactly here, before the - // parse gate, so the prefix recognition, `LANGUAGE WASM` branch, and syntax - // messages stay byte-identical. + // AST variant — the router dispatches all of them by string prefix + // from the raw SQL / token slice, before the parse gate (including the + // `LANGUAGE WASM` branch). if upper.starts_with("CREATE OR REPLACE AGGREGATE FUNCTION ") || upper.starts_with("CREATE AGGREGATE FUNCTION ") { - return Some(function::create_wasm_aggregate(state, identity, sql)); + return Some(function::create_wasm_aggregate(state, identity, sql).await); } if upper.starts_with("CREATE OR REPLACE FUNCTION ") || upper.starts_with("CREATE FUNCTION ") { if upper.contains("LANGUAGE WASM") { - return Some(function::create_wasm_function(state, identity, sql)); + return Some(function::create_wasm_function(state, identity, sql).await); } - return Some(function::create_function(state, identity, sql)); + return Some(function::create_function(state, identity, sql).await); } if upper.starts_with("DROP FUNCTION ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(function::drop_function(state, identity, &parts)); + return Some(function::drop_function(state, identity, &parts).await); } if upper.starts_with("ALTER FUNCTION ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(function::alter_function(state, identity, &parts)); + return Some(function::alter_function(state, identity, &parts).await); } if upper == "SHOW FUNCTIONS" || upper.starts_with("SHOW FUNCTIONS") { return Some(function::show_functions(state, identity, database_id)); @@ -74,19 +71,18 @@ pub(super) async fn try_string( // Constraint DDL. `ALTER COLLECTION ... ADD CONSTRAINT` / `ADD TRANSITION // CHECK` do not parse into any typed AST variant (the `parse_alter_operation` // path returns `None` for them, so `ddl_ast::parse` yields `None`), and - // `DROP CONSTRAINT` / `SHOW CONSTRAINTS ON` were dispatched by string prefix - // from the pgwire collaborative router. Replicate that exactly here, before - // the parse gate, so the prefix recognition and syntax messages stay - // byte-identical. Guard ordering (TRANSITIONS before the general CHECK arm, - // which excludes both TRANSITIONS and TRANSITION CHECK) is preserved verbatim. + // `DROP CONSTRAINT` / `SHOW CONSTRAINTS ON` are dispatched by string prefix + // here, before the parse gate. Guard ordering matters: TRANSITIONS comes + // before the general CHECK arm, which excludes both TRANSITIONS and + // TRANSITION CHECK. if upper.starts_with("ALTER COLLECTION ") && upper.contains("ADD CONSTRAINT") && upper.contains("TRANSITIONS") { - return Some(constraint::add_state_constraint(state, identity, sql)); + return Some(constraint::add_state_constraint(state, identity, sql).await); } if upper.starts_with("ALTER COLLECTION ") && upper.contains("ADD TRANSITION CHECK") { - return Some(constraint::add_transition_check(state, identity, sql)); + return Some(constraint::add_transition_check(state, identity, sql).await); } if upper.starts_with("ALTER COLLECTION ") && upper.contains("ADD CONSTRAINT") @@ -94,57 +90,56 @@ pub(super) async fn try_string( && !upper.contains("TRANSITIONS") && !upper.contains("TRANSITION CHECK") { - return Some(constraint::add_check_constraint(state, identity, sql)); + return Some(constraint::add_check_constraint(state, identity, sql).await); } if upper.starts_with("SHOW CONSTRAINTS ON ") { return Some(constraint::show_constraints(state, identity, sql)); } if upper.starts_with("DROP CONSTRAINT ") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(constraint::drop_constraint(state, identity, &parts)); + return Some(constraint::drop_constraint(state, identity, &parts).await); } // Permission tree management. `ALTER COLLECTION … SET PERMISSION_TREE` and // `… DROP PERMISSION_TREE` do not parse into any typed AST variant (the // `parse_alter_operation` path returns `None` for both, so `ddl_ast::parse` - // yields `None`) — the pgwire collaborative router dispatched both from the - // raw SQL by string prefix + `contains`. Replicate that exactly here, before - // the parse gate, so the recognition and syntax messages stay byte-identical. + // yields `None`) — the router dispatches both from the raw SQL by string + // prefix + `contains`, before the parse gate. if upper.starts_with("ALTER COLLECTION ") && upper.contains("SET PERMISSION_TREE") { - return Some(permission_tree::set_permission_tree(state, identity, sql).await); + return Some(permission_tree::set_permission_tree(state, identity, database_id, sql).await); } if upper.starts_with("ALTER COLLECTION ") && upper.contains("DROP PERMISSION_TREE") { - return Some(permission_tree::drop_permission_tree(state, identity, sql).await); + return Some( + permission_tree::drop_permission_tree(state, identity, database_id, sql).await, + ); } // Period lock management. `ALTER COLLECTION … ADD PERIOD LOCK` and `… DROP // PERIOD LOCK` do not parse into any typed AST variant (the // `parse_alter_operation` path returns `None` for both, so `ddl_ast::parse` - // yields `None`) — the pgwire collaborative router dispatched both from the - // raw SQL by string prefix + `contains`. Replicate that exactly here, before - // the parse gate, so the recognition and syntax messages stay byte-identical. + // yields `None`) — the router dispatches both from the raw SQL by string + // prefix + `contains`, before the parse gate. if upper.starts_with("ALTER COLLECTION ") && upper.contains("ADD PERIOD LOCK") { - return Some(period_lock::add_period_lock(state, identity, sql)); + return Some(period_lock::add_period_lock(state, identity, sql).await); } if upper.starts_with("ALTER COLLECTION ") && upper.contains("DROP PERIOD LOCK") { let parts: Vec<&str> = sql.split_whitespace().collect(); - return Some(period_lock::drop_period_lock(state, identity, &parts)); + return Some(period_lock::drop_period_lock(state, identity, &parts).await); } // TYPEGUARD DDL. None of these statements are dispatched from a typed AST - // variant — the pgwire router recognized all of them by string prefix from + // variant — the router recognizes all of them by string prefix from // the raw SQL (the `SHOW TYPEGUARD…` prefix does parse into a typed - // `MiscStmt::ShowTypeGuards`, but the pgwire string dispatch claimed it - // before the parse gate). Replicate that exactly here, before the parse - // gate, so the prefix recognition and syntax messages stay byte-identical. + // `MiscStmt::ShowTypeGuards`, but string dispatch claims it before the + // parse gate). if upper.starts_with("CREATE TYPEGUARD ") || upper.starts_with("CREATE OR REPLACE TYPEGUARD ") { - return Some(typeguard::create_typeguard(state, identity, sql)); + return Some(typeguard::create_typeguard(state, identity, sql).await); } if upper.starts_with("ALTER TYPEGUARD ") { - return Some(typeguard::alter_typeguard(state, identity, sql)); + return Some(typeguard::alter_typeguard(state, identity, sql).await); } if upper.starts_with("DROP TYPEGUARD ") { - return Some(typeguard::drop_typeguard(state, identity, sql)); + return Some(typeguard::drop_typeguard(state, identity, sql).await); } if upper.starts_with("VALIDATE TYPEGUARD ON ") { return Some(typeguard::validate_typeguard(state, identity, sql).await); diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/string_streaming.rs b/nodedb/src/control/server/shared/ddl/neutral/router/string_streaming.rs index 27388da67..490120346 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/string_streaming.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/string_streaming.rs @@ -4,6 +4,7 @@ //! consumer groups, topics, stream/topic consumption, and pub/sub subscribe. use crate::control::security::identity::AuthenticatedIdentity; +use crate::control::server::shared::session::DmlTxnCtx; use crate::control::state::SharedState; use crate::types::DatabaseId; @@ -22,14 +23,15 @@ pub(super) async fn try_string( sql: &str, upper: &str, database_id: DatabaseId, + txn_ctx: &DmlTxnCtx<'_>, ) -> Option, DdlError>> { // Schedule SHOW. `SHOW SCHEDULE HISTORY ` parses into a typed // `AutomationStmt::ShowScheduleHistory` and `SHOW SCHEDULES` into - // `AutomationStmt::ShowSchedules`, but the pgwire router dispatched both from - // the raw token slice by string prefix (the `SHOW SCHEDULE` prefix also - // captures the bare-singular `SHOW SCHEDULE` input, which parses into no - // typed variant). Replicate that exactly here, before the parse gate, so the - // prefix recognition and `parts.get(3)` name extraction stay byte-identical. + // `AutomationStmt::ShowSchedules`, but the router dispatches both from + // the raw token slice by string prefix, before the parse gate (the + // `SHOW SCHEDULE` prefix also captures the bare-singular `SHOW SCHEDULE` + // input, which parses into no typed variant). The name comes from + // `parts.get(3)`. if upper.starts_with("SHOW SCHEDULE HISTORY ") { let parts: Vec<&str> = sql.split_whitespace().collect(); let name = parts.get(3).copied().unwrap_or(""); @@ -46,12 +48,11 @@ pub(super) async fn try_string( // Alert SHOW. `SHOW ALERT STATUS ` parses into a typed // `AutomationStmt::ShowAlertStatus` and `SHOW ALERTS` into - // `AutomationStmt::ShowAlerts`, but the pgwire admin router dispatched both - // from the raw token slice by string prefix (the `SHOW ALERT` prefix also - // captures the bare-singular `SHOW ALERT` input, which parses into - // `ShowAlerts`). Replicate that exactly here, before the parse gate, so the - // prefix recognition (STATUS checked first) and the `parts.get(4)` name - // extraction (name after `ON`) stay byte-identical. + // `AutomationStmt::ShowAlerts`, but the router dispatches both from the + // raw token slice by string prefix, before the parse gate (the + // `SHOW ALERT` prefix also captures the bare-singular `SHOW ALERT` input, + // which parses into `ShowAlerts`). STATUS is checked first, and the name + // comes from `parts.get(4)` (name after `ON`). if upper.starts_with("SHOW ALERT STATUS ") { let parts: Vec<&str> = sql.split_whitespace().collect(); let name = parts.get(4).copied().unwrap_or(""); @@ -62,11 +63,10 @@ pub(super) async fn try_string( } // Change streams: `SHOW CHANGE STREAM(S)`. This parses into a typed - // `StreamViewStmt::ShowChangeStreams`, but the pgwire router dispatched it - // from the raw SQL by string prefix (the `SHOW CHANGE STREAM` prefix, which - // captures both the plural `SHOW CHANGE STREAMS` and the bare-singular - // input). Replicate that exactly here, before the parse gate, so the prefix - // recognition stays byte-identical. + // `StreamViewStmt::ShowChangeStreams`, but the router dispatches it from + // the raw SQL by string prefix, before the parse gate (the + // `SHOW CHANGE STREAM` prefix captures both the plural + // `SHOW CHANGE STREAMS` and the bare-singular input). if upper.starts_with("SHOW CHANGE STREAM") { return Some(change_stream::show_change_streams( state, @@ -76,15 +76,13 @@ pub(super) async fn try_string( } // Consumer groups: `SHOW CONSUMER GROUPS ON `, `SHOW PARTITIONS ON - // `, and `COMMIT OFFSET(S) …`. The pgwire streaming router dispatched - // all four by string prefix from the raw token slice. `SHOW CONSUMER GROUPS` - // parses into a typed `StreamViewStmt::ShowConsumerGroups`, but the pgwire - // string dispatch claimed it before any typed arm ran; `SHOW PARTITIONS` and - // `COMMIT OFFSET(S)` parse into no typed variant at all. Replicate that - // exactly here, before the parse gate, so the prefix recognition and the - // `parts`-based syntax messages stay byte-identical. (`SHOW PARTITIONS ` also - // shadows the timeseries `show_partitions` handler exactly as the pgwire - // streaming router — which ran before engine_ops — did.) + // `, and `COMMIT OFFSET(S) …`. The router dispatches all four by + // string prefix from the raw token slice, before the parse gate. + // `SHOW CONSUMER GROUPS` parses into a typed + // `StreamViewStmt::ShowConsumerGroups`, but string dispatch claims it + // before any typed arm runs; `SHOW PARTITIONS` and `COMMIT OFFSET(S)` + // parse into no typed variant at all. (`SHOW PARTITIONS ` also shadows the + // timeseries `show_partitions` handler.) if upper.starts_with("SHOW CONSUMER GROUPS ") { let parts: Vec<&str> = sql.split_whitespace().collect(); return Some(consumer_group::show_consumer_groups( @@ -109,12 +107,10 @@ pub(super) async fn try_string( } // Topics: `CREATE TOPIC`, `DROP TOPIC`, `SHOW TOPIC(S)`, and `PUBLISH TO`. - // None of these parse into any typed AST variant — the pgwire streaming - // router dispatched all four by string prefix from the raw token slice / - // SQL. Replicate that exactly here, before the parse gate, so the prefix - // recognition (including the trailing-space-less `SHOW TOPIC`, which - // captures both `SHOW TOPICS` and the bare-singular input) and the - // `parts`-based syntax messages stay byte-identical. + // None of these parse into any typed AST variant — the router dispatches + // all four by string prefix from the raw token slice / SQL, before the + // parse gate. The `SHOW TOPIC` prefix has no trailing space, so it + // captures both `SHOW TOPICS` and the bare-singular input. if upper.starts_with("CREATE TOPIC ") { let parts: Vec<&str> = sql.split_whitespace().collect(); return Some(topic::create_topic(state, identity, database_id, &parts, sql).await); @@ -127,14 +123,13 @@ pub(super) async fn try_string( return Some(topic::show_topics(state, identity, database_id)); } if upper.starts_with("PUBLISH TO ") { - return Some(topic::handle_publish(state, identity, database_id, sql).await); + return Some(topic::handle_publish(state, identity, database_id, sql, txn_ctx).await); } // Stream consumption: `SELECT * FROM STREAM CONSUMER GROUP // [PARTITION

] [LIMIT ]`. Parses into no typed AST variant — the - // pgwire streaming router recognized it by string prefix from the raw - // token slice. Replicate that exactly here, before the parse gate, so the - // prefix recognition and the `parts`-based extraction stay byte-identical. + // router recognizes it by string prefix from the raw token slice, before + // the parse gate. if upper.starts_with("SELECT ") && upper.contains("FROM STREAM ") && upper.contains("CONSUMER GROUP") @@ -144,10 +139,9 @@ pub(super) async fn try_string( } // Stream/Topic consumption: `SELECT * FROM TOPIC CONSUMER GROUP - // [LIMIT ]`. Topics use "topic:" buffer keys; the pgwire - // streaming router rewrote the token slice (TOPIC → STREAM, name → - // "topic:") and delegated to the stream-consume handler. Replicate - // that rewrite exactly here, before the parse gate. + // [LIMIT ]`. Topics use "topic:" buffer keys; the router + // rewrites the token slice (TOPIC → STREAM, name → "topic:") and + // delegates to the stream-consume handler, before the parse gate. if upper.starts_with("SELECT ") && upper.contains("FROM TOPIC ") && upper.contains("CONSUMER GROUP") diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/string_versioning.rs b/nodedb/src/control/server/shared/ddl/neutral/router/string_versioning.rs index bd566e950..1d5334675 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/string_versioning.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/string_versioning.rs @@ -27,23 +27,18 @@ pub(super) async fn try_string( // Version history. None of `CREATE CHECKPOINT`, `DROP CHECKPOINT`, `SHOW // VERSIONS OF`, `SELECT … AT VERSION`, `SELECT DIFF(…)`, `RESTORE … SET // VERSION`, or `COMPACT HISTORY ON` parse into any typed AST variant — the - // pgwire collaborative router dispatched all of them by string prefix from - // the raw SQL. Replicate that exactly here, before the parse gate, so the - // prefix recognition (including the `RESTORE … SET VERSION` guard that keeps - // `RESTORE TENANT` / `RESTORE DATABASE` on the typed path) and syntax - // messages stay byte-identical. Guard ordering mirrors the pgwire router. + // router dispatches all of them by string prefix from the raw SQL, + // before the parse gate. The `RESTORE … SET VERSION` guard keeps + // `RESTORE TENANT` / `RESTORE DATABASE` on the typed path. if upper.starts_with("CREATE CHECKPOINT ") { return Some( version_history::checkpoint::create_checkpoint(state, identity, database_id, sql).await, ); } if upper.starts_with("DROP CHECKPOINT ") { - return Some(version_history::checkpoint::drop_checkpoint( - state, - identity, - database_id, - sql, - )); + return Some( + version_history::checkpoint::drop_checkpoint(state, identity, database_id, sql).await, + ); } if upper.starts_with("SHOW VERSIONS OF ") { return Some(version_history::show_versions::show_versions( @@ -78,15 +73,13 @@ pub(super) async fn try_string( } // Maintenance: ANALYZE / COMPACT / SHOW STORAGE / SHOW COMPACTION STATUS. - // These parse into typed `ClusterStmt` variants, but the pgwire router - // dispatched all four by string prefix from the raw SQL / token slice (the - // pgwire typed-AST path has no arm for them). Replicate that exactly here, - // before the parse gate, so the prefix recognition (trailing space on - // `ANALYZE ` / `COMPACT `, and the `SHOW COMPACTION STATUS` exact / prefix - // forms) and the `parts`-based name extraction stay byte-identical. The - // `COMPACT ` prefix is placed after the version-history `COMPACT HISTORY ON` - // guard above, preserving that `COMPACT HISTORY ON …` routes to - // version_history exactly as the pgwire dispatch (neutral-first) did. + // These parse into typed `ClusterStmt` variants, but the router + // dispatches all four by string prefix from the raw SQL / token slice, + // before the parse gate (trailing space on `ANALYZE ` / `COMPACT `, and the + // `SHOW COMPACTION STATUS` exact / prefix forms; names come from `parts`). + // The `COMPACT ` prefix is placed after the version-history + // `COMPACT HISTORY ON` guard above, so `COMPACT HISTORY ON …` routes to + // version_history. if upper.starts_with("ANALYZE ") { return Some(maintenance::handle_analyze(state, identity, sql, database_id).await); } @@ -116,13 +109,10 @@ pub(super) async fn try_string( // SHOW RAFT GROUP , SHOW MIGRATIONS, REBALANCE, SHOW PEER HEALTH, // SHOW NODES, SHOW NODE , REMOVE NODE , SHOW RANGES, SHOW // ROUTING, SHOW SCHEMA VERSION. All of these parse into typed - // `ClusterStmt` variants, but the pgwire admin router dispatched them by - // string prefix from the raw SQL / token slice (the pgwire typed-AST path - // only had an arm for `ALTER RAFT GROUP`). Replicate that exactly here, - // before the parse gate, so the prefix recognition (order matters: `SHOW - // RAFT GROUPS` before `SHOW RAFT GROUP `) and the `parts`-based - // extraction stay byte-identical. `ALTER RAFT GROUP` is dispatched via - // the typed match below, exactly as the pgwire router did. + // `ClusterStmt` variants, but the router dispatches them by + // string prefix from the raw SQL / token slice, before the parse gate. + // Order matters: `SHOW RAFT GROUPS` comes before `SHOW RAFT GROUP `. + // `ALTER RAFT GROUP` is dispatched via the typed match below. if upper.starts_with("SHOW CLUSTER") { return Some(cluster::show_cluster(state, identity)); } @@ -162,13 +152,18 @@ pub(super) async fn try_string( if upper.starts_with("SHOW SCHEMA VERSION") { return Some(cluster::show_schema_version(state, identity)); } + if upper == "CREATE RESTORE POINT" || upper.starts_with("CREATE RESTORE POINT ") { + return Some(cluster::create_restore_point(state, identity).await); + } + if upper.starts_with("SHOW RESTORE POINTS") { + return Some(cluster::show_restore_points(state, identity)); + } // Vector index lifecycle: SHOW VECTOR INDEX / ALTER VECTOR INDEX. None of - // these are dispatched from a typed AST arm — the pgwire engine_ops router - // recognized all four by string prefix from the raw SQL. Replicate that - // exactly here, before the parse gate, so the prefix recognition (and the - // ` SEAL` / ` COMPACT` / ` SET ` sub-clause guards, checked in this order) - // stays byte-identical. + // these are dispatched from a typed AST arm — the router recognizes all + // four by string prefix from the raw SQL, before the parse gate. The + // ` SEAL` / ` COMPACT` / ` SET ` sub-clause guards are checked in this + // order. if upper.starts_with("SHOW VECTOR INDEX ") { return Some( maintenance::handle_show_vector_index(state, identity, database_id, sql).await, @@ -193,21 +188,14 @@ pub(super) async fn try_string( // Vector model metadata. None of these are dispatched from a typed AST arm — // `ALTER COLLECTION ... SET VECTOR METADATA ON` parses into no // `AlterCollectionOp` variant, and `SHOW VECTOR MODELS` / `SELECT - // VECTOR_METADATA(...)` parse into no typed DDL AST at all. The pgwire - // engine_ops router recognized all three by string prefix from the raw SQL. - // Replicate that exactly here, before the parse gate, so the prefix - // recognition (and the `ALTER COLLECTION ... SET VECTOR METADATA ON` guard - // running before the typed `AlterCollection` parse handling) stays - // byte-identical. The `SET VECTOR METADATA ON` guard precedes the typed - // parse gate below, so it is never shadowed by the migrated typed - // `AlterCollection` dispatch. + // VECTOR_METADATA(...)` parse into no typed DDL AST at all. The router + // recognizes all three by string prefix from the raw SQL, before the parse + // gate. The `SET VECTOR METADATA ON` guard precedes the typed parse gate + // below, so the typed `AlterCollection` dispatch never shadows it. if upper.starts_with("ALTER COLLECTION ") && upper.contains("SET VECTOR METADATA ON") { - return Some(collection::handle_set_vector_metadata( - state, - identity, - sql, - database_id, - )); + return Some( + collection::handle_set_vector_metadata(state, identity, sql, database_id).await, + ); } if upper.starts_with("SHOW VECTOR MODELS") { return Some(collection::handle_show_vector_models( @@ -244,19 +232,36 @@ pub(super) async fn try_string( // Graph index and tree operations: CREATE GRAPH INDEX / TREE_SUM / // TREE_CHILDREN. None of these are dispatched from a typed AST arm — the - // pgwire engine_ops router recognized all three by string prefix from the - // raw SQL (the `SELECT TREE_SUM` / bare `TREE_SUM` and `SELECT + // router recognizes all three by string prefix from the raw SQL, before the + // parse gate (the `SELECT TREE_SUM` / bare `TREE_SUM` and `SELECT // TREE_CHILDREN` / bare `TREE_CHILDREN` forms never parse into a typed DDL - // AST). Replicate that exactly here, before the parse gate, so the prefix - // recognition and syntax messages stay byte-identical. + // AST). if upper.starts_with("CREATE GRAPH INDEX ") { return Some(tree_ops::create_graph_index(state, identity, database_id, sql).await); } if upper.starts_with("SELECT TREE_SUM") || upper.starts_with("TREE_SUM") { - return Some(tree_ops::tree_sum(state, identity, database_id, sql).await); + return Some( + tree_ops::tree_sum( + state, + identity, + database_id, + sql, + txn_ctx.linearizable_reads(), + ) + .await, + ); } if upper.starts_with("SELECT TREE_CHILDREN") || upper.starts_with("TREE_CHILDREN") { - return Some(tree_ops::tree_children(state, identity, database_id, sql).await); + return Some( + tree_ops::tree_children( + state, + identity, + database_id, + sql, + txn_ctx.linearizable_reads(), + ) + .await, + ); } None diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/typed_auth.rs b/nodedb/src/control/server/shared/ddl/neutral/router/typed_auth.rs index de719e1b7..44637edd5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/typed_auth.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/typed_auth.rs @@ -30,30 +30,33 @@ pub(super) async fn try_typed( role, tenant, if_not_exists, - }) => Some(user::create_user( - state, - identity, - username, - password, - role.as_deref(), - tenant.as_ref(), - *if_not_exists, - )), + }) => Some( + user::create_user( + state, + identity, + username, + password, + role.as_deref(), + tenant.as_ref(), + *if_not_exists, + ) + .await, + ), NodedbStatement::Auth(AuthStmt::AlterUser { username, op }) => { - Some(user::alter_user(state, identity, username, op)) + Some(user::alter_user(state, identity, username, op).await) } NodedbStatement::Auth(AuthStmt::AlterRole { name, sub_op }) => { - Some(role::alter_role_typed(state, identity, name, sub_op)) + Some(role::alter_role_typed(state, identity, database_id, name, sub_op).await) } NodedbStatement::Auth(AuthStmt::GrantRole { roles, grantee }) => { - Some(grant::role::grant_role(state, identity, roles, grantee)) + Some(grant::role::grant_role(state, identity, roles, grantee).await) } NodedbStatement::Auth(AuthStmt::RevokeRole { roles, grantee }) => { - Some(grant::role::revoke_role(state, identity, roles, grantee)) + Some(grant::role::revoke_role(state, identity, roles, grantee).await) } NodedbStatement::Auth(AuthStmt::GrantPermission { @@ -61,44 +64,58 @@ pub(super) async fn try_typed( target_type, target_name, grantee, - }) => Some(grant::permission::grant_permission( - state, - identity, - permissions, - target_type, - target_name, - grantee, - )), + }) => Some( + grant::permission::grant_permission( + state, + identity, + database_id, + permissions, + target_type, + target_name, + grantee, + ) + .await, + ), NodedbStatement::Auth(AuthStmt::RevokePermission { permissions, target_type, target_name, grantee, - }) => Some(grant::permission::revoke_permission( - state, - identity, - permissions, - target_type, - target_name, - grantee, - )), + }) => Some( + grant::permission::revoke_permission( + state, + identity, + database_id, + permissions, + target_type, + target_name, + grantee, + ) + .await, + ), NodedbStatement::Auth(AuthStmt::GrantDatabasePermission { permission, db_name, grantee, - }) => Some(grant::database_permission::grant_database( - state, identity, permission, db_name, grantee, - )), + }) => Some( + grant::database_permission::grant_database( + state, identity, permission, db_name, grantee, + ) + .await, + ), NodedbStatement::Auth(AuthStmt::RevokeDatabasePermission { permission, db_name, grantee, - }) => Some(grant::database_permission::revoke_database( - state, identity, permission, db_name, grantee, - )), + }) => Some( + grant::database_permission::revoke_database( + state, identity, permission, db_name, grantee, + ) + .await, + ), NodedbStatement::Auth(AuthStmt::CreateOidcProvider { name, @@ -107,31 +124,31 @@ pub(super) async fn try_typed( tenant_id, audience, claim_mappings, - }) => Some(oidc::create_oidc_provider( - state, - identity, - oidc::CreateOidcProviderParams { - name, - issuer, - jwks_uri, - tenant_id: *tenant_id, - audience: audience.as_deref(), - claim_mappings, - }, - )), + }) => Some( + oidc::create_oidc_provider( + state, + identity, + oidc::CreateOidcProviderParams { + name, + issuer, + jwks_uri, + tenant_id: *tenant_id, + audience: audience.as_deref(), + claim_mappings, + }, + ) + .await, + ), NodedbStatement::Auth(AuthStmt::AlterOidcProviderClaimMapping { name, claim_mappings, - }) => Some(oidc::alter_oidc_provider_claim_mapping( - state, - identity, - name, - claim_mappings, - )), + }) => Some( + oidc::alter_oidc_provider_claim_mapping(state, identity, name, claim_mappings).await, + ), NodedbStatement::Auth(AuthStmt::DropOidcProvider { name, if_exists }) => { - Some(oidc::drop_oidc_provider(state, identity, name, *if_exists)) + Some(oidc::drop_oidc_provider(state, identity, name, *if_exists).await) } NodedbStatement::Auth(AuthStmt::ShowOidcProviders) => { @@ -139,9 +156,8 @@ pub(super) async fn try_typed( } // SHOW PERMISSIONS [ON ] [FOR ]. Parses into a - // typed `AuthStmt::ShowPermissions` and was dispatched from the pgwire - // typed-AST sync router (`sync_ops`). The permission-store reads are - // preserved verbatim in `inspect`. + // typed `AuthStmt::ShowPermissions`. The permission-store reads live + // in `inspect`. NodedbStatement::Auth(AuthStmt::ShowPermissions { on_collection, for_grantee, diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/typed_automation.rs b/nodedb/src/control/server/shared/ddl/neutral/router/typed_automation.rs index 58d1406de..53ea90ab0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/typed_automation.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/typed_automation.rs @@ -35,47 +35,45 @@ pub(super) async fn try_typed( priority, security, body_sql, - }) => Some(trigger::create_trigger( - state, - identity, - trigger::create::CreateTriggerRequest { - or_replace: *or_replace, - execution_mode, - name, - timing, - events_insert: *events_insert, - events_update: *events_update, - events_delete: *events_delete, - collection, - granularity, - when_condition: when_condition.as_deref(), - priority: *priority, - security: security.as_deref(), - body_sql, - }, - )), + }) => Some( + trigger::create_trigger( + state, + identity, + trigger::create::CreateTriggerRequest { + or_replace: *or_replace, + execution_mode, + name, + timing, + events_insert: *events_insert, + events_update: *events_update, + events_delete: *events_delete, + collection, + granularity, + when_condition: when_condition.as_deref(), + priority: *priority, + security: security.as_deref(), + body_sql, + }, + ) + .await, + ), NodedbStatement::Automation(AutomationStmt::AlterTrigger { name, action, new_owner, - }) => Some(trigger::alter_trigger( - state, - identity, - name, - action, - new_owner.as_deref(), - )), + }) => { + Some(trigger::alter_trigger(state, identity, name, action, new_owner.as_deref()).await) + } NodedbStatement::Automation(AutomationStmt::DropTrigger { name, if_exists, .. }) => { - // IF EXISTS short-circuit folded from the pgwire guard: a DROP of a + // IF EXISTS short-circuit: a DROP of a // non-existing trigger returns the tag before the token handler runs // (and before any catalog-read error surfaces). The `if_exists: // false` case and the existing-trigger case fall through to - // `drop_trigger`, which re-derives the name / IF EXISTS from `parts` - // exactly as the pgwire schema string dispatch did. + // `drop_trigger`, which re-derives the name / IF EXISTS from `parts`. if *if_exists && !trigger::trigger_exists(state, identity, name) { return Some(Ok(vec![DdlResult::Status { command: "DROP TRIGGER".to_string(), @@ -83,7 +81,7 @@ pub(super) async fn try_typed( }])); } let parts: Vec<&str> = sql.split_whitespace().collect(); - Some(trigger::drop_trigger(state, identity, &parts)) + Some(trigger::drop_trigger(state, identity, &parts).await) } NodedbStatement::Automation(AutomationStmt::ShowTriggers { .. }) => { @@ -98,40 +96,45 @@ pub(super) async fn try_typed( scope, missed_policy, allow_overlap, - }) => Some(schedule::create_schedule( - state, - identity, - database_id, - &CreateScheduleRequest { - name, - cron_expr, - body_sql, - scope, - missed_policy, - allow_overlap: *allow_overlap, - }, - )), + }) => Some( + schedule::create_schedule( + state, + identity, + database_id, + &CreateScheduleRequest { + name, + cron_expr, + body_sql, + scope, + missed_policy, + allow_overlap: *allow_overlap, + }, + ) + .await, + ), NodedbStatement::Automation(AutomationStmt::AlterSchedule { name, action, cron_expr, - }) => Some(schedule::alter_schedule( - state, - identity, - database_id, - name, - action, - cron_expr.as_deref(), - )), + }) => Some( + schedule::alter_schedule( + state, + identity, + database_id, + name, + action, + cron_expr.as_deref(), + ) + .await, + ), NodedbStatement::Automation(AutomationStmt::DropSchedule { name, if_exists }) => { - // IF EXISTS short-circuit folded from the pgwire guard: a DROP of a + // IF EXISTS short-circuit: a DROP of a // non-existing schedule returns the tag before the token handler runs // (and before the tenant-admin gate). The `if_exists: false` case and // the existing-schedule case fall through to `drop_schedule`, which - // re-derives the name / IF EXISTS from `parts` exactly as the pgwire - // admin string dispatch did. + // re-derives the name / IF EXISTS from `parts`. if *if_exists && !schedule::schedule_exists(state, identity, database_id, name) { return Some(Ok(vec![DdlResult::Status { command: "DROP SCHEDULE".to_string(), @@ -139,12 +142,7 @@ pub(super) async fn try_typed( }])); } let parts: Vec<&str> = sql.split_whitespace().collect(); - Some(schedule::drop_schedule( - state, - identity, - database_id, - &parts, - )) + Some(schedule::drop_schedule(state, identity, database_id, &parts).await) } NodedbStatement::Automation(AutomationStmt::CreateAlert { @@ -158,35 +156,37 @@ pub(super) async fn try_typed( recover_after, severity, notify_targets_raw, - }) => Some(alert::create_alert( - state, - identity, - &CreateAlertRequest { - name, - collection, - where_filter: where_filter.as_deref(), - condition_raw, - group_by, - window_raw, - fire_after: *fire_after, - recover_after: *recover_after, - severity, - notify_targets_raw, - database_id, - }, - )), - - NodedbStatement::Automation(AutomationStmt::AlterAlert { name, action }) => Some( - alert::alter_alert(state, identity, database_id, name, action), + }) => Some( + alert::create_alert( + state, + identity, + &CreateAlertRequest { + name, + collection, + where_filter: where_filter.as_deref(), + condition_raw, + group_by, + window_raw, + fire_after: *fire_after, + recover_after: *recover_after, + severity, + notify_targets_raw, + database_id, + }, + ) + .await, ), + NodedbStatement::Automation(AutomationStmt::AlterAlert { name, action }) => { + Some(alert::alter_alert(state, identity, database_id, name, action).await) + } + NodedbStatement::Automation(AutomationStmt::DropAlert { name, if_exists }) => { - // IF EXISTS short-circuit folded from the pgwire guard: a DROP of a + // IF EXISTS short-circuit: a DROP of a // non-existing alert returns the tag before the token handler runs // (and before the tenant-admin gate). The `if_exists: false` case and // the existing-alert case fall through to `drop_alert`, which - // re-derives the name from `parts[2]` exactly as the pgwire admin - // string dispatch did. + // re-derives the name from `parts[2]`. if *if_exists && !alert::alert_exists(state, identity, database_id, name) { return Some(Ok(vec![DdlResult::Status { command: "DROP ALERT".to_string(), @@ -194,7 +194,7 @@ pub(super) async fn try_typed( }])); } let parts: Vec<&str> = sql.split_whitespace().collect(); - Some(alert::drop_alert(state, identity, database_id, &parts)) + Some(alert::drop_alert(state, identity, database_id, &parts).await) } _ => None, diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/typed_collection.rs b/nodedb/src/control/server/shared/ddl/neutral/router/typed_collection.rs index 6cc918345..cbe28673e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/typed_collection.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/typed_collection.rs @@ -23,14 +23,12 @@ pub(super) async fn try_typed( stmt: &NodedbStatement, ) -> Option, DdlError>> { match stmt { - // CREATE COLLECTION / CREATE TABLE. Migrated from the pgwire typed-AST - // async router (`async_ops`) plus the `if_not_exists: true` guard - // short-circuit, formerly in the pgwire `guards` module - // (checked here, inline, before the create handler runs — same - // ordering). `build_and_persist` (name/duplicate/engine validation, + // CREATE COLLECTION / CREATE TABLE. The `if_not_exists: true` guard + // short-circuit runs inline here, before the create handler. + // `build_and_persist` (name/duplicate/engine validation, // schema construction, `StoredCollection` assembly, propose+apply, - // SERIAL sequence auto-creation) and the `dispatch_register_by_name` - // follow-up dispatch are preserved verbatim in `collection::create`. + // SERIAL sequence auto-creation) lives in `collection::create`. + // `register_created` registers the collection for a buffered CREATE. NodedbStatement::Collection(CollectionStmt::CreateCollection { name, if_not_exists, @@ -60,16 +58,7 @@ pub(super) async fn try_typed( database_id, ) .await; - let result = match result { - Ok(resp) => { - collection::dispatch_register_by_name(state, identity, name, database_id) - .await - .map(|()| resp) - .map_err(|e| DdlError::from_error(&e)) - } - Err(e) => Err(e), - }; - Some(result) + Some(register_created(state, result).await) } NodedbStatement::Collection(CollectionStmt::CreateTable { @@ -101,50 +90,42 @@ pub(super) async fn try_typed( database_id, ) .await; - let result = match result { - Ok(resp) => { - collection::dispatch_register_by_name(state, identity, name, database_id) - .await - .map(|()| resp) - .map_err(|e| DdlError::from_error(&e)) - } - Err(e) => Err(e), - }; - Some(result) + Some(register_created(state, result).await) } // DROP { COLLECTION | TABLE } [IF EXISTS] [PURGE] [CASCADE // [FORCE]] — parser folds both spellings into `DropCollection`. - // Migrated from the pgwire typed-AST sync router (`sync_ops`). The - // handler honours `if_exists` internally via its existence-check + // The handler honours `if_exists` internally via its existence-check // matrix (no guard short-circuit); the catalog propose + single-node // fallback, cascade dependent enumeration, soft vs hard delete, the - // implicit-sequence sweep, and the audit pair are preserved verbatim - // in `collection::drop`. + // implicit-sequence sweep, and the audit pair live in + // `collection::drop`. NodedbStatement::Collection(CollectionStmt::DropCollection { name, if_exists, purge, cascade, cascade_force, - }) => Some(collection::drop_collection( - state, - identity, - &collection::DropCollectionRequest { - name, - if_exists: *if_exists, - purge: *purge, - cascade: *cascade, - cascade_force: *cascade_force, - database_id, - }, - )), + }) => Some( + collection::drop_collection( + state, + identity, + &collection::DropCollectionRequest { + name, + if_exists: *if_exists, + purge: *purge, + cascade: *cascade, + cascade_force: *cascade_force, + database_id, + }, + ) + .await, + ), // CREATE [UNIQUE] INDEX [IF NOT EXISTS] [name] ON // () [WHERE ...]. - // Migrated from the pgwire typed-AST async router (`async_ops`). The - // two-phase Building→Ready backfill, peer fan-out, Register refresh, - // and owner-ledger propose are preserved verbatim in `collection::index`. + // The two-phase Building→Ready backfill, peer fan-out, Register + // refresh, and owner-ledger propose live in `collection::index`. NodedbStatement::Collection(CollectionStmt::CreateIndex { unique, index_name, @@ -183,11 +164,11 @@ pub(super) async fn try_typed( format_template_raw, reset_period_raw, gap_free, - scope, + scope: _, }) => { // IF NOT EXISTS on a non-existing sequence falls through to the - // planner today (the pgwire guard returned None and no create arm - // matched `if_not_exists: true`). Replicate by returning None. + // planner (no create arm matches `if_not_exists: true`), so this + // returns None. let tenant_id = identity.tenant_id.as_u64(); if *if_not_exists && !state @@ -196,44 +177,49 @@ pub(super) async fn try_typed( { return None; } - Some(sequence::create_sequence( - state, - identity, - database_id, - &CreateSequenceRequest { - name, - if_not_exists: *if_not_exists, - start: *start, - increment: *increment, - min_value: *min_value, - max_value: *max_value, - cycle: *cycle, - cache: *cache, - format_template_raw: format_template_raw.as_deref(), - reset_period_raw: reset_period_raw.as_deref(), - gap_free: *gap_free, - scope: scope.as_deref(), - }, - )) + Some( + sequence::create_sequence( + state, + identity, + database_id, + &CreateSequenceRequest { + name, + if_not_exists: *if_not_exists, + start: *start, + increment: *increment, + min_value: *min_value, + max_value: *max_value, + cycle: *cycle, + cache: *cache, + format_template_raw: format_template_raw.as_deref(), + reset_period_raw: reset_period_raw.as_deref(), + gap_free: *gap_free, + }, + ) + .await, + ) } NodedbStatement::Collection(CollectionStmt::AlterSequence { name, action, with_value, - }) => Some(sequence::alter_sequence( - state, - identity, - database_id, - name, - action, - with_value.as_deref(), - )), - - NodedbStatement::Collection(CollectionStmt::DropSequence { name, if_exists }) => Some( - sequence::drop_sequence(state, identity, database_id, name, *if_exists), + }) => Some( + sequence::alter_sequence( + state, + identity, + database_id, + name, + action, + with_value.as_deref(), + ) + .await, ), + NodedbStatement::Collection(CollectionStmt::DropSequence { name, if_exists }) => { + Some(sequence::drop_sequence(state, identity, database_id, name, *if_exists).await) + } + NodedbStatement::Collection(CollectionStmt::ShowSequences) => { Some(sequence::show_sequences(state, identity, database_id)) } @@ -246,11 +232,10 @@ pub(super) async fn try_typed( // `AlterCollectionOp` variant (ADD/DROP/RENAME/ALTER COLUMN, OWNER TO, // SET RETENTION / APPEND_ONLY / LAST_VALUE_CACHE / LEGAL_HOLD, ADD // MATERIALIZED_SUM, SET ON CONFLICT). `dispatch_alter_collection` is a - // total match over `AlterCollectionOp` — no variant falls through — so - // the pgwire path never sees an `AlterCollection` statement. Each + // total match over `AlterCollectionOp` — no variant falls through. Each // sub-handler's catalog / register / audit side effects and command tag - // (`ALTER TABLE` for ADD COLUMN, `ALTER COLLECTION` otherwise) are - // preserved verbatim in `collection::alter`. + // (`ALTER TABLE` for ADD COLUMN, `ALTER COLLECTION` otherwise) live in + // `collection::alter`. NodedbStatement::Collection(CollectionStmt::AlterCollection { name, operation }) => Some( collection::dispatch_alter_collection(state, identity, database_id, name, operation) .await, @@ -275,3 +260,20 @@ pub(super) async fn try_typed( _ => None, } } + +/// Finish a CREATE COLLECTION / CREATE TABLE: register the new collection on +/// this node's Data Plane when the proposal left that to the handler. +/// +/// A durable outcome ran the entry's awaited post-apply, which registered the +/// collection on every core. A buffered outcome registered nothing, and the +/// open transaction's later statements read the new shape. +async fn register_created( + state: &SharedState, + created: Result, +) -> Result, DdlError> { + let created = created?; + collection::register_proposed_collection(state, created.outcome, &created.collection) + .await + .map_err(|e| DdlError::from_error(&e))?; + Ok(created.results) +} diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/typed_database.rs b/nodedb/src/control/server/shared/ddl/neutral/router/typed_database.rs index 72eb6ec46..b180cdb38 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/typed_database.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/typed_database.rs @@ -25,9 +25,8 @@ pub(super) async fn try_typed( match stmt { // Database DDL family (CREATE / DROP / ALTER DATABASE, SHOW DATABASES / // QUOTA / USAGE / LINEAGE, CLONE / MIRROR / PROMOTE, BACKUP / RESTORE, - // SHOW DATABASE MIRROR STATUS). Migrated from the pgwire typed-AST - // database router (`database_ops`); all catalog / audit / gate side - // effects are preserved verbatim in `database`. + // SHOW DATABASE MIRROR STATUS). All catalog / audit / gate side + // effects live in `database`. // // NOT here: `UseDatabase` (session-coupled, intercepted before the DDL // router). `AlterTenant` / `ShowTenantQuotaInDatabase` / @@ -37,25 +36,21 @@ pub(super) async fn try_typed( name, if_not_exists, options, - }) => Some(database::create::create_database( - state, - identity, - name, - *if_not_exists, - options, - )), + }) => Some( + database::create::create_database(state, identity, name, *if_not_exists, options).await, + ), NodedbStatement::Database(DatabaseStmt::DropDatabase { name, if_exists, cascade, - }) => Some(database::drop::drop_database( - state, identity, name, *if_exists, *cascade, - )), + }) => { + Some(database::drop::drop_database(state, identity, name, *if_exists, *cascade).await) + } - NodedbStatement::Database(DatabaseStmt::AlterDatabase { name, operation }) => Some( - database::alter::alter_database(state, identity, name, operation), - ), + NodedbStatement::Database(DatabaseStmt::AlterDatabase { name, operation }) => { + Some(database::alter::alter_database(state, identity, name, operation).await) + } NodedbStatement::Database(DatabaseStmt::ShowDatabases) => { Some(database::show::show_databases(state, identity)) @@ -95,39 +90,49 @@ pub(super) async fn try_typed( source_cluster, source_database, mode, - }) => Some(database::mirror::create::mirror_database( - state, - identity, - local_name, - source_cluster, - source_database, - *mode, - )), + }) => Some( + database::mirror::create::mirror_database( + state, + identity, + local_name, + source_cluster, + source_database, + *mode, + ) + .await, + ), NodedbStatement::Database(DatabaseStmt::ShowDatabaseMirrorStatus { name }) => Some( database::mirror::show::show_database_mirror_status(state, identity, name.as_deref()), ), - NodedbStatement::Database(DatabaseStmt::BackupDatabase { name, .. }) => Some( - database::backup_restore::backup_database(state, identity, name), - ), + NodedbStatement::Database(DatabaseStmt::BackupDatabase { name, uri }) => { + Some(database::backup_restore::backup_database(state, identity, name, uri).await) + } - NodedbStatement::Database(DatabaseStmt::RestoreDatabase { name, .. }) => Some( - database::backup_restore::restore_database(state, identity, name), + NodedbStatement::Database(DatabaseStmt::RestoreDatabase { + name, + uri, + force, + dry_run, + }) => Some( + database::backup_restore::restore_database( + state, identity, name, uri, *force, *dry_run, + ) + .await, ), // Tenant DDL family (`ALTER TENANT ... IN DATABASE ... SET QUOTA`, // `SHOW TENANT QUOTA|USAGE FOR ... IN DATABASE ...`). These parse into - // typed `DatabaseStmt` variants and were dispatched from the pgwire - // typed-AST database router (`database_ops`); all catalog / audit / - // gate side effects are preserved verbatim in `tenant`. + // typed `DatabaseStmt` variants. All catalog / audit / gate side + // effects live in `tenant`. NodedbStatement::Database(DatabaseStmt::AlterTenant { name, database, operation, - }) => Some(tenant::handle_alter_tenant_quota( - state, identity, name, database, operation, - )), + }) => Some( + tenant::handle_alter_tenant_quota(state, identity, name, database, operation).await, + ), NodedbStatement::Database(DatabaseStmt::ShowTenantQuotaInDatabase { name, database }) => { Some(tenant::handle_show_tenant_quota_in_database( @@ -143,10 +148,9 @@ pub(super) async fn try_typed( // `MOVE TENANT FROM TO ` — async, // 5-phase re-parenting sequence. Parses into a typed `DatabaseStmt` - // variant and was dispatched from the pgwire typed-AST async router - // (`async_ops`); every phase (pre-flight, drain, snapshot, cutover, - // resume), the journal, and the compensation paths are preserved - // verbatim in `tenant::move_tenant`. + // variant. Every phase (pre-flight, drain, snapshot, cutover, + // resume), the journal, and the compensation paths live in + // `tenant::move_tenant`. NodedbStatement::Database(DatabaseStmt::MoveTenant { tenant_name, from_db, @@ -154,9 +158,8 @@ pub(super) async fn try_typed( }) => Some(tenant::handle_move_tenant(state, identity, tenant_name, from_db, to_db).await), // Tenant introspection by identifier / name filter. These parse into - // typed `DatabaseStmt` variants and were dispatched from the pgwire - // typed-AST database router (`database_ops`). The credential / usage - // reads are preserved verbatim in `inspect`. + // typed `DatabaseStmt` variants. The credential / usage reads live in + // `inspect`. NodedbStatement::Database(DatabaseStmt::ShowTenantByIdentifier { ident }) => { Some(inspect::show_tenant_by_identifier(state, identity, ident)) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/typed_misc.rs b/nodedb/src/control/server/shared/ddl/neutral/router/typed_misc.rs index bf12aec64..77e73478a 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/typed_misc.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/typed_misc.rs @@ -23,8 +23,7 @@ pub(super) async fn try_typed( txn_ctx: &DmlTxnCtx<'_>, ) -> Option, DdlError>> { match stmt { - // `COPY FROM '' [WITH (...)]` bulk import. Ported - // from the pgwire `ast::async_ops::try_dispatch_async` typed arm. + // `COPY FROM '' [WITH (...)]` bulk import. NodedbStatement::Misc(MiscStmt::CopyFromFile { collection, path, @@ -48,8 +47,7 @@ pub(super) async fn try_typed( .await, ), - // `COPY TO '' [WITH (...)]` bulk export. Ported from - // the pgwire `ast::async_ops::try_dispatch_async` typed arm. + // `COPY TO '' [WITH (...)]` bulk export. NodedbStatement::Misc(MiscStmt::CopyToFile { source, path, diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/typed_policy.rs b/nodedb/src/control/server/shared/ddl/neutral/router/typed_policy.rs index e7d3b367b..04b5414f0 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/typed_policy.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/typed_policy.rs @@ -26,9 +26,8 @@ pub(super) async fn try_typed( ) -> Option, DdlError>> { match stmt { // `SHOW CONFLICT POLICY ON `. Parses into a typed - // `PolicyStmt::ShowConflictPolicy` and was dispatched from the pgwire - // typed-AST async router. The Data Plane `GetPolicy` read is preserved - // verbatim in `conflict_policy`. + // `PolicyStmt::ShowConflictPolicy`. The Data Plane `GetPolicy` read + // lives in `conflict_policy`. NodedbStatement::Policy(PolicyStmt::ShowConflictPolicy { collection }) => Some( conflict_policy::show_conflict_policy(state, identity, database_id, collection).await, ), @@ -41,35 +40,41 @@ pub(super) async fn try_typed( is_restrictive, on_deny_raw, tenant_id_override, - }) => Some(rls::create_rls_policy( - state, - identity, - &CreateRlsPolicyRequest { - name, - collection, - policy_type_raw: policy_type, - predicate_raw, - is_restrictive: *is_restrictive, - on_deny_raw: on_deny_raw.as_deref(), - tenant_id_override: *tenant_id_override, - database_id, - }, - )), + }) => Some( + rls::create_rls_policy( + state, + identity, + &CreateRlsPolicyRequest { + name, + collection, + policy_type_raw: policy_type, + predicate_raw, + is_restrictive: *is_restrictive, + on_deny_raw: on_deny_raw.as_deref(), + tenant_id_override: *tenant_id_override, + database_id, + }, + ) + .await, + ), NodedbStatement::Policy(PolicyStmt::DropRlsPolicy { name, collection, if_exists, tenant_id_override, - }) => Some(rls::drop_rls_policy( - state, - identity, - database_id, - name, - collection, - *if_exists, - *tenant_id_override, - )), + }) => Some( + rls::drop_rls_policy( + state, + identity, + database_id, + name, + collection, + *if_exists, + *tenant_id_override, + ) + .await, + ), NodedbStatement::Policy(PolicyStmt::ShowRlsPolicies { collection, @@ -89,34 +94,40 @@ pub(super) async fn try_typed( rules, if_not_exists, tenant_id_override, - }) => Some(redaction::create_redaction_policy( - state, - identity, - &CreateRedactionPolicyRequest { - name, - collection, - for_role, - rules, - if_not_exists: *if_not_exists, - tenant_id_override: *tenant_id_override, - database_id, - }, - )), + }) => Some( + redaction::create_redaction_policy( + state, + identity, + &CreateRedactionPolicyRequest { + name, + collection, + for_role, + rules, + if_not_exists: *if_not_exists, + tenant_id_override: *tenant_id_override, + database_id, + }, + ) + .await, + ), NodedbStatement::Policy(PolicyStmt::DropRedactionPolicy { collection, for_role, if_exists, tenant_id_override, - }) => Some(redaction::drop_redaction_policy( - state, - identity, - database_id, - collection, - for_role, - *if_exists, - *tenant_id_override, - )), + }) => Some( + redaction::drop_redaction_policy( + state, + identity, + database_id, + collection, + for_role, + *if_exists, + *tenant_id_override, + ) + .await, + ), NodedbStatement::Policy(PolicyStmt::ShowRedactionPolicies { collection, @@ -129,20 +140,20 @@ pub(super) async fn try_typed( *tenant_id_override, )), - NodedbStatement::Policy(PolicyStmt::CreateEnumType { name, labels }) => Some( - custom_type::create_enum_type(state, identity, database_id, name, labels), - ), + NodedbStatement::Policy(PolicyStmt::CreateEnumType { name, labels }) => { + Some(custom_type::create_enum_type(state, identity, database_id, name, labels).await) + } NodedbStatement::Policy(PolicyStmt::CreateCompositeType { name, fields }) => Some( - custom_type::create_composite_type(state, identity, database_id, name, fields), + custom_type::create_composite_type(state, identity, database_id, name, fields).await, ), - NodedbStatement::Policy(PolicyStmt::DropType { name, if_exists }) => Some( - custom_type::drop_type(state, identity, database_id, name, *if_exists), - ), + NodedbStatement::Policy(PolicyStmt::DropType { name, if_exists }) => { + Some(custom_type::drop_type(state, identity, database_id, name, *if_exists).await) + } NodedbStatement::Policy(PolicyStmt::AlterTypeAddValue { type_name, label }) => Some( - custom_type::alter_type_add_value(state, identity, database_id, type_name, label), + custom_type::alter_type_add_value(state, identity, database_id, type_name, label).await, ), NodedbStatement::Policy(PolicyStmt::ShowTypes) => { @@ -172,18 +183,21 @@ pub(super) async fn try_typed( action, set_key, set_value, - }) => Some(retention_policy::alter_retention_policy( - state, - identity, - database_id, - name, - action, - set_key.as_deref(), - set_value.as_deref(), - )), + }) => Some( + retention_policy::alter_retention_policy( + state, + identity, + database_id, + name, + action, + set_key.as_deref(), + set_value.as_deref(), + ) + .await, + ), NodedbStatement::Policy(PolicyStmt::DropRetentionPolicy { name, if_exists }) => { - // IF EXISTS short-circuit folded from the pgwire guard: a DROP of a + // IF EXISTS short-circuit: a DROP of a // non-existing retention policy returns the tag before the handler // runs (and before the tenant-admin gate). The existence check reads // the in-memory registry for the identity tenant scoped to the diff --git a/nodedb/src/control/server/shared/ddl/neutral/router/typed_streamview.rs b/nodedb/src/control/server/shared/ddl/neutral/router/typed_streamview.rs index 0f20ada5a..9f6eb21e8 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/router/typed_streamview.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/router/typed_streamview.rs @@ -28,26 +28,28 @@ pub(super) async fn try_typed( name, collection, with_clause_raw, - }) => Some(change_stream::create_change_stream( - state, - identity, - database_id, - name, - collection, - with_clause_raw, - )), + }) => Some( + change_stream::create_change_stream( + state, + identity, + database_id, + name, + collection, + with_clause_raw, + ) + .await, + ), NodedbStatement::StreamView(StreamViewStmt::AlterChangeStream { name, action }) => Some( change_stream::alter_change_stream(state, identity, name, action), ), NodedbStatement::StreamView(StreamViewStmt::DropChangeStream { name, if_exists }) => { - // IF EXISTS short-circuit folded from the pgwire guard: a DROP of a + // IF EXISTS short-circuit: a DROP of a // non-existing change stream returns the tag before the token // handler runs. The `if_exists: false` case and the existing-stream // case fall through to `drop_change_stream`, which re-derives the - // name / IF EXISTS from `parts` exactly as the pgwire streaming - // string dispatch did. + // name / IF EXISTS from `parts`. if *if_exists && !change_stream::change_stream_exists(state, identity, database_id, name) { @@ -57,12 +59,7 @@ pub(super) async fn try_typed( }])); } let parts: Vec<&str> = sql.split_whitespace().collect(); - Some(change_stream::drop_change_stream( - state, - identity, - database_id, - &parts, - )) + Some(change_stream::drop_change_stream(state, identity, database_id, &parts).await) } NodedbStatement::StreamView(StreamViewStmt::CreateConsumerGroup { @@ -84,12 +81,11 @@ pub(super) async fn try_typed( stream, if_exists, }) => { - // IF EXISTS short-circuit folded from the pgwire guard: a DROP of a + // IF EXISTS short-circuit: a DROP of a // non-existing consumer group returns the tag before the token // handler runs. The `if_exists: false` case and the existing-group // case fall through to `drop_consumer_group`, which re-derives the - // name / stream from `parts` exactly as the pgwire streaming string - // dispatch did. The guard checks the in-memory group registry for the + // name / stream from `parts`. The guard checks the in-memory group registry for the // identity tenant using the parsed name / stream verbatim. let tid = identity.tenant_id.as_u64(); let requested_stream = match parse_stream_ident_token(stream) { @@ -140,14 +136,13 @@ pub(super) async fn try_typed( ), NodedbStatement::StreamView(StreamViewStmt::DropMaterializedView { name, if_exists }) => { - // IF EXISTS short-circuit folded from the pgwire guard: a DROP of a + // IF EXISTS short-circuit: a DROP of a // non-existing materialized view returns the tag before the token // handler runs. The existence check reads the in-memory registry - // (`mv_registry`) for the identity tenant exactly as the pgwire guard - // did. The `if_exists: false` case and the existing-view case fall - // through to `drop_materialized_view`, which re-derives the name / IF - // EXISTS from `parts` (and runs its own catalog-based existence check) - // exactly as the pgwire admin string dispatch did. + // (`mv_registry`) for the identity tenant. The `if_exists: false` + // case and the existing-view case fall through to + // `drop_materialized_view`, which re-derives the name / IF EXISTS + // from `parts` (and runs its own catalog-based existence check). if *if_exists && !materialized_view::materialized_view_exists(state, identity, database_id, name) { @@ -157,12 +152,10 @@ pub(super) async fn try_typed( }])); } let parts: Vec<&str> = sql.split_whitespace().collect(); - Some(materialized_view::drop_materialized_view( - state, - identity, - database_id, - &parts, - )) + Some( + materialized_view::drop_materialized_view(state, identity, database_id, &parts) + .await, + ) } NodedbStatement::StreamView(StreamViewStmt::CreateContinuousAggregate { @@ -193,14 +186,13 @@ pub(super) async fn try_typed( name, if_exists, }) => { - // IF EXISTS short-circuit folded from the pgwire guard: a DROP of a + // IF EXISTS short-circuit: a DROP of a // non-existing continuous aggregate returns the tag before the token // handler runs. The existence check reads the in-memory registry - // (`mv_registry`) for the identity tenant exactly as the pgwire guard - // did. The `if_exists: false` case and the existing-aggregate case - // fall through to `drop_continuous_aggregate`, which re-derives the - // name from `parts[3]` exactly as the pgwire admin string dispatch - // did. + // (`mv_registry`) for the identity tenant. The `if_exists: false` + // case and the existing-aggregate case fall through to + // `drop_continuous_aggregate`, which re-derives the name from + // `parts[3]`. if *if_exists && !continuous_agg::continuous_aggregate_exists(state, identity, database_id, name) { diff --git a/nodedb/src/control/server/shared/ddl/neutral/schedule/alter.rs b/nodedb/src/control/server/shared/ddl/neutral/schedule/alter.rs index 2ddf29142..bd8d47a66 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/schedule/alter.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/schedule/alter.rs @@ -2,11 +2,9 @@ //! Protocol-neutral `ALTER SCHEDULE` DDL handler. //! -//! Ported from the pgwire `ddl::schedule::alter` handler. The registry lookup, -//! the `propose_and_apply` catalog write, and the in-memory registry update are -//! preserved verbatim; only the result construction changed from pgwire -//! `Response` / `PgWireError` to the protocol-neutral [`DdlResult`] / -//! [`DdlError`]. +//! The registry lookup, the `propose_and_apply` catalog write, and the +//! in-memory registry update run here. The result is the protocol-neutral +//! [`DdlResult`] / [`DdlError`]. //! //! Supports: ENABLE, DISABLE, SET CRON 'expr'. @@ -21,7 +19,7 @@ use super::super::auth_support::status; /// /// `name`, `action`, and `cron_expr` come from the typed /// `AutomationStmt::AlterSchedule` variant. -pub fn alter_schedule( +pub async fn alter_schedule( state: &SharedState, identity: &AuthenticatedIdentity, database_id: crate::types::DatabaseId, @@ -72,7 +70,7 @@ pub fn alter_schedule( // OWNERS row. The earlier direct `catalog.put_schedule(&def)` // call did neither — divergence on replicas, orphan on disk. let entry = crate::control::catalog_entry::CatalogEntry::PutSchedule(Box::new(def.clone())); - super::super::super::catalog::propose_and_apply(state, &entry)?; + super::super::super::catalog::propose_and_apply_async(state, &entry).await?; // Update in-memory registry. state.schedule_registry.update(def); diff --git a/nodedb/src/control/server/shared/ddl/neutral/schedule/create.rs b/nodedb/src/control/server/shared/ddl/neutral/schedule/create.rs index 59be7aa7b..f7e2d9a54 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/schedule/create.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/schedule/create.rs @@ -2,12 +2,9 @@ //! Protocol-neutral `CREATE SCHEDULE` DDL handler. //! -//! Ported from the pgwire `ddl::schedule::create` handler. The catalog path -//! (`propose_and_apply` + `LocalOnly` local registry refresh, the -//! `_schedules` CRDT-sync delta enqueue, and the `audit_record` call) is -//! preserved verbatim; only the result construction changed from pgwire -//! `Response` / `PgWireError` to the protocol-neutral [`DdlResult`] / -//! [`DdlError`]. +//! The catalog path (`propose_and_apply`, the `_schedules` CRDT-sync delta +//! enqueue, and the `audit_record` call) runs here. The result is the +//! protocol-neutral [`DdlResult`] / [`DdlError`]. //! //! Syntax: //! ```sql @@ -38,7 +35,7 @@ pub struct CreateScheduleRequest<'a> { } /// Handle `CREATE SCHEDULE`. -pub fn create_schedule( +pub async fn create_schedule( state: &SharedState, identity: &AuthenticatedIdentity, database_id: crate::types::DatabaseId, @@ -108,10 +105,7 @@ pub fn create_schedule( }; let entry = crate::control::catalog_entry::CatalogEntry::PutSchedule(Box::new(def.clone())); - let outcome = super::super::super::catalog::propose_and_apply(state, &entry)?; - if outcome.needs_local_apply() { - state.schedule_registry.register(def.clone()); - } + super::super::super::catalog::propose_and_apply_async(state, &entry).await?; { let delta_payload = zerompk::to_msgpack_vec(&def).unwrap_or_default(); diff --git a/nodedb/src/control/server/shared/ddl/neutral/schedule/drop.rs b/nodedb/src/control/server/shared/ddl/neutral/schedule/drop.rs index 6848c0c3c..3ec0cc12e 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/schedule/drop.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/schedule/drop.rs @@ -2,12 +2,9 @@ //! Protocol-neutral `DROP SCHEDULE` DDL handler. //! -//! Ported from the pgwire `ddl::schedule::drop` handler. The original -//! `propose_catalog_entry` + `LocalOnly` local-delete fallback (direct -//! `catalog.delete_schedule` + in-memory registry unregister), the `_schedules` -//! CRDT-sync tombstone delta, and the `audit_record` call are preserved -//! verbatim; only the result construction changed from pgwire `Response` / -//! `PgWireError` to the protocol-neutral [`DdlResult`] / [`DdlError`]. +//! The `propose_catalog_entry`, the `_schedules` CRDT-sync tombstone delta, +//! and the `audit_record` call run here. The result is the protocol-neutral +//! [`DdlResult`] / [`DdlError`]. use crate::control::security::identity::AuthenticatedIdentity; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; @@ -17,7 +14,7 @@ use super::super::super::result::{DdlError, DdlResult}; use super::super::auth_support::{require_tenant_admin, status}; /// Existence check used by the `DROP SCHEDULE IF EXISTS` short-circuit in the -/// neutral router. Mirrors the pgwire `exists::schedule_exists` helper. +/// neutral router. pub fn schedule_exists( state: &SharedState, identity: &AuthenticatedIdentity, @@ -31,7 +28,7 @@ pub fn schedule_exists( } /// Handle `DROP SCHEDULE [IF EXISTS] ` -pub fn drop_schedule( +pub async fn drop_schedule( state: &SharedState, identity: &AuthenticatedIdentity, database_id: crate::types::DatabaseId, @@ -56,8 +53,6 @@ pub fn drop_schedule( let tenant_id = identity.tenant_id.as_u64(); - let catalog = state.credentials.catalog(); - // Pre-check existence: `IF EXISTS` + missing is a no-op that // doesn't touch raft. Check via the in-memory registry since // `schedules.rs` has no `get_schedule` method today. @@ -80,25 +75,9 @@ pub fn drop_schedule( tenant_id, name: name.clone(), }; - let outcome = crate::control::metadata_proposer::propose_catalog_entry(state, &entry) + crate::control::metadata_proposer::propose_catalog_entry_async(state, &entry) + .await .map_err(|e| DdlError::from_error_in_context("metadata propose", &e))?; - if outcome.needs_local_apply() { - let _ = catalog - .delete_schedule_in_database(database_id, tenant_id, &name) - .map_err(|e| DdlError::from_error_in_context("catalog delete", &e))?; - catalog - .delete_owner("schedule", database_id.as_u64(), tenant_id, &name) - .map_err(|e| DdlError::from_error_in_context("catalog owner delete", &e))?; - state - .schedule_registry - .unregister(database_id, tenant_id, &name); - state.permissions.install_replicated_remove_owner( - "schedule", - database_id.as_u64(), - tenant_id, - &name, - ); - } // Emit tombstone delta for Lite visibility (removes schedule from Lite catalog). { diff --git a/nodedb/src/control/server/shared/ddl/neutral/schedule/show.rs b/nodedb/src/control/server/shared/ddl/neutral/schedule/show.rs index 4bd6dfc72..5a04e9ce3 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/schedule/show.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/schedule/show.rs @@ -2,10 +2,8 @@ //! Protocol-neutral `SHOW SCHEDULES` and `SHOW SCHEDULE HISTORY` DDL handlers. //! -//! Ported from the pgwire `ddl::schedule::show` handlers. The registry read, -//! `job_history` lookups, next-fire computation, and the exact column set / -//! per-column text encoding are preserved verbatim; only the result -//! construction changed from pgwire `Response` / `QueryResponse` to the +//! The registry read, `job_history` lookups, next-fire computation, and the +//! exact column set / per-column text encoding run here. The result is the //! protocol-neutral [`DdlResult::Rows`] over [`ShapedRows`]. use serde_json::{Map, Value as JsonValue}; diff --git a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/define.rs b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/define.rs index 1897ea1a0..cdc9bed18 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/define.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/define.rs @@ -50,8 +50,9 @@ pub fn define_scope( return Err(err("42601", "INCLUDE requires a scope name")); } } - "READ" | "WRITE" | "CREATE" | "DROP" | "ALTER" | "ADMIN" => { - // ON + "READ" | "WRITE" | "CREATE" | "DROP" | "ALTER" | "ADMIN" | "BACKUP" => { + // ON . `BACKUP ON 'tenant:'` covers + // the backups and restores of that tenant. if i + 2 < def_parts.len() && def_parts[i + 1].to_uppercase() == "ON" { let coll = def_parts[i + 2].trim_matches('\'').trim_end_matches(','); grants.push((token.to_lowercase(), coll.to_string())); diff --git a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/grant.rs b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/grant.rs index d4fb7d62a..7cc254ca4 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/grant.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/grant.rs @@ -22,18 +22,24 @@ use super::support::{err, status}; /// The replication itself lives with the rest of the scope-grant apply path so /// the expiry sweep and the DDL handlers cannot drift apart; this wrapper only /// translates the error into what pgwire reports. -fn propose_scope_grant(state: &SharedState, stored: &StoredScopeGrant) -> Result<(), DdlError> { - propose_grant(state, stored).map_err(|e| DdlError::from_error(&e)) +async fn propose_scope_grant( + state: &SharedState, + stored: &StoredScopeGrant, +) -> Result<(), DdlError> { + propose_grant(state, stored) + .await + .map_err(|e| DdlError::from_error(&e)) } /// Replicate a scope-grant removal. Same dual path as [`propose_scope_grant`]. -fn propose_scope_revoke( +async fn propose_scope_revoke( state: &SharedState, scope_name: &str, grantee_type: &str, grantee_id: &str, ) -> Result<(), DdlError> { propose_revoke(state, scope_name, grantee_type, grantee_id) + .await .map_err(|e| DdlError::from_error(&e)) } @@ -45,7 +51,7 @@ fn propose_scope_revoke( /// /// The expiry clauses retire the grant on a wall clock; the condition /// clauses leave it granted but decide, per request, whether it applies. -pub fn grant_scope( +pub async fn grant_scope( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -93,7 +99,7 @@ pub fn grant_scope( conditions, }) .map_err(|e| DdlError::from_error(&e))?; - propose_scope_grant(state, &stored)?; + propose_scope_grant(state, &stored).await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, @@ -109,7 +115,7 @@ pub fn grant_scope( } /// REVOKE SCOPE '' FROM '' -pub fn revoke_scope( +pub async fn revoke_scope( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -127,7 +133,7 @@ pub fn revoke_scope( let grantee_type = parts[4].to_lowercase(); let grantee_id = parts[5].trim_matches('\''); - propose_scope_revoke(state, scope_name, &grantee_type, grantee_id)?; + propose_scope_revoke(state, scope_name, &grantee_type, grantee_id).await?; state.audit_record( crate::control::security::audit::AuditEvent::AdminAction, @@ -140,7 +146,7 @@ pub fn revoke_scope( } /// RENEW SCOPE '' FOR '' EXTEND BY -pub fn renew_scope( +pub async fn renew_scope( state: &SharedState, identity: &AuthenticatedIdentity, parts: &[&str], @@ -173,7 +179,7 @@ pub fn renew_scope( RenewOutcome::NotFound => return Err(err("42704", "scope grant not found")), // Nothing to move: a permanent grant has no deadline to extend. RenewOutcome::AlreadyPermanent => {} - RenewOutcome::Extend(stored) => propose_scope_grant(state, &stored)?, + RenewOutcome::Extend(stored) => propose_scope_grant(state, &stored).await?, } state.audit_record( @@ -240,7 +246,7 @@ pub fn show_scope_grants( }), ); // An operator debugging "why isn't this grant applying?" needs - // to see the conditions attached to it, not just its expiry. + // to see the conditions attached to it, not only its expiry. row.insert( "conditions".to_string(), JsonValue::String(render_conditions(&g.conditions)), diff --git a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/mod.rs b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/mod.rs index 54096cb74..c14137eba 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/mod.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/mod.rs @@ -17,10 +17,8 @@ //! SHOW SCOPE GRANTS //! ``` //! -//! Ported from the pgwire `ddl::scope_ddl` handlers. The superuser gate, -//! `scope_defs` / `scope_grants` catalog mutations, and `audit_record` side -//! effects are preserved verbatim; only the result construction changed from -//! pgwire `Response` / `QueryResponse` / `Tag` to the protocol-neutral +//! The superuser gate, `scope_defs` / `scope_grants` catalog mutations, and +//! `audit_record` side effects run here. The result is the protocol-neutral //! [`super::super::result::DdlResult`] over //! [`crate::control::server::response_shape::types::ShapedRows`]. diff --git a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/support.rs b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/support.rs index b4ff0ad86..3456e98bd 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/support.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/scope_ddl/support.rs @@ -4,8 +4,7 @@ use super::super::super::result::{DdlError, DdlResult}; -/// Construct a [`DdlError`], preserving the exact SQLSTATE codes and messages -/// the pgwire scope handlers produced (via `sqlstate_error`). +/// Construct a [`DdlError`] from a SQLSTATE code and a message. pub(super) fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/scope_query_ddl.rs b/nodedb/src/control/server/shared/ddl/neutral/scope_query_ddl.rs index 178c66df8..cb31e5d18 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/scope_query_ddl.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/scope_query_ddl.rs @@ -3,10 +3,8 @@ //! Protocol-neutral scope query DDL commands: ALTER SCOPE, SHOW MY SCOPES, //! SHOW SCOPES FOR. //! -//! Ported from the pgwire `ddl::scope_query_ddl` handlers. The superuser gate, -//! `scope_defs` / `scope_grants` / `orgs` catalog reads and mutations, and -//! `audit_record` side effects are preserved verbatim; only the result -//! construction changed from pgwire `Response` / `QueryResponse` / `Tag` to the +//! The superuser gate, `scope_defs` / `scope_grants` / `orgs` catalog reads +//! and mutations, and `audit_record` side effects run here. The result is the //! protocol-neutral [`DdlResult`] over [`ShapedRows`]. use serde_json::{Map, Value as JsonValue}; @@ -17,8 +15,7 @@ use crate::control::state::SharedState; use super::super::result::{DdlError, DdlResult}; -/// Construct a [`DdlError`], preserving the exact SQLSTATE codes and messages -/// the pgwire handlers produced (via `sqlstate_error`). +/// Construct a [`DdlError`] from a SQLSTATE code and a message. fn err(sqlstate: &str, message: impl Into) -> DdlError { DdlError::new(sqlstate, message) } diff --git a/nodedb/src/control/server/shared/ddl/neutral/sequence.rs b/nodedb/src/control/server/shared/ddl/neutral/sequence.rs index fdde1d9a9..0ac2795b5 100644 --- a/nodedb/src/control/server/shared/ddl/neutral/sequence.rs +++ b/nodedb/src/control/server/shared/ddl/neutral/sequence.rs @@ -2,24 +2,20 @@ //! Protocol-neutral sequence DDL — CREATE / ALTER / DROP / SHOW / DESCRIBE. //! -//! Ported from the pgwire `ddl::sequence` handlers. All non-return logic -//! (StoredSequence build, validation, sequence-registry ops, catalog proposes, -//! `schema_version.bump()`, tenant scoping) is preserved verbatim; only the -//! result construction changed from pgwire `Response` / `PgWireError` to the -//! protocol-neutral [`DdlResult`] / [`DdlError`]. +//! The StoredSequence build, validation, sequence-registry ops, catalog +//! proposes, `schema_version.bump()`, and tenant scoping run here. The result +//! is the protocol-neutral [`DdlResult`] / [`DdlError`]. use serde_json::{Map, Value as JsonValue}; use crate::control::security::catalog::sequence_types::StoredSequence; use crate::control::security::identity::AuthenticatedIdentity; -use crate::control::sequence::SequenceError; -use crate::control::sequence::error_map::sequence_error_to_error; use crate::control::server::response_shape::types::ShapedRows; use crate::control::server::shared::ddl::sql_parse::parse_ident_token; use crate::control::state::SharedState; use crate::types::DatabaseId; -use super::super::catalog::propose_and_apply; +use super::super::catalog::propose_and_apply_async; use super::super::result::{DdlError, DdlResult}; /// A `CREATE SEQUENCE` request. Fields mirror the typed @@ -36,10 +32,6 @@ pub struct CreateSequenceRequest<'a> { pub format_template_raw: Option<&'a str>, pub reset_period_raw: Option<&'a str>, pub gap_free: bool, - /// Parsed but not yet consumed by the handler (mirrors the pgwire - /// handler's ignored `_scope` argument); retained for field fidelity. - #[allow(dead_code)] - pub scope: Option<&'a str>, } /// Build a single-tag status result. @@ -52,12 +44,11 @@ fn status(command: &str) -> Vec { /// Handle `CREATE [IF NOT EXISTS] SEQUENCE [options…]`. /// -/// The `IF NOT EXISTS` existence short-circuit is folded in from the pgwire -/// guard: an already-existing sequence returns the tag before any option -/// parsing, preserving the guard's error-free early return. The `IF NOT +/// The `IF NOT EXISTS` existence short-circuit runs first: an already-existing +/// sequence returns the tag before any option parsing, without an error. The `IF NOT /// EXISTS` + non-existing case never reaches here — the neutral router returns /// `None` for it so dispatch falls through to the planner, matching today. -pub fn create_sequence( +pub async fn create_sequence( state: &SharedState, identity: &AuthenticatedIdentity, database_id: DatabaseId, @@ -66,9 +57,9 @@ pub fn create_sequence( let tenant_id = identity.tenant_id.as_u64(); let db = database_id.as_u64(); - // IF NOT EXISTS: swallow duplicate-creation (folded from the pgwire guard). + // IF NOT EXISTS: swallow duplicate-creation. // Placed before any building so an existing sequence returns the tag - // without option parsing — matching the guard's behavior exactly. + // without option parsing. if req.if_not_exists && state.sequence_registry.exists(db, tenant_id, req.name) { return Ok(status("CREATE SEQUENCE")); } @@ -131,41 +122,16 @@ pub fn create_sequence( )); } - let entry = crate::control::catalog_entry::CatalogEntry::PutSequence(Box::new(def.clone())); - let outcome = propose_and_apply(state, &entry)?; - if outcome.needs_local_apply() { - let name = def.name.clone(); - state - .sequence_registry - .create(def) - .map_err(|e| create_refusal(&name, e))?; - } + let entry = crate::control::catalog_entry::CatalogEntry::PutSequence(Box::new(def)); + propose_and_apply_async(state, &entry).await?; state.schema_version.bump(); Ok(status("CREATE SEQUENCE")) } -/// The DDL error for a registry refusal of a new sequence. A name taken since -/// the existence check is a duplicate object. Every other refusal keeps the -/// SQLSTATE the sequence error map gives it. -fn create_refusal(name: &str, error: SequenceError) -> DdlError { - match error { - SequenceError::AlreadyExists { .. } => DdlError::new("42P07", error.to_string()), - other @ (SequenceError::Exhausted { .. } - | SequenceError::NotYetCalled { .. } - | SequenceError::OutOfRange { .. } - | SequenceError::NotFound { .. } - | SequenceError::InvalidDefinition { .. } - | SequenceError::FormatParse { .. } - | SequenceError::InvalidResetScope { .. }) => { - DdlError::from_error(&sequence_error_to_error(name, other)) - } - } -} - /// Handle `ALTER SEQUENCE RESTART [WITH ] | FORMAT '